Match search terms at word starts/ends and rank results in tiers
This commit is contained in:
parent
dadba06f8c
commit
aaa9a0c1e5
4 changed files with 185 additions and 109 deletions
|
|
@ -1,68 +1,54 @@
|
|||
import {
|
||||
findPhraseSpans,
|
||||
findMatchSpans,
|
||||
tokenize,
|
||||
type Span,
|
||||
} from "$lib/textMatch";
|
||||
|
||||
const HIGHLIGHT_NAME = "search-terms";
|
||||
|
||||
function supported(): boolean {
|
||||
return typeof CSS !== "undefined" && "highlights" in CSS;
|
||||
}
|
||||
|
||||
function escapeRe(s: string): string {
|
||||
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
}
|
||||
|
||||
// Highlight the query inside the given roots using the CSS Custom Highlight
|
||||
// API, which paints ranges without touching the DOM, so it never conflicts
|
||||
// with Svelte re-renders. When the whole query appears contiguously anywhere
|
||||
// on the page, only those phrase occurrences are painted; per-word marks are
|
||||
// a fallback for pages that matched on scattered terms. Returns the first
|
||||
// match in document order (for scrolling), or null. No-op on unsupported
|
||||
// browsers.
|
||||
// on the page, only those phrase occurrences are painted; otherwise the
|
||||
// textMatch fallback (contiguous runs as blocks, then single words) is used.
|
||||
// Returns the first match in document order (for scrolling), or null. No-op
|
||||
// on unsupported browsers.
|
||||
export function applyHighlights(
|
||||
roots: Iterable<Element>,
|
||||
query: string,
|
||||
): Range | null {
|
||||
if (!supported()) return null;
|
||||
CSS.highlights.delete(HIGHLIGHT_NAME);
|
||||
const terms = [
|
||||
...new Set(query.toLowerCase().split(/\s+/).filter((t) => t.length >= 2)),
|
||||
];
|
||||
if (terms.length === 0) return null;
|
||||
const phraseRe =
|
||||
terms.length > 1
|
||||
? new RegExp(terms.map(escapeRe).join("\\s+"), "gi")
|
||||
: null;
|
||||
if (tokenize(query).length === 0) return null;
|
||||
|
||||
const phraseRanges: Range[] = [];
|
||||
const termRanges: Range[] = [];
|
||||
const nodes: Node[] = [];
|
||||
for (const root of roots) {
|
||||
const walker = document.createTreeWalker(root, NodeFilter.SHOW_TEXT);
|
||||
let node: Node | null;
|
||||
while ((node = walker.nextNode())) {
|
||||
const text = node.textContent ?? "";
|
||||
|
||||
if (phraseRe) {
|
||||
phraseRe.lastIndex = 0;
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = phraseRe.exec(text))) {
|
||||
const range = new Range();
|
||||
range.setStart(node, m.index);
|
||||
range.setEnd(node, m.index + m[0].length);
|
||||
phraseRanges.push(range);
|
||||
}
|
||||
}
|
||||
|
||||
const lower = text.toLowerCase();
|
||||
for (const term of terms) {
|
||||
let i = 0;
|
||||
while ((i = lower.indexOf(term, i)) !== -1) {
|
||||
const range = new Range();
|
||||
range.setStart(node, i);
|
||||
range.setEnd(node, i + term.length);
|
||||
termRanges.push(range);
|
||||
i += term.length;
|
||||
}
|
||||
}
|
||||
}
|
||||
while ((node = walker.nextNode())) nodes.push(node);
|
||||
}
|
||||
const ranges = phraseRanges.length > 0 ? phraseRanges : termRanges;
|
||||
|
||||
let perNode: Span[][] = nodes.map((n) =>
|
||||
findPhraseSpans(n.textContent ?? "", query),
|
||||
);
|
||||
if (perNode.every((spans) => spans.length === 0)) {
|
||||
perNode = nodes.map((n) => findMatchSpans(n.textContent ?? "", query));
|
||||
}
|
||||
|
||||
const ranges: Range[] = [];
|
||||
nodes.forEach((node, i) => {
|
||||
for (const span of perNode[i]) {
|
||||
const range = new Range();
|
||||
range.setStart(node, span.start);
|
||||
range.setEnd(node, span.end);
|
||||
ranges.push(range);
|
||||
}
|
||||
});
|
||||
if (ranges.length === 0) return null;
|
||||
|
||||
CSS.highlights.set(HIGHLIGHT_NAME, new Highlight(...ranges));
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import type { Event } from "@nostr/tools/core";
|
||||
import type { Filter } from "@nostr/tools/filter";
|
||||
import { queryForum } from "$lib/relay";
|
||||
import { bestMatch } from "$lib/textMatch";
|
||||
import { GROUP_ID, MODE } from "$lib/config";
|
||||
import { groupsStore } from "$lib/groups.svelte";
|
||||
|
||||
|
|
@ -12,49 +13,17 @@ export type SearchResult = {
|
|||
createdAt: number;
|
||||
};
|
||||
|
||||
// Window the snippet around the best match: a full-phrase occurrence when
|
||||
// present, otherwise the term cluster covering the most distinct query
|
||||
// terms. The score ranks how well this content matched (phrase beats any
|
||||
// scattered cluster), so dedupe can keep the best snippet per thread.
|
||||
// Window the snippet around the best match (full phrase > longest
|
||||
// contiguous run > densest single-term cluster, per textMatch rules). The
|
||||
// score ranks how well this content matched, so dedupe can keep the best
|
||||
// snippet per thread and the result list can order its tiers.
|
||||
function snippetOf(
|
||||
content: string,
|
||||
query: string,
|
||||
): { text: string; score: number } {
|
||||
const flat = content.replace(/\s+/g, " ").trim();
|
||||
const MAX = 140;
|
||||
const lower = flat.toLowerCase();
|
||||
const phrase = query.toLowerCase().trim().replace(/\s+/g, " ");
|
||||
const terms = [...new Set(phrase.split(" "))].filter(Boolean);
|
||||
|
||||
const occurrences: { i: number; term: string }[] = [];
|
||||
for (const t of terms) {
|
||||
let i = 0;
|
||||
while ((i = lower.indexOf(t, i)) !== -1) {
|
||||
occurrences.push({ i, term: t });
|
||||
i += t.length;
|
||||
}
|
||||
}
|
||||
occurrences.sort((a, b) => a.i - b.i);
|
||||
|
||||
let anchor = occurrences[0]?.i ?? -1;
|
||||
let score = 0;
|
||||
const phraseIdx = terms.length > 1 ? lower.indexOf(phrase) : -1;
|
||||
if (phraseIdx !== -1) {
|
||||
anchor = phraseIdx;
|
||||
score = terms.length + 100; // Full phrase beats any scattered cluster
|
||||
} else {
|
||||
const span = MAX - 40; // Visible chars from the anchor to the window's end
|
||||
for (const o of occurrences) {
|
||||
const seen = new Set<string>();
|
||||
for (const p of occurrences) {
|
||||
if (p.i >= o.i && p.i + p.term.length <= o.i + span) seen.add(p.term);
|
||||
}
|
||||
if (seen.size > score) {
|
||||
score = seen.size;
|
||||
anchor = o.i;
|
||||
}
|
||||
}
|
||||
}
|
||||
const { anchor, score } = bestMatch(flat, query, MAX - 40);
|
||||
|
||||
if (flat.length <= MAX) return { text: flat, score };
|
||||
if (anchor <= 40) return { text: flat.slice(0, MAX) + "…", score };
|
||||
|
|
@ -129,7 +98,13 @@ export async function searchThreads(query: string): Promise<SearchResult[]> {
|
|||
}),
|
||||
);
|
||||
|
||||
// Keep the relay's relevance order (arrival order); a thread keeps the
|
||||
// position of its best-ranked match
|
||||
return [...byThread.values()];
|
||||
// Tiered ordering: full-phrase matches, then contiguous multi-word runs
|
||||
// (longest first), then single-word matches capped to keep noise down.
|
||||
// Within a tier the relay's relevance (arrival) order is preserved —
|
||||
// Array.prototype.sort is stable.
|
||||
const SINGLES_LIMIT = 5;
|
||||
let singles = 0;
|
||||
return [...byThread.values()]
|
||||
.sort((a, b) => b.score - a.score)
|
||||
.filter((r) => r.score >= 100 || ++singles <= SINGLES_LIMIT);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import { searchThreads, type SearchResult } from "$lib/search";
|
||||
import { findPhraseSpans, findMatchSpans } from "$lib/textMatch";
|
||||
|
||||
// Debounced-search state machine shared by the inline (homepage) and modal
|
||||
// search shells, so behavior lives in one place and the shells only differ
|
||||
|
|
@ -12,10 +13,6 @@ export function createSearchState() {
|
|||
let timer: ReturnType<typeof setTimeout> | null = null;
|
||||
let seq = 0;
|
||||
|
||||
const terms = $derived(
|
||||
resultsQuery.split(/\s+/).filter((t) => t.length >= 2),
|
||||
);
|
||||
|
||||
function schedule() {
|
||||
const q = query.trim();
|
||||
if (timer) clearTimeout(timer);
|
||||
|
|
@ -41,29 +38,26 @@ export function createSearchState() {
|
|||
}, 300);
|
||||
}
|
||||
|
||||
function escapeRe(s: string): string {
|
||||
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
}
|
||||
|
||||
// Split into alternating plain/matched segments for <mark> rendering.
|
||||
// When the text contains the whole query as a contiguous phrase, only the
|
||||
// phrase is marked; scattered single words are marked only as a fallback,
|
||||
// to show why a phrase-less text matched at all.
|
||||
// phrase is marked; otherwise contiguous runs (as single blocks) and
|
||||
// single words are marked as a fallback, to show why the text matched.
|
||||
function highlight(text: string): { text: string; hit: boolean }[] {
|
||||
if (!text || terms.length === 0) return [{ text, hit: false }];
|
||||
const phrase =
|
||||
terms.length > 1
|
||||
? escapeRe(resultsQuery.trim()).replace(/\s+/g, "\\s+")
|
||||
: null;
|
||||
const alts =
|
||||
phrase && new RegExp(phrase, "i").test(text)
|
||||
? phrase
|
||||
: terms.map(escapeRe).join("|");
|
||||
const exact = new RegExp(`^(${alts})$`, "i");
|
||||
return text
|
||||
.split(new RegExp(`(${alts})`, "gi"))
|
||||
.filter((s) => s !== "")
|
||||
.map((s) => ({ text: s, hit: exact.test(s) }));
|
||||
if (!text) return [{ text, hit: false }];
|
||||
const phrase = findPhraseSpans(text, resultsQuery);
|
||||
const spans =
|
||||
phrase.length > 0 ? phrase : findMatchSpans(text, resultsQuery);
|
||||
if (spans.length === 0) return [{ text, hit: false }];
|
||||
const out: { text: string; hit: boolean }[] = [];
|
||||
let pos = 0;
|
||||
for (const s of spans) {
|
||||
if (s.start > pos)
|
||||
out.push({ text: text.slice(pos, s.start), hit: false });
|
||||
out.push({ text: text.slice(s.start, s.end), hit: true });
|
||||
pos = s.end;
|
||||
}
|
||||
if (pos < text.length) out.push({ text: text.slice(pos), hit: false });
|
||||
return out;
|
||||
}
|
||||
|
||||
// Arrow/Enter handling; returns true when the event was consumed
|
||||
|
|
|
|||
121
src/lib/textMatch.ts
Normal file
121
src/lib/textMatch.ts
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
// Shared matching rules for search highlighting and ranking. The query is
|
||||
// tokenized on any non-alphanumeric character (so "-_|!./" all act as word
|
||||
// separators, in the query and in the text), and a token matches where a
|
||||
// word starts or ends with it — "id" marks "id" or "identify" but never the
|
||||
// middle of "gravida", while still allowing prefix searches like the first
|
||||
// characters of an npub.
|
||||
|
||||
export type Span = { start: number; end: number };
|
||||
|
||||
// Separator between adjacent tokens of a run: one or more non-word chars
|
||||
const SEP = "[^\\p{L}\\p{N}]+";
|
||||
|
||||
export function tokenize(query: string): string[] {
|
||||
return query
|
||||
.toLowerCase()
|
||||
.split(/[^\p{L}\p{N}]+/u)
|
||||
.filter((t) => t.length >= 2);
|
||||
}
|
||||
|
||||
function isWordChar(ch: string | undefined): boolean {
|
||||
return ch !== undefined && /[\p{L}\p{N}]/u.test(ch);
|
||||
}
|
||||
|
||||
// Start-with or end-with: the match must begin at a word start or finish at
|
||||
// a word end.
|
||||
function validBoundaries(text: string, span: Span): boolean {
|
||||
return !isWordChar(text[span.start - 1]) || !isWordChar(text[span.end]);
|
||||
}
|
||||
|
||||
function overlaps(taken: Span[], s: Span): boolean {
|
||||
return taken.some((t) => s.start < t.end && s.end > t.start);
|
||||
}
|
||||
|
||||
// All spans where the given tokens appear consecutively (any separators
|
||||
// between them). Tokens are alphanumeric-only, so no regex escaping needed.
|
||||
function findRuns(text: string, tokens: string[]): Span[] {
|
||||
const re = new RegExp(tokens.join(SEP), "giu");
|
||||
const out: Span[] = [];
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = re.exec(text))) {
|
||||
const span = { start: m.index, end: m.index + m[0].length };
|
||||
if (validBoundaries(text, span)) out.push(span);
|
||||
if (m[0].length === 0) re.lastIndex++;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// Occurrences of the whole query as one contiguous phrase
|
||||
export function findPhraseSpans(text: string, query: string): Span[] {
|
||||
const tokens = tokenize(query);
|
||||
if (tokens.length < 2) return [];
|
||||
return findRuns(text, tokens);
|
||||
}
|
||||
|
||||
// Fallback matches: longest contiguous multi-token runs claim their spans
|
||||
// first (each painted as one block), then single tokens outside them.
|
||||
export function findMatchSpans(text: string, query: string): Span[] {
|
||||
const tokens = tokenize(query);
|
||||
if (tokens.length === 0) return [];
|
||||
const spans: Span[] = [];
|
||||
for (let len = tokens.length; len >= 2; len--) {
|
||||
for (let s = 0; s + len <= tokens.length; s++) {
|
||||
for (const span of findRuns(text, tokens.slice(s, s + len))) {
|
||||
if (!overlaps(spans, span)) spans.push(span);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const t of new Set(tokens)) {
|
||||
for (const span of findRuns(text, [t])) {
|
||||
if (!overlaps(spans, span)) spans.push(span);
|
||||
}
|
||||
}
|
||||
return spans.sort((a, b) => a.start - b.start);
|
||||
}
|
||||
|
||||
// Rank how well a text matches the query, and where to anchor an excerpt:
|
||||
// the full query (1000) beats any shorter contiguous run (100 + length),
|
||||
// which beats scattered singles (count of distinct tokens inside a window of
|
||||
// `windowSpan` chars, always < 100).
|
||||
export function bestMatch(
|
||||
text: string,
|
||||
query: string,
|
||||
windowSpan: number,
|
||||
): { anchor: number; score: number } {
|
||||
const tokens = tokenize(query);
|
||||
if (tokens.length === 0) return { anchor: -1, score: 0 };
|
||||
|
||||
for (let len = tokens.length; len >= 2; len--) {
|
||||
for (let s = 0; s + len <= tokens.length; s++) {
|
||||
const runs = findRuns(text, tokens.slice(s, s + len));
|
||||
if (runs.length > 0)
|
||||
return {
|
||||
anchor: runs[0].start,
|
||||
score: len === tokens.length ? 1000 : 100 + len,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
const singles: { start: number; token: string }[] = [];
|
||||
for (const t of new Set(tokens)) {
|
||||
for (const span of findRuns(text, [t]))
|
||||
singles.push({ start: span.start, token: t });
|
||||
}
|
||||
if (singles.length === 0) return { anchor: -1, score: 0 };
|
||||
singles.sort((a, b) => a.start - b.start);
|
||||
|
||||
let anchor = singles[0].start;
|
||||
let score = 0;
|
||||
for (const o of singles) {
|
||||
const seen = new Set<string>();
|
||||
for (const p of singles) {
|
||||
if (p.start >= o.start && p.start <= o.start + windowSpan)
|
||||
seen.add(p.token);
|
||||
}
|
||||
if (seen.size > score) {
|
||||
score = seen.size;
|
||||
anchor = o.start;
|
||||
}
|
||||
}
|
||||
return { anchor, score };
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue