diff --git a/src/lib/pageHighlight.ts b/src/lib/pageHighlight.ts index b230ab4..48824c8 100644 --- a/src/lib/pageHighlight.ts +++ b/src/lib/pageHighlight.ts @@ -1,68 +1,54 @@ +import { + findPhraseSpans, + findMatchSpans, + tokenize, + type Span, +} from "$lib/textMatch"; + const HIGHLIGHT_NAME = "search-terms"; function supported(): boolean { return typeof CSS !== "undefined" && "highlights" in CSS; } -function escapeRe(s: string): string { - return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); -} - // Highlight the query inside the given roots using the CSS Custom Highlight // API, which paints ranges without touching the DOM, so it never conflicts // with Svelte re-renders. When the whole query appears contiguously anywhere -// on the page, only those phrase occurrences are painted; per-word marks are -// a fallback for pages that matched on scattered terms. Returns the first -// match in document order (for scrolling), or null. No-op on unsupported -// browsers. +// on the page, only those phrase occurrences are painted; otherwise the +// textMatch fallback (contiguous runs as blocks, then single words) is used. +// Returns the first match in document order (for scrolling), or null. No-op +// on unsupported browsers. export function applyHighlights( roots: Iterable, query: string, ): Range | null { if (!supported()) return null; CSS.highlights.delete(HIGHLIGHT_NAME); - const terms = [ - ...new Set(query.toLowerCase().split(/\s+/).filter((t) => t.length >= 2)), - ]; - if (terms.length === 0) return null; - const phraseRe = - terms.length > 1 - ? new RegExp(terms.map(escapeRe).join("\\s+"), "gi") - : null; + if (tokenize(query).length === 0) return null; - const phraseRanges: Range[] = []; - const termRanges: Range[] = []; + const nodes: Node[] = []; for (const root of roots) { const walker = document.createTreeWalker(root, NodeFilter.SHOW_TEXT); let node: Node | null; - while ((node = walker.nextNode())) { - const text = node.textContent ?? ""; - - if (phraseRe) { - phraseRe.lastIndex = 0; - let m: RegExpExecArray | null; - while ((m = phraseRe.exec(text))) { - const range = new Range(); - range.setStart(node, m.index); - range.setEnd(node, m.index + m[0].length); - phraseRanges.push(range); - } - } - - const lower = text.toLowerCase(); - for (const term of terms) { - let i = 0; - while ((i = lower.indexOf(term, i)) !== -1) { - const range = new Range(); - range.setStart(node, i); - range.setEnd(node, i + term.length); - termRanges.push(range); - i += term.length; - } - } - } + while ((node = walker.nextNode())) nodes.push(node); } - const ranges = phraseRanges.length > 0 ? phraseRanges : termRanges; + + let perNode: Span[][] = nodes.map((n) => + findPhraseSpans(n.textContent ?? "", query), + ); + if (perNode.every((spans) => spans.length === 0)) { + perNode = nodes.map((n) => findMatchSpans(n.textContent ?? "", query)); + } + + const ranges: Range[] = []; + nodes.forEach((node, i) => { + for (const span of perNode[i]) { + const range = new Range(); + range.setStart(node, span.start); + range.setEnd(node, span.end); + ranges.push(range); + } + }); if (ranges.length === 0) return null; CSS.highlights.set(HIGHLIGHT_NAME, new Highlight(...ranges)); diff --git a/src/lib/search.ts b/src/lib/search.ts index e9a38a9..c779206 100644 --- a/src/lib/search.ts +++ b/src/lib/search.ts @@ -1,6 +1,7 @@ import type { Event } from "@nostr/tools/core"; import type { Filter } from "@nostr/tools/filter"; import { queryForum } from "$lib/relay"; +import { bestMatch } from "$lib/textMatch"; import { GROUP_ID, MODE } from "$lib/config"; import { groupsStore } from "$lib/groups.svelte"; @@ -12,49 +13,17 @@ export type SearchResult = { createdAt: number; }; -// Window the snippet around the best match: a full-phrase occurrence when -// present, otherwise the term cluster covering the most distinct query -// terms. The score ranks how well this content matched (phrase beats any -// scattered cluster), so dedupe can keep the best snippet per thread. +// Window the snippet around the best match (full phrase > longest +// contiguous run > densest single-term cluster, per textMatch rules). The +// score ranks how well this content matched, so dedupe can keep the best +// snippet per thread and the result list can order its tiers. function snippetOf( content: string, query: string, ): { text: string; score: number } { const flat = content.replace(/\s+/g, " ").trim(); const MAX = 140; - const lower = flat.toLowerCase(); - const phrase = query.toLowerCase().trim().replace(/\s+/g, " "); - const terms = [...new Set(phrase.split(" "))].filter(Boolean); - - const occurrences: { i: number; term: string }[] = []; - for (const t of terms) { - let i = 0; - while ((i = lower.indexOf(t, i)) !== -1) { - occurrences.push({ i, term: t }); - i += t.length; - } - } - occurrences.sort((a, b) => a.i - b.i); - - let anchor = occurrences[0]?.i ?? -1; - let score = 0; - const phraseIdx = terms.length > 1 ? lower.indexOf(phrase) : -1; - if (phraseIdx !== -1) { - anchor = phraseIdx; - score = terms.length + 100; // Full phrase beats any scattered cluster - } else { - const span = MAX - 40; // Visible chars from the anchor to the window's end - for (const o of occurrences) { - const seen = new Set(); - for (const p of occurrences) { - if (p.i >= o.i && p.i + p.term.length <= o.i + span) seen.add(p.term); - } - if (seen.size > score) { - score = seen.size; - anchor = o.i; - } - } - } + const { anchor, score } = bestMatch(flat, query, MAX - 40); if (flat.length <= MAX) return { text: flat, score }; if (anchor <= 40) return { text: flat.slice(0, MAX) + "…", score }; @@ -129,7 +98,13 @@ export async function searchThreads(query: string): Promise { }), ); - // Keep the relay's relevance order (arrival order); a thread keeps the - // position of its best-ranked match - return [...byThread.values()]; + // Tiered ordering: full-phrase matches, then contiguous multi-word runs + // (longest first), then single-word matches capped to keep noise down. + // Within a tier the relay's relevance (arrival) order is preserved — + // Array.prototype.sort is stable. + const SINGLES_LIMIT = 5; + let singles = 0; + return [...byThread.values()] + .sort((a, b) => b.score - a.score) + .filter((r) => r.score >= 100 || ++singles <= SINGLES_LIMIT); } diff --git a/src/lib/searchState.svelte.ts b/src/lib/searchState.svelte.ts index ef8c186..61899e0 100644 --- a/src/lib/searchState.svelte.ts +++ b/src/lib/searchState.svelte.ts @@ -1,4 +1,5 @@ import { searchThreads, type SearchResult } from "$lib/search"; +import { findPhraseSpans, findMatchSpans } from "$lib/textMatch"; // Debounced-search state machine shared by the inline (homepage) and modal // search shells, so behavior lives in one place and the shells only differ @@ -12,10 +13,6 @@ export function createSearchState() { let timer: ReturnType | null = null; let seq = 0; - const terms = $derived( - resultsQuery.split(/\s+/).filter((t) => t.length >= 2), - ); - function schedule() { const q = query.trim(); if (timer) clearTimeout(timer); @@ -41,29 +38,26 @@ export function createSearchState() { }, 300); } - function escapeRe(s: string): string { - return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); - } - // Split into alternating plain/matched segments for rendering. // When the text contains the whole query as a contiguous phrase, only the - // phrase is marked; scattered single words are marked only as a fallback, - // to show why a phrase-less text matched at all. + // phrase is marked; otherwise contiguous runs (as single blocks) and + // single words are marked as a fallback, to show why the text matched. function highlight(text: string): { text: string; hit: boolean }[] { - if (!text || terms.length === 0) return [{ text, hit: false }]; - const phrase = - terms.length > 1 - ? escapeRe(resultsQuery.trim()).replace(/\s+/g, "\\s+") - : null; - const alts = - phrase && new RegExp(phrase, "i").test(text) - ? phrase - : terms.map(escapeRe).join("|"); - const exact = new RegExp(`^(${alts})$`, "i"); - return text - .split(new RegExp(`(${alts})`, "gi")) - .filter((s) => s !== "") - .map((s) => ({ text: s, hit: exact.test(s) })); + if (!text) return [{ text, hit: false }]; + const phrase = findPhraseSpans(text, resultsQuery); + const spans = + phrase.length > 0 ? phrase : findMatchSpans(text, resultsQuery); + if (spans.length === 0) return [{ text, hit: false }]; + const out: { text: string; hit: boolean }[] = []; + let pos = 0; + for (const s of spans) { + if (s.start > pos) + out.push({ text: text.slice(pos, s.start), hit: false }); + out.push({ text: text.slice(s.start, s.end), hit: true }); + pos = s.end; + } + if (pos < text.length) out.push({ text: text.slice(pos), hit: false }); + return out; } // Arrow/Enter handling; returns true when the event was consumed diff --git a/src/lib/textMatch.ts b/src/lib/textMatch.ts new file mode 100644 index 0000000..33962aa --- /dev/null +++ b/src/lib/textMatch.ts @@ -0,0 +1,121 @@ +// Shared matching rules for search highlighting and ranking. The query is +// tokenized on any non-alphanumeric character (so "-_|!./" all act as word +// separators, in the query and in the text), and a token matches where a +// word starts or ends with it — "id" marks "id" or "identify" but never the +// middle of "gravida", while still allowing prefix searches like the first +// characters of an npub. + +export type Span = { start: number; end: number }; + +// Separator between adjacent tokens of a run: one or more non-word chars +const SEP = "[^\\p{L}\\p{N}]+"; + +export function tokenize(query: string): string[] { + return query + .toLowerCase() + .split(/[^\p{L}\p{N}]+/u) + .filter((t) => t.length >= 2); +} + +function isWordChar(ch: string | undefined): boolean { + return ch !== undefined && /[\p{L}\p{N}]/u.test(ch); +} + +// Start-with or end-with: the match must begin at a word start or finish at +// a word end. +function validBoundaries(text: string, span: Span): boolean { + return !isWordChar(text[span.start - 1]) || !isWordChar(text[span.end]); +} + +function overlaps(taken: Span[], s: Span): boolean { + return taken.some((t) => s.start < t.end && s.end > t.start); +} + +// All spans where the given tokens appear consecutively (any separators +// between them). Tokens are alphanumeric-only, so no regex escaping needed. +function findRuns(text: string, tokens: string[]): Span[] { + const re = new RegExp(tokens.join(SEP), "giu"); + const out: Span[] = []; + let m: RegExpExecArray | null; + while ((m = re.exec(text))) { + const span = { start: m.index, end: m.index + m[0].length }; + if (validBoundaries(text, span)) out.push(span); + if (m[0].length === 0) re.lastIndex++; + } + return out; +} + +// Occurrences of the whole query as one contiguous phrase +export function findPhraseSpans(text: string, query: string): Span[] { + const tokens = tokenize(query); + if (tokens.length < 2) return []; + return findRuns(text, tokens); +} + +// Fallback matches: longest contiguous multi-token runs claim their spans +// first (each painted as one block), then single tokens outside them. +export function findMatchSpans(text: string, query: string): Span[] { + const tokens = tokenize(query); + if (tokens.length === 0) return []; + const spans: Span[] = []; + for (let len = tokens.length; len >= 2; len--) { + for (let s = 0; s + len <= tokens.length; s++) { + for (const span of findRuns(text, tokens.slice(s, s + len))) { + if (!overlaps(spans, span)) spans.push(span); + } + } + } + for (const t of new Set(tokens)) { + for (const span of findRuns(text, [t])) { + if (!overlaps(spans, span)) spans.push(span); + } + } + return spans.sort((a, b) => a.start - b.start); +} + +// Rank how well a text matches the query, and where to anchor an excerpt: +// the full query (1000) beats any shorter contiguous run (100 + length), +// which beats scattered singles (count of distinct tokens inside a window of +// `windowSpan` chars, always < 100). +export function bestMatch( + text: string, + query: string, + windowSpan: number, +): { anchor: number; score: number } { + const tokens = tokenize(query); + if (tokens.length === 0) return { anchor: -1, score: 0 }; + + for (let len = tokens.length; len >= 2; len--) { + for (let s = 0; s + len <= tokens.length; s++) { + const runs = findRuns(text, tokens.slice(s, s + len)); + if (runs.length > 0) + return { + anchor: runs[0].start, + score: len === tokens.length ? 1000 : 100 + len, + }; + } + } + + const singles: { start: number; token: string }[] = []; + for (const t of new Set(tokens)) { + for (const span of findRuns(text, [t])) + singles.push({ start: span.start, token: t }); + } + if (singles.length === 0) return { anchor: -1, score: 0 }; + singles.sort((a, b) => a.start - b.start); + + let anchor = singles[0].start; + let score = 0; + for (const o of singles) { + const seen = new Set(); + for (const p of singles) { + if (p.start >= o.start && p.start <= o.start + windowSpan) + seen.add(p.token); + } + if (seen.size > score) { + score = seen.size; + anchor = o.start; + } + } + return { anchor, score }; +}