Match search terms at word starts/ends and rank results in tiers

This commit is contained in:
dtonon 2026-08-13 21:37:38 +01:00
parent dadba06f8c
commit aaa9a0c1e5
4 changed files with 185 additions and 109 deletions

View file

@ -1,68 +1,54 @@
import {
findPhraseSpans,
findMatchSpans,
tokenize,
type Span,
} from "$lib/textMatch";
const HIGHLIGHT_NAME = "search-terms"; const HIGHLIGHT_NAME = "search-terms";
function supported(): boolean { function supported(): boolean {
return typeof CSS !== "undefined" && "highlights" in CSS; return typeof CSS !== "undefined" && "highlights" in CSS;
} }
function escapeRe(s: string): string {
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}
// Highlight the query inside the given roots using the CSS Custom Highlight // Highlight the query inside the given roots using the CSS Custom Highlight
// API, which paints ranges without touching the DOM, so it never conflicts // API, which paints ranges without touching the DOM, so it never conflicts
// with Svelte re-renders. When the whole query appears contiguously anywhere // with Svelte re-renders. When the whole query appears contiguously anywhere
// on the page, only those phrase occurrences are painted; per-word marks are // on the page, only those phrase occurrences are painted; otherwise the
// a fallback for pages that matched on scattered terms. Returns the first // textMatch fallback (contiguous runs as blocks, then single words) is used.
// match in document order (for scrolling), or null. No-op on unsupported // Returns the first match in document order (for scrolling), or null. No-op
// browsers. // on unsupported browsers.
export function applyHighlights( export function applyHighlights(
roots: Iterable<Element>, roots: Iterable<Element>,
query: string, query: string,
): Range | null { ): Range | null {
if (!supported()) return null; if (!supported()) return null;
CSS.highlights.delete(HIGHLIGHT_NAME); CSS.highlights.delete(HIGHLIGHT_NAME);
const terms = [ if (tokenize(query).length === 0) return null;
...new Set(query.toLowerCase().split(/\s+/).filter((t) => t.length >= 2)),
];
if (terms.length === 0) return null;
const phraseRe =
terms.length > 1
? new RegExp(terms.map(escapeRe).join("\\s+"), "gi")
: null;
const phraseRanges: Range[] = []; const nodes: Node[] = [];
const termRanges: Range[] = [];
for (const root of roots) { for (const root of roots) {
const walker = document.createTreeWalker(root, NodeFilter.SHOW_TEXT); const walker = document.createTreeWalker(root, NodeFilter.SHOW_TEXT);
let node: Node | null; let node: Node | null;
while ((node = walker.nextNode())) { while ((node = walker.nextNode())) nodes.push(node);
const text = node.textContent ?? "";
if (phraseRe) {
phraseRe.lastIndex = 0;
let m: RegExpExecArray | null;
while ((m = phraseRe.exec(text))) {
const range = new Range();
range.setStart(node, m.index);
range.setEnd(node, m.index + m[0].length);
phraseRanges.push(range);
}
} }
const lower = text.toLowerCase(); let perNode: Span[][] = nodes.map((n) =>
for (const term of terms) { findPhraseSpans(n.textContent ?? "", query),
let i = 0; );
while ((i = lower.indexOf(term, i)) !== -1) { if (perNode.every((spans) => spans.length === 0)) {
perNode = nodes.map((n) => findMatchSpans(n.textContent ?? "", query));
}
const ranges: Range[] = [];
nodes.forEach((node, i) => {
for (const span of perNode[i]) {
const range = new Range(); const range = new Range();
range.setStart(node, i); range.setStart(node, span.start);
range.setEnd(node, i + term.length); range.setEnd(node, span.end);
termRanges.push(range); ranges.push(range);
i += term.length;
} }
} });
}
}
const ranges = phraseRanges.length > 0 ? phraseRanges : termRanges;
if (ranges.length === 0) return null; if (ranges.length === 0) return null;
CSS.highlights.set(HIGHLIGHT_NAME, new Highlight(...ranges)); CSS.highlights.set(HIGHLIGHT_NAME, new Highlight(...ranges));

View file

@ -1,6 +1,7 @@
import type { Event } from "@nostr/tools/core"; import type { Event } from "@nostr/tools/core";
import type { Filter } from "@nostr/tools/filter"; import type { Filter } from "@nostr/tools/filter";
import { queryForum } from "$lib/relay"; import { queryForum } from "$lib/relay";
import { bestMatch } from "$lib/textMatch";
import { GROUP_ID, MODE } from "$lib/config"; import { GROUP_ID, MODE } from "$lib/config";
import { groupsStore } from "$lib/groups.svelte"; import { groupsStore } from "$lib/groups.svelte";
@ -12,49 +13,17 @@ export type SearchResult = {
createdAt: number; createdAt: number;
}; };
// Window the snippet around the best match: a full-phrase occurrence when // Window the snippet around the best match (full phrase > longest
// present, otherwise the term cluster covering the most distinct query // contiguous run > densest single-term cluster, per textMatch rules). The
// terms. The score ranks how well this content matched (phrase beats any // score ranks how well this content matched, so dedupe can keep the best
// scattered cluster), so dedupe can keep the best snippet per thread. // snippet per thread and the result list can order its tiers.
function snippetOf( function snippetOf(
content: string, content: string,
query: string, query: string,
): { text: string; score: number } { ): { text: string; score: number } {
const flat = content.replace(/\s+/g, " ").trim(); const flat = content.replace(/\s+/g, " ").trim();
const MAX = 140; const MAX = 140;
const lower = flat.toLowerCase(); const { anchor, score } = bestMatch(flat, query, MAX - 40);
const phrase = query.toLowerCase().trim().replace(/\s+/g, " ");
const terms = [...new Set(phrase.split(" "))].filter(Boolean);
const occurrences: { i: number; term: string }[] = [];
for (const t of terms) {
let i = 0;
while ((i = lower.indexOf(t, i)) !== -1) {
occurrences.push({ i, term: t });
i += t.length;
}
}
occurrences.sort((a, b) => a.i - b.i);
let anchor = occurrences[0]?.i ?? -1;
let score = 0;
const phraseIdx = terms.length > 1 ? lower.indexOf(phrase) : -1;
if (phraseIdx !== -1) {
anchor = phraseIdx;
score = terms.length + 100; // Full phrase beats any scattered cluster
} else {
const span = MAX - 40; // Visible chars from the anchor to the window's end
for (const o of occurrences) {
const seen = new Set<string>();
for (const p of occurrences) {
if (p.i >= o.i && p.i + p.term.length <= o.i + span) seen.add(p.term);
}
if (seen.size > score) {
score = seen.size;
anchor = o.i;
}
}
}
if (flat.length <= MAX) return { text: flat, score }; if (flat.length <= MAX) return { text: flat, score };
if (anchor <= 40) return { text: flat.slice(0, MAX) + "…", score }; if (anchor <= 40) return { text: flat.slice(0, MAX) + "…", score };
@ -129,7 +98,13 @@ export async function searchThreads(query: string): Promise<SearchResult[]> {
}), }),
); );
// Keep the relay's relevance order (arrival order); a thread keeps the // Tiered ordering: full-phrase matches, then contiguous multi-word runs
// position of its best-ranked match // (longest first), then single-word matches capped to keep noise down.
return [...byThread.values()]; // Within a tier the relay's relevance (arrival) order is preserved —
// Array.prototype.sort is stable.
const SINGLES_LIMIT = 5;
let singles = 0;
return [...byThread.values()]
.sort((a, b) => b.score - a.score)
.filter((r) => r.score >= 100 || ++singles <= SINGLES_LIMIT);
} }

View file

@ -1,4 +1,5 @@
import { searchThreads, type SearchResult } from "$lib/search"; import { searchThreads, type SearchResult } from "$lib/search";
import { findPhraseSpans, findMatchSpans } from "$lib/textMatch";
// Debounced-search state machine shared by the inline (homepage) and modal // Debounced-search state machine shared by the inline (homepage) and modal
// search shells, so behavior lives in one place and the shells only differ // search shells, so behavior lives in one place and the shells only differ
@ -12,10 +13,6 @@ export function createSearchState() {
let timer: ReturnType<typeof setTimeout> | null = null; let timer: ReturnType<typeof setTimeout> | null = null;
let seq = 0; let seq = 0;
const terms = $derived(
resultsQuery.split(/\s+/).filter((t) => t.length >= 2),
);
function schedule() { function schedule() {
const q = query.trim(); const q = query.trim();
if (timer) clearTimeout(timer); if (timer) clearTimeout(timer);
@ -41,29 +38,26 @@ export function createSearchState() {
}, 300); }, 300);
} }
function escapeRe(s: string): string {
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}
// Split into alternating plain/matched segments for <mark> rendering. // Split into alternating plain/matched segments for <mark> rendering.
// When the text contains the whole query as a contiguous phrase, only the // When the text contains the whole query as a contiguous phrase, only the
// phrase is marked; scattered single words are marked only as a fallback, // phrase is marked; otherwise contiguous runs (as single blocks) and
// to show why a phrase-less text matched at all. // single words are marked as a fallback, to show why the text matched.
function highlight(text: string): { text: string; hit: boolean }[] { function highlight(text: string): { text: string; hit: boolean }[] {
if (!text || terms.length === 0) return [{ text, hit: false }]; if (!text) return [{ text, hit: false }];
const phrase = const phrase = findPhraseSpans(text, resultsQuery);
terms.length > 1 const spans =
? escapeRe(resultsQuery.trim()).replace(/\s+/g, "\\s+") phrase.length > 0 ? phrase : findMatchSpans(text, resultsQuery);
: null; if (spans.length === 0) return [{ text, hit: false }];
const alts = const out: { text: string; hit: boolean }[] = [];
phrase && new RegExp(phrase, "i").test(text) let pos = 0;
? phrase for (const s of spans) {
: terms.map(escapeRe).join("|"); if (s.start > pos)
const exact = new RegExp(`^(${alts})$`, "i"); out.push({ text: text.slice(pos, s.start), hit: false });
return text out.push({ text: text.slice(s.start, s.end), hit: true });
.split(new RegExp(`(${alts})`, "gi")) pos = s.end;
.filter((s) => s !== "") }
.map((s) => ({ text: s, hit: exact.test(s) })); if (pos < text.length) out.push({ text: text.slice(pos), hit: false });
return out;
} }
// Arrow/Enter handling; returns true when the event was consumed // Arrow/Enter handling; returns true when the event was consumed

121
src/lib/textMatch.ts Normal file
View file

@ -0,0 +1,121 @@
// Shared matching rules for search highlighting and ranking. The query is
// tokenized on any non-alphanumeric character (so "-_|!./" all act as word
// separators, in the query and in the text), and a token matches where a
// word starts or ends with it — "id" marks "id" or "identify" but never the
// middle of "gravida", while still allowing prefix searches like the first
// characters of an npub.
export type Span = { start: number; end: number };
// Separator between adjacent tokens of a run: one or more non-word chars
const SEP = "[^\\p{L}\\p{N}]+";
export function tokenize(query: string): string[] {
return query
.toLowerCase()
.split(/[^\p{L}\p{N}]+/u)
.filter((t) => t.length >= 2);
}
function isWordChar(ch: string | undefined): boolean {
return ch !== undefined && /[\p{L}\p{N}]/u.test(ch);
}
// Start-with or end-with: the match must begin at a word start or finish at
// a word end.
function validBoundaries(text: string, span: Span): boolean {
return !isWordChar(text[span.start - 1]) || !isWordChar(text[span.end]);
}
function overlaps(taken: Span[], s: Span): boolean {
return taken.some((t) => s.start < t.end && s.end > t.start);
}
// All spans where the given tokens appear consecutively (any separators
// between them). Tokens are alphanumeric-only, so no regex escaping needed.
function findRuns(text: string, tokens: string[]): Span[] {
const re = new RegExp(tokens.join(SEP), "giu");
const out: Span[] = [];
let m: RegExpExecArray | null;
while ((m = re.exec(text))) {
const span = { start: m.index, end: m.index + m[0].length };
if (validBoundaries(text, span)) out.push(span);
if (m[0].length === 0) re.lastIndex++;
}
return out;
}
// Occurrences of the whole query as one contiguous phrase
export function findPhraseSpans(text: string, query: string): Span[] {
const tokens = tokenize(query);
if (tokens.length < 2) return [];
return findRuns(text, tokens);
}
// Fallback matches: longest contiguous multi-token runs claim their spans
// first (each painted as one block), then single tokens outside them.
export function findMatchSpans(text: string, query: string): Span[] {
const tokens = tokenize(query);
if (tokens.length === 0) return [];
const spans: Span[] = [];
for (let len = tokens.length; len >= 2; len--) {
for (let s = 0; s + len <= tokens.length; s++) {
for (const span of findRuns(text, tokens.slice(s, s + len))) {
if (!overlaps(spans, span)) spans.push(span);
}
}
}
for (const t of new Set(tokens)) {
for (const span of findRuns(text, [t])) {
if (!overlaps(spans, span)) spans.push(span);
}
}
return spans.sort((a, b) => a.start - b.start);
}
// Rank how well a text matches the query, and where to anchor an excerpt:
// the full query (1000) beats any shorter contiguous run (100 + length),
// which beats scattered singles (count of distinct tokens inside a window of
// `windowSpan` chars, always < 100).
export function bestMatch(
text: string,
query: string,
windowSpan: number,
): { anchor: number; score: number } {
const tokens = tokenize(query);
if (tokens.length === 0) return { anchor: -1, score: 0 };
for (let len = tokens.length; len >= 2; len--) {
for (let s = 0; s + len <= tokens.length; s++) {
const runs = findRuns(text, tokens.slice(s, s + len));
if (runs.length > 0)
return {
anchor: runs[0].start,
score: len === tokens.length ? 1000 : 100 + len,
};
}
}
const singles: { start: number; token: string }[] = [];
for (const t of new Set(tokens)) {
for (const span of findRuns(text, [t]))
singles.push({ start: span.start, token: t });
}
if (singles.length === 0) return { anchor: -1, score: 0 };
singles.sort((a, b) => a.start - b.start);
let anchor = singles[0].start;
let score = 0;
for (const o of singles) {
const seen = new Set<string>();
for (const p of singles) {
if (p.start >= o.start && p.start <= o.start + windowSpan)
seen.add(p.token);
}
if (seen.size > score) {
score = seen.size;
anchor = o.start;
}
}
return { anchor, score };
}