Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 7fc1c1f9c46eb358b36b2b7e165abe8db5d15cf8
parent d849b172ca086ff74f5dd20aaee2ddf379827720
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  7 Jul 2026 14:55:28 -0400

Fix alias matching: multi-word triggers + regex-mode literals

matchAliases tokenized the query into a Set and fired only when a whole
trigger string equalled one token, so any trigger containing a space (e.g.
"graham platner") never matched. Switch to an ordered contiguous-token-
subsequence test so phrase triggers fire as whole words, in order, case- and
whitespace-insensitively; single-token triggers behave exactly as before.

Also stop suppressing suggestions entirely in regex mode: add a regexMode
option that fires only when the whole query equals a trigger literal, so a
bare typed term still upgrades to the curated regex while a hand-written
pattern (or an already-applied suggestion) is left alone.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

Diffstat:
Mcommon/components/QueryLeafView.tsx | 16++++++++--------
Mcommon/lib/searchAliases.test.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/searchAliases.ts | 52+++++++++++++++++++++++++++++++++++++++++++++++-----
Mexport/CHANGELOG.md | 1+
Mexport/e2e/alias-suggestion.spec.ts | 28++++++++++++++++++++++++++++
5 files changed, 147 insertions(+), 13 deletions(-)

diff --git a/common/components/QueryLeafView.tsx b/common/components/QueryLeafView.tsx @@ -59,20 +59,20 @@ export default function QueryLeafView({ leaf.useRegex && leaf.query.trim() !== "" && !isValidRegex(leaf.query); // Known-alias suggestions. A curated concept ("loli") that a typed term - // matches offers a better regex the user can Apply — never forced. Suppressed - // in regex mode (the user is writing their own pattern). Dismissals are scoped - // to the current query text so a new term re-offers. See ../lib/searchAliases. + // matches offers a better regex the user can Apply — never forced. In regex + // mode we only offer when the whole query is exactly a trigger (a bare + // literal), so a hand-written pattern — or an already-applied suggestion — + // isn't second-guessed. Dismissals are scoped to the current query text so a + // new term re-offers. See ../lib/searchAliases. const { aliases } = useSearchData(); const [dismissed, setDismissed] = useState<{ q: string; ids: Set<string> }>({ q: "", ids: new Set(), }); const activeDismissed = dismissed.q === leaf.query ? dismissed.ids : null; - const suggestions = leaf.useRegex - ? [] - : matchAliases(leaf.query, leaf.scope, aliases).filter( - (a) => !activeDismissed?.has(a.id), - ); + const suggestions = matchAliases(leaf.query, leaf.scope, aliases, { + regexMode: leaf.useRegex, + }).filter((a) => !activeDismissed?.has(a.id)); const dismissAlias = (id: string) => setDismissed((prev) => { const ids = diff --git a/common/lib/searchAliases.test.ts b/common/lib/searchAliases.test.ts @@ -37,6 +37,69 @@ test("matchAliases does NOT fire on a substring (whole-token only)", () => { assert.deepEqual(matchAliases("deloli", "transcripts", [loli]), []); }); +const platner: SearchAlias = { + id: "platner", + label: "Graham Platner", + triggers: ["graham platner"], + suggestion: "graham\\s+platner", + useRegex: true, +}; + +test("matchAliases fires on a multi-word (phrase) trigger", () => { + assert.deepEqual( + matchAliases("graham platner", "transcripts", [platner]), + [platner], + ); + // Case-insensitive and whitespace-normalized. + assert.deepEqual( + matchAliases("Graham Platner", "transcripts", [platner]), + [platner], + ); + // Fires when the phrase is a run of tokens among others (in order). + assert.deepEqual( + matchAliases("the graham platner interview", "transcripts", [platner]), + [platner], + ); +}); + +test("matchAliases phrase trigger needs the whole phrase, in order", () => { + // A single word of the phrase is not enough. + assert.deepEqual(matchAliases("graham", "transcripts", [platner]), []); + assert.deepEqual(matchAliases("platner", "transcripts", [platner]), []); + // Wrong order does not match (contiguous, ordered). + assert.deepEqual(matchAliases("platner graham", "transcripts", [platner]), []); +}); + +test("matchAliases in regexMode fires only on a bare-literal trigger", () => { + // A plain typed literal equal to a trigger still gets upgraded. + assert.deepEqual( + matchAliases("graham platner", "transcripts", [platner], { regexMode: true }), + [platner], + ); + assert.deepEqual( + matchAliases("loli", "transcripts", [loli], { regexMode: true }), + [loli], + ); + // A hand-written pattern or extra tokens is left alone. + assert.deepEqual( + matchAliases("graham\\s+platner", "transcripts", [platner], { + regexMode: true, + }), + [], + ); + assert.deepEqual( + matchAliases("graham platner clips", "transcripts", [platner], { + regexMode: true, + }), + [], + ); + // An already-applied suggestion does not re-offer. + assert.deepEqual( + matchAliases("\\blol(i|ly)", "transcripts", [loli], { regexMode: true }), + [], + ); +}); + test("matchAliases respects the enabled flag", () => { const off = { ...loli, enabled: false }; assert.deepEqual(matchAliases("loli", "transcripts", [off]), []); diff --git a/common/lib/searchAliases.ts b/common/lib/searchAliases.ts @@ -50,6 +50,9 @@ function isLayerScope(v: unknown): v is LayerScope { // Split a query into lowercased word tokens on any non-alphanumeric boundary. // Whole-token matching is deliberate: "loli" fires inside "loli clips" but NOT // inside "lolight" (substring matching would be too noisy and erode trust). +// Triggers are tokenized the same way, so a multi-word trigger like +// "graham platner" becomes ["graham", "platner"] and matching is case- and +// whitespace-insensitive for free. export function tokenizeQuery(query: string): string[] { return query .toLowerCase() @@ -57,21 +60,60 @@ export function tokenizeQuery(query: string): string[] { .filter(Boolean); } +// True if `needle` (non-empty) appears as a contiguous run of whole tokens in +// `haystack`. A single-token needle reduces to "token present anywhere", which +// preserves the original whole-word behavior (fires on "loli clips", not on +// "lolight"); a multi-token needle requires the phrase in order. +function containsTokenSequence(haystack: string[], needle: string[]): boolean { + if (needle.length === 0) return false; + if (needle.length > haystack.length) return false; + for (let i = 0; i + needle.length <= haystack.length; i++) { + let all = true; + for (let j = 0; j < needle.length; j++) { + if (haystack[i + j] !== needle[j]) { + all = false; + break; + } + } + if (all) return true; + } + return false; +} + +function tokensEqual(a: string[], b: string[]): boolean { + return a.length === b.length && a.every((t, i) => t === b[i]); +} + // Which aliases match the typed query for a given scope. Returns every match -// (usually 0 or 1). The caller is responsible for suppressing suggestions when -// the leaf is already in regex mode — the user is crafting their own pattern. +// (usually 0 or 1). +// +// By default a trigger fires when its token sequence appears as whole tokens in +// the query (phrase-aware). In `regexMode` the user is crafting their own +// pattern, so we fire ONLY when the whole query is exactly one of the triggers +// (a bare literal, not a hand-written pattern) — this lets a plain typed word +// still be upgraded to the curated regex, while an applied suggestion (whose +// tokens don't equal any trigger) stays suppressed. export function matchAliases( query: string, scope: LayerScope, aliases: SearchAlias[], + opts?: { regexMode?: boolean }, ): SearchAlias[] { - const tokens = new Set(tokenizeQuery(query)); - if (tokens.size === 0) return []; + const qTokens = tokenizeQuery(query); + if (qTokens.length === 0) return []; + const regexMode = opts?.regexMode === true; const out: SearchAlias[] = []; for (const a of aliases) { if (a.enabled === false) continue; if (a.scopes && a.scopes.length > 0 && !a.scopes.includes(scope)) continue; - if (a.triggers.some((t) => tokens.has(t.toLowerCase()))) out.push(a); + const fires = a.triggers.some((t) => { + const tTokens = tokenizeQuery(t); + if (tTokens.length === 0) return false; + return regexMode + ? tokensEqual(qTokens, tTokens) + : containsTokenSequence(qTokens, tTokens); + }); + if (fires) out.push(a); } return out; } diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -2,6 +2,7 @@ ## [Unreleased] - **Search now suggests a better query when you type a known term.** When a search term matches a curated alias — e.g. typing `loli`, `lolly`, or `loly` — a quiet chip appears under the box offering a more robust regex like `\blol(i|ly)`. Click **Apply** to swap it in (and switch that layer to regex mode) or **Dismiss** to ignore it; a new term re-offers. It's never forced, matching is whole-word (so `lolight` won't trigger it), and it's suppressed while you're already writing a regex. The dictionary is authored in the editor's new Search aliases page and shipped per site as `/search-aliases.json` (the global list merged with per-site overrides). See `common/components/QueryLeafView.tsx`, `common/components/{aliasesCache,SearchDataContext}.tsx`, `common/lib/searchAliases.ts`, and `export/e2e/alias-suggestion.spec.ts`. +- **Alias suggestions now handle multi-word terms and regex mode.** A trigger with a space in it (e.g. `graham platner`) now offers its suggestion the same way single words do — matching is whole-word, in order, and case-insensitive. And a suggestion still surfaces when the box is already in regex mode *as long as you've typed a plain term equal to the alias* (a hand-written pattern, or an already-applied suggestion, is left untouched). See `common/lib/searchAliases.ts` (`matchAliases`). ## [0.7.0] - 2026-07-06 - **Bring-your-own-AI: the archive is now machine-navigable for AI tools.** Every site publishes a small fixed set of discovery files — `llms.txt` (an LLM-readable overview) and `corpus.json` (a documented index of the channels and *how to fetch any transcript* from the existing paginated JSON shards), plus `robots.txt` and a `sitemap.xml`. Nothing is generated per video (the shard scheme is documented instead), so the file count stays constant no matter how large the corpus grows. This lets Claude Code and other tools browse and answer questions about the archive by fetching a couple of URLs. The federated hub publishes an aggregate `corpus.json`/`llms.txt` spanning every member site. diff --git a/export/e2e/alias-suggestion.spec.ts b/export/e2e/alias-suggestion.spec.ts @@ -15,6 +15,13 @@ const ALIASES = { useRegex: true, note: "AI transcription often expands this.", }, + { + id: "platner", + label: "Graham Platner", + triggers: ["graham platner"], + suggestion: "graham\\s+platner", + useRegex: true, + }, ], }; @@ -61,6 +68,27 @@ test.describe("search alias suggestions", () => { await expect(chip).toHaveCount(0); }); + test("a multi-word trigger offers the suggestion; Apply swaps in the regex", async ({ + page, + }) => { + await page.goto("/"); + await firstLeaf(page).fill("graham platner"); + + const chip = page.locator('[data-testid^="leaf-alias-suggestions-"]'); + await expect(chip).toBeVisible(); + await expect(chip).toContainText("Graham Platner"); + await expect(chip.locator("code")).toHaveText("graham\\s+platner"); + + await page.locator('[data-testid^="leaf-alias-apply-"]').click(); + await expect(firstLeaf(page)).toHaveValue("graham\\s+platner"); + await expect( + page.locator('[data-testid^="leaf-regex-"]').first(), + ).toHaveAttribute("aria-checked", "true"); + + // The applied pattern is not a bare literal, so it is not re-offered. + await expect(chip).toHaveCount(0); + }); + test("does not fire on a mere substring (whole-token match only)", async ({ page, }) => {