commit b35da36055ad5b3192a2a587df47c79455ca8ba9
parent e9e404562426ce2197c50d2c650e7f7a7208bcf3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 7 Jul 2026 14:55:35 -0400
Merge branch 'worktree-alias-suggestions': fix multi-word + regex-mode alias matching
Diffstat:
5 files changed, 147 insertions(+), 13 deletions(-)
diff --git a/common/components/QueryLeafView.tsx b/common/components/QueryLeafView.tsx
@@ -59,20 +59,20 @@ export default function QueryLeafView({
leaf.useRegex && leaf.query.trim() !== "" && !isValidRegex(leaf.query);
// Known-alias suggestions. A curated concept ("loli") that a typed term
- // matches offers a better regex the user can Apply — never forced. Suppressed
- // in regex mode (the user is writing their own pattern). Dismissals are scoped
- // to the current query text so a new term re-offers. See ../lib/searchAliases.
+ // matches offers a better regex the user can Apply — never forced. In regex
+ // mode we only offer when the whole query is exactly a trigger (a bare
+ // literal), so a hand-written pattern — or an already-applied suggestion —
+ // isn't second-guessed. Dismissals are scoped to the current query text so a
+ // new term re-offers. See ../lib/searchAliases.
const { aliases } = useSearchData();
const [dismissed, setDismissed] = useState<{ q: string; ids: Set<string> }>({
q: "",
ids: new Set(),
});
const activeDismissed = dismissed.q === leaf.query ? dismissed.ids : null;
- const suggestions = leaf.useRegex
- ? []
- : matchAliases(leaf.query, leaf.scope, aliases).filter(
- (a) => !activeDismissed?.has(a.id),
- );
+ const suggestions = matchAliases(leaf.query, leaf.scope, aliases, {
+ regexMode: leaf.useRegex,
+ }).filter((a) => !activeDismissed?.has(a.id));
const dismissAlias = (id: string) =>
setDismissed((prev) => {
const ids =
diff --git a/common/lib/searchAliases.test.ts b/common/lib/searchAliases.test.ts
@@ -37,6 +37,69 @@ test("matchAliases does NOT fire on a substring (whole-token only)", () => {
assert.deepEqual(matchAliases("deloli", "transcripts", [loli]), []);
});
+const platner: SearchAlias = {
+ id: "platner",
+ label: "Graham Platner",
+ triggers: ["graham platner"],
+ suggestion: "graham\\s+platner",
+ useRegex: true,
+};
+
+test("matchAliases fires on a multi-word (phrase) trigger", () => {
+ assert.deepEqual(
+ matchAliases("graham platner", "transcripts", [platner]),
+ [platner],
+ );
+ // Case-insensitive and whitespace-normalized.
+ assert.deepEqual(
+ matchAliases("Graham Platner", "transcripts", [platner]),
+ [platner],
+ );
+ // Fires when the phrase is a run of tokens among others (in order).
+ assert.deepEqual(
+ matchAliases("the graham platner interview", "transcripts", [platner]),
+ [platner],
+ );
+});
+
+test("matchAliases phrase trigger needs the whole phrase, in order", () => {
+ // A single word of the phrase is not enough.
+ assert.deepEqual(matchAliases("graham", "transcripts", [platner]), []);
+ assert.deepEqual(matchAliases("platner", "transcripts", [platner]), []);
+ // Wrong order does not match (contiguous, ordered).
+ assert.deepEqual(matchAliases("platner graham", "transcripts", [platner]), []);
+});
+
+test("matchAliases in regexMode fires only on a bare-literal trigger", () => {
+ // A plain typed literal equal to a trigger still gets upgraded.
+ assert.deepEqual(
+ matchAliases("graham platner", "transcripts", [platner], { regexMode: true }),
+ [platner],
+ );
+ assert.deepEqual(
+ matchAliases("loli", "transcripts", [loli], { regexMode: true }),
+ [loli],
+ );
+ // A hand-written pattern or extra tokens is left alone.
+ assert.deepEqual(
+ matchAliases("graham\\s+platner", "transcripts", [platner], {
+ regexMode: true,
+ }),
+ [],
+ );
+ assert.deepEqual(
+ matchAliases("graham platner clips", "transcripts", [platner], {
+ regexMode: true,
+ }),
+ [],
+ );
+ // An already-applied suggestion does not re-offer.
+ assert.deepEqual(
+ matchAliases("\\blol(i|ly)", "transcripts", [loli], { regexMode: true }),
+ [],
+ );
+});
+
test("matchAliases respects the enabled flag", () => {
const off = { ...loli, enabled: false };
assert.deepEqual(matchAliases("loli", "transcripts", [off]), []);
diff --git a/common/lib/searchAliases.ts b/common/lib/searchAliases.ts
@@ -50,6 +50,9 @@ function isLayerScope(v: unknown): v is LayerScope {
// Split a query into lowercased word tokens on any non-alphanumeric boundary.
// Whole-token matching is deliberate: "loli" fires inside "loli clips" but NOT
// inside "lolight" (substring matching would be too noisy and erode trust).
+// Triggers are tokenized the same way, so a multi-word trigger like
+// "graham platner" becomes ["graham", "platner"] and matching is case- and
+// whitespace-insensitive for free.
export function tokenizeQuery(query: string): string[] {
return query
.toLowerCase()
@@ -57,21 +60,60 @@ export function tokenizeQuery(query: string): string[] {
.filter(Boolean);
}
+// True if `needle` (non-empty) appears as a contiguous run of whole tokens in
+// `haystack`. A single-token needle reduces to "token present anywhere", which
+// preserves the original whole-word behavior (fires on "loli clips", not on
+// "lolight"); a multi-token needle requires the phrase in order.
+function containsTokenSequence(haystack: string[], needle: string[]): boolean {
+ if (needle.length === 0) return false;
+ if (needle.length > haystack.length) return false;
+ for (let i = 0; i + needle.length <= haystack.length; i++) {
+ let all = true;
+ for (let j = 0; j < needle.length; j++) {
+ if (haystack[i + j] !== needle[j]) {
+ all = false;
+ break;
+ }
+ }
+ if (all) return true;
+ }
+ return false;
+}
+
+function tokensEqual(a: string[], b: string[]): boolean {
+ return a.length === b.length && a.every((t, i) => t === b[i]);
+}
+
// Which aliases match the typed query for a given scope. Returns every match
-// (usually 0 or 1). The caller is responsible for suppressing suggestions when
-// the leaf is already in regex mode — the user is crafting their own pattern.
+// (usually 0 or 1).
+//
+// By default a trigger fires when its token sequence appears as whole tokens in
+// the query (phrase-aware). In `regexMode` the user is crafting their own
+// pattern, so we fire ONLY when the whole query is exactly one of the triggers
+// (a bare literal, not a hand-written pattern) — this lets a plain typed word
+// still be upgraded to the curated regex, while an applied suggestion (whose
+// tokens don't equal any trigger) stays suppressed.
export function matchAliases(
query: string,
scope: LayerScope,
aliases: SearchAlias[],
+ opts?: { regexMode?: boolean },
): SearchAlias[] {
- const tokens = new Set(tokenizeQuery(query));
- if (tokens.size === 0) return [];
+ const qTokens = tokenizeQuery(query);
+ if (qTokens.length === 0) return [];
+ const regexMode = opts?.regexMode === true;
const out: SearchAlias[] = [];
for (const a of aliases) {
if (a.enabled === false) continue;
if (a.scopes && a.scopes.length > 0 && !a.scopes.includes(scope)) continue;
- if (a.triggers.some((t) => tokens.has(t.toLowerCase()))) out.push(a);
+ const fires = a.triggers.some((t) => {
+ const tTokens = tokenizeQuery(t);
+ if (tTokens.length === 0) return false;
+ return regexMode
+ ? tokensEqual(qTokens, tTokens)
+ : containsTokenSequence(qTokens, tTokens);
+ });
+ if (fires) out.push(a);
}
return out;
}
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -13,6 +13,7 @@
## [0.7.1] - 2026-07-06
- **The "Ask AI" chat now finds the right videos.** The `/ask` retrieval previously skimmed the first few transcript shards and cut off early, so it kept answering from the same handful of videos no matter what you asked. It now uses the **site's own transcript search engine** (the one behind the search box): your question's keywords rank matches across the whole archive, snippets carry real timestamps, and it reuses the cached index — so different questions surface different, relevant videos, and on the federated hub it searches every member site.
- **Search now suggests a better query when you type a known term.** When a search term matches a curated alias — e.g. typing `loli`, `lolly`, or `loly` — a quiet chip appears under the box offering a more robust regex like `\blol(i|ly)`. Click **Apply** to swap it in (and switch that layer to regex mode) or **Dismiss** to ignore it; a new term re-offers. It's never forced, matching is whole-word (so `lolight` won't trigger it), and it's suppressed while you're already writing a regex. The dictionary is authored in the editor's new Search aliases page and shipped per site as `/search-aliases.json` (the global list merged with per-site overrides). See `common/components/QueryLeafView.tsx`, `common/components/{aliasesCache,SearchDataContext}.tsx`, `common/lib/searchAliases.ts`, and `export/e2e/alias-suggestion.spec.ts`.
+- **Alias suggestions now handle multi-word terms and regex mode.** A trigger with a space in it (e.g. `graham platner`) now offers its suggestion the same way single words do — matching is whole-word, in order, and case-insensitive. And a suggestion still surfaces when the box is already in regex mode *as long as you've typed a plain term equal to the alias* (a hand-written pattern, or an already-applied suggestion, is left untouched). See `common/lib/searchAliases.ts` (`matchAliases`).
## [0.7.0] - 2026-07-06
- **Bring-your-own-AI: the archive is now machine-navigable for AI tools.** Every site publishes a small fixed set of discovery files — `llms.txt` (an LLM-readable overview) and `corpus.json` (a documented index of the channels and *how to fetch any transcript* from the existing paginated JSON shards), plus `robots.txt` and a `sitemap.xml`. Nothing is generated per video (the shard scheme is documented instead), so the file count stays constant no matter how large the corpus grows. This lets Claude Code and other tools browse and answer questions about the archive by fetching a couple of URLs. The federated hub publishes an aggregate `corpus.json`/`llms.txt` spanning every member site.
diff --git a/export/e2e/alias-suggestion.spec.ts b/export/e2e/alias-suggestion.spec.ts
@@ -15,6 +15,13 @@ const ALIASES = {
useRegex: true,
note: "AI transcription often expands this.",
},
+ {
+ id: "platner",
+ label: "Graham Platner",
+ triggers: ["graham platner"],
+ suggestion: "graham\\s+platner",
+ useRegex: true,
+ },
],
};
@@ -61,6 +68,27 @@ test.describe("search alias suggestions", () => {
await expect(chip).toHaveCount(0);
});
+ test("a multi-word trigger offers the suggestion; Apply swaps in the regex", async ({
+ page,
+ }) => {
+ await page.goto("/");
+ await firstLeaf(page).fill("graham platner");
+
+ const chip = page.locator('[data-testid^="leaf-alias-suggestions-"]');
+ await expect(chip).toBeVisible();
+ await expect(chip).toContainText("Graham Platner");
+ await expect(chip.locator("code")).toHaveText("graham\\s+platner");
+
+ await page.locator('[data-testid^="leaf-alias-apply-"]').click();
+ await expect(firstLeaf(page)).toHaveValue("graham\\s+platner");
+ await expect(
+ page.locator('[data-testid^="leaf-regex-"]').first(),
+ ).toHaveAttribute("aria-checked", "true");
+
+ // The applied pattern is not a bare literal, so it is not re-offered.
+ await expect(chip).toHaveCount(0);
+ });
+
test("does not fire on a mere substring (whole-token match only)", async ({
page,
}) => {