commit f8152a9c3ec6f66a0f04b8a1186fd51a9a0c6d5f
parent eea0500c33c263cd20736945f5db023ad6bb9bee
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 29 Jul 2026 11:01:38 -0400
Ship the digest layer: build, page tree, and the viewer's third panel
The 102 digests on disk now travel generation → LMDB → a shared
`/digests/<slug>/` page tree → compose → the viewer, where a reader opens a
Digest panel, clicks a chapter and the player seeks to it.
Build: three new sub-DBs (`digests`, `digestPageHashes`, `channelDigestStats`)
at `SCHEMA_VERSION` 13, `digestMs` folded into the mtime record and the channel
signature so a digest-only change can't leave an archive believing the channel
is unchanged, compose reconcile, and `CORPUS_SPEC_VERSION` 3 with a
`digestScheme` so an AI tool reading the corpus knows an absent video means
"not yet generated".
Viewer: `?vm=digest` (not `?vm=summary` — "summary" already means two other
things here), a separate IndexedDB database rather than another store in the
transcript one, and a client cache versioned by PAGE CONTENT HASH rather than
`generatedAt`, because digests are regenerated in place.
What ships is `effectiveDigest()` — human overrides applied, rejected chapters
dropped. `warnings[]` and `history[]` stay in the editor: they are operator
telemetry, not something to put in front of readers. The control is hidden
where there is no digest (coverage is 0.13%), gated on the manifest so it costs
one cached request per channel and never a per-video fetch.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
23 files changed, 2049 insertions(+), 58 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -55,6 +55,7 @@ yarn-error.log*
# them, leaving generated data showing up as untracked in every worktree.
/export/public/subs
/export/public/posts
+/export/public/digests
/export/public/summaries
/export/public/transcripts
/export/public/stats
diff --git a/PLAN.md b/PLAN.md
@@ -248,7 +248,14 @@ Markdown with optional frontmatter: `hosts`, `recurring_guests`,
- Hash the resolved context into `ai-digest.json.contextHash` so editing notes correctly
marks digests stale. **This is why backfill must not start before 1.5 exists.**
-### Phase 2 — Digests as a first-class corpus
+### Phase 2 — Digests as a first-class corpus — **DONE (2026-07-29)**
+
+> Shipped as specified, plus three things this spec omitted: `exportDigestsDir` /
+> `exportSharedDigestsDir` in `paths.ts` (compose cannot be written without them), a
+> per-SITE digests manifest so `corpus.json` can source per-channel counts the way it does
+> for posts, and `digests` added to the service worker's `SHARD_RE` (without it the tree has
+> no offline story — the gap already recorded for `duplicates.json`). Note `?vm=digest`, not
+> `?vm=summary`; see the naming decision in `plans/STATE.md`.
An earlier draft inlined the digest into `TranscriptDetail`. That is wrong: under
`summary-only` visibility a transcript page ships **no cues**, so a digest must never be
@@ -308,7 +315,13 @@ tune or safely interrupt.
- `channelSnapshot.ts` gains `noDigest` **here**, not in Phase 5; Phase 5 then consumes what
already exists.
-### Phase 3 — Viewer
+### Phase 3 — Viewer — **DONE (2026-07-29), as `?vm=digest`**
+
+> Every `"summary"` below reads `"digest"` in the shipped code: "summary" already means a
+> video listing card AND the `/summaries/` page tree, and this file's own naming-hazard rule
+> forbids a third meaning. Also note the control is HIDDEN where no digest exists (0.1%
+> coverage makes an always-present button a dead end), and the client cache is versioned by
+> page content hash rather than `generatedAt` — reasons for both in `plans/STATE.md`.
- `urlState.ts` — add `"summary"` to `ModalMode`, the parse chain, and the `writeUrlParams`
`vm` branch. `ModalMode` appears in only three files.
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -30,6 +30,7 @@ import {
} from "../lib/duplicates";
import type { Manifest, SubsManifest } from "../lib/manifest";
import type { PostsManifest } from "../lib/posts";
+import type { DigestsManifest } from "../lib/digests";
import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescriptor";
import { effectiveSiteAliases } from "../lib/aliasesStore";
import {
@@ -150,7 +151,25 @@ async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
/* no posts manifest for this site */
}
- const corpus = buildSiteCorpus(descriptor, { hasArchives, postCounts });
+ // Per-channel digest counts, from the site digests manifest composed above.
+ // Absent (no digested channels) leaves corpus.json without a digest scheme.
+ const digestCounts: Record<string, number> = {};
+ try {
+ const raw = await readFile(
+ path.join(paths.exportDigestsDir, "manifest.json"),
+ "utf8",
+ );
+ const dm = JSON.parse(raw) as DigestsManifest;
+ for (const ch of dm.channels ?? []) digestCounts[ch.slug] = ch.digestCount;
+ } catch {
+ /* no digests manifest for this site */
+ }
+
+ const corpus = buildSiteCorpus(descriptor, {
+ hasArchives,
+ postCounts,
+ digestCounts,
+ });
await writeFile(
path.join(paths.exportPublicDir, "corpus.json"),
JSON.stringify(corpus),
@@ -492,6 +511,9 @@ type ComposeCache = {
// Optional for backwards compat: a cache written before the posts corpus
// existed simply has no entry, so every social channel composes once.
posts?: Record<string, string>;
+ // Same, for the AI-digest corpus: an existing cache composes digests once
+ // rather than erroring on a missing key.
+ digests?: Record<string, string>;
summaries?: string;
stats?: string;
duplicates?: string;
@@ -515,6 +537,7 @@ async function readComposeCache(p: string): Promise<ComposeCache> {
transcripts: parsed?.transcripts ?? {},
subs: parsed?.subs ?? {},
posts: parsed?.posts ?? {},
+ digests: parsed?.digests ?? {},
summaries: parsed?.summaries,
stats: parsed?.stats,
duplicates: parsed?.duplicates,
@@ -694,6 +717,17 @@ async function main(): Promise<void> {
cache.posts ?? {},
console.log,
);
+ // The AI-digest corpus: same shared-tree shape again. Sparse — only channels
+ // with at least one digest have a source dir, and reconcileChannelTree treats
+ // a missing one as "nothing to copy", so passing every member slug is right.
+ cache.digests = await reconcileChannelTree(
+ "digests",
+ paths.exportSharedDigestsDir,
+ paths.exportDigestsDir,
+ memberSlugs,
+ cache.digests ?? {},
+ console.log,
+ );
// Subs also carries a per-site manifest.json (tiny — copied every build).
const subsManifestSrc = path.join(
paths.exportSitesIndexDir,
@@ -715,6 +749,20 @@ async function main(): Promise<void> {
await mkdir(paths.exportPostsDir, { recursive: true });
await cp(postsManifestSrc, path.join(paths.exportPostsDir, "manifest.json"));
}
+ // Same for the per-site digests manifest (which channels carry digests).
+ const digestsManifestSrc = path.join(
+ paths.exportSitesIndexDir,
+ siteId,
+ "digests",
+ "manifest.json",
+ );
+ if (await exists(digestsManifestSrc)) {
+ await mkdir(paths.exportDigestsDir, { recursive: true });
+ await cp(
+ digestsManifestSrc,
+ path.join(paths.exportDigestsDir, "manifest.json"),
+ );
+ }
// --- charts dashboard ---
const templatesSrc = path.join(
diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx
@@ -14,6 +14,8 @@ import {
import type ReactPlayerType from "react-player";
import { fetchTranscript } from "./transcriptCache";
import { fetchSubs } from "./subsCache";
+import { fetchDigest, hasDigest } from "./digestCache";
+import type { VideoDigest } from "../lib/digests";
import { useUrlParams, writeUrlParams } from "./urlState";
import type { ModalMode } from "./urlState";
import { formatDate, formatDuration } from "../lib/format";
@@ -78,6 +80,9 @@ export type TranscriptData = {
type Status = "idle" | "loading" | "ready";
type ChatStatus = "idle" | "loading" | "ready" | "missing";
+// Parallel to ChatStatus. "missing" is a first-class outcome here rather than
+// an edge case: only ~0.1% of the corpus is digested.
+export type DigestStatus = "idle" | "loading" | "ready" | "missing";
export type DisplayMode = "modal" | "mini" | "hidden";
type Detail = {
@@ -111,7 +116,18 @@ type PlayerState = {
modalMode: ModalMode;
chatCues: Cue[] | null;
chatStatus: ChatStatus;
- chatNotice: string | null;
+ // The modal's transient snap-back banner, shared by every mode that can fail
+ // to load and fall back to the transcript (live chat, digest). One channel,
+ // because only one such message can be relevant at a time.
+ modalNotice: string | null;
+ digest: VideoDigest | null;
+ digestStatus: DigestStatus;
+ // Whether the ACTIVE video has a digest at all, resolved from the channel
+ // manifest without fetching a page. Gates the modal's Digest control: with
+ // coverage at 0.1% of the corpus an always-present button would be a dead end
+ // on almost every video. Same idea as hasDuplicates() gating the Duplicates
+ // nav link (export/app/lib/duplicates.ts).
+ digestAvailable: boolean;
clipStart: number | null;
clipEnd: number | null;
openTranscript: (
@@ -190,6 +206,16 @@ type Core = {
cues: Cue[] | null;
notice: string | null;
};
+ // The derived layer, shaped exactly like `chat` and fetched the same lazy
+ // way. `available` is resolved separately (and eagerly, on slug change) from
+ // the channel manifest, because the toolbar needs it before anyone asks for
+ // the panel.
+ digest: {
+ slug: string | null;
+ status: DigestStatus;
+ data: VideoDigest | null;
+ available: boolean;
+ };
};
const INITIAL_CORE: Core = {
@@ -197,6 +223,7 @@ const INITIAL_CORE: Core = {
displayState: null,
clip: { slug: null, start: null, end: null },
chat: { slug: null, status: "idle", cues: null, notice: null },
+ digest: { slug: null, status: "idle", data: null, available: false },
};
type CoreAction =
@@ -211,7 +238,12 @@ type CoreAction =
| { type: "CHAT_LOADING"; slug: string }
| { type: "CHAT_READY"; slug: string; cues: Cue[] }
| { type: "CHAT_MISSING"; slug: string; notice: string }
- | { type: "CHAT_NOTICE_CLEAR" };
+ | { type: "MODAL_NOTICE_CLEAR" }
+ | { type: "DIGEST_RESET" }
+ | { type: "DIGEST_AVAILABLE"; slug: string; available: boolean }
+ | { type: "DIGEST_LOADING"; slug: string }
+ | { type: "DIGEST_READY"; slug: string; data: VideoDigest }
+ | { type: "DIGEST_MISSING"; slug: string; notice: string };
function coreReducer(state: Core, action: CoreAction): Core {
switch (action.type) {
@@ -273,8 +305,53 @@ function coreReducer(state: Core, action: CoreAction): Core {
notice: action.notice,
},
};
- case "CHAT_NOTICE_CLEAR":
+ case "MODAL_NOTICE_CLEAR":
return { ...state, chat: { ...state.chat, notice: null } };
+ case "DIGEST_RESET":
+ return { ...state, digest: INITIAL_CORE.digest };
+ case "DIGEST_AVAILABLE":
+ return {
+ ...state,
+ digest: {
+ ...state.digest,
+ // Availability is per-video and arrives asynchronously; keep whatever
+ // slug the fetch/status fields already refer to unless this is the
+ // first thing we know about this video.
+ slug: state.digest.slug ?? action.slug,
+ available: action.available,
+ },
+ };
+ case "DIGEST_LOADING":
+ return {
+ ...state,
+ digest: {
+ slug: action.slug,
+ status: "loading",
+ data: null,
+ available: state.digest.available,
+ },
+ };
+ case "DIGEST_READY":
+ return {
+ ...state,
+ digest: {
+ slug: action.slug,
+ status: "ready",
+ data: action.data,
+ available: true,
+ },
+ };
+ case "DIGEST_MISSING":
+ return {
+ ...state,
+ chat: { ...state.chat, notice: action.notice },
+ digest: {
+ slug: action.slug,
+ status: "missing",
+ data: null,
+ available: false,
+ },
+ };
default:
return state;
}
@@ -287,7 +364,7 @@ export function PlayerProvider({
}) {
const { v: urlSlug, t: urlTime, vm: urlVm } = useUrlParams();
const [core, dispatch] = useReducer(coreReducer, INITIAL_CORE);
- const { detail, displayState, clip, chat } = core;
+ const { detail, displayState, clip, chat, digest } = core;
const [playing, setPlaying] = useState(false);
const [currentTime, setCurrentTime] = useState(0);
// Set when the Kick HLS manifest fails to load (usually an expired VOD), so
@@ -304,6 +381,13 @@ export function PlayerProvider({
// chat-fetch effect needs to decide "re-toggle on already-missing slug".
const chatStatusRef = useRef<ChatStatus>("idle");
chatStatusRef.current = chat.status;
+ // The digest slice's equivalents, for exactly the same reason: including
+ // `digest.slug`/`digest.status` in the fetch effect's deps makes React run
+ // cleanup before the fetch resolves, and the closure's cancelled-flag then
+ // drops the result.
+ const digestInFlightForRef = useRef<string | null>(null);
+ const digestStatusRef = useRef<DigestStatus>("idle");
+ digestStatusRef.current = digest.status;
const activeSlug = urlSlug;
const modalMode: ModalMode = urlVm;
@@ -354,12 +438,7 @@ export function PlayerProvider({
writeUrlParams({
v: slug,
t,
- vm:
- opts?.mode === "chat"
- ? "chat"
- : opts?.mode === "post"
- ? "post"
- : "transcript",
+ vm: opts?.mode ?? "transcript",
});
},
[activeSlug],
@@ -422,7 +501,9 @@ export function PlayerProvider({
const params = new URLSearchParams();
params.set("v", data.slug);
if (secs > 0) params.set("t", String(secs));
- if (modalMode === "chat") params.set("vm", "chat");
+ // Carry the current panel so a shared link reopens what the sharer was
+ // looking at. "transcript" is the default and stays absent from the URL.
+ if (modalMode !== "transcript") params.set("vm", modalMode);
const url = `${window.location.origin}${window.location.pathname}?${params.toString()}`;
try {
await navigator.clipboard.writeText(url);
@@ -535,6 +616,8 @@ export function PlayerProvider({
useEffect(() => {
dispatch({ type: "CHAT_RESET" });
chatInFlightForRef.current = null;
+ dispatch({ type: "DIGEST_RESET" });
+ digestInFlightForRef.current = null;
setKickError(false);
if (!urlSlug) return;
// Posts are not transcripts — never run the video fetch for one.
@@ -640,13 +723,86 @@ export function PlayerProvider({
};
}, [urlSlug, modalMode]);
+ // Does this video have a digest? Resolved eagerly on every slug change (one
+ // small, cached manifest request per channel) because the modal's toolbar has
+ // to decide whether to render the Digest control before anyone clicks it.
+ // Never fetches a page — see digestCache.hasDigest.
+ useEffect(() => {
+ if (!urlSlug) return;
+ if (urlVm === "post") return;
+ let cancelled = false;
+ void hasDigest(urlSlug).then((available) => {
+ if (cancelled) return;
+ dispatch({ type: "DIGEST_AVAILABLE", slug: urlSlug, available });
+ });
+ return () => {
+ cancelled = true;
+ };
+ // urlVm intentionally omitted for the same reason as the transcript fetch
+ // above: a plain mode toggle must not re-run this.
+ // eslint-disable-next-line react-hooks/exhaustive-deps
+ }, [urlSlug]);
+
+ // Lazy-fetch the digest only when the modal enters digest mode for the
+ // current video, and snap back to the transcript on missing/error so a reader
+ // is never stranded on an empty panel. Structurally identical to the chat
+ // effect above, including why the two refs exist instead of reducer state in
+ // the dep array (cleanup would run before the fetch resolves and the result
+ // would be dropped).
+ useEffect(() => {
+ if (!urlSlug) return;
+ if (modalMode !== "digest") return;
+ if (
+ digestInFlightForRef.current === urlSlug &&
+ digestStatusRef.current === "missing"
+ ) {
+ dispatch({
+ type: "DIGEST_MISSING",
+ slug: urlSlug,
+ notice: "No AI digest for this video — showing transcript.",
+ });
+ writeUrlParams({ vm: "transcript" });
+ return;
+ }
+ if (digestInFlightForRef.current === urlSlug) return;
+ digestInFlightForRef.current = urlSlug;
+ dispatch({ type: "DIGEST_LOADING", slug: urlSlug });
+ let cancelled = false;
+ fetchDigest(urlSlug)
+ .then((found) => {
+ if (cancelled) return;
+ if (found && (found.chapters.length > 0 || found.tags.length > 0)) {
+ dispatch({ type: "DIGEST_READY", slug: urlSlug, data: found });
+ } else {
+ dispatch({
+ type: "DIGEST_MISSING",
+ slug: urlSlug,
+ notice: "No AI digest for this video — showing transcript.",
+ });
+ writeUrlParams({ vm: "transcript" });
+ }
+ })
+ .catch(() => {
+ if (cancelled) return;
+ dispatch({
+ type: "DIGEST_MISSING",
+ slug: urlSlug,
+ notice: "Couldn't load the AI digest — showing transcript.",
+ });
+ writeUrlParams({ vm: "transcript" });
+ });
+ return () => {
+ cancelled = true;
+ };
+ }, [urlSlug, modalMode]);
+
// Auto-clear the snap-back notice after a brief window. Replaces the
// hand-managed setTimeout/clearTimeout ref dance — the effect's cleanup is
// the cancellation.
useEffect(() => {
if (!chat.notice) return;
const id = window.setTimeout(() => {
- dispatch({ type: "CHAT_NOTICE_CLEAR" });
+ dispatch({ type: "MODAL_NOTICE_CLEAR" });
}, 4000);
return () => {
window.clearTimeout(id);
@@ -710,7 +866,10 @@ export function PlayerProvider({
modalMode,
chatCues: chat.slug === activeSlug ? chat.cues : null,
chatStatus: chat.slug === activeSlug ? chat.status : "idle",
- chatNotice: chat.notice,
+ modalNotice: chat.notice,
+ digest: digest.slug === activeSlug ? digest.data : null,
+ digestStatus: digest.slug === activeSlug ? digest.status : "idle",
+ digestAvailable: digest.slug === activeSlug && digest.available,
clipStart,
clipEnd,
openTranscript,
@@ -735,6 +894,7 @@ export function PlayerProvider({
modalOpen,
modalMode,
chat,
+ digest,
clipStart,
clipEnd,
openTranscript,
diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx
@@ -7,7 +7,9 @@ import {
toHMS,
usePlayer,
usePlayerTime,
+ type DigestStatus,
} from "./PlayerProvider";
+import type { VideoDigest } from "../lib/digests";
import { AgeRestrictedBadge, LivestreamBadge } from "./badges";
import { VirtualRow } from "./VirtualRow";
import { formatTimestamp } from "../lib/vtt";
@@ -31,7 +33,10 @@ export default function TranscriptModal() {
setModalMode,
chatCues,
chatStatus,
- chatNotice,
+ modalNotice,
+ digest,
+ digestStatus,
+ digestAvailable,
clipStart,
clipEnd,
setDisplayMode,
@@ -63,6 +68,7 @@ export default function TranscriptModal() {
const scrollKindRef = useRef<"smooth" | "auto">("smooth");
const isChat = modalMode === "chat";
+ const isDigest = modalMode === "digest";
const canDownloadFile = isChat
? (chatCues?.length ?? 0) > 0
: (data?.cues?.length ?? 0) > 0;
@@ -93,7 +99,10 @@ export default function TranscriptModal() {
// doesn't redo `indexOf`/`slice` on every progress tick. Memo key is the
// identity of the underlying cues array.
const displayCues: DisplayCue[] = useMemo(() => {
- const raw = isChat ? (chatCues ?? []) : (data?.cues ?? []);
+ // The digest panel renders chapters, not cues, and is NOT virtualized — so
+ // the cue list is empty in that mode and the virtualizer below measures
+ // nothing rather than a list that isn't on screen.
+ const raw = isDigest ? [] : isChat ? (chatCues ?? []) : (data?.cues ?? []);
return raw.map((c) => {
const sepIdx = isChat ? c.text.indexOf(": ") : -1;
const author = sepIdx > 0 ? c.text.slice(0, sepIdx) : null;
@@ -106,7 +115,14 @@ export default function TranscriptModal() {
body,
};
});
- }, [isChat, chatCues, data?.cues]);
+ }, [isDigest, isChat, chatCues, data?.cues]);
+
+ // Chapters are tens of rows, so they render as a plain list. Their active-row
+ // highlight uses the same binary search as the cue list — and reads the clock
+ // from usePlayerTime(), which is why that context is split out: a 4 Hz tick
+ // must not re-render every usePlayer() consumer.
+ const chapters = digest?.chapters ?? [];
+ const activeChapter = findActiveIndex(chapters, currentTime);
const cueStatus: "loading" | "ready" =
isChat
@@ -164,11 +180,30 @@ export default function TranscriptModal() {
[seekTo],
);
- const onToggleMode = useCallback(() => {
+ // Three-way selector, expressed as two toggles that each fall back to the
+ // transcript. A cycling single button would make "get me back to the
+ // transcript" take a variable number of clicks.
+ const onToggleChat = useCallback(() => {
scrollKindRef.current = "smooth";
setModalMode(isChat ? "transcript" : "chat");
}, [isChat, setModalMode]);
+ const onToggleDigest = useCallback(() => {
+ scrollKindRef.current = "smooth";
+ setModalMode(isDigest ? "transcript" : "digest");
+ }, [isDigest, setModalMode]);
+
+ // Seek from a chapter. ALWAYS on `start` — the parser already snapped it to a
+ // real cue boundary — and never by re-parsing `clock`, which is the raw model
+ // output kept for auditing and can be wrong.
+ const onSeekChapter = useCallback(
+ (start: number) => {
+ scrollKindRef.current = "smooth";
+ seekTo(start);
+ },
+ [seekTo],
+ );
+
if (!modalOpen || !activeSlug) return null;
const canDownload = clipStart !== null && clipEnd !== null && clipEnd > clipStart;
@@ -238,9 +273,20 @@ export default function TranscriptModal() {
disabled={!canDownload}
/>
<div className="flex-1" />
+ {/* Only offered where a digest exists. ~0.1% of the corpus is
+ digested, so an always-present control would be a dead end on
+ almost every video. */}
+ {digestAvailable && (
+ <ControlButton
+ title={isDigest ? "Show transcript" : "Show AI chapters"}
+ onClick={onToggleDigest}
+ char={isDigest ? "📜" : "✦"}
+ highlight={isDigest}
+ />
+ )}
<ControlButton
title={isChat ? "Show transcript" : "Show live chat"}
- onClick={onToggleMode}
+ onClick={onToggleChat}
char={isChat ? "📜" : "💬"}
highlight={isChat}
/>
@@ -344,18 +390,27 @@ export default function TranscriptModal() {
)}
</div>
- {chatNotice && (
+ {modalNotice && (
<p
role="status"
className="text-xs text-amber-200 bg-amber-500/15 ring-1 ring-amber-400/40 rounded px-3 py-1.5"
>
- {chatNotice}
+ {modalNotice}
</p>
)}
<div
ref={scrollRef}
className="flex-1 overflow-y-auto rounded-lg border border-zinc-700 bg-zinc-900"
>
+ {isDigest ? (
+ <DigestPanel
+ digest={digest}
+ status={digestStatus}
+ activeChapter={activeChapter}
+ onSeek={onSeekChapter}
+ />
+ ) : (
+ <>
{cueStatus === "loading" && (
<p className="text-sm text-zinc-400 p-4">
{isChat ? "Loading live chat…" : "Loading transcript…"}
@@ -388,6 +443,8 @@ export default function TranscriptModal() {
})}
</ol>
)}
+ </>
+ )}
</div>
</div>
</div>
@@ -395,6 +452,140 @@ export default function TranscriptModal() {
);
}
+// The digest panel: chapters as a plain (non-virtualized) list, the topic tags,
+// a provenance line, and — when the digest was borrowed from another video — a
+// prominent notice saying so.
+//
+// Styling deliberately matches this modal's local hard-coded zinc/white palette
+// rather than the semantic tokens used elsewhere in the app. This component and
+// PlayerProvider are the one part of the codebase still on literal colors
+// (bg-zinc-900/80, ring-white/10, active row bg-blue-950/50); migrating them is
+// its own change, and half-migrating one panel inside them would just look
+// broken.
+function DigestPanel({
+ digest,
+ status,
+ activeChapter,
+ onSeek,
+}: {
+ digest: VideoDigest | null;
+ status: DigestStatus;
+ activeChapter: number;
+ onSeek: (start: number) => void;
+}) {
+ if (status === "loading" || status === "idle") {
+ return <p className="text-sm text-zinc-400 p-4">Loading AI digest…</p>;
+ }
+ if (!digest) {
+ return (
+ <p className="text-sm text-zinc-400 p-4">
+ No AI digest for this video.
+ </p>
+ );
+ }
+
+ const prov = digest.provenance.chapters ?? digest.provenance.tags;
+ const borrowed = digest.derivedFrom;
+
+ return (
+ <div className="text-zinc-100">
+ {borrowed && (
+ <div className="m-3 rounded px-3 py-2 text-xs bg-amber-500/15 ring-1 ring-amber-400/40 text-amber-100">
+ <p className="font-medium">Borrowed from a duplicate upload</p>
+ <p className="mt-1 text-amber-200/90">
+ These chapters were generated for{" "}
+ <span className="font-mono">{borrowed.slug}</span>, a near-identical
+ copy of this video, and copied here. Timings were measured to differ
+ by {formatOffset(borrowed.offsetSeconds)}, so they should line up —
+ but the titles describe that upload, not this one.
+ </p>
+ </div>
+ )}
+
+ {digest.chapters.length > 0 ? (
+ <ol className="divide-y divide-zinc-800">
+ {digest.chapters.map((c, i) => (
+ <li key={c.id} className={i === activeChapter ? "bg-blue-950/50" : ""}>
+ <button
+ type="button"
+ onClick={() => onSeek(c.start)}
+ // Announces the chapter the playhead is currently inside — for
+ // a screen reader, and it is also the only DOM-observable proof
+ // that a chapter click actually moved the player.
+ aria-current={i === activeChapter ? "true" : undefined}
+ className="w-full text-left flex gap-3 px-3 py-2 hover:bg-zinc-800"
+ >
+ <span className="text-xs font-mono text-zinc-400 shrink-0 w-16 pt-0.5">
+ {formatTimestamp(c.start)}
+ </span>
+ <span className="text-sm min-w-0 flex-1">{c.title}</span>
+ {c.decidedBy === "human" && (
+ <span
+ title="Written or corrected by a person"
+ className="text-[10px] uppercase tracking-wide text-emerald-300/80 shrink-0 pt-1"
+ >
+ edited
+ </span>
+ )}
+ </button>
+ </li>
+ ))}
+ </ol>
+ ) : (
+ <p className="text-sm text-zinc-400 px-3 py-4">
+ No chapters in this digest.
+ </p>
+ )}
+
+ {digest.tags.length > 0 && (
+ <div className="border-t border-zinc-800 px-3 py-3">
+ <p className="text-[11px] uppercase tracking-wide text-zinc-500 mb-1.5">
+ Topics
+ </p>
+ <ul className="flex flex-wrap gap-1.5">
+ {digest.tags.map((t) => (
+ <li
+ key={t.id}
+ className="rounded bg-white/10 px-2 py-0.5 text-xs text-zinc-200"
+ >
+ {t.tag}
+ </li>
+ ))}
+ </ul>
+ </div>
+ )}
+
+ {/* Always say where this came from. A derived layer that doesn't
+ announce itself as machine-generated is the one that misleads. */}
+ <p className="border-t border-zinc-800 px-3 py-2 text-[11px] leading-relaxed text-zinc-500">
+ {prov ? (
+ <>
+ Chapters and topics generated by{" "}
+ <span className="font-mono text-zinc-400">{prov.model}</span>
+ {prov.generatedAt ? ` on ${prov.generatedAt.slice(0, 10)}` : ""}
+ . AI-generated and not reviewed unless marked{" "}
+ <span className="text-emerald-300/80">edited</span>.
+ </>
+ ) : (
+ // No machine provenance at all: this digest is entirely hand-written
+ // (an overrides file with no generated counterpart). Claiming
+ // "AI-generated" here would be the same dishonesty in reverse.
+ "Written by hand."
+ )}
+ </p>
+ </div>
+ );
+}
+
+// The measured cue-timing offset between a borrowed digest's source video and
+// this one. Sharing only happens at near-zero offset, so this is normally well
+// under a second — say so precisely rather than rounding it away to "0s".
+function formatOffset(seconds: number): string {
+ const s = Math.abs(seconds);
+ if (s < 1) return `under a second`;
+ return `${s.toFixed(1)}s`;
+}
+
const CueRow = memo(function CueRow({
cue,
isActive,
diff --git a/common/components/digestCache.ts b/common/components/digestCache.ts
@@ -0,0 +1,167 @@
+"use client";
+
+import type { ChannelDigestsManifest, VideoDigest } from "../lib/digests";
+import { digestPageFileName, manifestHasDigest } from "../lib/digests";
+import { idbGet, idbPut } from "./digestStore";
+import { makeId, splitId, idBaseUrl } from "./originId";
+
+// Fetch layer for the derived corpus, mirroring transcriptCache.ts: a memory
+// map, in-flight dedupe, manifest -> page resolution and opportunistic warming
+// of every record in a fetched page.
+//
+// The one structural difference is that THE LAYER IS SPARSE. A transcript
+// exists for every indexed video, so transcriptCache treats a missing one as an
+// error. Digests cover 102 of ~76,000 videos, so "absent" is the normal answer
+// and must be a value, not a throw:
+//
+// hasDigest() -> boolean, answered from the channel manifest alone
+// fetchDigest() -> VideoDigest | null, null meaning "not digested"
+//
+// A channel with no digests has no manifest at all, so its 404 is also a normal
+// answer and is cached as `null` — otherwise every video opened in an
+// undigested channel would re-request the same missing file.
+
+const resolved = new Map<string, VideoDigest | null>();
+const inFlight = new Map<string, Promise<VideoDigest | null>>();
+const channelManifests = new Map<
+ string,
+ Promise<ChannelDigestsManifest | null>
+>();
+const pagePromises = new Map<string, Promise<VideoDigest[]>>();
+
+// `id` is an OriginId: a bare "channelSlug/videoId" for same-origin content, or
+// "origin\tchannelSlug/videoId" for a federated cross-origin video.
+export function fetchDigest(id: string): Promise<VideoDigest | null> {
+ const hit = resolved.get(id);
+ if (hit !== undefined) return Promise.resolve(hit);
+ const flying = inFlight.get(id);
+ if (flying) return flying;
+
+ const p = load(id).then((digest) => {
+ resolved.set(id, digest);
+ inFlight.delete(id);
+ return digest;
+ });
+ p.catch(() => {
+ inFlight.delete(id);
+ });
+ inFlight.set(id, p);
+ return p;
+}
+
+// Does this video have a digest? Answered from the channel manifest, which is
+// one small cached request per channel — never a per-video fetch. This is what
+// gates the viewer's Digest control: with coverage at 0.1% of the corpus, an
+// always-present button would be a dead end almost everywhere.
+export async function hasDigest(id: string): Promise<boolean> {
+ const parts = parseId(id);
+ if (!parts) return false;
+ try {
+ const manifest = await fetchChannelManifest(
+ parts.channelSlug,
+ parts.origin,
+ );
+ return manifestHasDigest(manifest, parts.videoId);
+ } catch {
+ return false;
+ }
+}
+
+function parseId(
+ id: string,
+): { origin: string; slug: string; channelSlug: string; videoId: string } | null {
+ const { origin, slug } = splitId(id);
+ const slashIdx = slug.indexOf("/");
+ if (slashIdx < 0) return null;
+ return {
+ origin,
+ slug,
+ channelSlug: slug.slice(0, slashIdx),
+ videoId: slug.slice(slashIdx + 1),
+ };
+}
+
+// Resolves to null (not a rejection) when the channel ships no digests, so the
+// absence is cached like any other answer.
+function fetchChannelManifest(
+ channelSlug: string,
+ origin: string,
+): Promise<ChannelDigestsManifest | null> {
+ const key = makeId(origin, channelSlug);
+ let p = channelManifests.get(key);
+ if (!p) {
+ p = fetch(`${idBaseUrl(origin)}/digests/${channelSlug}/manifest.json`)
+ .then((r) => {
+ if (r.status === 404) return null;
+ if (!r.ok) {
+ throw new Error(`Failed to fetch digests manifest for ${channelSlug}`);
+ }
+ return r.json() as Promise<ChannelDigestsManifest>;
+ })
+ .catch((err) => {
+ // A transport failure is not proof of absence, so it must not be
+ // cached as one — drop the entry and let the next caller retry.
+ channelManifests.delete(key);
+ throw err;
+ });
+ channelManifests.set(key, p);
+ }
+ return p;
+}
+
+function fetchPage(
+ channelSlug: string,
+ pageIndex: number,
+ origin: string,
+): Promise<VideoDigest[]> {
+ const key = `${makeId(origin, channelSlug)}:${pageIndex}`;
+ let p = pagePromises.get(key);
+ if (!p) {
+ p = fetch(
+ `${idBaseUrl(origin)}/digests/${channelSlug}/${digestPageFileName(pageIndex)}`,
+ ).then((r) => {
+ if (!r.ok) {
+ throw new Error(`Failed to fetch digest page ${channelSlug}/${pageIndex}`);
+ }
+ return r.json() as Promise<VideoDigest[]>;
+ });
+ p.catch(() => pagePromises.delete(key));
+ pagePromises.set(key, p);
+ }
+ return p;
+}
+
+async function load(id: string): Promise<VideoDigest | null> {
+ const parts = parseId(id);
+ if (!parts) return null;
+ const { origin, slug, channelSlug, videoId } = parts;
+
+ const manifest = await fetchChannelManifest(channelSlug, origin);
+ if (!manifest) return null;
+ const pageIndex = manifest.slugToPage[videoId];
+ // Not in slugToPage = not digested. The normal case, and not an error.
+ if (pageIndex === undefined) return null;
+
+ // The page's content hash is this record's cache version. Absent (an older
+ // manifest) means every cached entry misses, which is the safe direction.
+ const version = manifest.pageHashes?.[pageIndex] ?? "";
+ const stored = await idbGet(id, version);
+ if (stored) return stored;
+
+ const page = await fetchPage(channelSlug, pageIndex, origin);
+ let found: VideoDigest | null = null;
+ for (const entry of page) {
+ // Re-key warmed entries by OriginId so a cross-origin channelSlug/videoId
+ // can't shadow a same-origin one with the same slug.
+ const entryId = makeId(origin, entry.slug);
+ if (entry.slug === slug) found = entry;
+ resolved.set(entryId, entry);
+ idbPut(entryId, entry, version);
+ }
+ // In slugToPage but missing from the page means the manifest and the page
+ // tree disagree — a real build fault, not a coverage gap, so it throws.
+ if (!found) {
+ throw new Error(`Digest ${slug} missing from page ${pageIndex}`);
+ }
+ return found;
+}
diff --git a/common/components/digestStore.ts b/common/components/digestStore.ts
@@ -0,0 +1,204 @@
+"use client";
+
+import type { VideoDigest } from "../lib/digests";
+
+// A SEPARATE DATABASE, not a second store inside transcriptStore.ts's.
+//
+// `DB_VERSION` is a property of the DATABASE, not of an object store, and
+// transcriptStore.ts's `onupgradeneeded` does deleteObjectStore +
+// createObjectStore on every upgrade. Adding a `digests` store there would
+// force a version bump and WIPE every existing client's transcript cache — a
+// large, silent, entirely avoidable cost for shipping a new layer. Following
+// searchLayerCache.ts instead: own DB name, own version, independent evolution.
+const DB_NAME = "yt-dlp-transcript-browser:digests";
+const DB_VERSION = 1;
+const STORE = "digests";
+
+// PER-ENTRY VERSIONING, which transcripts deliberately do without.
+//
+// transcriptStore.ts gets away with an ad-hoc shape sniff (does `platform` look
+// valid?) because a transcript is near-immutable: once written it does not
+// change, so a structurally-valid cached copy is also a CURRENT one. Digests
+// break that assumption — they are regenerated in place whenever the prompt,
+// model or a human correction changes, and the regenerated record has exactly
+// the same shape. A shape sniff cannot see the difference, so a returning
+// reader would keep being served the superseded digest forever, which is the
+// one failure this whole layer exists to avoid (a corrected chapter that never
+// reaches anyone). We therefore store the record's own `generatedAt` alongside
+// it and treat any mismatch with the manifest-fresh copy as a miss.
+//
+// This is the same class of bug already recorded for the transcript cache when
+// TranscriptDetail's shape changed; here it is designed out rather than patched
+// with a version bump after the fact.
+//
+// The version we compare is the CONTENT HASH of the page the record came from
+// (ChannelDigestsManifest.pageHashes), not the record's own `generatedAt`. Both
+// detect a regeneration, but only the page hash is knowable from the manifest
+// alone — i.e. before paying for the fetch the cache exists to avoid — and it
+// additionally catches changes that leave `generatedAt` untouched, such as a
+// human retitling a chapter through the overrides file.
+type StoredEntry = {
+ digest: VideoDigest;
+ // The manifest pageHash this record was fetched under.
+ version: string;
+};
+
+type Mode = "pending" | "ok" | "unavailable";
+
+let mode: Mode = "pending";
+let dbPromise: Promise<IDBDatabase | null> | null = null;
+
+function openDb(): Promise<IDBDatabase | null> {
+ if (typeof indexedDB === "undefined") {
+ mode = "unavailable";
+ return Promise.resolve(null);
+ }
+ if (dbPromise) return dbPromise;
+ dbPromise = new Promise<IDBDatabase | null>((resolve) => {
+ let req: IDBOpenDBRequest;
+ try {
+ req = indexedDB.open(DB_NAME, DB_VERSION);
+ } catch {
+ downgrade("open threw");
+ resolve(null);
+ return;
+ }
+ req.onupgradeneeded = () => {
+ const db = req.result;
+ if (db.objectStoreNames.contains(STORE)) {
+ db.deleteObjectStore(STORE);
+ }
+ // Out-of-line keys: the caller supplies an OriginId (see idbPut).
+ db.createObjectStore(STORE);
+ };
+ req.onsuccess = () => {
+ mode = "ok";
+ resolve(req.result);
+ };
+ req.onerror = () => {
+ downgrade("open failed");
+ resolve(null);
+ };
+ req.onblocked = () => {
+ downgrade("open blocked");
+ resolve(null);
+ };
+ });
+ return dbPromise;
+}
+
+function downgrade(reason: string): void {
+ if (mode === "unavailable") return;
+ mode = "unavailable";
+ console.warn(
+ `[digestStore] IndexedDB unavailable (${reason}); falling back to network-only.`,
+ );
+}
+
+// Read a cached digest, but only if its recorded version matches `expected`
+// (the page hash the manifest says is current). A mismatch — or an entry
+// written before this field existed — resolves null, so the caller re-fetches.
+// An empty `expected` also misses: a record we cannot version is one we must
+// not serve from cache.
+export async function idbGet(
+ id: string,
+ expected: string,
+): Promise<VideoDigest | null> {
+ if (!expected) return null;
+ const db = await openDb();
+ if (!db) return null;
+ return new Promise<VideoDigest | null>((resolve) => {
+ let req: IDBRequest<StoredEntry | undefined>;
+ try {
+ const tx = db.transaction(STORE, "readonly");
+ req = tx.objectStore(STORE).get(id) as IDBRequest<StoredEntry | undefined>;
+ } catch {
+ downgrade("read tx threw");
+ resolve(null);
+ return;
+ }
+ req.onsuccess = () => {
+ const entry = req.result;
+ if (!entry || !entry.digest || entry.version !== expected) {
+ resolve(null);
+ return;
+ }
+ resolve(entry.digest);
+ };
+ req.onerror = () => resolve(null);
+ });
+}
+
+type PendingEntry = { id: string; entry: StoredEntry };
+let pending: PendingEntry[] = [];
+let flushScheduled = false;
+let flushInFlight: Promise<void> | null = null;
+
+// Queue a digest for persistence under the given OriginId, stamped with the
+// page hash it was fetched under. Batched via queueMicrotask exactly as
+// transcriptStore.ts does, so warming a whole page's worth of digests costs one
+// transaction. An unversioned write is dropped rather than stored — an entry
+// with no version could never be validated and would only ever waste quota.
+export function idbPut(id: string, digest: VideoDigest, version: string): void {
+ if (mode === "unavailable") return;
+ if (!version) return;
+ pending.push({ id, entry: { digest, version } });
+ if (flushScheduled) return;
+ flushScheduled = true;
+ queueMicrotask(() => {
+ flushScheduled = false;
+ void flush();
+ });
+}
+
+async function flush(): Promise<void> {
+ if (flushInFlight) {
+ await flushInFlight;
+ }
+ if (pending.length === 0) return;
+ const batch: PendingEntry[] = pending;
+ pending = [];
+ flushInFlight = writeBatch(batch).finally(() => {
+ flushInFlight = null;
+ if (pending.length > 0 && !flushScheduled) {
+ flushScheduled = true;
+ queueMicrotask(() => {
+ flushScheduled = false;
+ void flush();
+ });
+ }
+ });
+ await flushInFlight;
+}
+
+async function writeBatch(batch: PendingEntry[]): Promise<void> {
+ const db = await openDb();
+ if (!db) return;
+ return new Promise<void>((resolve) => {
+ let tx: IDBTransaction;
+ try {
+ tx = db.transaction(STORE, "readwrite");
+ } catch {
+ downgrade("write tx threw");
+ resolve();
+ return;
+ }
+ const store = tx.objectStore(STORE);
+ for (const item of batch) {
+ try {
+ store.put(item.entry, item.id);
+ } catch {
+ // Per-entry errors (e.g. unclonable values) shouldn't fail the batch.
+ }
+ }
+ tx.oncomplete = () => resolve();
+ tx.onerror = () => {
+ downgrade("write tx error");
+ resolve();
+ };
+ tx.onabort = () => {
+ downgrade("write tx abort");
+ resolve();
+ };
+ });
+}
diff --git a/common/components/urlState.ts b/common/components/urlState.ts
@@ -9,8 +9,14 @@ export type SearchMode = "transcripts" | "subs" | "posts";
// Per-video modal content mode. Independent from the search page's `mode` so
// the modal can be toggled without disturbing search state. Absence on the
-// URL means "transcript" — only `"chat"` is persisted.
-export type ModalMode = "transcript" | "chat" | "post";
+// URL means "transcript" — every other value is persisted verbatim.
+//
+// "digest" and NOT "summary", deliberately: `DisplaySummary` is already a video
+// LISTING CARD (common/lib/transcripts.ts) and `/summaries/` is already the
+// browse-index page tree the search index serves. A third meaning of "summary"
+// is exactly the naming hazard PLAN.md warns about. "digest" matches the
+// artifact, the generation stage, the settings section and the page tree.
+export type ModalMode = "transcript" | "chat" | "post" | "digest";
export type UrlParams = {
q: string;
@@ -62,7 +68,13 @@ function parse(search: string): UrlParams {
modeRaw === "subs" ? "subs" : modeRaw === "posts" ? "posts" : "transcripts";
const vmRaw = p.get("vm");
const vm: ModalMode =
- vmRaw === "chat" ? "chat" : vmRaw === "post" ? "post" : "transcript";
+ vmRaw === "chat"
+ ? "chat"
+ : vmRaw === "post"
+ ? "post"
+ : vmRaw === "digest"
+ ? "digest"
+ : "transcript";
return {
q: p.get("q") ?? "",
re: p.get("re") === "1",
@@ -126,6 +138,9 @@ export function writeUrlParams(patch: Patch) {
if (patch.vm !== undefined) {
if (patch.vm === "chat") params.set("vm", "chat");
else if (patch.vm === "post") params.set("vm", "post");
+ else if (patch.vm === "digest") params.set("vm", "digest");
+ // "transcript" is the fallthrough and DELETES the param rather than
+ // setting vm=transcript — the default must stay absent from the URL.
else params.delete("vm");
}
if (patch.ch !== undefined) {
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -67,6 +67,7 @@ import {
import { getSettings } from "../lib/settings";
import {
listSites,
+ siteDigestsDir,
sitePostsDir,
siteSummariesDir,
siteSubsDir,
@@ -104,6 +105,18 @@ import {
readPostShard,
} from "../lib/posts-server";
import { isSocialChannel } from "../lib/channelConfig";
+import {
+ DIGESTS_MANIFEST_VERSION,
+ SITE_DIGESTS_MANIFEST_VERSION,
+ digestPageFileName,
+ newestGeneratedAt,
+ type ChannelDigestsManifest,
+ type DigestsChannelEntry,
+ type DigestsManifest,
+ type VideoDigest,
+} from "../lib/digests";
+import { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME, effectiveDigest } from "../lib/digest";
+import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
// v10: multi-site build. Shared per-channel transcript/subs pages are written
// once; per-site summaries + subs manifests are filtered selections. Bumped to
@@ -113,7 +126,11 @@ import { isSocialChannel } from "../lib/channelConfig";
// v12: the social-post corpus. A `posts` sub-DB keyed [createdAt, channelSlug,
// id] (ISO-8601 sorts correctly, unlike the [uploadDate, …] tuple videos use)
// plus a shared /posts/<slug>/ page tree.
-const SCHEMA_VERSION = 12;
+// v13: the AI-digest corpus. A `digests` sub-DB holding the COMPOSED digest
+// (effectiveDigest of the machine sidecar + the human override file) plus a
+// shared /digests/<slug>/ page tree. Bumped so existing indexes populate the
+// new sub-DB — nothing re-derives it lazily.
+const SCHEMA_VERSION = 13;
// Per-channel post stats, persisted so per-site aggregates survive a no-op
// rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat.
@@ -126,6 +143,14 @@ type ChannelPostsStat = {
signature: string;
};
+// Per-channel digest stats, persisted so the per-site digests manifest survives
+// a no-op rebuild that doesn't re-encode the digest pages. Mirrors
+// ChannelSubsStat / ChannelPostsStat.
+type ChannelDigestStat = {
+ name: string;
+ digestCount: number;
+};
+
// Per-channel subtitle stats, collected while writing the shared subs pages and
// persisted to LMDB so per-site subs manifests can be assembled on a no-op
// rebuild without re-encoding every channel's pages.
@@ -148,6 +173,13 @@ type MtimeRecord = {
transcriptMs: number | null;
subsMs: number | null;
availabilityMs: number | null;
+ // Newest mtime across BOTH digest sidecars — ai-digest.json (machine) and
+ // ai-digest.overrides.json (human). Max-of-two, not a single stat, because a
+ // human correction only ever touches the overrides file: keying off the
+ // machine file alone would leave a corrected digest producing no index
+ // mutation, so the correction would never ship. That is the entire reason
+ // the overrides file exists. See the subsMs multi-file loop it follows.
+ digestMs: number | null;
isDeleted: boolean;
isUnlisted: boolean;
indexKey: IndexKey;
@@ -176,6 +208,7 @@ type LiveEntry = {
subTracks: SubTrack[];
subsMs: number | null;
availabilityMs: number | null;
+ digestMs: number | null;
};
async function exists(p: string): Promise<boolean> {
@@ -283,6 +316,19 @@ async function scanSource(
} catch {
availabilityMs = null;
}
+ // MAX of both digest sidecars, following the subsMs loop above rather
+ // than the single-stat availabilityMs below it. The overrides file is the
+ // one a human writes; if it did not move this number, a hand-corrected
+ // digest would produce no mutation and never reach a built site.
+ let digestMs: number | null = null;
+ for (const name of [DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME]) {
+ try {
+ const ms = (await stat(path.join(fullVideoDir, name))).mtimeMs;
+ if (digestMs === null || ms > digestMs) digestMs = ms;
+ } catch {
+ // Sidecar absent — the common case (102 of ~76,000 videos have one).
+ }
+ }
live.push({
channelSlug: ch.name,
handling: cfg.handling,
@@ -296,6 +342,7 @@ async function scanSource(
subTracks,
subsMs,
availabilityMs,
+ digestMs,
});
}
}
@@ -353,11 +400,13 @@ export async function buildIndex({
const transcriptsOutDir = paths.exportSharedTranscriptsDir;
const subsOutDir = paths.exportSharedSubsDir;
const postsOutDir = paths.exportSharedPostsDir;
+ const digestsOutDir = paths.exportSharedDigestsDir;
await mkdir(path.dirname(dbPath), { recursive: true });
await mkdir(transcriptsOutDir, { recursive: true });
await mkdir(subsOutDir, { recursive: true });
await mkdir(postsOutDir, { recursive: true });
+ await mkdir(digestsOutDir, { recursive: true });
await mkdir(paths.exportSitesIndexDir, { recursive: true });
const root = open({
@@ -414,6 +463,22 @@ export async function buildIndex({
name: "channelPostsStats",
encoding: "msgpack",
});
+ // The AI-digest corpus. Holds the COMPOSED digest (effectiveDigest of the
+ // machine sidecar + the human overrides), keyed like sums/cues/subs, so the
+ // page writer below can stream a channel's digests straight out of LMDB
+ // without re-reading 76k video dirs on a no-op rebuild.
+ const digests = root.openDB<VideoDigest, IndexKey>({
+ name: "digests",
+ encoding: "msgpack",
+ });
+ const digestPageHashes = root.openDB<PageHashRecord, PageHashKey>({
+ name: "digestPageHashes",
+ encoding: "msgpack",
+ });
+ const channelDigestStatsDb = root.openDB<ChannelDigestStat, string>({
+ name: "channelDigestStats",
+ encoding: "msgpack",
+ });
const meta = root.openDB<unknown, string>({
name: "meta",
encoding: "msgpack",
@@ -436,6 +501,9 @@ export async function buildIndex({
await posts.clearAsync();
await postPageHashes.clearAsync();
await channelPostsStatsDb.clearAsync();
+ await digests.clearAsync();
+ await digestPageHashes.clearAsync();
+ await channelDigestStatsDb.clearAsync();
await meta.put("schema", SCHEMA_VERSION);
}
@@ -461,7 +529,8 @@ export async function buildIndex({
prev.metaMs !== s.metaMs ||
prev.transcriptMs !== s.transcriptMs ||
(prev.subsMs ?? null) !== s.subsMs ||
- (prev.availabilityMs ?? null) !== s.availabilityMs
+ (prev.availabilityMs ?? null) !== s.availabilityMs ||
+ (prev.digestMs ?? null) !== s.digestMs
) {
changed.push(s);
}
@@ -580,6 +649,7 @@ export async function buildIndex({
sums.remove(prev.indexKey);
cues.remove(prev.indexKey);
subs.remove(prev.indexKey);
+ digests.remove(prev.indexKey);
byChannel.remove(indexToChannelKey(prev.indexKey));
}
@@ -619,6 +689,54 @@ export async function buildIndex({
if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs);
else subs.remove(indexKey);
+ // The derived layer. Only opened when the scan saw a sidecar, so the
+ // ~76k videos without one cost zero extra reads. What is stored is
+ // effectiveDigest(machine, overrides) — human corrections applied,
+ // `enabled: false` items dropped, chapters sorted by start — so the
+ // corpus ships what a human approved rather than raw model output.
+ if (s.digestMs !== null) {
+ const [machine, overrides] = await Promise.all([
+ loadDigest(videoFullDir),
+ loadDigestOverrides(videoFullDir),
+ ]);
+ const eff = effectiveDigest(machine, overrides);
+ if (eff.chapters.length > 0 || eff.tags.length > 0) {
+ const provenance = {
+ ...(eff.sections.chapters
+ ? { chapters: eff.sections.chapters.provenance }
+ : {}),
+ ...(eff.sections.tags
+ ? { tags: eff.sections.tags.provenance }
+ : {}),
+ };
+ digests.put(indexKey, {
+ slug: `${s.channelSlug}/${summary.id}`,
+ id: summary.id,
+ chapters: eff.chapters.map((c) => ({
+ id: c.id,
+ start: c.start,
+ clock: c.clock,
+ title: c.title,
+ decidedBy: c.decidedBy,
+ })),
+ tags: eff.tags.map((t) => ({
+ id: t.id,
+ tag: t.tag,
+ decidedBy: t.decidedBy,
+ })),
+ generatedAt: newestGeneratedAt(provenance),
+ provenance,
+ // Carried straight through: a shared digest must stay
+ // identifiable as borrowed all the way to the viewer.
+ ...(eff.derivedFrom ? { derivedFrom: eff.derivedFrom } : {}),
+ });
+ } else {
+ digests.remove(indexKey);
+ }
+ } else {
+ digests.remove(indexKey);
+ }
+
let isDeleted = false;
let isUnlisted = false;
if (s.availabilityMs !== null) {
@@ -633,6 +751,7 @@ export async function buildIndex({
transcriptMs: s.transcriptMs,
subsMs: s.subsMs,
availabilityMs: s.availabilityMs,
+ digestMs: s.digestMs,
isDeleted,
isUnlisted,
indexKey,
@@ -652,6 +771,7 @@ export async function buildIndex({
sums.remove(indexKey);
cues.remove(indexKey);
subs.remove(indexKey);
+ digests.remove(indexKey);
byChannel.remove(indexToChannelKey(indexKey));
mtimes.remove(pathKey);
}
@@ -659,6 +779,7 @@ export async function buildIndex({
await sums.flushed;
await cues.flushed;
await subs.flushed;
+ await digests.flushed;
await byChannel.flushed;
await mtimes.flushed;
@@ -689,6 +810,12 @@ export async function buildIndex({
pagesWritten: number;
pagesSkipped: number;
slugToPage: Record<string, number>;
+ // Content hash of each emitted page, by page index. Recorded for BOTH the
+ // written and the skipped branch (a skipped page's hash is by definition
+ // the one already on disk), so this is the true current content hash
+ // regardless of whether the page was rewritten this build. The digests tree
+ // publishes these in its manifest as a client cache version.
+ pageHashes: string[];
};
const createPageWriter = (opts: PageWriterOpts) => {
@@ -705,6 +832,7 @@ export async function buildIndex({
let pagesSkippedLocal = 0;
let dirEnsured = !opts.ensureDir;
const slugToPage: Record<string, number> = {};
+ const pageHashes: string[] = [];
const writeChunk = async (chunk: string): Promise<void> => {
const s = stream;
@@ -743,6 +871,7 @@ export async function buildIndex({
});
});
const digest = hash.digest("hex");
+ pageHashes[pageIdx] = digest;
const prev = opts.getPrevHash(pageIdx);
if (prev === digest && (await exists(outPath))) {
await rm(tmpPath, { force: true });
@@ -794,6 +923,7 @@ export async function buildIndex({
pagesWritten: pagesWrittenLocal,
pagesSkipped: pagesSkippedLocal,
slugToPage,
+ pageHashes,
};
};
@@ -1240,6 +1370,156 @@ export async function buildIndex({
}
// ---------------------------------------------------------------------------
+ // The AI-digest corpus: a third shared per-channel page tree, holding the
+ // chapters + topic tags derived from each transcript. Unlike transcripts and
+ // subs it is SPARSE — a channel emits a manifest only if at least one of its
+ // videos has been digested, and a manifest's slugToPage lists only those
+ // videos. That sparsity is load-bearing downstream: it is what lets the
+ // viewer decide whether to offer a Digest control without a per-video fetch.
+ // ---------------------------------------------------------------------------
+ const channelDigestStats = new Map<string, ChannelDigestStat>();
+ let digestPagesWritten = 0;
+ let digestPagesSkipped = 0;
+ let digestTotalCount = 0;
+
+ if (sharedNeedsBuild) {
+ for (const channelSlug of Array.from(channelConfigs.keys()).sort()) {
+ const cfg = channelConfigs.get(channelSlug)!;
+ const digestChannelDir = path.join(digestsOutDir, channelSlug);
+ let digestCount = 0;
+
+ const digestWriter = createPageWriter({
+ outDir: digestChannelDir,
+ fileName: digestPageFileName,
+ maxPageBytes: maxTranscriptPageBytes,
+ ensureDir: true,
+ getPrevHash: (idx) => digestPageHashes.get([channelSlug, idx])?.hash,
+ setHash: (idx, record) => {
+ digestPageHashes.put([channelSlug, idx], record);
+ },
+ onLog: log,
+ });
+
+ // Oldest-first (the byChannel key order), matching the transcript tree so
+ // adding a newer digest only dirties the last page.
+ for (const { key } of byChannel.getRange({
+ start: [channelSlug],
+ end: [channelSlug, ""],
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== channelSlug) continue;
+ const indexKey: IndexKey = [ck[1], ck[0], ck[2]];
+ const digest = digests.get(indexKey);
+ if (!digest) continue;
+ await digestWriter.push(JSON.stringify(digest), digest.id);
+ digestCount++;
+ }
+
+ const {
+ pageCount: digestPageCount,
+ pagesWritten: chDigestsWritten,
+ pagesSkipped: chDigestsSkipped,
+ slugToPage: digestSlugToPage,
+ pageHashes: digestHashes,
+ } = await digestWriter.finish();
+ digestPagesWritten += chDigestsWritten;
+ digestPagesSkipped += chDigestsSkipped;
+
+ if (digestCount === 0) {
+ // No digests in this channel — drop any stale tree + hashes so a
+ // channel whose digests were deleted stops advertising them.
+ await rm(digestChannelDir, { recursive: true, force: true });
+ for (const { key } of digestPageHashes.getRange({
+ start: [channelSlug],
+ end: [channelSlug, Number.MAX_SAFE_INTEGER],
+ })) {
+ digestPageHashes.remove(key as PageHashKey);
+ }
+ continue;
+ }
+
+ const digestKeep = new Set<string>(["manifest.json"]);
+ for (let i = 0; i < digestPageCount; i++) {
+ digestKeep.add(digestPageFileName(i));
+ }
+ for (const name of await readdir(digestChannelDir).catch(
+ () => [] as string[],
+ )) {
+ if (digestKeep.has(name)) continue;
+ await rm(path.join(digestChannelDir, name), { force: true });
+ }
+ for (const { key } of digestPageHashes.getRange({
+ start: [channelSlug, digestPageCount],
+ end: [channelSlug, Number.MAX_SAFE_INTEGER],
+ })) {
+ digestPageHashes.remove(key as PageHashKey);
+ }
+
+ const channelDigestsManifest: ChannelDigestsManifest = {
+ version: DIGESTS_MANIFEST_VERSION,
+ channelSlug,
+ pageCount: digestPageCount,
+ maxPageBytes: maxTranscriptPageBytes,
+ generatedAt,
+ slugToPage: digestSlugToPage,
+ pageHashes: digestHashes,
+ };
+ await writeJsonAtomic(
+ path.join(digestChannelDir, "manifest.json"),
+ channelDigestsManifest,
+ );
+ channelDigestStats.set(channelSlug, {
+ name: cfg.name ?? channelSlug,
+ digestCount,
+ });
+ digestTotalCount += digestCount;
+ }
+
+ await digestPageHashes.flushed;
+
+ // Drop shared digest dirs for channels that no longer have any.
+ const topDigestEntries = await readdir(digestsOutDir, {
+ withFileTypes: true,
+ }).catch(() => [] as Dirent[]);
+ for (const e of topDigestEntries) {
+ if (e.isDirectory()) {
+ if (!channelDigestStats.has(e.name)) {
+ await rm(path.join(digestsOutDir, e.name), {
+ recursive: true,
+ force: true,
+ });
+ }
+ } else if (e.isFile()) {
+ // The site-level digests manifest is per-site; the shared root holds
+ // only per-channel dirs.
+ await rm(path.join(digestsOutDir, e.name), { force: true });
+ }
+ }
+
+ await channelDigestStatsDb.clearAsync();
+ for (const [slug, statRec] of channelDigestStats) {
+ channelDigestStatsDb.put(slug, statRec);
+ }
+ await channelDigestStatsDb.flushed;
+
+ if (digestTotalCount > 0 || digestPagesWritten > 0) {
+ log(
+ `Digest pages: ${digestPagesWritten} written, ${digestPagesSkipped} unchanged ` +
+ `across ${channelDigestStats.size} channel(s) (${digestTotalCount} digests).`,
+ );
+ }
+ } else {
+ for (const { key, value } of channelDigestStatsDb.getRange()) {
+ channelDigestStats.set(key as string, value as ChannelDigestStat);
+ }
+ if (channelDigestStats.size > 0) {
+ log(
+ `Shared digest pages up to date; ${channelDigestStats.size} channel(s) with digests.`,
+ );
+ }
+ }
+
+ // ---------------------------------------------------------------------------
// Per-site aggregates: a filtered summaries index (pages + manifest) and a
// site-level subs manifest, one bundle per configured site. The heavy
// per-channel page trees above are shared; here we only select + regroup.
@@ -1277,9 +1557,11 @@ export async function buildIndex({
const summariesOut = siteSummariesDir(paths, site.siteId);
const subsOut = siteSubsDir(paths, site.siteId);
const postsOut = sitePostsDir(paths, site.siteId);
+ const digestsOut = siteDigestsDir(paths, site.siteId);
const summariesManifestPath = path.join(summariesOut, "manifest.json");
const subsManifestPath = path.join(subsOut, "manifest.json");
const postsManifestPath = path.join(postsOut, "manifest.json");
+ const digestsManifestPath = path.join(digestsOut, "manifest.json");
const fingerprint = JSON.stringify({
gen: generation,
@@ -1447,6 +1729,33 @@ export async function buildIndex({
await mkdir(postsOut, { recursive: true });
await writeJsonAtomic(postsManifestPath, sitePostsManifest);
+ // --- site-level digests manifest (filtered to member channels) ---
+ // Only channels that actually carry digests contribute, so a site with none
+ // ships a manifest with an empty channel list rather than no manifest —
+ // keeping the client's fetch unconditional, as with posts.
+ const digestEntries: DigestsChannelEntry[] = [];
+ let digestsTotalForSite = 0;
+ for (const slug of memberSlugs) {
+ const statRec = channelDigestStats.get(slug);
+ if (!statRec || statRec.digestCount === 0) continue;
+ digestEntries.push({
+ name: statRec.name,
+ slug,
+ digestCount: statRec.digestCount,
+ groupId: slugGroup.get(slug) ?? site.defaultGroupId,
+ });
+ digestsTotalForSite += statRec.digestCount;
+ }
+ const siteDigestsManifest: DigestsManifest = {
+ version: SITE_DIGESTS_MANIFEST_VERSION,
+ channels: digestEntries.sort((a, b) => a.name.localeCompare(b.name)),
+ totalCount: digestsTotalForSite,
+ generatedAt: new Date().toISOString(),
+ siteId: site.siteId,
+ };
+ await mkdir(digestsOut, { recursive: true });
+ await writeJsonAtomic(digestsManifestPath, siteDigestsManifest);
+
await meta.put(fpKey, fingerprint);
sitesBuilt++;
aggregateSummaryPages += pageIndex;
diff --git a/common/lib/channelSignature.ts b/common/lib/channelSignature.ts
@@ -20,10 +20,15 @@ import { existsSync } from "node:fs";
import { open } from "lmdb";
import type { Paths } from "./paths";
+// Structural subset of buildIndex.ts's MtimeRecord — only the fields that feed
+// the signature. `digestMs` is the newest mtime across BOTH digest sidecars
+// (machine + human overrides); without it in the hash below, a digest-only
+// change would leave every archive believing the channel was unchanged.
type MtimeRecord = {
metaMs: number;
transcriptMs: number | null;
subsMs: number | null;
+ digestMs: number | null;
};
type PathKey = [string, string];
@@ -74,7 +79,7 @@ export function openChannelSigner(paths: Paths): ChannelSigner {
if (k[0] !== slug) break;
sawAny = true;
h.update(
- `${k[1]}\t${value.metaMs}\t${value.transcriptMs ?? ""}\t${value.subsMs ?? ""}\n`,
+ `${k[1]}\t${value.metaMs}\t${value.transcriptMs ?? ""}\t${value.subsMs ?? ""}\t${value.digestMs ?? ""}\n`,
);
}
// A channel with zero indexed videos still gets a stable signature (schema
diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts
@@ -80,6 +80,48 @@ test("renderSiteLlmsTxt: title, corpus link, channels", () => {
assert.match(txt, /Bulk archives/);
});
+test("buildSiteCorpus: digest layer is advertised only where it exists", () => {
+ const corpus = buildSiteCorpus(descriptor(), {
+ hasArchives: false,
+ digestCounts: { alice: 7, bob: 0 },
+ });
+ // Advertised on the channel that has digests…
+ assert.equal(corpus.channels[0].digestCount, 7);
+ assert.equal(
+ corpus.channels[0].manifests.digests,
+ "/digests/alice/manifest.json",
+ );
+ // …and absent, not zero-valued, on the one that doesn't. A `digests` pointer
+ // to a manifest that was never written would send clients to a 404.
+ assert.equal(corpus.channels[1].digestCount, undefined);
+ assert.equal(corpus.channels[1].manifests.digests, undefined);
+ assert.ok(corpus.digestScheme, "some digests → digestScheme block");
+ // The sparsity contract is the part a client must not get wrong.
+ assert.match(corpus.digestScheme!.description, /SPARSE/);
+ assert.match(corpus.digestScheme!.chapterTiming, /SECONDS/);
+});
+
+test("buildSiteCorpus: no digests → no digestScheme, unchanged channels", () => {
+ const corpus = buildSiteCorpus(descriptor(), { hasArchives: false });
+ assert.equal(corpus.digestScheme, undefined);
+ assert.equal(corpus.channels[0].manifests.digests, undefined);
+ assert.equal(corpus.channels[0].digestCount, undefined);
+});
+
+test("renderSiteLlmsTxt: names the digest layer only when present", () => {
+ const withDigests = renderSiteLlmsTxt(
+ buildSiteCorpus(descriptor({ siteUrl: "https://demo.example" }), {
+ hasArchives: false,
+ digestCounts: { alice: 7 },
+ }),
+ );
+ assert.match(withDigests, /AI digests: 7 of these transcripts/);
+ const without = renderSiteLlmsTxt(
+ buildSiteCorpus(descriptor(), { hasArchives: false }),
+ );
+ assert.doesNotMatch(without, /AI digests/);
+});
+
test("buildHubCorpus: drops members with no siteUrl, links each corpus", () => {
const hub = buildHubCorpus(
[
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -13,7 +13,9 @@ import type { PublicSiteDescriptor } from "./siteDescriptor";
// v2: the corpus now also describes the parallel social-post layer
// (postScheme + per-channel posts manifest pointers).
-export const CORPUS_SPEC_VERSION = 2;
+// v3: …and the DERIVED layer — AI digests (chapters + topic tags) served under
+// the same shard scheme (digestScheme + per-channel digests manifest pointers).
+export const CORPUS_SPEC_VERSION = 3;
// How to resolve a single transcript from the paginated shards, described once
// and embedded in every corpus.json so any HTTP client can navigate without
@@ -63,6 +65,39 @@ const POST_SCHEME = {
permalink: "each post carries its own canonical `url`; no timestamp fragment applies",
} as const;
+// The AI-digest corpus: a DERIVED layer over transcripts, not a parallel source
+// like posts. Same paginated-shard scheme, but sparse — a video absent from a
+// digests manifest simply has not been digested, which is the normal case.
+const DIGEST_SCHEME = {
+ description:
+ "AI digests are chapters and topic tags DERIVED from a video's transcript " +
+ "by a local model, composed with any human corrections before publication. " +
+ "They are served as paginated JSON shards under the same scheme: (1) GET " +
+ "the channel's digests manifest; (2) look up the video id in its " +
+ "`slugToPage` map to get a page number N; (3) GET page-<NNNN>.json and take " +
+ "the record whose `id` matches. THE LAYER IS SPARSE: a video id absent from " +
+ "`slugToPage` has no digest, and a channel with no digests has no manifest " +
+ "at all. Absence means 'not yet generated', never 'nothing to say'.",
+ digestsManifest:
+ "<channel.manifests.digests> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }",
+ digestPage:
+ "/digests/<slug>/page-<NNNN>.json -> array of { id, slug, generatedAt, " +
+ "chapters: [{ id, start, clock, title, decidedBy }], tags: [{ id, tag, " +
+ "decidedBy }], provenance, derivedFrom? }",
+ chapterTiming:
+ "`start` is SECONDS, already snapped to a real transcript cue boundary — use " +
+ "it to seek. `clock` is the raw HH:MM:SS string the model emitted, kept for " +
+ "auditing; do not parse it for timing.",
+ attribution:
+ "`decidedBy` is \"ai\" or \"human\" per item, so machine output and human " +
+ "corrections stay distinguishable after composition.",
+ derivedFrom:
+ "When present, this digest was generated for a DIFFERENT video (the canonical " +
+ "member of a duplicate cluster) and shared onto this one; it names that video " +
+ "and the measured cue-timing offset in seconds. Treat its chapter titles as " +
+ "describing the canonical upload.",
+} as const;
+
export type CorpusChannel = {
slug: string;
name: string;
@@ -73,9 +108,16 @@ export type CorpusChannel = {
subs: string;
// Present only for social channels (the posts corpus).
posts?: string;
+ // Present only for channels with at least one digested video.
+ digests?: string;
};
// Present only for social channels.
postCount?: number;
+ // Number of this channel's videos that carry a digest. Present only when
+ // non-zero, and deliberately reported ALONGSIDE videoCount rather than
+ // instead of it: the ratio is the coverage of the derived layer, which is
+ // what tells a client whether to expect a digest for an arbitrary video.
+ digestCount?: number;
};
export type SiteCorpus = {
@@ -94,6 +136,8 @@ export type SiteCorpus = {
shardScheme: typeof SHARD_SCHEME;
// Present when this site includes at least one social channel.
postScheme?: typeof POST_SCHEME;
+ // Present when this site ships at least one digested video.
+ digestScheme?: typeof DIGEST_SCHEME;
// Present when this build ships bulk-download archives (whole-channel zips).
bulkArchives?: { manifest: string; note: string };
// Pointer to the human page and BYO-key chat.
@@ -144,24 +188,33 @@ export function buildSiteCorpus(
// slug -> archived post count, for the social channels in this site. Absent
// / empty means the site has no posts corpus and postScheme is omitted.
postCounts?: Record<string, number>;
+ // slug -> digested video count. Absent / empty means the site ships no
+ // derived layer and digestScheme is omitted.
+ digestCounts?: Record<string, number>;
},
): SiteCorpus {
const base = descriptor.siteUrl;
const postCounts = opts.postCounts ?? {};
+ const digestCounts = opts.digestCounts ?? {};
const channels: CorpusChannel[] = descriptor.channels.map((c) => {
const postCount = postCounts[c.slug];
+ const digestCount = digestCounts[c.slug];
return {
slug: c.slug,
name: c.name,
videoCount: c.count,
...(c.groupId ? { groupId: c.groupId } : {}),
...(postCount ? { postCount } : {}),
+ ...(digestCount ? { digestCount } : {}),
manifests: {
transcripts: join(base, `/transcripts/${c.slug}/manifest.json`),
subs: join(base, `/subs/${c.slug}/manifest.json`),
...(postCount
? { posts: join(base, `/posts/${c.slug}/manifest.json`) }
: {}),
+ ...(digestCount
+ ? { digests: join(base, `/digests/${c.slug}/manifest.json`) }
+ : {}),
},
};
});
@@ -188,6 +241,11 @@ export function buildSiteCorpus(
if (Object.values(postCounts).some((n) => n > 0)) {
corpus.postScheme = POST_SCHEME;
}
+ // Likewise for the derived layer: a site with no digests is unchanged apart
+ // from the spec bump.
+ if (Object.values(digestCounts).some((n) => n > 0)) {
+ corpus.digestScheme = DIGEST_SCHEME;
+ }
if (opts.hasArchives) {
corpus.bulkArchives = {
manifest: join(base, "/archives/manifest.json"),
@@ -263,6 +321,17 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
`- [corpus.json](${join(base, "/corpus.json")}): machine-readable index — ` +
`channels and how to fetch any transcript from the paginated JSON shards.`,
);
+ if (corpus.digestScheme) {
+ const digested = corpus.channels.reduce(
+ (n, c) => n + (c.digestCount ?? 0),
+ 0,
+ );
+ out.push(
+ `- AI digests: ${digested.toLocaleString()} of these transcripts also carry ` +
+ `machine-generated chapters and topic tags, served under /digests/ — ` +
+ `see corpus.json's digestScheme. Coverage is partial and growing.`,
+ );
+ }
if (corpus.bulkArchives) {
out.push(
`- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` +
diff --git a/common/lib/digests.test.ts b/common/lib/digests.test.ts
@@ -0,0 +1,180 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ digestPageFileName,
+ manifestHasDigest,
+ newestGeneratedAt,
+ type ChannelDigestsManifest,
+ type VideoDigest,
+} from "./digests";
+import {
+ effectiveDigest,
+ type DigestOverrides,
+ type DigestRecord,
+} from "./digest";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/digests.test.ts
+
+test("digestPageFileName: zero-padded to 4, matching every other page tree", () => {
+ assert.equal(digestPageFileName(0), "page-0000.json");
+ assert.equal(digestPageFileName(7), "page-0007.json");
+ assert.equal(digestPageFileName(1234), "page-1234.json");
+});
+
+function manifest(
+ slugToPage: Record<string, number>,
+): ChannelDigestsManifest {
+ return {
+ version: 1,
+ channelSlug: "alice",
+ pageCount: 1,
+ maxPageBytes: 1_000_000,
+ generatedAt: "2026-07-29T00:00:00.000Z",
+ slugToPage,
+ };
+}
+
+test("manifestHasDigest: slugToPage IS the existence check", () => {
+ const m = manifest({ vaaa: 0, vbbb: 0 });
+ assert.equal(manifestHasDigest(m, "vaaa"), true);
+ // Page 0 is a real page — a falsy page index must not read as "absent".
+ assert.equal(manifestHasDigest(m, "vbbb"), true);
+ assert.equal(manifestHasDigest(m, "vccc"), false);
+ // No manifest at all (channel has zero digests) is simply "no".
+ assert.equal(manifestHasDigest(null, "vaaa"), false);
+});
+
+test("newestGeneratedAt: max across sections, tolerant of missing ones", () => {
+ assert.equal(
+ newestGeneratedAt({
+ chapters: { generatedAt: "2026-07-01T00:00:00.000Z" },
+ tags: { generatedAt: "2026-07-20T00:00:00.000Z" },
+ }),
+ "2026-07-20T00:00:00.000Z",
+ );
+ // One section only.
+ assert.equal(
+ newestGeneratedAt({ chapters: { generatedAt: "2026-07-01T00:00:00.000Z" } }),
+ "2026-07-01T00:00:00.000Z",
+ );
+ // Nothing to date — the cache treats "" as "always a miss", which is the safe
+ // direction: it re-fetches rather than serving something it can't version.
+ assert.equal(newestGeneratedAt({}), "");
+});
+
+// --- the shipped record is the COMPOSED digest, not the raw sidecar ---------
+// This is the property the whole build edge depends on: what reaches a page
+// file is effectiveDigest(machine, overrides), so a human correction ships and
+// a human rejection does not.
+
+function record(): DigestRecord {
+ return {
+ digestSchemaVersion: 1,
+ promptVersion: 2,
+ contextHash: "",
+ warnings: [
+ { code: "out-of-range", section: "chapters", chunk: 2, value: "03:11:00" },
+ ],
+ sections: {
+ chapters: {
+ provenance: {
+ appId: "ollama-direct",
+ model: "qwen2.5:7b",
+ lane: "local-gpu",
+ generatedAt: "2026-07-20T00:00:00.000Z",
+ promptVersion: 2,
+ contextHash: "",
+ },
+ items: [
+ { id: "c300", start: 300, clock: "00:05:00", title: "Second", decidedBy: "ai" },
+ { id: "c0", start: 0, clock: "00:00:00", title: "Machine title", decidedBy: "ai" },
+ { id: "c900", start: 900, clock: "00:15:00", title: "Rejected", decidedBy: "ai" },
+ ],
+ },
+ },
+ };
+}
+
+// Mirror of the projection buildIndex.ts performs when it stores a record.
+function toVideoDigest(slug: string, id: string): VideoDigest {
+ const eff = effectiveDigest(record(), {
+ version: 1,
+ // Authored as PATCHES (id + title, no start/clock) — the shape
+ // digest-server.ts's sanitizeChapters actually emits for a hand-written
+ // overrides file, and which it casts the same way. mergeItems is built to
+ // spread the machine item underneath, so the snapped start survives.
+ chapters: [
+ // Replaces a generated item by id (a retitle)…
+ { id: "c0", title: "Human title", decidedBy: "human" },
+ // …and suppresses another without deleting it.
+ { id: "c900", title: "", decidedBy: "human", enabled: false },
+ ] as DigestOverrides["chapters"],
+ });
+ const provenance = {
+ ...(eff.sections.chapters
+ ? { chapters: eff.sections.chapters.provenance }
+ : {}),
+ ...(eff.sections.tags ? { tags: eff.sections.tags.provenance } : {}),
+ };
+ return {
+ slug,
+ id,
+ chapters: eff.chapters.map((c) => ({
+ id: c.id,
+ start: c.start,
+ clock: c.clock,
+ title: c.title,
+ decidedBy: c.decidedBy,
+ })),
+ tags: eff.tags.map((t) => ({ id: t.id, tag: t.tag, decidedBy: t.decidedBy })),
+ generatedAt: newestGeneratedAt(provenance),
+ provenance,
+ ...(eff.derivedFrom ? { derivedFrom: eff.derivedFrom } : {}),
+ };
+}
+
+test("shipped digest: overrides applied, suppressed items dropped, sorted by start", () => {
+ const d = toVideoDigest("alice/vaaa", "vaaa");
+ assert.deepEqual(
+ d.chapters.map((c) => c.title),
+ ["Human title", "Second"],
+ );
+ // Sorted by start, not by the order the model emitted them.
+ assert.deepEqual(d.chapters.map((c) => c.start), [0, 300]);
+ // A retitled chapter keeps the machine's snapped start — an override authored
+ // as a patch (id + title) must not blank it, or the seek target is lost.
+ assert.equal(d.chapters[0].start, 0);
+ assert.equal(d.chapters[0].decidedBy, "human");
+ assert.equal(d.chapters[1].decidedBy, "ai");
+});
+
+test("shipped digest: warnings are operator telemetry and never ship", () => {
+ const d = toVideoDigest("alice/vaaa", "vaaa") as VideoDigest &
+ Record<string, unknown>;
+ assert.equal(d.warnings, undefined);
+ assert.equal(d.history, undefined);
+ assert.equal(JSON.stringify(d).includes("out-of-range"), false);
+});
+
+test("shipped digest: provenance and the cache version come through", () => {
+ const d = toVideoDigest("alice/vaaa", "vaaa");
+ assert.equal(d.provenance.chapters?.model, "qwen2.5:7b");
+ assert.equal(d.provenance.chapters?.lane, "local-gpu");
+ assert.equal(d.generatedAt, "2026-07-20T00:00:00.000Z");
+});
+
+test("shipped digest: derivedFrom survives so a borrowed digest stays borrowed", () => {
+ const shared: DigestRecord = {
+ ...record(),
+ derivedFrom: {
+ slug: "bob/vzzz",
+ clusterId: "cl1",
+ sharedAt: "2026-07-21T00:00:00.000Z",
+ offsetSeconds: 0.4,
+ },
+ };
+ const eff = effectiveDigest(shared, null);
+ assert.equal(eff.derivedFrom?.slug, "bob/vzzz");
+ assert.equal(eff.derivedFrom?.offsetSeconds, 0.4);
+});
diff --git a/common/lib/digests.ts b/common/lib/digests.ts
@@ -0,0 +1,160 @@
+// The SHIPPED shape of the AI-digest corpus: a per-channel paginated page tree
+// at /digests/<slug>/{manifest,page-NNNN}.json, mirroring common/lib/manifest.ts
+// (transcripts) and common/lib/posts.ts (posts).
+//
+// This is deliberately NOT DigestRecord (common/lib/digest.ts). That type is the
+// on-disk sidecar the generator owns: two files per video, machine output plus a
+// human shadow, carrying operator telemetry. What ships is the COMPOSED result
+// of effectiveDigest() — human overrides applied, suppressed items dropped,
+// sorted — which is the only version a reader should ever see.
+//
+// Two things are deliberately absent from the shipped record:
+//
+// warnings[] operator telemetry for the editor panel and the Phase 11a review
+// queue. A reader has no use for "the model proposed 13 chapters
+// that were clamped away"; shipping it would put the corpus's
+// failures in front of the audience rather than the operator.
+// history[] the regeneration audit trail. Same reasoning, plus it is the
+// single largest field in a mature sidecar.
+//
+// SPARSE BY DESIGN. A channel's manifest lists only the videos that actually
+// have a digest — 102 of ~76,000 at the time of writing. `slugToPage` is
+// therefore also the existence check: a videoId absent from it has no digest,
+// which is what lets the viewer decide whether to render its Digest control
+// without a per-video fetch (the export/app/lib/duplicates.ts hasDuplicates()
+// idiom). A channel with zero digests gets no manifest at all.
+
+import type {
+ DecidedBy,
+ DigestDerivedFrom,
+ DigestProvenance,
+} from "./digest";
+
+// v1: the initial shipped shape.
+export const DIGESTS_MANIFEST_VERSION = 1;
+
+// Zero-padded to 4 digits, the convention every other page tree uses
+// (pageFileName / transcriptPageFileName / subsPageFileName / postsPageFileName).
+export function digestPageFileName(index: number): string {
+ return `page-${String(index).padStart(4, "0")}.json`;
+}
+
+// A titled moment. `start` is SECONDS and is what a seek uses: the parser has
+// already snapped it to a real cue boundary. `clock` is the raw HH:MM:SS string
+// the model emitted, carried for auditing — NEVER re-parse it to seek, or a
+// model that miscounted gets to move the playhead.
+export type DigestPageChapter = {
+ id: string;
+ start: number;
+ clock: string;
+ title: string;
+ // "human" when an override replaced or added this item, so the viewer can be
+ // honest about which chapters a person wrote.
+ decidedBy: DecidedBy;
+};
+
+export type DigestPageTag = {
+ id: string;
+ tag: string;
+ decidedBy: DecidedBy;
+};
+
+// One video's shipped digest — the unit a page file holds an array of.
+export type VideoDigest = {
+ // "channelSlug/videoId", matching TranscriptDetail.slug.
+ slug: string;
+ // The bare video id, matching the manifest's slugToPage key.
+ id: string;
+ chapters: DigestPageChapter[];
+ tags: DigestPageTag[];
+ // Newest section generatedAt in this record (ISO-8601), or "" when no section
+ // carries one. THE PER-ENTRY VERSION: digests are regenerated in place, so a
+ // client's cached copy stays structurally valid while going stale, and a
+ // structure sniff (what transcriptStore.ts gets away with) cannot detect that.
+ // The client cache compares this field and treats a mismatch as a miss —
+ // otherwise a corrected digest never reaches a returning reader.
+ generatedAt: string;
+ // Per-SECTION provenance, because one video can legitimately hold chapters
+ // from the local lane and tags from the metered one.
+ provenance: {
+ chapters?: DigestProvenance;
+ tags?: DigestProvenance;
+ };
+ // Set when this digest was generated for a DIFFERENT video (a duplicate
+ // cluster's canonical member) and shared onto this one. Presenting a borrowed
+ // digest as native is the failure mode that looks like success: every chapter
+ // reads plausibly while describing another upload. The viewer renders this.
+ derivedFrom?: DigestDerivedFrom;
+};
+
+export type DigestPage = VideoDigest[];
+
+export type ChannelDigestsManifest = {
+ version: number;
+ channelSlug: string;
+ pageCount: number;
+ maxPageBytes: number;
+ generatedAt: string;
+ // videoId -> page index. Only DIGESTED videos appear; see the sparse note above.
+ slugToPage: Record<string, number>;
+ // Content hash of each page, indexed by page number — the CACHE VERSION.
+ //
+ // A client cannot know whether its cached copy of one video's digest is
+ // current without something to compare, and the record's own generatedAt is
+ // only legible once the page has been fetched (which is the cost the cache
+ // exists to avoid). The build already computes these hashes for its own
+ // page-skip logic, so surfacing them is free and gives exact per-page
+ // versioning: any regeneration, retitle or suppression changes the hash of
+ // the page it lands on and nothing else.
+ //
+ // Optional so a manifest written without it still parses; a client that
+ // finds it absent simply treats every cached entry as a miss.
+ pageHashes?: string[];
+};
+
+// --- site level ------------------------------------------------------------
+// The per-SITE manifest at /digests/manifest.json: which member channels carry
+// digests, and how many. Mirrors the site posts manifest, and exists for the
+// same reason — compose-site.ts reads it to populate corpus.json's per-channel
+// digests pointer without walking every channel's page tree.
+export const SITE_DIGESTS_MANIFEST_VERSION = 1;
+
+export type DigestsChannelEntry = {
+ name: string;
+ slug: string;
+ digestCount: number;
+ groupId?: string;
+};
+
+export type DigestsManifest = {
+ version: number;
+ channels: DigestsChannelEntry[];
+ totalCount: number;
+ generatedAt: string;
+ siteId?: string;
+};
+
+// Whether a manifest covers a given video. The whole existence check, in one
+// place, so the viewer's control-gating and the fetch path agree on what
+// "has a digest" means.
+export function manifestHasDigest(
+ manifest: ChannelDigestsManifest | null,
+ videoId: string,
+): boolean {
+ return manifest?.slugToPage[videoId] !== undefined;
+}
+
+// The newest generatedAt across a record's sections — the value that lands in
+// VideoDigest.generatedAt. Pure, so both the builder and any test can call it.
+export function newestGeneratedAt(provenance: {
+ chapters?: Pick<DigestProvenance, "generatedAt">;
+ tags?: Pick<DigestProvenance, "generatedAt">;
+}): string {
+ const stamps = [
+ provenance.chapters?.generatedAt,
+ provenance.tags?.generatedAt,
+ ].filter((s): s is string => typeof s === "string" && s !== "");
+ if (stamps.length === 0) return "";
+ // ISO-8601 sorts lexicographically, so no Date parsing is needed.
+ return stamps.reduce((a, b) => (a > b ? a : b));
+}
diff --git a/common/lib/paths.ts b/common/lib/paths.ts
@@ -61,6 +61,13 @@ export type Paths = {
// Served per-channel social-post page tree (/posts/<slug>/{manifest,page-NNNN}.json).
// The parallel corpus to transcripts/subs — see common/lib/posts.ts.
exportPostsDir: string;
+ // Served per-channel AI-digest page tree
+ // (/digests/<slug>/{manifest,page-NNNN}.json). The DERIVED corpus: chapters
+ // and topic tags generated from a transcript, composed with human overrides
+ // at index time. Sparse by design — only videos that have been digested
+ // appear in a manifest's slugToPage, which is what lets the viewer hide its
+ // Digest control without a per-video fetch. See common/lib/digests.ts.
+ exportDigestsDir: string;
exportStatsDir: string;
// Staging area (NOT served) where the shared index + per-site aggregates are
// built before composition. Shared per-channel transcript/subs pages are
@@ -71,6 +78,7 @@ export type Paths = {
exportSharedTranscriptsDir: string;
exportSharedSubsDir: string;
exportSharedPostsDir: string;
+ exportSharedDigestsDir: string;
exportSitesIndexDir: string;
// Per-site extracted static output (`out/`) from an isolated (Docker) build,
// keyed exportBuildsDir/<siteId>. Sibling of .export-index. The wrangler deploy
@@ -162,12 +170,14 @@ export function getPaths(): Paths {
exportTranscriptsDir: path.join(exportPublicDir, "transcripts"),
exportSubsDir: path.join(exportPublicDir, "subs"),
exportPostsDir: path.join(exportPublicDir, "posts"),
+ exportDigestsDir: path.join(exportPublicDir, "digests"),
exportStatsDir: path.join(exportPublicDir, "stats"),
exportIndexDir,
exportSharedDir,
exportSharedTranscriptsDir: path.join(exportSharedDir, "transcripts"),
exportSharedSubsDir: path.join(exportSharedDir, "subs"),
exportSharedPostsDir: path.join(exportSharedDir, "posts"),
+ exportSharedDigestsDir: path.join(exportSharedDir, "digests"),
exportSitesIndexDir: path.join(exportIndexDir, "sites"),
exportBuildsDir:
process.env.EXPORT_BUILDS_DIR ??
diff --git a/common/lib/site.ts b/common/lib/site.ts
@@ -150,6 +150,10 @@ export function sitePostsDir(paths: Paths, siteId: string): string {
return path.join(siteIndexDir(paths, siteId), "posts");
}
+export function siteDigestsDir(paths: Paths, siteId: string): string {
+ return path.join(siteIndexDir(paths, siteId), "digests");
+}
+
export function siteStatsDir(paths: Paths, siteId: string): string {
return path.join(siteIndexDir(paths, siteId), "stats");
}
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **AI chapters: jump straight to the part of a video you want.** Videos that have been through the local-AI digest pass now ship their derived **chapters and topic tags** to the site, and the player gains a third panel beside Transcript and Live chat. Open it and you get a titled list of moments — click one and the player **seeks there**; the chapter you're currently inside stays marked as the video plays. The layer is **sparse on purpose and honest about it**: only a small, growing fraction of the archive has been digested (generation is a multi-week GPU pass), so the control simply **isn't shown** on a video that has no digest, rather than offering a button that opens an empty panel — and if a digest can't be loaded, the panel snaps back to the transcript with a notice instead of stranding you. What ships is the **composed** digest: any human correction is applied and any chapter a human rejected is dropped, so you see what a person approved rather than raw model output, with hand-edited chapters marked **edited** and a provenance line naming the model that wrote the rest. **A digest borrowed from a duplicate upload says so, prominently** — when the same recording exists twice in the archive, one copy's chapters can be shared onto the other, and the panel names the source video and the measured timing offset rather than passing them off as native (plausible chapters describing a *different* upload is the failure that looks like success). Digests live at `/digests/<channel>/` under the same paginated-shard scheme as transcripts and posts, are offline-cached by the service worker, and are described in `corpus.json` — which bumps to **spec 3** with a `digestScheme` and per-channel `digests` manifest pointers, so an AI tool reading the corpus can navigate them and knows that an absent video means "not yet generated" rather than "nothing to say". Deep links carry the panel (`?vm=digest`), and share links reopen on it. Operator telemetry (why the model's proposals were rejected, the regeneration history) is deliberately **not** shipped — that stays in the editor. See `common/lib/digests.ts`, `common/components/{digestCache,digestStore}.ts`, `common/components/{PlayerProvider,TranscriptModal,urlState}.tsx/ts`, and `export/e2e/modal-digest.spec.ts`.
- **Search results tell you when a video exists elsewhere in the archive, and take you there.** A result that belongs to a duplicate cluster now carries a **Dupe** badge, and a strip under the card header offers one button per other copy — the same recording mirrored to another platform, or re-uploaded on another channel. Clicking one opens that copy in the player. **The jump is honest about what it knows:** matching content does not imply matching timings (a mirror with a longer intro carries the same words at shifted times), so a button only carries your current timestamp when detection *measured* the two as aligned; otherwise it says so and opens the other copy from the start. A copy whose alignment was never measured is treated as not aligned. Only clusters whose transcripts were actually compared reach the site — a pair that merely shares a title and a runtime stays an internal review item and is never asserted to you. In hub mode the badge is limited to same-origin results, since the duplicate index is per-site. Sites with no duplicate report are entirely unaffected.
- **One search now covers video transcripts *and* social posts.** Archived X/Twitter and Bluesky posts ship as a parallel corpus beside transcripts and live chat, and compose into the same boolean query tree — so `(transcripts:"foo" OR posts:"foo")` returns both kinds in one ranked, newest-first result set. `LayerScope` gains `"posts"` (whitelisted in `qt=` deserialization, so a shared link round-trips a posts leaf), the leaf scope selector gains **Posts**, and the filter row gains a **Posts** media kind beside Videos and Livestreams — a third kind, because a post is neither, and folding it into the video toggle would silently drop the whole corpus. Post and video slugs live in disjoint namespaces, partitioned per-leaf by the eval engine so a transcripts leaf never fetches a post and an AND across the two can't collapse to nothing. Post result cards drop what doesn't apply (no seek gutter, no livestream/age badges, no VOD expiry) and lead with the post body; opening one shows a new **PostModal** — a sibling of the transcript reader, not a generalization of it — with the post, its archived thread, its outbound links and its engagement counts. Date filters work unchanged: every post carries a derived `uploadDate`. Posts are cached and served under `/posts/`, offline-cached by the service worker, and CORS-readable so a federating hub merges them across origins.
- **"Ask AI" is grounded in posts as well as transcripts.** Retrieval adds a posts leaf per keyword alongside the transcript and metadata leaves, sharing the same term key so ranking still counts a keyword once rather than three times. Post excerpts render without a `[clock]` line and are cited as a bare `[n]` (never `[n @ mm:ss]` — a post has no timeline), the source list shows a date instead of a meaningless 0:00 seek button, and "load more context" on a post returns its **thread** rather than a time window.
diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts
@@ -269,6 +269,93 @@ export function transcriptPage() {
];
}
+// ─── AI-digest fixtures ───
+// The derived layer is SPARSE: only VIDEO_TRANSCRIPT_ONLY carries a digest, so
+// the same fixture set covers both "has a digest" and "does not" without a
+// second channel. VIDEO_CHAT_SMALL deliberately has none — that is what proves
+// the Digest control stays hidden rather than becoming a dead end.
+//
+// Chapter starts are chosen to sit ON the transcript cue starts above (5 / 50 /
+// 100), which is what the real parser guarantees by snapping to cue boundaries.
+export function channelDigestsManifest() {
+ return {
+ version: 1,
+ channelSlug: CHANNEL_SLUG,
+ pageCount: 1,
+ maxPageBytes: 8388608,
+ generatedAt: new Date().toISOString(),
+ slugToPage: { [VIDEO_TRANSCRIPT_ONLY]: 0 },
+ pageHashes: ["fixture-hash-0"],
+ };
+}
+
+export function digestPage() {
+ return [
+ {
+ slug: slug(VIDEO_TRANSCRIPT_ONLY),
+ id: VIDEO_TRANSCRIPT_ONLY,
+ generatedAt: "2026-07-20T00:00:00.000Z",
+ chapters: [
+ {
+ id: "c5",
+ start: 5,
+ clock: "00:00:05",
+ title: "Opening remarks",
+ decidedBy: "ai" as const,
+ },
+ {
+ id: "c50",
+ start: 50,
+ clock: "00:00:50",
+ title: "The main argument",
+ decidedBy: "ai" as const,
+ },
+ {
+ id: "c100",
+ start: 100,
+ clock: "00:01:40",
+ // A human-corrected title, so the "edited" marker has something to
+ // render and the ai/human distinction is exercised.
+ title: "Corrected by hand",
+ decidedBy: "human" as const,
+ },
+ ],
+ tags: [
+ { id: "tnews", tag: "news", decidedBy: "ai" as const },
+ { id: "tpolicy", tag: "policy", decidedBy: "ai" as const },
+ ],
+ provenance: {
+ chapters: {
+ appId: "ollama-direct",
+ model: "qwen2.5:7b",
+ lane: "local-gpu" as const,
+ generatedAt: "2026-07-20T00:00:00.000Z",
+ promptVersion: 2,
+ contextHash: "",
+ },
+ },
+ },
+ ];
+}
+
+// The same digest, marked as borrowed from a duplicate cluster's canonical
+// member. Used to prove a shared digest is presented AS shared — the failure
+// mode here looks like success, so it needs its own assertion.
+export function borrowedDigestPage() {
+ const [entry] = digestPage();
+ return [
+ {
+ ...entry,
+ derivedFrom: {
+ slug: "other-channel/vid-canonical",
+ clusterId: "cluster-1",
+ sharedAt: "2026-07-21T00:00:00.000Z",
+ offsetSeconds: 0.4,
+ },
+ },
+ ];
+}
+
export function channelTranscriptsManifest() {
return {
version: 1,
diff --git a/export/e2e/helpers.ts b/export/e2e/helpers.ts
@@ -66,6 +66,17 @@ export async function installRoutes(page: Page) {
await page.route(/\/posts\/[^/]+\/page-\d+\.json$/, async (route) => {
await fulfillJson(route, postsPage());
});
+ // AI digests — absent by default, which is also the real default: only ~0.1%
+ // of the corpus is digested, so most channels ship no digests tree at all and
+ // the manifest 404s. Tests that want the Digest control register their own
+ // route afterwards, which takes precedence.
+ await page.route(/\/digests\/[^/]+\/manifest\.json$/, async (route) => {
+ await route.fulfill({
+ status: 404,
+ contentType: "application/json",
+ body: "{}",
+ });
+ });
// Search-alias dictionary — empty by default; alias-suggestion.spec overrides
// this with a populated list. Kept here so other search tests get a clean
// intercept instead of a real 404.
diff --git a/export/e2e/modal-digest.spec.ts b/export/e2e/modal-digest.spec.ts
@@ -0,0 +1,198 @@
+import { expect, test, type Page } from "@playwright/test";
+import {
+ CHANNEL_SLUG,
+ VIDEO_CHAT_SMALL,
+ VIDEO_TRANSCRIPT_ONLY,
+ borrowedDigestPage,
+ channelDigestsManifest,
+ digestPage,
+} from "./fixtures/data";
+import { expectModalOpen, installRoutes, urlParams } from "./helpers";
+
+test.use({
+ permissions: ["clipboard-read", "clipboard-write"],
+});
+
+// Register the digest tree AFTER installRoutes so these take precedence over
+// its default 404s (the duplicates.json idiom).
+async function installDigestRoutes(
+ page: Page,
+ page0: unknown = digestPage(),
+): Promise<void> {
+ await page.route(/\/digests\/[^/]+\/manifest\.json$/, async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(channelDigestsManifest()),
+ });
+ });
+ await page.route(/\/digests\/[^/]+\/page-\d+\.json$/, async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(page0),
+ });
+ });
+}
+
+test.describe("video modal — AI digest panel", () => {
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ });
+
+ test("control is hidden on a video with no digest", async ({ page }) => {
+ await installDigestRoutes(page);
+ // VIDEO_CHAT_SMALL is absent from the manifest's slugToPage.
+ await page.goto(`/?v=${CHANNEL_SLUG}/${VIDEO_CHAT_SMALL}`);
+ await expectModalOpen(page);
+ await expect(page.getByText(/alpha line/)).toBeVisible();
+ await expect(
+ page.getByRole("button", { name: "Show AI chapters" }),
+ ).toHaveCount(0);
+ });
+
+ test("control is hidden when the channel ships no digests at all", async ({
+ page,
+ }) => {
+ // No installDigestRoutes → the manifest 404s, as for most of the corpus.
+ await page.goto(`/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}`);
+ await expectModalOpen(page);
+ await expect(
+ page.getByRole("button", { name: "Show AI chapters" }),
+ ).toHaveCount(0);
+ });
+
+ test("toggle sets vm=digest and back, touching nothing else", async ({
+ page,
+ }) => {
+ await installDigestRoutes(page);
+ await page.goto(
+ `/?m=subs&tk=foo&v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}`,
+ );
+ await expectModalOpen(page);
+
+ const toggle = page.getByRole("button", { name: "Show AI chapters" });
+ await expect(toggle).toBeVisible();
+ await toggle.click();
+
+ await expect(page.getByText("The main argument")).toBeVisible();
+ let params = await urlParams(page);
+ expect(params.get("vm")).toBe("digest");
+ expect(params.get("m")).toBe("subs");
+ expect(params.get("tk")).toBe("foo");
+
+ // Back to the transcript — vm is DELETED, not set to "transcript".
+ await page.getByRole("button", { name: "Show transcript" }).click();
+ await expect(page.getByText(/alpha line/)).toBeVisible();
+ params = await urlParams(page);
+ expect(params.get("vm")).toBeNull();
+ });
+
+ test("deep link with ?vm=digest opens straight into the panel", async ({
+ page,
+ }) => {
+ await installDigestRoutes(page);
+ await page.goto(
+ `/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}&vm=digest`,
+ );
+ await expectModalOpen(page);
+ await expect(page.getByText("Opening remarks")).toBeVisible();
+ await expect(page.getByText("The main argument")).toBeVisible();
+ // Topic tags and the provenance line ship alongside the chapters.
+ // Exact, because the provenance sentence also contains the word "topics".
+ await expect(page.getByText("Topics", { exact: true })).toBeVisible();
+ await expect(page.getByText(/qwen2\.5:7b/)).toBeVisible();
+ });
+
+ test("a chapter click seeks the player to that moment", async ({ page }) => {
+ await installDigestRoutes(page);
+ await page.goto(`/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}&vm=digest`);
+ await expectModalOpen(page);
+
+ // The playhead starts at 0, before the first chapter's 00:05 — so nothing
+ // is active yet. (A chapter list need not start at zero.)
+ await expect(
+ page.getByRole("button", { name: /Opening remarks/ }),
+ ).not.toHaveAttribute("aria-current", "true");
+
+ // Click the 00:50 chapter; the active-chapter marker follows the playhead,
+ // which is what proves the seek took effect.
+ await page.getByRole("button", { name: /The main argument/ }).click();
+ await expect(
+ page.getByRole("button", { name: /The main argument/ }),
+ ).toHaveAttribute("aria-current", "true");
+
+ // And back — the marker tracks the player, it isn't just a click highlight.
+ await page.getByRole("button", { name: /Opening remarks/ }).click();
+ await expect(
+ page.getByRole("button", { name: /Opening remarks/ }),
+ ).toHaveAttribute("aria-current", "true");
+ await expect(
+ page.getByRole("button", { name: /The main argument/ }),
+ ).not.toHaveAttribute("aria-current", "true");
+ });
+
+ test("share link carries the digest panel", async ({ page, context }) => {
+ await context.grantPermissions(["clipboard-read", "clipboard-write"]);
+ await installDigestRoutes(page);
+ await page.goto(
+ `/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}&vm=digest&t=50`,
+ );
+ await expectModalOpen(page);
+ await expect(page.getByText("The main argument")).toBeVisible();
+
+ await page
+ .getByRole("button", { name: "Copy share link at current time" })
+ .click();
+ const url = new URL(
+ await page.evaluate(() => navigator.clipboard.readText()),
+ );
+ expect(url.searchParams.get("vm")).toBe("digest");
+ expect(url.searchParams.get("t")).toBe("50");
+ });
+
+ test("warnings never reach the viewer", async ({ page }) => {
+ await installDigestRoutes(page);
+ await page.goto(`/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}&vm=digest`);
+ await expectModalOpen(page);
+ await expect(page.getByText("Opening remarks")).toBeVisible();
+ // Operator telemetry (out-of-range clamps etc.) is for the editor panel and
+ // the review queue, not for readers.
+ await expect(page.getByText(/out-of-range/)).toHaveCount(0);
+ });
+
+ test("a borrowed digest is presented as borrowed", async ({ page }) => {
+ await installDigestRoutes(page, borrowedDigestPage());
+ await page.goto(`/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}&vm=digest`);
+ await expectModalOpen(page);
+ // The failure mode this guards against looks like success: plausible
+ // chapters that describe a different upload.
+ await expect(page.getByText("Borrowed from a duplicate upload")).toBeVisible();
+ await expect(page.getByText(/other-channel\/vid-canonical/)).toBeVisible();
+ });
+
+ test("snaps back to the transcript when the page is unreachable", async ({
+ page,
+ }) => {
+ // Manifest says the video is digested, but the page fetch fails.
+ await page.route(/\/digests\/[^/]+\/manifest\.json$/, async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(channelDigestsManifest()),
+ });
+ });
+ await page.route(/\/digests\/[^/]+\/page-\d+\.json$/, async (route) => {
+ await route.fulfill({ status: 500, body: "boom" });
+ });
+
+ await page.goto(`/?v=${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}&vm=digest`);
+ await expectModalOpen(page);
+ await expect(page.getByRole("status")).toContainText(
+ /Couldn't load the AI digest/,
+ );
+ await expect(page.getByText(/alpha line/)).toBeVisible();
+ const params = await urlParams(page);
+ expect(params.get("vm")).toBeNull();
+ });
+});
diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js
@@ -24,7 +24,7 @@ const PAGES = `pages-${VERSION}`;
const META = `meta-${VERSION}`;
// Matches /transcripts/<slug>/... and the parallel /subs, /summaries trees.
-const SHARD_RE = /^\/(transcripts|subs|posts|summaries)\/([^/]+)\/(.+)$/;
+const SHARD_RE = /^\/(transcripts|subs|posts|digests|summaries)\/([^/]+)\/(.+)$/;
self.addEventListener("install", () => {
// Activate immediately — no precache list (corpus is too large to bundle).
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -29,35 +29,37 @@ and `transcriptWindow.test.ts` already exists.
`maxDbs` caps the named DBs opened *by one handle*, not the DBs present in the environment.
-- `common/controller/buildIndex.ts:365` — `maxDbs: 17`, opens **12** (`sums`, `cues`,
- `subs`, `mtimes`, `byChannel`, `pageHashes`, `subPageHashes`, `channelStats`, `posts`,
- `postPageHashes`, `channelPostsStats`, `meta`). Three more → 15, still under.
+- `common/controller/buildIndex.ts` — `maxDbs: 17`, now opens **15**: the original 12
+ (`sums`, `cues`, `subs`, `mtimes`, `byChannel`, `pageHashes`, `subPageHashes`,
+ `channelStats`, `posts`, `postPageHashes`, `channelPostsStats`, `meta`) plus the three
+ Phase 2 added (`digests`, `digestPageHashes`, `channelDigestStats`). Still under.
- `common/lib/channelSignature.ts:48` — `maxDbs: 14`, opens **2** (`mtimes`, `meta`).
- `common/controller/duplicateShorts.ts:111` — `maxDbs: 12`, opens **3**.
None of these break. The "17 vs 14 inconsistency" is a red herring.
-→ What **is** required and easy to miss: the schema-invalidation block at
-`buildIndex.ts:425-439` enumerates every sub-DB by hand with `clearAsync()`. Three new lines
-there, plus `SCHEMA_VERSION` (`buildIndex.ts:116`, currently **12**) → 13.
+→ What **was** required and easy to miss: the schema-invalidation block enumerates every
+sub-DB by hand with `clearAsync()`. **Done (2026-07-29)** — the three digest sub-DBs are
+listed there and `SCHEMA_VERSION` is **13**. The next person to add a sub-DB still has to
+add its `clearAsync()` line by hand; nothing enforces it.
-### `JobProgressMetric` is not a closed union in practice
+### `JobProgressMetric` is not a closed union in practice — **RESOLVED, table was stale**
-`editor/app/jobs/components/RunningJobsList.tsx:41` re-spells
-`metric: "downloads" | "transcripts"` as a **literal**, not an import of
-`JobProgressMetric`. TypeScript will therefore **not** flag it when the union is extended.
+**This entry is kept only to stop the next session re-doing it. Verified 2026-07-29: all of
+it is already done.** The union is extended and every site is correct:
-Fix that line first, then extend. All six sites:
+- `common/jobs/registry.ts:17` — `JobProgressMetric` is `"downloads" | "transcripts" |
+ "digests"`, and `:25` `JobTaskKind` is `"download" | "transcribe" | "digest"`.
+- `editor/app/jobs/components/RunningJobsList.tsx:6-9` **imports** `JobProgressMetric` and
+ `JobTaskKind` rather than re-spelling them, so TypeScript now does flag the next member.
+ A comment on the union at `registry.ts:14` records that this is why.
+- `editor/app/widget/components/MonitorWidget.tsx:628` is a `METRIC_PREFIX` lookup table
+ (`Record<JobProgressMetric, string>`), not the two copies of a `metric === "downloads" ? …`
+ binary the old table described — so it is exhaustiveness-checked and there is nothing
+ left to duplicate.
-| File | Line |
-| --- | --- |
-| `common/jobs/registry.ts` | 14 (the union), 22 (`JobTaskKind`) |
-| `editor/app/jobs/components/RunningJobsList.tsx` | 41 (literal), 313, 326 |
-| `editor/app/widget/components/MonitorWidget.tsx` | 615, 636 |
-| `editor/app/jobs/active/buildActiveJobs.ts` | 66 |
-
-`MonitorWidget.tsx:615` and `:636` are two copies of the same
-`metric === "downloads" ? … : …` binary, in `jobProgressText()` and `JobProgressBar()`.
+PLAN.md's Phase 2.5 opens by demanding this fix "first". It is done; start 2.5 at its
+second item.
### Queue concurrency is not declared per key
@@ -691,6 +693,21 @@ worth noting: `cut-release` failed **once** rather than twice, and `undownloaded
appeared in place of the second `cut-release` — it passes in isolation. Export suite:
**141/141**.
+**Re-verified again 2026-07-29** after Phases 2/3: **366 passed / 24 failed of 390**. The 24
+are the 22 in the table above (`cut-release` failed **both** this time — the working tree was
+dirty, which is exactly when it does) plus **two more, each verified passing in isolation**:
+
+| Spec | Isolation result |
+| --- | --- |
+| `auto-queue.spec:209` — strict priority drains the high-priority channel first | passes (13/13 in 1.8 min) |
+| `scheduler.spec:239` — schedule page edits the global controls | passes (all 8 scheduler tests) |
+
+Both are the **CPU-contention flake** already recorded for `undownloaded.spec:53` and
+`sync-break-on-existing`: this run shared the box with an unrelated containerized playwright
+suite at load ~10. Add them to the "passes in isolation" set rather than the base-failure
+set. Export suite the same day: **150/150** (141 + 9 new digest specs). `digest.spec.ts`
+(12) and the `settings.spec` digest tests all passed.
+
The `deploy-page` (×3), `new-channel-onboarding` (×5) and `site-scope` failures — **9 specs**
— are all *strict-mode locator violations* ("resolved to N elements"): specs written against
an earlier form structure that has since gained sibling controls with overlapping accessible
diff --git a/plans/STATE.md b/plans/STATE.md
@@ -3,7 +3,21 @@
The working memory for the local-AI derived-corpus work. Rewritten at the end of every
session, before context is cleared. See [`README.md`](README.md) for the protocol.
-**Last updated:** 2026-07-27 — Stage B2: the digest layer is now INSPECTABLE and its defaults
+**Last updated:** 2026-07-29 — **Phases 2 and 3: the digest layer SHIPS.** The 102 digests
+already on disk now travel generation → LMDB → a shared `/digests/<slug>/` page tree →
+compose → the viewer, where a reader can open a Digest panel, click a chapter and seek to it.
+Proven end to end against the real corpus, not fixtures: `community-notes/v2cywen` renders its
+20 real chapters (first at 01:09:48 — the validation run's worst coverage gap) and the
+playhead follows a click. The control is hidden on the 99.9% of videos with no digest, and a
+borrowed digest is labelled as borrowed. Deliberately independent of the three backfill gates
+(1.5 context, 2.5 observability, 11a review), none of which moved.
+
+Verification that matters: common 356 tests (346 + 10 new), `tsc` clean in all three
+packages, export playwright **150/150** (141 + 9 new), a no-op rebuild skips digest pages,
+and a hand-written `ai-digest.overrides.json` produces exactly `~1 changed` and ships the
+corrected title — the `digestMs` max-of-two-sidecars requirement, which nothing else tests.
+
+**Previous entry —** 2026-07-27 — Stage B2: the digest layer is now INSPECTABLE and its defaults
are the ones the bake-off measured best. Settings gained a real Digest section, each video
page gained a review panel (provenance, warnings, `derivedFrom`, staleness), the two digest
counters were made to agree, `chunk-local`/8192/600 shipped as defaults behind a
@@ -20,9 +34,9 @@ the 0.6 duplicate near-threshold was measured and found to be rejecting real mir
| 0 · Benchmark transcription engines | not started | Still unmeasured. The DIGEST bake-off is done; this is the separate transcription one. |
| 1 · Generation harness | **done + validated** | Spine in `8c041fd`; correctness + bake-off + e2e in Stage B1; operator surfaces + measured-best defaults + 102-video validation run in Stage B2. |
| 1.5 · Channel context | not started | `contextHash` is plumbed and empty, so notes can be added without invalidating the corpus. |
-| 2 · Digest corpus in build | not started | |
-| 2.5 · Observability | not started | **Blocks backfill.** |
-| 3 · Viewer `?vm=summary` | not started | |
+| 2 · Digest corpus in build | **done** | Shared `/digests/<slug>/` page tree, `digests` + `digestPageHashes` + `channelDigestStats` sub-DBs at `SCHEMA_VERSION` 13, `digestMs` in the mtime record and the channel signature, compose reconcile, `CORPUS_SPEC_VERSION` 3. |
+| 2.5 · Observability | **partly landed** | Its stated first step is already done — see the correction below. |
+| 3 · Viewer `?vm=digest` | **done** | Not `?vm=summary` — see the naming decision below. |
| 4 · Search indexing | not started | |
| 5 · Auto-queue | not started | |
| 6 · Ollama `/ask` provider | not started | **No dependencies.** |
@@ -33,6 +47,22 @@ the 0.6 duplicate near-threshold was measured and found to be rejecting real mir
| 11a · Review queue | not started | **Land before backfilling.** `warnings[]` is the data it reads. |
| 11b · Viewer feedback | not started | |
+**Three corrections to this document's own claims, verified against the code 2026-07-29.**
+Exploration for Phase 2 found the planning docs describing more unbuilt work than exists:
+
+- **Phase 2.5 is not "not started".** `JobProgressMetric` already carries `"digests"` and
+ `JobTaskKind` already carries `"digest"` (`common/jobs/registry.ts:17,25`).
+- **PLAN.md's Phase 2.5 opens by demanding the duplicated-metric-union fix "first". It is
+ done.** `RunningJobsList.tsx:6-9` imports the union instead of re-spelling it, and
+ `MonitorWidget.tsx:628` is a `METRIC_PREFIX` lookup table, not the two copies of a binary
+ ternary the old FACTS.md table described. That table is now marked RESOLVED there.
+- **`listChannels` already pays for a digest count and then throws it away.**
+ `common/controller/channels.ts:192` calls `countDataFiles`, which computes `counts.digests`
+ — but unlike `readChannelStat` (`:139`), `listChannels` omits `digestCount` from the
+ `ChannelStat` it emits. The field is already optional on the type, so recovering it is one
+ line and every cross-channel surface gets a counter it is already paying for. Left
+ unchanged here deliberately: it is Phase 2.5's to take, not a digest-corpus side effect.
+
**Recommended next**, in the order the validation run argues for:
1. **Phase 2.5 observability, and treat the throughput finding as its first requirement.**
@@ -63,6 +93,45 @@ relitigate. Earlier entries (fabric deferred, ollama-direct as the structured de
Claude Code on a second queue key, digests in their own page tree, per-section provenance,
transcripts never rewritten) still stand and are unchanged.
+**The viewer mode is `?vm=digest`, NOT `?vm=summary` (2026-07-29).** "Summary" was already
+taken twice — `DisplaySummary` is a video *listing card* (`common/lib/transcripts.ts`) and
+`/summaries/` is the browse-index page tree the search index serves. PLAN.md's own naming
+rule ("`report` already means three different things — do not add a fourth") applies exactly
+as well to a third meaning of "summary". `digest` already names the artifact, the stage, the
+settings section, the editor panel and the page tree.
+
+**What ships is `effectiveDigest()`, and `warnings[]` is deliberately NOT in it
+(2026-07-29).** The page tree carries the COMPOSED digest — human overrides applied,
+`enabled: false` items dropped, chapters sorted — so a reader sees what a human approved
+rather than raw model output. `warnings[]` and `history[]` stay behind: they are operator
+telemetry for the editor panel and the Phase 11a review queue. Shipping "the model proposed
+13 chapters that were clamped away" puts the corpus's failures in front of the audience
+instead of the operator, and `history[]` is the largest field in a mature sidecar.
+
+**The digest control is HIDDEN where there is no digest, and the manifest is what says so
+(2026-07-29).** Coverage is 102 of ~77,108 videos — 0.1% — so an always-present button is a
+dead end almost everywhere. `slugToPage` doubles as the existence check (a channel with no
+digests ships no manifest at all), so gating costs one small cached request per channel and
+never a per-video fetch. Same shape as `hasDuplicates()` gating the Duplicates nav link.
+
+**The client cache is versioned by PAGE CONTENT HASH, not `provenance.generatedAt`
+(2026-07-29).** Digests are regenerated in place, so a cached entry stays structurally valid
+while going stale — the shape sniff `transcriptStore.ts` gets away with cannot see that.
+The plan called for storing `generatedAt`, but that is only legible *after* fetching the
+page, which is the cost the cache exists to avoid. The build already computes per-page
+hashes for its own skip logic, so `ChannelDigestsManifest.pageHashes` publishes them: exact
+versioning, knowable from the manifest alone, and it also catches an overrides-only retitle
+that leaves `generatedAt` untouched. **This was not theoretical** — the first build emitted
+`pageHashes: [""]` because an LMDB read-back immediately after `put()` returned nothing.
+The hashes are now returned directly by `createPageWriter` (recorded on BOTH the written and
+the skipped branch), which is correct by construction rather than by LMDB semantics.
+
+**Digests get their own IndexedDB DATABASE, not another store in the transcript one
+(2026-07-29).** `DB_VERSION` is a property of the database and `transcriptStore.ts`'s
+`onupgradeneeded` does `deleteObjectStore` + `createObjectStore`, so adding a store there
+would force a version bump and wipe every client's transcript cache. `searchLayerCache.ts`
+already establishes the separate-`DB_NAME` idiom; digests follow it.
+
**`chunk-local` timestamps are adopted, on measurement — and (2026-07-27) SHIPPED as the
default.** Each chunk's transcript is re-based to `00:00:00` and the parser adds the offset
back before any guard runs. Measured on the long tail it cut zero-yield chunks from 33.3% to
@@ -279,13 +348,30 @@ playwright `webServer` with `OLLAMA_URL` pointed at it in **both** `dev:test` an
Established against this branch:
```bash
-pnpm --filter yt-dlp-transcript-common run test # 344 tests
-pnpm --filter {yt-dlp-transcript-common,editor,export} exec tsc --noEmit
+pnpm --filter yt-dlp-transcript-common run test # 356 tests (was 344)
pnpm --filter editor build
pnpm e2e # default dev mode; E2E_MODE=start serves a stale build
-pnpm --filter export exec playwright test # 141 tests, separate suite on :3020
+pnpm --filter export exec playwright test # 150 tests, separate suite on :3020
```
+Run `tsc` per package — the brace form is mangled by fish:
+
+```bash
+cd common && pnpm exec tsc --noEmit
+cd editor && pnpm exec tsc --noEmit
+cd export && pnpm exec tsc --noEmit
+```
+
+A `SCHEMA_VERSION` bump means a FULL index rebuild: 77k videos, **~35 min** and it re-reads
+every transcript. Subsequent incremental builds are ~6 min (the scan of 77k dirs dominates;
+digest pages skip on content hash). Budget for it before bumping.
+
+**`pkill -f <pattern>` can kill its own shell.** `pkill -f 'next dev --port 3020'` matches
+the `bash -c` process whose command line *contains that string* — so chaining it before a
+test run silently kills the run (exit 144, no output). Bracket one character
+(`'next [d]ev --port 3020'`), or run the pkill as its own command. This is the same bracket
+trick already noted for `fixtures/bin/fake[-]`, for a different reason.
+
Duplicate detection (writes `transcripts/duplicates.json`; back it up first if the current
one matters):
@@ -340,6 +426,19 @@ first save after the settings form lands would have silently reset `timestampMod
invalidated every digest made under it. Both are fixed. The lesson is that the digest
layer's failure modes live at the *wiring*, not in the pure functions.
+**Phase 2/3 (2026-07-29): the two things that would have shipped broken and looked fine.**
+Both were caught by building the real thing and looking at the output, not by types or tests:
+
+- **`pageHashes: [""]`.** Reading `digestPageHashes` back immediately after `put()` in the
+ same build returned nothing. The manifest was structurally valid and the site worked — the
+ cache version was just empty, which fails *open* (every read a miss) rather than loudly.
+ Had it failed the other way, clients would have pinned stale digests forever.
+- **A video's directory name is not its video id.** `community-notes/v2fkbw7` — the id the
+ plan named, and the one the editor's URL shows — holds video `v2cywen`. Page trees key
+ `slugToPage` by `summary.id`, so the digest manifest correctly contains `v2cywen` and not
+ `v2fkbw7`; the transcripts manifest has always done the same. Anyone hand-checking a
+ digest manifest against a `data/` listing will "find" a missing video that is not missing.
+
**The bake-off's most useful finding was one nobody asked for.** The plan framed the choice
as model-vs-model with chunk-local as the variable. The data showed gemma2's clean sweep was
confounded with its forced 8k context, and testing chunk size directly turned out to matter