// Where a clip's caption cues come from. // // A clip window is widened from a cue span to a whole sentence, which needs cue // END times. Nothing else in the pipeline carries them: a report citation is a // single start second, and the MCP `Snippet` type has no `end` field. So this is // the one place that answers "what are the real cue boundaries for this video". // // TWO SOURCES, SAME SHAPE. A local corpus stores each video as // `//data//transcript.cues.json`, and a *published* // archive serves the same record inside a paginated shard. The two carry the // same fields — `{ slug, id, channelSlug, title, uploadDate, duration, channel, // description, platform, webpageUrl, cues: [{start, end, text}] }` — so one // resolver serves both `loadCues` and `videoMeta`, and a caller cannot tell // which it got beyond the `from` marker. // // That parity is what makes a corpus optional. Clone the repo, point a manifest // at a public instance, and the video pipeline can cut clips without mirroring a // single channel: the cue windows come over HTTP, and the media itself was // always a network fetch (`yt-dlp --download-sections`). // // The shard walk is the contract published at `/corpus.json` under `shardScheme`: // 1. GET /corpus.json -> channels[].manifests.transcripts // 2. GET that manifest -> { pageCount, slugToPage: { : N } } // 3. GET page-.json (N zero-padded to 4) -> array of records // 4. take the record whose `id` matches // // Local wins when present: it is faster, works offline, and is the operator's own // data. HTTP is the fallback, not a preference — and only for a channel this // corpus does not hold. A channel it DOES hold whose media cannot be reached (a // relocated `data/` whose drive is not mounted) is an error, never a quiet // downgrade to the archive; see checkChannelReachable below. // // THE TWO SOURCES CAN DISAGREE, AND IT IS NOT ROUNDING. A published archive is a // snapshot; a live corpus keeps moving. Re-synced platform captions, an // auto-caption replacement or a re-transcription all rewrite a video's cues in // place, and the archive keeps the text it was built from until it is rebuilt. // Measured on this corpus (local 2026-08-13 against a 2026-08-07 publish): of // four videos checked, three were byte-identical and one had 65 of its 84 cue // texts changed with timings shifted by up to **2.24 s** — enough to cut a clip // in the wrong place. // // So `prefer` is a real decision, not a micro-optimisation: // "auto" (default) local when present, else HTTP. Right for an operator. // "local" never fall back. Fail loudly instead of silently cutting from // different cues than the ones a window was authored against. // "http" always the archive. Right when you want the windows to match what a // reader following the citation will actually see, and the only // option that is reproducible on a machine with no corpus. // Whatever answers, the returned record carries `from` so a caller can record it. // // THE ON-DISK CACHE IS SHARED, AND ONE READER REFRESHES IT. post-links.mjs (a // post's archive channel, for its QR) reads `/corpus.json` and the posts // manifests through createJsonCache below, the same files on disk, and on a // post it cannot find it fetches them again and rewrites them. The cue walk // never refreshes, but it can read a corpus.json post-links rewrote: newer, so // a channel the stale copy lacked is found. Manifest and shard URLs carry no // version, so a cue window does not move because of it. import { readFile, writeFile, mkdir, lstat, readlink, readdir, stat } from "node:fs/promises"; import path from "node:path"; import os from "node:os"; import { createHash } from "node:crypto"; import { fileURLToPath } from "node:url"; // This file lives at /umtool/report-to-video/, so the corpus a plain // checkout would have is two levels up. Previously this defaulted to an absolute // path inside the original author's home directory, which meant every other // clone silently looked in a directory that does not exist. const REPO_ROOT = path.resolve(/* turbopackIgnore: true */ path.dirname(/* turbopackIgnore: true */ fileURLToPath(import.meta.url)), "..", ".."); export const DEFAULT_CHANNELS_DIR = process.env.CHANNELS_DIR ?? path.join(/* turbopackIgnore: true */ REPO_ROOT, "transcripts", "channels"); export const DEFAULT_CACHE_DIR = process.env.REPORT_CACHE_DIR ?? path.join(os.homedir(), ".cache", "archilyzer-report-to-video"); // A shard page is capped at 8 MB and holds ~100 videos, so refetching one per // clip — across two separate processes, resolve-windows then build-video — is // the difference between usable and painful. Cached by URL on disk; archives are // rebuilt rarely and a stale page only matters if the cues themselves changed. function cacheKey(url) { return createHash("sha1").update(url).digest("hex") + ".json"; } export function pageFileName(pageNumber) { return `page-${String(pageNumber).padStart(4, "0")}.json`; } export function pageUrlFrom(manifestUrl, pageNumber) { const u = new URL(manifestUrl); u.pathname = u.pathname.replace(/[^/]+$/, pageFileName(pageNumber)); return u.toString(); } // The origin of the archive a manifest was built against. Every manifest already // records this — `siteOrigin` explicitly, and `corpus` as `remote:` or // `local:` — so the common case needs no configuration at all. export function siteOriginFromManifest(manifest) { const p = manifest?.provenance ?? {}; if (typeof p.siteOrigin === "string" && p.siteOrigin.trim()) { return p.siteOrigin.replace(/\/+$/, ""); } if (typeof p.corpus === "string" && p.corpus.startsWith("remote:")) { return p.corpus.slice("remote:".length).replace(/\/+$/, ""); } if (typeof p.shareLink === "string" && /^https?:/.test(p.shareLink)) { try { return new URL(p.shareLink).origin; } catch { /* fall through */ } } return null; } // A TWIN OF `CAPTION_TRACK_RULE_VERSION` (`common/lib/videoStatus.ts`) — the // caption-track rule a `transcript.cues.json` records as `captionTrackRule`. // Copied for the same reason as the text guard below: umtool's bins run under // plain node. Change one, change the other. export const CAPTION_TRACK_RULE_VERSION = 1; const ENGLISH_VTT_RE = /^transcript\.en(?:-[^.]+)?\.vtt$/; // A local caption record normalized under an older caption-track rule, where // the rule now reads other words — the same test as `isCuesJsonFresh` // (`common/controller/normalizeTranscript.ts`, followsCaptionTrackRule): no // cues at all (a cue-block VTT used to parse to nothing; a lone track with word // timing is genuinely empty and is let through), or a transcript.en.vtt and // transcript.en-orig.vtt that differ (the older rule read en; en-orig is read // now). Cutting from it would widen clips on text the corpus no longer // publishes, so it is refused with the fix, not used. async function staleCaptionRecord(videoDir, record) { if (record?.source !== "vtt" || record.captionTrackRule === CAPTION_TRACK_RULE_VERSION) { return false; } const entries = await readdir(videoDir).catch(() => []); const english = entries.filter((e) => ENGLISH_VTT_RE.test(e)); if (!Array.isArray(record.cues) || record.cues.length === 0) { if (english.length !== 1) return true; const head = (await readFile(path.join(videoDir, english[0]), "utf8").catch(() => "")).slice(0, 2048); return !/<\d{2}:\d{2}:\d{2}\.\d{3}>/.test(head); } if (!entries.includes("transcript.en.vtt") || !entries.includes("transcript.en-orig.vtt")) return false; const [en, orig] = await Promise.all([ stat(path.join(videoDir, "transcript.en.vtt")), stat(path.join(videoDir, "transcript.en-orig.vtt")), ]); return en.size !== orig.size; } export class CueLookupError extends Error { constructor(message, { channelSlug, videoId, tried }) { super(message); this.name = "CueLookupError"; this.channelSlug = channelSlug; this.videoId = videoId; this.tried = tried; } } /** * A JSON GET with an in-memory and an on-disk cache, keyed by URL: the cache * the cue walk reads the archive through, and the one the post-link resolver * (post-links.mjs) reads it through too: the same files on disk, so an * archive's corpus.json fetched for one is on disk for the other. * * `getJson(url, { refresh: true })` goes back to the network for a URL -- * unless this cache already fetched it from the network less than * `refreshAfterMs` ago. The default, Infinity, makes that "at most once per * process": a build re-reads a stale cached document once, never twice. A * long-lived server passes a window instead. */ export function createJsonCache({ cacheDir = DEFAULT_CACHE_DIR, fetchImpl = globalThis.fetch, log = () => {}, refreshAfterMs = Infinity, } = {}) { const mem = new Map(); /** URL -> when this cache last fetched it from the network (ms). */ const fetchedAt = new Map(); /** URL -> the network fetch of it in progress. */ const inflight = new Map(); async function getJson(url, { refresh = false } = {}) { const recent = fetchedAt.has(url) && Date.now() - fetchedAt.get(url) < refreshAfterMs; const useCache = !refresh || recent; if (useCache && mem.has(url)) return mem.get(url); const disk = cacheDir ? path.join(cacheDir, cacheKey(url)) : null; if (useCache && disk) { try { const cached = JSON.parse(await readFile(disk, "utf8")); mem.set(url, cached); return cached; } catch { /* cold cache */ } } // Callers that ask for the same URL while it is on its way share the one // fetch (umtool's preview routes run side by side). if (inflight.has(url)) return inflight.get(url); const pending = (async () => { log(`fetch ${url}`); const res = await fetchImpl(url); if (!res.ok) throw new Error(`GET ${url} -> ${res.status}`); const json = await res.json(); mem.set(url, json); fetchedAt.set(url, Date.now()); if (disk) { try { await mkdir(path.dirname(disk), { recursive: true }); await writeFile(disk, JSON.stringify(json)); } catch { // A cache we cannot write is a slow run, not a failed one. } } return json; })(); inflight.set(url, pending); try { return await pending; } finally { inflight.delete(url); } } return { getJson }; } export function createCueSource({ channelsDir = DEFAULT_CHANNELS_DIR, siteOrigin = null, cacheDir = DEFAULT_CACHE_DIR, fetchImpl = globalThis.fetch, log = () => {}, // "auto" | "local" | "http" — see the note on divergence above. prefer = "auto", // Opt-in, because it is expensive: see resolveSiteId below. resolveSiteIds = false, } = {}) { const { getJson } = createJsonCache({ cacheDir, fetchImpl, log }); async function channelEntry(origin, channelSlug) { const corpus = await getJson(`${origin}/corpus.json`); // A cited report site (corpus spec 5, `site.scope: "cited"`) publishes its // reports and the moments they cite — no channels, no transcript shards — // so there are no cues to walk to. Say that, not "channel not found". if (corpus.site?.scope === "cited") { throw new CueLookupError( `${origin} is a cited report site (corpus.json site.scope "cited"): it publishes no transcripts, ` + `so cues cannot be read from it — point the manifest at a full archive or use a local corpus`, { channelSlug, videoId: null, tried: [`${origin}/corpus.json`] }, ); } const found = (corpus.channels ?? []).find((c) => c.slug === channelSlug); if (!found) { throw new CueLookupError( `channel "${channelSlug}" is not in ${origin}/corpus.json`, { channelSlug, videoId: null, tried: [`${origin}/corpus.json`] }, ); } return found; } // Map a LOCAL video id onto the id the site serves, by scanning the channel's // pages for a record whose `webpageUrl` contains it. // // WHY THIS EXISTS: a Rumble video has two ids. The site (and the MCP) key it by // the EMBED id; the local cue directory is named for the URL SLUG. A manifest // hand-authored against local cue dirs therefore carries slugs that are absent // from the published `slugToPage` — every Rumble clip misses. // // WHY IT IS OPT-IN: it downloads a channel's shards until it hits a match, and // a shard is up to 8 MB. That is a reasonable price to pay knowingly and a // terrible one to pay silently, so the direct lookup fails with instructions // instead and this runs only when asked. async function resolveSiteId(origin, entry, wanted) { const manifest = await getJson(entry.manifests.transcripts); log(`resolving "${wanted}" by scanning ${manifest.pageCount} shard(s) of ${entry.slug}`); for (let n = 0; n < manifest.pageCount; n += 1) { const page = await getJson(pageUrlFrom(entry.manifests.transcripts, n)); const hit = page.find( (r) => r.id === wanted || r.slug === wanted || String(r.webpageUrl ?? "").includes(wanted), ); if (hit) return hit; } return null; } async function fromHttp(channelSlug, videoId, hints) { const origin = hints.siteOrigin ?? siteOrigin; if (!origin) { throw new CueLookupError( `no local cues for ${channelSlug}/${videoId} and no archive origin to fetch them from ` + `(set provenance.siteOrigin in the manifest, or pass --site-origin / SITE_ORIGIN)`, { channelSlug, videoId, tried: ["local"] }, ); } const siteChannel = hints.siteChannel ?? channelSlug; const siteVideo = hints.siteVideo ?? videoId; const entry = await channelEntry(origin, siteChannel); const manifest = await getJson(entry.manifests.transcripts); const pageNumber = manifest.slugToPage?.[siteVideo]; if (pageNumber === undefined) { if (resolveSiteIds) { const hit = await resolveSiteId(origin, entry, siteVideo); if (hit) return { ...hit, from: "http" }; } throw new CueLookupError( `"${siteVideo}" is not in ${siteChannel}'s published slugToPage on ${origin}.\n` + ` If this is a Rumble clip, the archive is keyed by the EMBED id while a local cue\n` + ` directory is named for the URL SLUG — they differ. Either add "siteVideo" (and\n` + ` "siteChannel" if it also differs) to this clip in the manifest, or re-run with\n` + ` --resolve-site-ids to find it by scanning the channel's shards (slow: downloads\n` + ` up to 8 MB per shard until it matches).\n` + ` Note that a clip's citeUrl is NOT usable here — it may deliberately cite a\n` + ` different recording (a mirror that reads better), whose clock is not the same.`, { channelSlug, videoId, tried: [entry.manifests.transcripts] }, ); } const page = await getJson(pageUrlFrom(entry.manifests.transcripts, pageNumber)); const record = page.find((r) => r.id === siteVideo || r.slug === siteVideo); if (!record) { throw new CueLookupError( `${siteChannel}/${siteVideo} is on shard ${pageNumber} per the manifest, but no record ` + `there has that id — the published archive is inconsistent`, { channelSlug, videoId, tried: [pageUrlFrom(entry.manifests.transcripts, pageNumber)] }, ); } return { ...record, from: "http" }; } // A TWIN OF THE EDITOR'S TEXT GUARD. `assertChannelTextReadable` // (`common/lib/channelMedia.ts`) is the same check in TypeScript, and that // file carries a pointer back here — change one, change the other. It is // copied rather than imported because umtool's bins run under plain node // with no `tsx` and no build step, and that stays true for now. // // WHY IT EXISTS. This resolver reads TEXT — `transcript.cues.json`. Since // release 17 a channel's text stays on the corpus disk whatever its media is // doing (only the big files move, into `channels//media`), so an // unmounted MEDIA drive does not concern it. The one layout whose text is on // another drive is the RETIRED whole-directory one — `data/` an absolute // symlink to `//data`, `config.json` recording `dataDir` — and // there an unmounted drive reads as a plain ENOENT, which `load` used to // swallow as "no local copy" and answer from the published archive instead: // silently cutting from a snapshot whose cues can differ from the corpus by // seconds (see the note at the top of this file). So a `legacy` channel is // refused, loudly, with its way out — mounted or not. And a marker is // refused only when its `scope` is `tier-migration` (the migration rebuilds // `data/` itself); a media move leaves the text where it is. // // SCOPE, DELIBERATELY NARROW. This fires only for a channel the local corpus // actually holds. No `channels/` dir at all (a clone with no corpus), or a // channel this corpus does not mirror, leaves `data/` absent with nothing // recorded — which the editor calls readable and passes, and which here still // falls through to HTTP. That is the supported archive-only case, not a // failure. const checkedChannels = new Map(); function unreachable(channelSlug, dataDir, detail) { return new CueLookupError( `channel "${channelSlug}": its local text is not readable — ${detail}`, { channelSlug, videoId: null, tried: [dataDir] }, ); } async function checkChannelReachable(channelSlug) { const channelDir = path.join(channelsDir, channelSlug); const dataDir = path.join(channelDir, "data"); const fail = (detail) => { throw unreachable(channelSlug, dataDir, detail); }; let retired; try { const parsed = JSON.parse(await readFile(path.join(channelDir, "config.json"), "utf8")); if (typeof parsed.dataDir === "string" && parsed.dataDir.trim()) { retired = parsed.dataDir.trim(); } } catch { // No config.json, or one that is not JSON: nothing records a layout. } // A tier migration in flight (or interrupted) is rebuilding `data/`: half // of two places at once. The editor refuses such a channel; so does this. // A media move leaves the text alone. let marker = null; try { marker = JSON.parse(await readFile(path.join(channelDir, ".relocating.json"), "utf8")); } catch { /* no marker: the normal case */ } // `markerHoldsText` in channelMedia.ts: a tier migration, or a scope-less // marker aimed at the retired `//data` shape (the old mover, // copying the whole `data/`). const holdsText = marker && typeof marker.target === "string" && marker.target.trim() && (marker.scope === "tier-migration" || (marker.scope !== "media" && path.basename(marker.target.trim().replace(/\/+$/, "")) === "data")); if (holdsText) { fail( `its media layout is being migrated (phase "${marker.phase ?? "copy"}") — ` + `wait for archilyzer storage migrate-tier to finish`, ); } let link = null; try { link = await lstat(dataDir); } catch { /* no data/ at all: below */ } // THE RETIRED LAYOUT: a `data` link, or a recorded `dataDir`. Refused // whether or not its drive is mounted, never followed. if ((link && link.isSymbolicLink()) || retired) { let target = retired ?? ""; if (link && link.isSymbolicLink() && !target) { try { target = await readlink(dataDir); } catch { /* unreadable link: named as such */ } } fail( `its media layout is the retired whole-directory one${target ? ` (${target})` : ""} — ` + `run archilyzer storage migrate-tier ${channelSlug}`, ); } // No `data/`: a channel that has downloaded nothing — or is not mirrored // here — and not an error. if (!link) return; if (!link.isDirectory()) fail(`${dataDir} is neither a directory nor a symlink`); } function assertChannelReachable(channelSlug) { // One check per channel per run, cached as a promise so a rejection replays // for every clip of that channel instead of re-statting for each. let pending = checkedChannels.get(channelSlug); if (!pending) { pending = checkChannelReachable(channelSlug); checkedChannels.set(channelSlug, pending); } return pending; } async function fromLocal(channelSlug, videoId) { // Throws a CueLookupError — which carries no `code`, so `load`'s ENOENT // fallback rethrows it rather than reaching for the archive. await assertChannelReachable(channelSlug); const p = path.join(channelsDir, channelSlug, "data", videoId, "transcript.cues.json"); const parsed = JSON.parse(await readFile(p, "utf8")); if (await staleCaptionRecord(path.dirname(p), parsed)) { throw new CueLookupError( `${channelSlug}/${videoId}: transcript.cues.json was normalized under an older caption-track rule ` + `(it holds the served en track's words where en-orig is read now, or no words at all). ` + `Run Normalize for channel ${channelSlug} in the editor, or pass --cue-source http.`, { channelSlug, videoId, tried: [p] }, ); } return { ...parsed, from: "local" }; } return { channelsDir, /** * The full record for one video: cues plus the metadata build-video needs. * `hints` may carry `siteChannel` / `siteVideo` (when the published archive * keys this recording differently) and `siteOrigin` (per-manifest override). */ prefer, async load(channelSlug, videoId, hints = {}) { if (prefer === "http") return await fromHttp(channelSlug, videoId, hints); try { return await fromLocal(channelSlug, videoId); } catch (err) { if (err?.code !== "ENOENT" && err?.code !== "ENOTDIR") throw err; if (prefer === "local") { throw new CueLookupError( `no local cues for ${channelSlug}/${videoId} under ${channelsDir}, and ` + `--cue-source local forbids falling back to the archive`, { channelSlug, videoId, tried: [channelsDir] }, ); } return await fromHttp(channelSlug, videoId, hints); } }, }; }