commit 5e74ea228e1f71f22464b417e8c862eb29bff3f3
parent 44528b0e124392aafabcfd76aa72a989d77037ce
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 20 Aug 2026 01:18:48 -0400
report-to-video: read cue windows from a published archive, not just local disk
A clip window is widened from a cue span to a whole sentence, and that needs cue
END times. Nothing else in the pipeline carries them — a report citation is a
single start second and the MCP Snippet type has no `end` — so the two scripts
read transcript.cues.json off local disk. That quietly made a corpus mandatory
for the whole video path, which is the one thing standing between "clone the
repo" and "cut clips from a public archive".
It never needed to be. A published archive already serves the same record:
`shardScheme` in /corpus.json documents transcript pages as
{ id, title, …, cues: [{ start, end, text }] }, and they are really there
(verified against jeralyzer.pages.dev). The published record carries the same
fields a local cue file does — title, uploadDate, duration, webpageUrl — so one
resolver serves both loadCues and videoMeta and a caller cannot tell which
answered beyond the `from` marker.
cues.mjs walks the published contract: corpus.json -> the channel's transcripts
manifest -> slugToPage -> page-<NNNN>.json (zero-padded to four; page-0.json is
a 404) -> the record whose id matches. Local wins when present; HTTP is the
fallback. Manifests, corpus and shards are cached in memory and on disk, because
a shard is up to 8 MB and resolve-windows and build-video are separate
processes.
No new configuration in the common case: a manifest already records the archive
it was built against, as provenance.siteOrigin (or corpus: "remote:<url>", or
the shareLink's origin).
THE SOURCES CAN DISAGREE, and not by rounding. An archive is a snapshot; a
corpus keeps moving. Measured here — local 2026-08-13 against a 2026-08-07
publish — three of four videos were byte-identical and the fourth had 65 of its
84 cue texts rewritten with timings shifted by up to 2.24 s, which is enough to
cut in the wrong place. So --cue-source auto|local|http makes the choice
explicit: `local` refuses to fall back rather than silently cut from other cues,
`http` is the reproducible option on a machine with no corpus.
THE RUMBLE TWO-ID TRAP. A Rumble video has two ids: the archive keys it by the
EMBED id while a local cue directory is named for the URL SLUG, so a manifest
authored against local dirs misses on every Rumble clip. That fails loudly with
the diagnosis, an offer of --resolve-site-ids (find it by scanning the channel's
shards — opt-in, because a shard is 8 MB), and a per-clip siteVideo/siteChannel
escape hatch. It deliberately does NOT derive the id from citeUrl: a citeUrl may
point at a different recording on purpose (a mirror that reads better), whose
clock is explicitly not assumed to match.
Also fixes three scripts that defaulted CHANNELS_DIR to an absolute path inside
the original author's home directory — every other clone looked in a directory
that does not exist. It is now resolved relative to the repo.
15 tests, stubbed so they run offline, plus an opt-in LIVE=1 test that hits a
real archive so the stubs cannot drift from the published shape unnoticed.
Verified end-to-end: the same manifest resolves identical windows with a local
corpus and with CHANNELS_DIR pointed at nothing.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat:
6 files changed, 638 insertions(+), 24 deletions(-)
diff --git a/scripts/report-to-video/build-video.mjs b/scripts/report-to-video/build-video.mjs
@@ -39,6 +39,12 @@
// --continue-on-error Record a failed entry and carry on, instead of aborting
// --fetch-only <id> Fetch one clip's window into clips-raw and stop
// --pad <s> Override render.fetchPad (the clip bench fetches wide)
+// --site-origin <url> Archive to read cue windows from when there is no local
+// corpus (defaults to the manifest's provenance.siteOrigin)
+// --resolve-site-ids On a published-id miss, find the record by scanning the
+// channel's shards. Slow; see cues.mjs.
+// --cue-source <which> auto (default) | local | http. The two can disagree
+// once a corpus moves past its last publish — see cues.mjs.
// --no-rail Skip the claim rail even when the manifest configures one
// --rail-only Re-run just the rail over out/<slug>.prerail.mp4
// --preview <s> <d> Rail-only, over a <d>-second window starting at <s>
@@ -55,6 +61,7 @@ import {
renderLedgerCard, ledgerRevealAt, ledgerSeconds,
cardWidth, contentWidth, reservedFooterHeight,
} from "./render-cards.mjs";
+import { createCueSource, siteOriginFromManifest } from "./cues.mjs";
const execFileP = promisify(execFile);
@@ -63,9 +70,11 @@ const FFMPEG = process.env.FFMPEG_BIN ?? "ffmpeg";
const FFPROBE = process.env.FFPROBE_BIN ?? "ffprobe";
const QRENCODE = process.env.QRENCODE_BIN ?? "qrencode";
-const CHANNELS_DIR =
- process.env.CHANNELS_DIR ??
- "/home/user/Projects/yt-dlp-transcript-browser/transcripts/channels";
+// Cue windows and per-video metadata come from a local corpus when there is one
+// and from the published archive otherwise, so this runs in a clone with no
+// `transcripts/` directory. Built once main() has the manifest (it carries the
+// archive origin); see cues.mjs.
+let CUES = null;
const exists = (p) => access(p).then(() => true, () => false);
@@ -227,9 +236,10 @@ function wrap(text, cols) {
return lines.join("\n");
}
-async function videoMeta(videoId, channelSlug) {
- const p = path.join(CHANNELS_DIR, channelSlug, "data", videoId, "transcript.cues.json");
- const d = JSON.parse(await readFile(p, "utf8"));
+// The published shard record carries the same fields as a local cue file, so this
+// reads identically whichever source answered.
+async function videoMeta(videoId, channelSlug, hints = {}) {
+ const d = await CUES.load(channelSlug, videoId, hints);
return { title: d.title, uploadDate: d.uploadDate, webpageUrl: d.webpageUrl, duration: d.duration };
}
@@ -1417,7 +1427,7 @@ async function chapterTitle(entry, index, provenance) {
if (entry.chapter) return entry.chapter;
if (entry.type !== "clip") return entry.title ?? entry.heading ?? `Card ${index + 1}`;
try {
- const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug);
+ const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug, { siteChannel: entry.siteChannel, siteVideo: entry.siteVideo });
const d = String(meta.uploadDate ?? "");
const date = /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : d;
const title = String(meta.title ?? entry.video);
@@ -1508,6 +1518,15 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly
const whole = JSON.parse(await readFile(manifestPath, "utf8"));
const manifest = selectVariant(whole, variant);
const { render, provenance } = manifest;
+
+ // The manifest already records which archive it was built against, so a clone
+ // with no corpus needs no extra configuration to read cue windows.
+ CUES = createCueSource({
+ siteOrigin: opts.siteOrigin ?? process.env.SITE_ORIGIN ?? siteOriginFromManifest(whole),
+ resolveSiteIds: opts.resolveSiteIds === true,
+ prefer: opts.cueSource ?? "auto",
+ log: (m) => EMIT("log", { message: m }),
+ });
const outRoot = out ?? path.join(path.dirname(path.resolve(manifestPath)), "out");
const dirs = variantPaths(outRoot, manifest.slug, variant);
const outDir = dirs.dir;
@@ -1553,7 +1572,7 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly
end: at + 1,
};
}
- const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug);
+ const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug, { siteChannel: entry.siteChannel, siteVideo: entry.siteVideo });
const r = await fetchClip(entry, meta, render, dirs.rawDir, opts);
EMIT("done", { out: r.path, fetchStart: r.fetchStart, cached: r.cached });
return { out: r.path, failures: [] };
@@ -1651,7 +1670,7 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly
: await buildLedgerSegment(entry, render, outDir, manifest.ledger, availability),
);
} else {
- const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug);
+ const meta = await videoMeta(entry.video, entry.channel ?? provenance.channelSlug, { siteChannel: entry.siteChannel, siteVideo: entry.siteVideo });
EMIT("clip", {
id: entry.id, i, n: entries.length, video: entry.video,
section: entry.section, sectionEnter: !!entry.sectionEnter,
@@ -1761,7 +1780,8 @@ async function main() {
" [--only <id>] [--fetch-only <id>]\n" +
" [--pad <s>] [--skip-fetch] [--no-xfade] [--no-chapters] [--chapters-only]\n" +
" [--progress ndjson] [--continue-on-error] [--no-reuse]\n" +
- " [--no-rail] [--rail-only] [--preview <start> <dur>]",
+ " [--no-rail] [--rail-only] [--preview <start> <dur>]\n" +
+ " [--site-origin <url>] [--resolve-site-ids] [--cue-source auto|local|http]",
);
process.exit(2);
}
@@ -1783,6 +1803,9 @@ async function main() {
noRail: argv.includes("--no-rail"),
railOnly: argv.includes("--rail-only"),
pad: padArg === undefined ? undefined : Number(padArg),
+ siteOrigin: flag("--site-origin"),
+ resolveSiteIds: argv.includes("--resolve-site-ids"),
+ cueSource: flag("--cue-source"),
};
const pv = argv.indexOf("--preview");
if (pv >= 0) {
diff --git a/scripts/report-to-video/check-availability.mjs b/scripts/report-to-video/check-availability.mjs
@@ -25,12 +25,14 @@ import { promisify } from "node:util";
import { mkdir, readFile, writeFile } from "node:fs/promises";
import path from "node:path";
+import { DEFAULT_CHANNELS_DIR } from "./cues.mjs";
+
const execFileP = promisify(execFile);
const YTDLP = process.env.YTDLP_BIN ?? "yt-dlp";
-const CHANNELS_DIR =
- process.env.CHANNELS_DIR ??
- "/home/user/Projects/yt-dlp-transcript-browser/transcripts/channels";
+// Resolved relative to the repo (see cues.mjs) rather than an absolute path in
+// one machine's home directory, which every other clone would miss.
+const CHANNELS_DIR = DEFAULT_CHANNELS_DIR;
// yt-dlp says why in prose, and the distinction matters editorially: a private
// or removed video needs the clip converting to a quote card, while a network
diff --git a/scripts/report-to-video/cues.mjs b/scripts/report-to-video/cues.mjs
@@ -0,0 +1,270 @@
+// Where a clip's caption cues come from.
+//
+// A clip window is widened from a cue span to a whole sentence, which needs cue
+// END times. Nothing else in the pipeline carries them: a report citation is a
+// single start second, and the MCP `Snippet` type has no `end` field. So this is
+// the one place that answers "what are the real cue boundaries for this video".
+//
+// TWO SOURCES, SAME SHAPE. A local corpus stores each video as
+// `<CHANNELS_DIR>/<slug>/data/<id>/transcript.cues.json`, and a *published*
+// archive serves the same record inside a paginated shard. The two carry the
+// same fields — `{ slug, id, channelSlug, title, uploadDate, duration, channel,
+// description, platform, webpageUrl, cues: [{start, end, text}] }` — so one
+// resolver serves both `loadCues` and `videoMeta`, and a caller cannot tell
+// which it got beyond the `from` marker.
+//
+// That parity is what makes a corpus optional. Clone the repo, point a manifest
+// at a public instance, and the video pipeline can cut clips without mirroring a
+// single channel: the cue windows come over HTTP, and the media itself was
+// always a network fetch (`yt-dlp --download-sections`).
+//
+// The shard walk is the contract published at `/corpus.json` under `shardScheme`:
+// 1. GET <origin>/corpus.json -> channels[].manifests.transcripts
+// 2. GET that manifest -> { pageCount, slugToPage: { <id>: N } }
+// 3. GET page-<NNNN>.json (N zero-padded to 4) -> array of records
+// 4. take the record whose `id` matches
+//
+// Local wins when present: it is faster, works offline, and is the operator's own
+// data. HTTP is the fallback, not a preference.
+//
+// THE TWO SOURCES CAN DISAGREE, AND IT IS NOT ROUNDING. A published archive is a
+// snapshot; a live corpus keeps moving. Re-synced platform captions, an
+// auto-caption replacement or a re-transcription all rewrite a video's cues in
+// place, and the archive keeps the text it was built from until it is rebuilt.
+// Measured on this corpus (local 2026-08-13 against a 2026-08-07 publish): of
+// four videos checked, three were byte-identical and one had 65 of its 84 cue
+// texts changed with timings shifted by up to **2.24 s** — enough to cut a clip
+// in the wrong place.
+//
+// So `prefer` is a real decision, not a micro-optimisation:
+// "auto" (default) local when present, else HTTP. Right for an operator.
+// "local" never fall back. Fail loudly instead of silently cutting from
+// different cues than the ones a window was authored against.
+// "http" always the archive. Right when you want the windows to match what a
+// reader following the citation will actually see, and the only
+// option that is reproducible on a machine with no corpus.
+// Whatever answers, the returned record carries `from` so a caller can record it.
+
+import { readFile, writeFile, mkdir } from "node:fs/promises";
+import path from "node:path";
+import os from "node:os";
+import { createHash } from "node:crypto";
+import { fileURLToPath } from "node:url";
+
+// This file lives at <repo>/scripts/report-to-video/, so the corpus a plain
+// checkout would have is two levels up. Previously this defaulted to an absolute
+// path inside the original author's home directory, which meant every other
+// clone silently looked in a directory that does not exist.
+const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+
+export const DEFAULT_CHANNELS_DIR =
+ process.env.CHANNELS_DIR ?? path.join(REPO_ROOT, "transcripts", "channels");
+
+const DEFAULT_CACHE_DIR =
+ process.env.REPORT_CACHE_DIR ??
+ path.join(os.homedir(), ".cache", "archilyzer-report-to-video");
+
+// A shard page is capped at 8 MB and holds ~100 videos, so refetching one per
+// clip — across two separate processes, resolve-windows then build-video — is
+// the difference between usable and painful. Cached by URL on disk; archives are
+// rebuilt rarely and a stale page only matters if the cues themselves changed.
+function cacheKey(url) {
+ return createHash("sha1").update(url).digest("hex") + ".json";
+}
+
+export function pageFileName(pageNumber) {
+ return `page-${String(pageNumber).padStart(4, "0")}.json`;
+}
+
+export function pageUrlFrom(manifestUrl, pageNumber) {
+ const u = new URL(manifestUrl);
+ u.pathname = u.pathname.replace(/[^/]+$/, pageFileName(pageNumber));
+ return u.toString();
+}
+
+// The origin of the archive a manifest was built against. Every manifest already
+// records this — `siteOrigin` explicitly, and `corpus` as `remote:<url>` or
+// `local:<path>` — so the common case needs no configuration at all.
+export function siteOriginFromManifest(manifest) {
+ const p = manifest?.provenance ?? {};
+ if (typeof p.siteOrigin === "string" && p.siteOrigin.trim()) {
+ return p.siteOrigin.replace(/\/+$/, "");
+ }
+ if (typeof p.corpus === "string" && p.corpus.startsWith("remote:")) {
+ return p.corpus.slice("remote:".length).replace(/\/+$/, "");
+ }
+ if (typeof p.shareLink === "string" && /^https?:/.test(p.shareLink)) {
+ try {
+ return new URL(p.shareLink).origin;
+ } catch {
+ /* fall through */
+ }
+ }
+ return null;
+}
+
+export class CueLookupError extends Error {
+ constructor(message, { channelSlug, videoId, tried }) {
+ super(message);
+ this.name = "CueLookupError";
+ this.channelSlug = channelSlug;
+ this.videoId = videoId;
+ this.tried = tried;
+ }
+}
+
+export function createCueSource({
+ channelsDir = DEFAULT_CHANNELS_DIR,
+ siteOrigin = null,
+ cacheDir = DEFAULT_CACHE_DIR,
+ fetchImpl = globalThis.fetch,
+ log = () => {},
+ // "auto" | "local" | "http" — see the note on divergence above.
+ prefer = "auto",
+ // Opt-in, because it is expensive: see resolveSiteId below.
+ resolveSiteIds = false,
+} = {}) {
+ const mem = new Map();
+
+ async function getJson(url) {
+ if (mem.has(url)) return mem.get(url);
+ const disk = cacheDir ? path.join(cacheDir, cacheKey(url)) : null;
+ if (disk) {
+ try {
+ const cached = JSON.parse(await readFile(disk, "utf8"));
+ mem.set(url, cached);
+ return cached;
+ } catch {
+ /* cold cache */
+ }
+ }
+ log(`fetch ${url}`);
+ const res = await fetchImpl(url);
+ if (!res.ok) throw new Error(`GET ${url} -> ${res.status}`);
+ const json = await res.json();
+ mem.set(url, json);
+ if (disk) {
+ try {
+ await mkdir(path.dirname(disk), { recursive: true });
+ await writeFile(disk, JSON.stringify(json));
+ } catch {
+ // A cache we cannot write is a slow run, not a failed one.
+ }
+ }
+ return json;
+ }
+
+ async function channelEntry(origin, channelSlug) {
+ const corpus = await getJson(`${origin}/corpus.json`);
+ const found = (corpus.channels ?? []).find((c) => c.slug === channelSlug);
+ if (!found) {
+ throw new CueLookupError(
+ `channel "${channelSlug}" is not in ${origin}/corpus.json`,
+ { channelSlug, videoId: null, tried: [`${origin}/corpus.json`] },
+ );
+ }
+ return found;
+ }
+
+ // Map a LOCAL video id onto the id the site serves, by scanning the channel's
+ // pages for a record whose `webpageUrl` contains it.
+ //
+ // WHY THIS EXISTS: a Rumble video has two ids. The site (and the MCP) key it by
+ // the EMBED id; the local cue directory is named for the URL SLUG. A manifest
+ // hand-authored against local cue dirs therefore carries slugs that are absent
+ // from the published `slugToPage` — every Rumble clip misses.
+ //
+ // WHY IT IS OPT-IN: it downloads a channel's shards until it hits a match, and
+ // a shard is up to 8 MB. That is a reasonable price to pay knowingly and a
+ // terrible one to pay silently, so the direct lookup fails with instructions
+ // instead and this runs only when asked.
+ async function resolveSiteId(origin, entry, wanted) {
+ const manifest = await getJson(entry.manifests.transcripts);
+ log(`resolving "${wanted}" by scanning ${manifest.pageCount} shard(s) of ${entry.slug}`);
+ for (let n = 0; n < manifest.pageCount; n += 1) {
+ const page = await getJson(pageUrlFrom(entry.manifests.transcripts, n));
+ const hit = page.find(
+ (r) => r.id === wanted || r.slug === wanted || String(r.webpageUrl ?? "").includes(wanted),
+ );
+ if (hit) return hit;
+ }
+ return null;
+ }
+
+ async function fromHttp(channelSlug, videoId, hints) {
+ const origin = hints.siteOrigin ?? siteOrigin;
+ if (!origin) {
+ throw new CueLookupError(
+ `no local cues for ${channelSlug}/${videoId} and no archive origin to fetch them from ` +
+ `(set provenance.siteOrigin in the manifest, or pass --site-origin / SITE_ORIGIN)`,
+ { channelSlug, videoId, tried: ["local"] },
+ );
+ }
+ const siteChannel = hints.siteChannel ?? channelSlug;
+ const siteVideo = hints.siteVideo ?? videoId;
+ const entry = await channelEntry(origin, siteChannel);
+ const manifest = await getJson(entry.manifests.transcripts);
+ const pageNumber = manifest.slugToPage?.[siteVideo];
+
+ if (pageNumber === undefined) {
+ if (resolveSiteIds) {
+ const hit = await resolveSiteId(origin, entry, siteVideo);
+ if (hit) return { ...hit, from: "http" };
+ }
+ throw new CueLookupError(
+ `"${siteVideo}" is not in ${siteChannel}'s published slugToPage on ${origin}.\n` +
+ ` If this is a Rumble clip, the archive is keyed by the EMBED id while a local cue\n` +
+ ` directory is named for the URL SLUG — they differ. Either add "siteVideo" (and\n` +
+ ` "siteChannel" if it also differs) to this clip in the manifest, or re-run with\n` +
+ ` --resolve-site-ids to find it by scanning the channel's shards (slow: downloads\n` +
+ ` up to 8 MB per shard until it matches).\n` +
+ ` Note that a clip's citeUrl is NOT usable here — it may deliberately cite a\n` +
+ ` different recording (a mirror that reads better), whose clock is not the same.`,
+ { channelSlug, videoId, tried: [entry.manifests.transcripts] },
+ );
+ }
+
+ const page = await getJson(pageUrlFrom(entry.manifests.transcripts, pageNumber));
+ const record = page.find((r) => r.id === siteVideo || r.slug === siteVideo);
+ if (!record) {
+ throw new CueLookupError(
+ `${siteChannel}/${siteVideo} is on shard ${pageNumber} per the manifest, but no record ` +
+ `there has that id — the published archive is inconsistent`,
+ { channelSlug, videoId, tried: [pageUrlFrom(entry.manifests.transcripts, pageNumber)] },
+ );
+ }
+ return { ...record, from: "http" };
+ }
+
+ async function fromLocal(channelSlug, videoId) {
+ const p = path.join(channelsDir, channelSlug, "data", videoId, "transcript.cues.json");
+ const parsed = JSON.parse(await readFile(p, "utf8"));
+ return { ...parsed, from: "local" };
+ }
+
+ return {
+ channelsDir,
+ /**
+ * The full record for one video: cues plus the metadata build-video needs.
+ * `hints` may carry `siteChannel` / `siteVideo` (when the published archive
+ * keys this recording differently) and `siteOrigin` (per-manifest override).
+ */
+ prefer,
+ async load(channelSlug, videoId, hints = {}) {
+ if (prefer === "http") return await fromHttp(channelSlug, videoId, hints);
+ try {
+ return await fromLocal(channelSlug, videoId);
+ } catch (err) {
+ if (err?.code !== "ENOENT" && err?.code !== "ENOTDIR") throw err;
+ if (prefer === "local") {
+ throw new CueLookupError(
+ `no local cues for ${channelSlug}/${videoId} under ${channelsDir}, and ` +
+ `--cue-source local forbids falling back to the archive`,
+ { channelSlug, videoId, tried: [channelsDir] },
+ );
+ }
+ return await fromHttp(channelSlug, videoId, hints);
+ }
+ },
+ };
+}
diff --git a/scripts/report-to-video/cues.test.mjs b/scripts/report-to-video/cues.test.mjs
@@ -0,0 +1,297 @@
+// Tests for cues.mjs — the local-or-published cue resolver.
+//
+// The fetches are stubbed against a miniature of the real published shape, so
+// these run offline and in CI. One separate, opt-in test hits a live archive to
+// prove the miniature has not drifted from reality; see LIVE below.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import test from "node:test";
+import { mkdtemp, mkdir, writeFile, rm } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+import {
+ createCueSource,
+ pageFileName,
+ pageUrlFrom,
+ siteOriginFromManifest,
+} from "./cues.mjs";
+
+const ORIGIN = "https://example.pages.dev";
+
+// A published record. Deliberately the SAME field set a local transcript.cues.json
+// carries — that parity is the whole reason one resolver can serve both.
+const RECORD = {
+ slug: "chan/vid1",
+ id: "vid1",
+ channelSlug: "chan",
+ title: "A Title",
+ uploadDate: "20260101",
+ duration: 1200,
+ webpageUrl: "https://rumble.com/localslug-a-title.html",
+ cues: [
+ { start: 0, end: 2.5, text: "first line" },
+ { start: 2.5, end: 5, text: "second line." },
+ ],
+};
+
+function stubFetch(routes, seen = []) {
+ return async (url) => {
+ seen.push(url);
+ if (!(url in routes)) return { ok: false, status: 404 };
+ return { ok: true, status: 200, json: async () => routes[url] };
+ };
+}
+
+const ROUTES = {
+ [`${ORIGIN}/corpus.json`]: {
+ channels: [
+ { slug: "chan", manifests: { transcripts: `${ORIGIN}/transcripts/chan/manifest.json` } },
+ ],
+ },
+ [`${ORIGIN}/transcripts/chan/manifest.json`]: {
+ pageCount: 2,
+ slugToPage: { vid1: 0, other: 1 },
+ },
+ [`${ORIGIN}/transcripts/chan/page-0000.json`]: [RECORD],
+ [`${ORIGIN}/transcripts/chan/page-0001.json`]: [
+ { ...RECORD, id: "other", slug: "chan/other", webpageUrl: "https://rumble.com/v94hyv-x.html" },
+ ],
+};
+
+// cacheDir:null keeps every test off the real disk cache, so one test cannot
+// poison another (or a developer's home directory).
+function source(extra = {}) {
+ return createCueSource({
+ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"),
+ siteOrigin: ORIGIN,
+ cacheDir: null,
+ fetchImpl: stubFetch(ROUTES),
+ ...extra,
+ });
+}
+
+// --- the shard walk ---------------------------------------------------------
+
+test("page numbers are zero-padded to four digits", () => {
+ assert.equal(pageFileName(0), "page-0000.json");
+ assert.equal(pageFileName(7), "page-0007.json");
+ assert.equal(pageFileName(1234), "page-1234.json");
+ // Not a cosmetic detail: page-0.json is a 404 on a real archive.
+ assert.notEqual(pageFileName(0), "page-0.json");
+});
+
+test("a page URL replaces only the manifest's last segment", () => {
+ assert.equal(
+ pageUrlFrom("https://x.dev/transcripts/some-chan/manifest.json", 3),
+ "https://x.dev/transcripts/some-chan/page-0003.json",
+ );
+});
+
+test("resolves a video over HTTP with cues intact", async () => {
+ const got = await source().load("chan", "vid1");
+ assert.equal(got.from, "http");
+ assert.equal(got.title, "A Title");
+ assert.equal(got.duration, 1200);
+ assert.equal(got.cues.length, 2);
+ assert.deepEqual(got.cues[0], { start: 0, end: 2.5, text: "first line" });
+ // The end times are the entire point: they are what widens a clip to a
+ // whole sentence, and nothing else in the pipeline carries them.
+ assert.ok(got.cues.every((c) => typeof c.end === "number"));
+});
+
+test("fetches each URL once, however many videos are read", async () => {
+ const seen = [];
+ const src = createCueSource({
+ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"),
+ siteOrigin: ORIGIN,
+ cacheDir: null,
+ fetchImpl: stubFetch(ROUTES, seen),
+ });
+ await src.load("chan", "vid1");
+ await src.load("chan", "vid1");
+ await src.load("chan", "other");
+ // corpus + manifest once each, then one shard per distinct page.
+ assert.deepEqual(seen, [
+ `${ORIGIN}/corpus.json`,
+ `${ORIGIN}/transcripts/chan/manifest.json`,
+ `${ORIGIN}/transcripts/chan/page-0000.json`,
+ `${ORIGIN}/transcripts/chan/page-0001.json`,
+ ]);
+});
+
+// --- local wins -------------------------------------------------------------
+
+test("a local corpus is preferred over the network", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "cues-local-"));
+ try {
+ const vdir = path.join(dir, "chan", "data", "vid1");
+ await mkdir(vdir, { recursive: true });
+ await writeFile(
+ path.join(vdir, "transcript.cues.json"),
+ JSON.stringify({ ...RECORD, title: "LOCAL COPY" }),
+ );
+ const seen = [];
+ const src = createCueSource({
+ channelsDir: dir,
+ siteOrigin: ORIGIN,
+ cacheDir: null,
+ fetchImpl: stubFetch(ROUTES, seen),
+ });
+ const got = await src.load("chan", "vid1");
+ assert.equal(got.from, "local");
+ assert.equal(got.title, "LOCAL COPY");
+ assert.deepEqual(seen, [], "must not touch the network when local data exists");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+// --- the Rumble two-id trap -------------------------------------------------
+
+test("a published-id miss fails loudly, naming the two-id trap", async () => {
+ // A Rumble video has two ids: the archive keys it by the EMBED id, while a
+ // local cue directory is named for the URL SLUG. A manifest authored against
+ // local dirs therefore carries an id the archive has never heard of.
+ await assert.rejects(
+ () => source().load("chan", "localslug"),
+ (err) => {
+ assert.equal(err.name, "CueLookupError");
+ assert.match(err.message, /EMBED id/);
+ assert.match(err.message, /siteVideo/);
+ assert.match(err.message, /--resolve-site-ids/);
+ // It must also warn off the tempting wrong fix.
+ assert.match(err.message, /citeUrl is NOT usable/);
+ return true;
+ },
+ );
+});
+
+test("an explicit siteVideo hint resolves the mismatch", async () => {
+ const got = await source().load("chan", "localslug", { siteVideo: "vid1" });
+ assert.equal(got.id, "vid1");
+ assert.equal(got.cues.length, 2);
+});
+
+test("--resolve-site-ids finds the record by scanning shards", async () => {
+ // `other`'s webpageUrl embeds v94hyv, mirroring how a Rumble URL carries the
+ // slug while the archive is keyed by the embed id.
+ const got = await source({ resolveSiteIds: true }).load("chan", "v94hyv");
+ assert.equal(got.id, "other");
+ assert.equal(got.from, "http");
+});
+
+test("scanning is off by default, because a shard is up to 8 MB", async () => {
+ await assert.rejects(() => source().load("chan", "v94hyv"), { name: "CueLookupError" });
+});
+
+// --- prefer: the two sources can genuinely disagree --------------------------
+
+test("prefer:http ignores a local copy entirely", async () => {
+ // Not a micro-optimisation. A published archive is a snapshot and a corpus
+ // keeps moving: measured on the real corpus, one video of four had 65 of its
+ // 84 cue texts rewritten and timings shifted by up to 2.24s between a
+ // 2026-08-07 publish and the local copy six days later. Which source answered
+ // decides where a clip gets cut.
+ const dir = await mkdtemp(path.join(tmpdir(), "cues-prefer-"));
+ try {
+ const vdir = path.join(dir, "chan", "data", "vid1");
+ await mkdir(vdir, { recursive: true });
+ await writeFile(
+ path.join(vdir, "transcript.cues.json"),
+ JSON.stringify({ ...RECORD, title: "LOCAL COPY" }),
+ );
+ const src = createCueSource({
+ channelsDir: dir,
+ siteOrigin: ORIGIN,
+ cacheDir: null,
+ fetchImpl: stubFetch(ROUTES),
+ prefer: "http",
+ });
+ const got = await src.load("chan", "vid1");
+ assert.equal(got.from, "http");
+ assert.equal(got.title, "A Title", "must be the archive's copy, not the local one");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("prefer:local refuses to fall back rather than cut from other cues", async () => {
+ const src = createCueSource({
+ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"),
+ siteOrigin: ORIGIN,
+ cacheDir: null,
+ fetchImpl: stubFetch(ROUTES),
+ prefer: "local",
+ });
+ await assert.rejects(() => src.load("chan", "vid1"), (err) => {
+ assert.equal(err.name, "CueLookupError");
+ assert.match(err.message, /forbids falling back/);
+ return true;
+ });
+});
+
+// --- origin discovery -------------------------------------------------------
+
+test("the archive origin comes from the manifest, in priority order", () => {
+ assert.equal(
+ siteOriginFromManifest({ provenance: { siteOrigin: "https://a.dev/" } }),
+ "https://a.dev",
+ "trailing slash trimmed",
+ );
+ assert.equal(
+ siteOriginFromManifest({ provenance: { corpus: "remote:https://b.dev" } }),
+ "https://b.dev",
+ );
+ assert.equal(
+ siteOriginFromManifest({ provenance: { shareLink: "https://c.dev/?v=x&t=1" } }),
+ "https://c.dev",
+ );
+ // A local-corpus manifest names no remote origin, and must not invent one.
+ assert.equal(siteOriginFromManifest({ provenance: { corpus: "local:/srv/x" } }), null);
+ assert.equal(siteOriginFromManifest({}), null);
+});
+
+test("no origin and no local copy is a clear error, not a crash", async () => {
+ const src = createCueSource({
+ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"),
+ siteOrigin: null,
+ cacheDir: null,
+ fetchImpl: stubFetch({}),
+ });
+ await assert.rejects(() => src.load("chan", "vid1"), (err) => {
+ assert.equal(err.name, "CueLookupError");
+ assert.match(err.message, /no archive origin/);
+ return true;
+ });
+});
+
+test("a channel absent from corpus.json is reported as such", async () => {
+ await assert.rejects(() => source().load("nosuch", "vid1"), (err) => {
+ assert.match(err.message, /not in .*corpus\.json/);
+ return true;
+ });
+});
+
+// --- reality check ----------------------------------------------------------
+
+// Opt-in: `LIVE=1 pnpm test:scripts`. The stubs above encode assumptions about a
+// published archive's shape; this is the only thing that can catch them going
+// stale. Skipped by default so the suite stays offline and deterministic.
+test("LIVE: a real archive still matches the shape these stubs assume", {
+ skip: process.env.LIVE === "1" ? false : "set LIVE=1 to hit the network",
+}, async () => {
+ const src = createCueSource({
+ channelsDir: path.join(tmpdir(), "definitely-no-corpus-here"),
+ siteOrigin: "https://jeralyzer.pages.dev",
+ cacheDir: null,
+ });
+ const got = await src.load("chrissie-mayr", "2Pn_rMrHmEs");
+ assert.equal(got.from, "http");
+ assert.ok(got.cues.length > 0);
+ assert.ok(got.cues.every((c) => typeof c.start === "number" && typeof c.end === "number"));
+ for (const field of ["title", "uploadDate", "duration", "webpageUrl"]) {
+ assert.ok(got[field] !== undefined, `published record should carry ${field}`);
+ }
+});
diff --git a/scripts/report-to-video/package.json b/scripts/report-to-video/package.json
@@ -15,10 +15,11 @@
"./build-video": "./build-video.mjs",
"./check-availability": "./check-availability.mjs",
"./compose-chrome": "./compose-chrome.mjs",
+ "./cues": "./cues.mjs",
"./ledger-totals": "./ledger-totals.mjs",
+ "./package.json": "./package.json",
"./render-cards": "./render-cards.mjs",
"./resolve-windows": "./resolve-windows.mjs",
- "./verify-build": "./verify-build.mjs",
- "./package.json": "./package.json"
+ "./verify-build": "./verify-build.mjs"
}
}
diff --git a/scripts/report-to-video/resolve-windows.mjs b/scripts/report-to-video/resolve-windows.mjs
@@ -21,15 +21,19 @@
// --write Rewrite the manifest in place (default: dry run, print a table)
// --max-lead <s> Max seconds to expand backwards (default 9)
// --max-tail <s> Max seconds to expand forwards (default 12)
+// --site-origin <url> Archive to read cues from when there is no local corpus
+// (defaults to the manifest's provenance.siteOrigin)
+// --resolve-site-ids On a published-id miss, find the record by scanning the
+// channel's shards. Slow; see cues.mjs.
+// --cue-source <which> auto (default) | local | http. The two can disagree
+// once a corpus moves past its last publish — see cues.mjs.
//
// A clip entry may set `lockStart` / `lockEnd` to pin that edge exactly.
import { readFile, writeFile } from "node:fs/promises";
-import path from "node:path";
-const CHANNELS_DIR =
- process.env.CHANNELS_DIR ??
- "/home/user/Projects/yt-dlp-transcript-browser/transcripts/channels";
+import { createCueSource, siteOriginFromManifest } from "./cues.mjs";
+
const ENDS_SENTENCE = /[.!?]["'”’)\]]*\s*$/;
@@ -43,10 +47,6 @@ const IS_FILLER = /^\s*(\[[^\]]*\]|>>|♪|—|-)*\s*$/;
// the clip grows a little every time this is run — it has to be a fixed point.
const EPS = 0.02;
-function loadCues(videoId, channelSlug) {
- const p = path.join(CHANNELS_DIR, channelSlug, "data", videoId, "transcript.cues.json");
- return readFile(p, "utf8").then((s) => JSON.parse(s).cues);
-}
function indexAt(cues, t, which) {
// First cue whose span contains t, else the nearest one on the right side.
@@ -120,6 +120,22 @@ async function main() {
const slug = manifest.provenance.channelSlug;
const cache = new Map();
+ // Cues come from a local corpus when there is one, and from the published
+ // archive the manifest was built against when there is not — so this runs in a
+ // clone with no `transcripts/` at all. See cues.mjs.
+ const flag = (name) => {
+ const i = argv.indexOf(name);
+ return i >= 0 ? argv[i + 1] : undefined;
+ };
+ const cues = createCueSource({
+ siteOrigin: flag("--site-origin") ?? process.env.SITE_ORIGIN ?? siteOriginFromManifest(manifest),
+ resolveSiteIds: argv.includes("--resolve-site-ids"),
+ prefer: flag("--cue-source") ?? "auto",
+ log: (m) => console.error(` · ${m}`),
+ });
+ const loadCues = (videoId, channelSlug, hints) =>
+ cues.load(channelSlug, videoId, hints).then((r) => r.cues);
+
let changed = 0;
for (const e of manifest.timeline) {
if (e.type !== "clip") continue;
@@ -135,7 +151,12 @@ async function main() {
}
const chan = e.channel ?? slug;
const key = `${chan}/${e.video}`;
- if (!cache.has(key)) cache.set(key, await loadCues(e.video, chan));
+ if (!cache.has(key)) {
+ cache.set(
+ key,
+ await loadCues(e.video, chan, { siteChannel: e.siteChannel, siteVideo: e.siteVideo }),
+ );
+ }
const cues = cache.get(key);
const before = { start: e.start, end: e.end };