commit a756154f67b26cf2c9f91d5e273ae2f6d7b446ef
parent 1cf1a7f7410854c0cf8160f256b762f8ce95ba48
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 12:08:02 -0400
Merge main into r18/integration (multi-track captions, en-track fallback, Wayback, Odysee/BitChute spacing); the changelog keeps both sides, release 18's bullets first
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
76 files changed, 4230 insertions(+), 176 deletions(-)
diff --git a/PUBLISH.md b/PUBLISH.md
@@ -38,6 +38,16 @@ the stats datasets and the chart templates — `archilyzer index`, `build stats`
`export/public`, plus its download archives — `archilyzer compose site <id>`) and
`next build`. `--nodata` skips the data phase and reuses the last one's staging.
+A transcript record in the shared pages carries its other English caption tracks
+(`altTracks`, with the transcript's own `track`) only where one's words differ from
+the transcript's — the served `en` beside `en-orig`, a regional or auto-translated
+track, the captions a local transcription replaced (`common/lib/captionTracks.ts`).
+Identical tracks add nothing, so most records' bytes are what they were. Search reads
+those tracks with the transcript and names the track of a hit only one holds; the
+reader switches to them. English VTTs are not published as subtitle tracks. The first
+index build after this reads them once (`Alternate tracks v1: N record(s) re-read.`),
+for exactly the records that can hold one.
+
## The three ways to drive it
The editor's **/sites** page, `pnpm ops` (HTTP to a running editor, with its
@@ -116,8 +126,9 @@ mirror is (a fresh bare clone, `repack -a -d`, `update-server-info`, then only `
`packed-refs`, `info/refs`, `objects/info/packs`, the packs and `refs/heads/main`), so
`git clone <siteUrl>/reports/<id>/history/repo` works. The deploy audit refuses any other
file in a clone or one over 24 MiB, and counts the clone's files toward Pages' 20,000. Every quote is checked as it is
-composed: a span's against its cues within 5 s either side (a record whose `en` track
-has no cues is read from `en-orig`), a post's against its text, scored as the share of
+composed: a span's against its cues within 5 s either side (read under the caption-track
+rule — `en-orig` first, then the first English track with cues — and against every
+other English track, the best match shown), a post's against its text, scored as the share of
the quote's words found there; the score, the time and the method are written into the
citation, replacing any typed by hand. Compose fails, before it writes any of it, with
the list of everything wrong: an invalid report, a citation of a channel outside the
diff --git a/README.md b/README.md
@@ -212,6 +212,29 @@ nothing asked while BitChute is in a rate-limit cooldown, which a 429 starts; an
on disk never fetched again (`common/ytdlp/platformArgs.mjs`,
`common/controller/bitchuteImport.ts`).
+### Wayback Machine captures
+
+**Import video** takes a Wayback Machine capture
+(`https://web.archive.org/web/<timestamp>/<original>`, with or without a replay
+modifier such as `id_`) and yt-dlp downloads it: an archived YouTube page through its
+Wayback extractor, a raw media file as the file. The record is named by what the capture
+is OF: an archived YouTube page by its YouTube id, a JW Player file
+(`…/videos/<id>-<rendition>.mp4` on `cdn.jwplayer.com`, `content.jwplatform.com`,
+`videos-fms.jwpsrv.com`) by its media id. Every download of a capture writes
+`wayback.json` beside its metadata: the original URL (an archived YouTube page's watch
+URL), the capture's timestamp, the capture as a page that plays and as its raw bytes.
+
+The video page says "Archived copy (Wayback Machine, <capture date>) of <original>". A
+citation of one links the original, marked as the original and as possibly gone, and the
+Wayback copy; its moment link is the capture, which plays.
+
+`pnpm archilyzer wayback refresh <channel> [--titles <file>] [--dry-run]` brings records
+imported before this up to it, offline: `wayback.json`, the dir renamed to its id through
+the snapshot's own reconcile pass (the roster entry moves with it), and with `--titles` (a
+JSON file of `id → {title, upload_date}`) the title and date of a raw file that has none.
+A record a running job holds is skipped and named. It prints old → new; a second run
+changes nothing.
+
### Requirements
Always needed, to install and run the apps:
@@ -359,7 +382,13 @@ differs is emphasis:
- **Captions you already publish are reused.** Channels whose platform publishes
captions need no transcription backend at all; those are taken directly. There is
also an opt-in lane that *replaces* platform auto-captions with your own transcripts
- where you would rather have the better text.
+ where you would rather have the better text. Where YouTube serves both, the
+ original-audio captions (`en-orig`) are read before the served `en` track, which can
+ reword what was said; a track with no text falls through to the next, and a video's
+ page can pin another (`common/lib/videoStatus.ts`, "the caption-track rule"). The
+ other English tracks are kept where their words differ — uploaded captions are not
+ always what was said: search reads them too and says which track a hit is in, and
+ the transcript reader switches to them (`common/lib/captionTracks.ts`).
- **The output is yours to brand.** Site title, header, description, tagline, social
links and channel grouping are all per-site configuration.
- **One corpus can publish several sites.** Channels are grouped into sites, so a
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -434,6 +434,25 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["wayback", "refresh"],
+ usage:
+ "<slug> [--titles <json>] [--dry-run] bring a channel's Wayback Machine copies up to the Wayback rules: wayback.json (original URL, capture time), the dir renamed to its canonical id through the snapshot's reconcile (roster moved with it), and with --titles (a file of id → {title, upload_date}) a raw file's title and date; offline, skips a record a live job holds; prints old → new",
+ flags: { titles: "string", "dry-run": "boolean" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags }) => {
+ const [slug] = positionals;
+ if (!slug) {
+ console.error("wayback refresh: which channel? Pass its slug.");
+ return 2;
+ }
+ return (await import("./wayback-refresh")).main({
+ slug,
+ dryRun: flags["dry-run"] === true,
+ ...(typeof flags.titles === "string" ? { titlesFile: flags.titles } : {}),
+ });
+ },
+ },
+ {
path: ["feeds", "backfill-metadata"],
usage:
"<slug> [--feed <url>] [--dry-run] complete a podcast channel's records (title, date, description, duration) from its RSS feed: one fetch of the feed (default: the channel's url), no media; --dry-run counts matched / unmatched / already complete and writes nothing",
diff --git a/common/bin/wayback-refresh.ts b/common/bin/wayback-refresh.ts
@@ -0,0 +1,70 @@
+// `archilyzer wayback refresh <slug> [--titles <json>] [--dry-run]` — every
+// Wayback Machine copy of a channel brought up to the Wayback rules
+// (controller/waybackRefresh.ts): its `wayback.json`, its dir under its
+// canonical id (through the snapshot's own reconcile pass, roster moved with
+// it), and with `--titles` a file mapping id → {title, upload_date} for a raw
+// file that has neither. Offline. Prints old → new per record; a second run
+// changes nothing.
+
+import { readFile } from "node:fs/promises";
+import { isValidChannelSlug } from "../controller/channels";
+import { refreshWaybackRecords, type WaybackTitle } from "../controller/waybackRefresh";
+import { getPaths, type Paths } from "../lib/paths";
+
+async function readTitles(file: string): Promise<Record<string, WaybackTitle>> {
+ const v = JSON.parse(await readFile(file, "utf8")) as unknown;
+ if (!v || typeof v !== "object" || Array.isArray(v)) {
+ throw new Error(`${file}: expected an object of id → {title, upload_date}`);
+ }
+ const out: Record<string, WaybackTitle> = {};
+ for (const [id, raw] of Object.entries(v as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error(`${file}: "${id}" is not an object`);
+ const r = raw as Record<string, unknown>;
+ out[id] = {
+ ...(typeof r.title === "string" ? { title: r.title } : {}),
+ ...(typeof r.upload_date === "string" ? { upload_date: r.upload_date } : {}),
+ };
+ }
+ return out;
+}
+
+export async function main(opts: {
+ slug: string;
+ dryRun: boolean;
+ titlesFile?: string;
+ paths?: Paths;
+}): Promise<number> {
+ if (!isValidChannelSlug(opts.slug)) {
+ console.error(`wayback refresh: "${opts.slug}" is not a channel slug`);
+ return 2;
+ }
+ try {
+ const titles = opts.titlesFile ? await readTitles(opts.titlesFile) : undefined;
+ const r = await refreshWaybackRecords({
+ slug: opts.slug,
+ paths: opts.paths ?? getPaths(),
+ dryRun: opts.dryRun,
+ titles,
+ onLog: (line) => console.log(line.replace(/\n$/, "")),
+ });
+ const would = opts.dryRun ? "would be " : "";
+ const renamed = r.records.filter((x) => x.rename === "renamed" || x.rename === "merged").length;
+ const held = r.records.filter((x) => x.rename === "held" || x.rename === "conflict").length;
+ const sidecars = r.records.filter((x) => x.sidecar).length;
+ const patched = r.records.filter((x) => Object.keys(x.changes).length > 0).length;
+ for (const id of r.unmatchedTitles) console.log(`--titles: no Wayback record "${id}"`);
+ for (const f of r.failed) console.error(`${f.id}: failed — ${f.error}`);
+ console.log(
+ `${opts.dryRun ? "dry run: " : ""}${r.records.length} Wayback records, ` +
+ `${renamed} ${would}renamed, ${sidecars} wayback.json ${would}written, ` +
+ `${patched} ${would}retitled` +
+ (held > 0 ? `, ${held} not renamed` : "") +
+ (r.failed.length > 0 ? `, ${r.failed.length} failed` : "") +
+ ".",
+ );
+ return r.failed.length > 0 ? 1 : 0;
+ } catch (err) {
+ console.error(`wayback refresh: ${(err as Error).message}`);
+ return 1;
+ }
+}
diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx
@@ -25,6 +25,7 @@ import { cuesToSrt, cuesToText } from "../lib/vtt";
import { transcriptToMarkdown } from "../lib/transcriptToMarkdown";
import type { Platform } from "../lib/transcripts";
import { vodExpiry } from "../lib/vodExpiry";
+import { cuesOfTrack, recordTracks, type TrackFields } from "../lib/captionTracks";
import { VodExpiredBadge } from "./badges";
import type { RumblePlayerHandle } from "./RumblePlayer";
import type { OdyseePlayerHandle } from "./OdyseePlayer";
@@ -92,7 +93,15 @@ export type TranscriptData = {
webpageUrl: string;
hlsUrl?: string;
mediaUrl?: string;
+ // The cues of the track on show: the primary, or the alternate the viewer
+ // picked (`track`).
cues?: Cue[];
+ // The record's tracks, primary first (lib/captionTracks.ts) — more than one
+ // only when an alternate's words differ from the primary's. Empty when the
+ // record names none.
+ tracks: string[];
+ // The track on show; null when the record names no tracks.
+ track: string | null;
};
type Status = "idle" | "loading" | "ready";
@@ -117,7 +126,7 @@ type Detail = {
hlsUrl?: string;
mediaUrl?: string;
cues?: Cue[];
-};
+} & TrackFields;
type ClipState = {
slug: string | null;
@@ -164,8 +173,11 @@ type PlayerState = {
openTranscript: (
slug: string,
start?: number,
- opts?: { mode?: ModalMode },
+ opts?: { mode?: ModalMode; track?: string },
) => void;
+ // Show another of the record's tracks (null: the primary). A viewer's
+ // choice — it changes the URL (`vt`), never the record.
+ setViewTrack: (track: string | null) => void;
setDisplayMode: (mode: DisplayMode) => void;
setModalMode: (mode: ModalMode) => void;
closePlayer: () => void;
@@ -395,7 +407,7 @@ export function PlayerProvider({
children: React.ReactNode;
features?: PlayerFeatures;
}) {
- const { v: urlSlug, t: urlTime, vm: urlVm } = useUrlParams();
+ const { v: urlSlug, t: urlTime, vm: urlVm, vt: urlVt } = useUrlParams();
const [core, dispatch] = useReducer(coreReducer, INITIAL_CORE);
const { detail, displayState, clip, chat, digest } = core;
const [playing, setPlaying] = useState(false);
@@ -450,9 +462,15 @@ export function PlayerProvider({
webpageUrl: detail.webpageUrl,
hlsUrl: detail.hlsUrl,
mediaUrl: detail.mediaUrl,
- cues: detail.cues,
+ // A `vt` the record does not hold reads as the primary.
+ cues: cuesOfTrack(detail, urlVt) ?? detail.cues,
+ tracks: recordTracks(detail),
+ track:
+ urlVt && cuesOfTrack(detail, urlVt) !== undefined
+ ? urlVt
+ : (detail.track ?? null),
};
- }, [detail, detailMatches]);
+ }, [detail, detailMatches, urlVt]);
const status: Status = !activeSlug
? "idle"
: detailMatches
@@ -463,7 +481,7 @@ export function PlayerProvider({
const clipEnd = clip.slug && clip.slug === activeSlug ? clip.end : null;
const openTranscript = useCallback(
- (slug: string, start?: number, opts?: { mode?: ModalMode }) => {
+ (slug: string, start?: number, opts?: { mode?: ModalMode; track?: string }) => {
dispatch({ type: "DISPLAY_SET", slug, mode: "modal" });
const t = typeof start === "number" ? Math.round(start) : null;
if (slug === activeSlug && t !== null && playerRef.current) {
@@ -475,11 +493,18 @@ export function PlayerProvider({
v: slug,
t,
vm: opts?.mode ?? "transcript",
+ // A hit from an alternate track opens on that track; anything else
+ // opens on the primary.
+ vt: opts?.track ?? null,
});
},
[activeSlug],
);
+ const setViewTrack = useCallback((track: string | null) => {
+ writeUrlParams({ vt: track });
+ }, []);
+
const setModalMode = useCallback((mode: ModalMode) => {
writeUrlParams({ vm: mode });
}, []);
@@ -494,7 +519,7 @@ export function PlayerProvider({
const closePlayer = useCallback(() => {
dispatch({ type: "DISPLAY_CLEAR" });
- writeUrlParams({ v: null, t: null, vm: "transcript" });
+ writeUrlParams({ v: null, t: null, vm: "transcript", vt: null });
}, []);
const markClipStart = useCallback(() => {
@@ -542,6 +567,10 @@ export function PlayerProvider({
// Carry the current panel so a shared link reopens what the sharer was
// looking at. "transcript" is the default and stays absent from the URL.
if (modalMode !== "transcript") params.set("vm", modalMode);
+ // And the track, when it is not the primary.
+ if (data.track && data.tracks.length > 1 && data.track !== data.tracks[0]) {
+ params.set("vt", data.track);
+ }
const url = `${window.location.origin}${window.location.pathname}?${params.toString()}`;
try {
await navigator.clipboard.writeText(url);
@@ -688,6 +717,9 @@ export function PlayerProvider({
hlsUrl: full.hlsUrl,
mediaUrl: full.mediaUrl,
cues: full.cues,
+ ...(full.altTracks?.length
+ ? { track: full.track, altTracks: full.altTracks }
+ : {}),
},
});
})
@@ -922,6 +954,7 @@ export function PlayerProvider({
clipStart,
clipEnd,
openTranscript,
+ setViewTrack,
setDisplayMode,
setModalMode,
closePlayer,
@@ -948,6 +981,7 @@ export function PlayerProvider({
clipStart,
clipEnd,
openTranscript,
+ setViewTrack,
setDisplayMode,
setModalMode,
closePlayer,
diff --git a/common/components/SearchResults.tsx b/common/components/SearchResults.tsx
@@ -45,6 +45,7 @@ import {
} from "./SearchSessionContext";
import { useSearchData } from "./SearchDataContext";
import type { PublishedTag } from "../lib/curatedTags";
+import { inTrackLabel } from "../lib/captionTracks";
// Rough first-paint guess for one full result card (header + a couple of
// leaf sections + a handful of hits). After mount, ResizeObserver measures
@@ -969,7 +970,9 @@ function PostBadge({ platform }: { platform: string }) {
}
function TrackBadge({ track }: { track: string }) {
- const label = track === "live_chat" ? "live chat" : track;
+ // An alternate English track of a transcript (lib/captionTracks.ts) says
+ // which one: "in uploaded captions".
+ const label = track === "live_chat" ? "live chat" : inTrackLabel(track);
return (
<span className="shrink-0 text-[10px] uppercase tracking-wide font-medium px-1.5 py-0.5 rounded bg-muted text-muted-foreground self-center">
{label}
diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx
@@ -1925,7 +1925,12 @@ function useSearchSessionState() {
: hit?.scope === "chat"
? "chat"
: "transcript";
- openTranscript(slug, hit?.start, { mode: modalMode });
+ // A transcript hit from an alternate track (lib/captionTracks.ts) opens
+ // the reader on that track, where its words are.
+ openTranscript(slug, hit?.start, {
+ mode: modalMode,
+ ...(modalMode === "transcript" && hit?.track ? { track: hit.track } : {}),
+ });
},
[openTranscript],
);
diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx
@@ -36,6 +36,7 @@ import { AgeRestrictedBadge, LivestreamBadge } from "./badges";
import { VirtualRow } from "./VirtualRow";
import { formatTimestamp } from "../lib/vtt";
import { formatDate } from "../lib/format";
+import { trackLabels } from "../lib/captionTracks";
type DisplayCue = {
start: number;
@@ -53,6 +54,7 @@ export default function TranscriptModal() {
status,
modalMode,
setModalMode,
+ setViewTrack,
chatCues,
chatStatus,
modalNotice,
@@ -320,7 +322,22 @@ export default function TranscriptModal() {
)}
</div>
- <div className="px-3 sm:px-0">{modeStrip}</div>
+ <div className="px-3 sm:px-0 flex flex-wrap items-center gap-x-3 gap-y-1">
+ {modeStrip}
+ {/* The record's other English tracks (lib/captionTracks.ts), only
+ where one's words differ from the primary's. A viewer's choice:
+ it swaps the cues on show and changes nothing else. */}
+ {!isChat && !isDigest && data && data.tracks.length > 1 && (
+ <TrackSwitcher
+ tracks={data.tracks}
+ track={data.track ?? data.tracks[0]}
+ onChange={(t) => {
+ scrollKindRef.current = "smooth";
+ setViewTrack(t === data.tracks[0] ? null : t);
+ }}
+ />
+ )}
+ </div>
{modalNotice && (
<p
@@ -793,3 +810,34 @@ function findActiveIndex(
}
return found;
}
+
+// "Track: original audio captions ▾" — small and quiet, beside the mode strip.
+// The first track is the primary (the default); its label says so in the menu.
+function TrackSwitcher({
+ tracks,
+ track,
+ onChange,
+}: {
+ tracks: string[];
+ track: string;
+ onChange: (track: string) => void;
+}) {
+ return (
+ <label className="inline-flex items-center gap-1.5 text-xs text-zinc-400 pointer-events-auto">
+ <span>Track:</span>
+ <select
+ data-testid="track-switcher"
+ value={track}
+ onChange={(e) => onChange(e.target.value)}
+ className="bg-transparent text-zinc-200 rounded border border-zinc-700 px-1 py-0.5 text-xs focus:outline-none focus:ring-1 focus:ring-zinc-500"
+ >
+ {trackLabels(tracks).map((label, i) => (
+ <option key={tracks[i]} value={tracks[i]} className="bg-zinc-900">
+ {label}
+ {i === 0 ? " (default)" : ""}
+ </option>
+ ))}
+ </select>
+ </label>
+ );
+}
diff --git a/common/components/searchIndex.worker.ts b/common/components/searchIndex.worker.ts
@@ -51,7 +51,7 @@ function mkIndex(): Index {
}
type RawCue = { start?: number; text?: unknown };
-type RawEntry = { slug?: unknown; cues?: unknown };
+type RawEntry = { slug?: unknown; cues?: unknown; altTracks?: unknown };
async function build(reqId: number, slug: string): Promise<void> {
// Manifest → page count.
@@ -75,13 +75,29 @@ async function build(reqId: number, slug: string): Promise<void> {
for (const entry of entries) {
const eslug = typeof entry.slug === "string" ? entry.slug : slug;
const cues = Array.isArray(entry.cues) ? entry.cues : [];
+ const said = new Set<string>();
for (const c of cues as RawCue[]) {
const text = typeof c.text === "string" ? c.text : "";
if (!text) continue;
+ said.add(text);
idx.add(id, text);
meta[id] = { slug: eslug, start: c.start, text };
id++;
}
+ // The record's alternate English tracks (lib/captionTracks.ts): a line
+ // the transcript itself does not say is indexed too, under its track.
+ const alts = Array.isArray(entry.altTracks) ? entry.altTracks : [];
+ for (const alt of alts as { track?: unknown; cues?: unknown }[]) {
+ const track = typeof alt.track === "string" ? alt.track : "";
+ if (!track || !Array.isArray(alt.cues)) continue;
+ for (const c of alt.cues as RawCue[]) {
+ const text = typeof c.text === "string" ? c.text : "";
+ if (!text || said.has(text)) continue;
+ idx.add(id, text);
+ meta[id] = { slug: eslug, start: c.start, text, track };
+ id++;
+ }
+ }
}
post({ type: "progress", reqId, slug, done: p + 1, total: pageCount });
}
diff --git a/common/components/searchIndexWorkerProtocol.ts b/common/components/searchIndexWorkerProtocol.ts
@@ -2,7 +2,9 @@
// the FlexSearch web worker (searchIndex.worker). Types only — safe to import
// from both sides.
-export type IndexHit = { slug: string; start?: number; text: string };
+// `track`: a cue from one of the record's alternate English tracks
+// (lib/captionTracks.ts) — absent for the transcript's own.
+export type IndexHit = { slug: string; start?: number; text: string; track?: string };
// Main thread → worker.
export type WorkerRequest =
diff --git a/common/components/urlState.ts b/common/components/urlState.ts
@@ -26,6 +26,10 @@ export type UrlParams = {
v: string | null;
t: number | null;
vm: ModalMode;
+ // The transcript TRACK the reader shows (lib/captionTracks.ts) when it is
+ // not the record's primary — a viewer's choice, never a data change. Absent
+ // (null) for the primary, which is the default and stays off the URL.
+ vt: string | null;
ch: string[];
nov: boolean;
nol: boolean;
@@ -94,6 +98,7 @@ function parse(search: string): UrlParams {
v: p.get("v"),
t: t !== null && Number.isFinite(t) ? t : null,
vm,
+ vt: p.get("vt") || null,
ch: p.getAll("ch"),
nov: p.get("nov") === "1",
nol: p.get("nol") === "1",
@@ -126,6 +131,7 @@ type Patch = Partial<{
v: string | null;
t: number | null;
vm: ModalMode;
+ vt: string | null;
ch: string[];
nov: boolean;
nol: boolean;
@@ -165,6 +171,10 @@ export function writeUrlParams(patch: Patch) {
// setting vm=transcript — the default must stay absent from the URL.
else params.delete("vm");
}
+ if (patch.vt !== undefined) {
+ if (patch.vt) params.set("vt", patch.vt);
+ else params.delete("vt");
+ }
if (patch.ch !== undefined) {
params.delete("ch");
for (const v of patch.ch) params.append("ch", v);
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -65,6 +65,8 @@ import type {
DisplaySummary,
} from "../lib/transcripts";
import type { StoredSubs, SubsDetail } from "../lib/subs";
+import type { TrackFields } from "../lib/captionTracks";
+import { readTrackFields } from "../lib/captionTracks-server";
import type {
Manifest,
ChannelEntry,
@@ -107,7 +109,11 @@ import {
import { resolveChannelGroupId } from "../lib/channelGroups";
import type { Paths } from "../lib/paths";
import {
+ CAPTION_TRACK_RULE_VERSION,
+ captionInputs,
+ englishVttsByPreference,
pickIndexTranscript,
+ readEnglishVttCues,
readSubTracks,
readVideoFiles,
type IndexTranscript,
@@ -205,6 +211,61 @@ const SCHEMA_VERSION = 13;
const PLATFORM_LABELS_VERSION = 1;
const PLATFORM_LABELS_KEY = "platformLabels";
+// CAPTION TRACK — the same one-shot shape, for the caption-track rule
+// (CAPTION_TRACK_RULE_VERSION, lib/videoStatus.ts). When the rule changes, the
+// first build that sees the new version re-reads, from disk, every caption
+// record the change can reach — one with more than one English VTT (the rule
+// may now pick another), or whose stored cues are empty (the cue-block shape
+// used to parse to nothing) — and nothing else; then records the version,
+// again only when no channel is held. Each re-read is reported per channel:
+// how many now read different text, how many had none and now do.
+const CAPTION_TRACK_KEY = "captionTrackRule";
+
+// A cue list's text, for "did the words change" — timing alone is not a
+// different transcript.
+function cueText(list: readonly Cue[]): string {
+ return list.map((c) => c.text).join("\n");
+}
+
+// Whether the cue list stored under a key is empty or absent, WITHOUT decoding
+// it: a non-empty list of cues is far longer than the few bytes an empty
+// msgpack array takes, and decoding every caption record of the corpus to ask
+// this would cost the full read the pass exists to avoid.
+function storedCuesEmpty(db: { getBinaryFast(key: IndexKey): Buffer | undefined }, key: IndexKey): boolean {
+ const raw = db.getBinaryFast(key);
+ return raw === undefined || raw.length <= 4;
+}
+
+// ALTERNATE TRACKS — the same one-shot shape again (lib/captionTracks.ts). A
+// record's other English tracks, kept where their words differ from the
+// primary's, live in the `alts` sub-DB and ride on its transcript page
+// (`track` + `altTracks`). The first build that sees a new ALT_TRACKS_VERSION
+// re-reads every record that CAN hold one — a caption record with two or more
+// English VTTs, a transcribed record with any — and nothing else; then records
+// the version, only when no channel is held. Bump it when what an alternate is
+// changes.
+const ALT_TRACKS_VERSION = 1;
+const ALT_TRACKS_KEY = "altTracks";
+
+// Whether a scanned record can hold an alternate track at all — no file read.
+function canHoldAltTracks(s: {
+ transcriptKind: IndexTranscript["kind"] | null;
+ englishVttCount: number;
+}): boolean {
+ return (
+ (s.transcriptKind === "vtt" && s.englishVttCount >= 2) ||
+ (s.transcriptKind === "whisper" && s.englishVttCount >= 1)
+ );
+}
+
+export type CaptionTrackChannelReport = {
+ reread: number;
+ // Re-read records whose caption text is not what the index held.
+ changed: number;
+ // Of those, the ones the index held no text for.
+ zeroToText: number;
+};
+
// Per-channel post stats, persisted so per-site aggregates survive a no-op
// rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat.
type ChannelPostsStat = {
@@ -299,6 +360,8 @@ type LiveEntry = {
transcriptPath: string;
transcriptMs: number | null;
transcriptKind: IndexTranscript["kind"] | null;
+ // How many English VTTs the dir holds (the caption-track pass's question).
+ englishVttCount: number;
subTracks: SubTrack[];
subsMs: number | null;
availabilityMs: number | null;
@@ -429,10 +492,21 @@ async function scanSource(
: path.join(fullVideoDir, "transcript.json");
let transcriptMs: number | null = null;
if (picked) {
- try {
- transcriptMs = (await stat(transcriptPath)).mtimeMs;
- } catch {
- transcriptMs = null;
+ // Captions: the newest of every caption input (each English VTT and
+ // the operator's pin), since the cues may come from any of them. A
+ // transcribed record's captions are its alternate tracks
+ // (lib/captionTracks.ts), so they count for it too.
+ const inputs =
+ picked.kind === "vtt"
+ ? captionInputs(files.entries)
+ : [picked.filename, ...captionInputs(files.entries)];
+ for (const name of inputs) {
+ try {
+ const ms = (await stat(path.join(fullVideoDir, name))).mtimeMs;
+ if (transcriptMs === null || ms > transcriptMs) transcriptMs = ms;
+ } catch {
+ // ignore
+ }
}
}
const subTracks = await readSubTracks(fullVideoDir);
@@ -482,6 +556,7 @@ async function scanSource(
transcriptPath,
transcriptMs,
transcriptKind: picked?.kind ?? null,
+ englishVttCount: englishVttsByPreference(files.entries).length,
subTracks,
subsMs,
availabilityMs,
@@ -542,6 +617,9 @@ export type BuildIndexResult = {
// Channels whose media could not be read, so they were not rescanned: their
// index records and shared pages were kept as they were (see the header).
heldChannels: string[];
+ // Per channel, what a caption-track pass re-read and changed; empty when no
+ // pass ran (CAPTION_TRACK_KEY).
+ captionTrack: Record<string, CaptionTrackChannelReport>;
};
export type BuildIndexOptions = {
@@ -601,6 +679,12 @@ export async function buildIndex({
name: "subs",
encoding: "msgpack",
});
+ // A record's alternate English tracks (ALT_TRACKS_KEY), only where one
+ // differs from the primary — sparse, like `subs`.
+ const alts = root.openDB<TrackFields, IndexKey>({
+ name: "alts",
+ encoding: "msgpack",
+ });
const mtimes = root.openDB<MtimeRecord, PathKey>({
name: "mtimes",
encoding: "msgpack",
@@ -711,6 +795,7 @@ export async function buildIndex({
await sums.clearAsync();
await cues.clearAsync();
await subs.clearAsync();
+ await alts.clearAsync();
await mtimes.clearAsync();
await byChannel.clearAsync();
await pageHashes.clearAsync();
@@ -803,6 +888,50 @@ export async function buildIndex({
);
}
+ // Caption records the caption-track rule may now read differently
+ // (CAPTION_TRACK_KEY). A schema bump re-reads everything anyway.
+ const captionTrackDue =
+ meta.get(CAPTION_TRACK_KEY) !== CAPTION_TRACK_RULE_VERSION;
+ // pathKeyIds of the records this pass re-reads.
+ const captionReread = new Set<string>();
+ if (captionTrackDue && !schemaBumped) {
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ for (const s of live) {
+ if (s.transcriptKind !== "vtt") continue;
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ const prev = mtimes.get(pk);
+ if (!prev) continue;
+ if (s.englishVttCount < 2 && !storedCuesEmpty(cues, prev.indexKey)) continue;
+ captionReread.add(pathKeyId(pk));
+ if (!queued.has(pathKeyId(pk))) changed.push(s);
+ }
+ log(
+ `Caption track v${CAPTION_TRACK_RULE_VERSION}: ${captionReread.size} record(s) re-read.`,
+ );
+ }
+ const captionReport = new Map<string, CaptionTrackChannelReport>();
+
+ // Records that can hold an alternate track (ALT_TRACKS_KEY). A schema bump
+ // re-reads everything anyway.
+ const altTracksDue = meta.get(ALT_TRACKS_KEY) !== ALT_TRACKS_VERSION;
+ if (altTracksDue && !schemaBumped) {
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ let reread = 0;
+ for (const s of live) {
+ if (!canHoldAltTracks(s)) continue;
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ if (!mtimes.get(pk)) continue;
+ reread++;
+ if (!queued.has(pathKeyId(pk))) changed.push(s);
+ }
+ log(`Alternate tracks v${ALT_TRACKS_VERSION}: ${reread} record(s) re-read.`);
+ }
+ let altTrackRecords = 0;
+
const anyMutations =
added.length > 0 || changed.length > 0 || removed.length > 0;
@@ -901,11 +1030,11 @@ export async function buildIndex({
);
if (cueList === undefined && s.transcriptMs !== null && s.transcriptKind) {
try {
- const raw = await readFile(s.transcriptPath, "utf8");
cueList =
s.transcriptKind === "vtt"
- ? parseVtt(raw)
- : parseTranscriptJson(raw);
+ ? // The caption-track rule (lib/videoStatus.ts).
+ (await readEnglishVttCues(videoFullDir))?.cues
+ : parseTranscriptJson(await readFile(s.transcriptPath, "utf8"));
} catch {
cueList = undefined;
}
@@ -923,6 +1052,12 @@ export async function buildIndex({
const pk: PathKey = [s.channelSlug, s.videoDir];
const prev = mtimes.get(pk);
+ // What the last build held for this video's captions, for the
+ // caption-track report — read before the put below replaces it.
+ const heldBefore =
+ prev && captionReread.has(pathKeyId(pk))
+ ? cueText(cues.get(prev.indexKey) ?? [])
+ : undefined;
// What the last build held for this video's sub tracks, read before
// a re-keyed record is removed: a tiered live chat whose raw cannot
// be read now keeps the cues it had (below).
@@ -931,6 +1066,7 @@ export async function buildIndex({
sums.remove(prev.indexKey);
cues.remove(prev.indexKey);
subs.remove(prev.indexKey);
+ alts.remove(prev.indexKey);
digests.remove(prev.indexKey);
byChannel.remove(indexToChannelKey(prev.indexKey));
}
@@ -938,6 +1074,29 @@ export async function buildIndex({
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
+ // The other English tracks, where their words differ from the
+ // primary's (lib/captionTracks.ts). Only a record that can hold one
+ // reads anything.
+ const trackFields: TrackFields =
+ canHoldAltTracks(s) && s.transcriptKind
+ ? await readTrackFields(videoFullDir, s.transcriptKind, cueList)
+ : {};
+ if (trackFields.altTracks) {
+ alts.put(indexKey, trackFields);
+ altTrackRecords++;
+ } else alts.remove(indexKey);
+
+ if (heldBefore !== undefined) {
+ const r = captionReport.get(s.channelSlug) ?? { reread: 0, changed: 0, zeroToText: 0 };
+ r.reread++;
+ const now = cueText(cueList ?? []);
+ if (heldBefore !== now) {
+ r.changed++;
+ if (heldBefore === "" && now !== "") r.zeroToText++;
+ }
+ captionReport.set(s.channelSlug, r);
+ }
+
const parsedSubs: StoredSubs = [];
// A tiered track that could be neither read nor kept: `subsMs` is
// stored null so the next build's scan sees a change and retries.
@@ -1112,10 +1271,22 @@ export async function buildIndex({
}
}
+ // What the caption-track pass changed, per channel (CAPTION_TRACK_KEY).
+ for (const [slug, r] of [...captionReport].sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) {
+ log(
+ `Caption track v${CAPTION_TRACK_RULE_VERSION}: ${slug}: ${r.reread} re-read, ${r.changed} now read different text, ${r.zeroToText} had no text and now do.`,
+ );
+ }
+
+ if (altTrackRecords > 0) {
+ log(`Alternate tracks: ${altTrackRecords} record(s) read this build hold a track whose words differ from the primary's.`);
+ }
+
for (const { pathKey, indexKey } of removed) {
sums.remove(indexKey);
cues.remove(indexKey);
subs.remove(indexKey);
+ alts.remove(indexKey);
digests.remove(indexKey);
byChannel.remove(indexToChannelKey(indexKey));
mtimes.remove(pathKey);
@@ -1155,6 +1326,7 @@ export async function buildIndex({
await sums.flushed;
await cues.flushed;
await subs.flushed;
+ await alts.flushed;
await digests.flushed;
await byChannel.flushed;
await mtimes.flushed;
@@ -1337,7 +1509,16 @@ export async function buildIndex({
const summary = sums.get(indexKey);
if (!summary) continue;
const cueList = cues.get(indexKey);
- const detail: TranscriptDetail = { ...summary, cues: cueList };
+ // `track` + `altTracks` only on a record that has an alternate, so every
+ // other record's page bytes are what they were.
+ const trackFields = alts.get(indexKey);
+ const detail: TranscriptDetail = {
+ ...summary,
+ cues: cueList,
+ ...(trackFields?.altTracks?.length
+ ? { track: trackFields.track, altTracks: trackFields.altTracks }
+ : {}),
+ };
const encoded = JSON.stringify(detail);
await writer.push(encoded, summary.id);
}
@@ -2332,6 +2513,12 @@ export async function buildIndex({
if (relabelDue && held.size === 0) {
await meta.put(PLATFORM_LABELS_KEY, PLATFORM_LABELS_VERSION);
}
+ if (captionTrackDue && held.size === 0) {
+ await meta.put(CAPTION_TRACK_KEY, CAPTION_TRACK_RULE_VERSION);
+ }
+ if (altTracksDue && held.size === 0) {
+ await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION);
+ }
await meta.flushed;
await root.close();
@@ -2356,5 +2543,6 @@ export async function buildIndex({
removed: removed.length,
shortCircuited: !sharedNeedsBuild && sitesBuilt === 0,
heldChannels: [...held.keys()],
+ captionTrack: Object.fromEntries(captionReport),
};
}
diff --git a/common/controller/buildIndexAltTracks.test.ts b/common/controller/buildIndexAltTracks.test.ts
@@ -0,0 +1,211 @@
+// Integration: ALTERNATE TRACKS (lib/captionTracks.ts) through the REAL
+// buildIndex over a temp corpus. A record whose served `en` says something its
+// en-orig does not carries that track on its transcript page — so a word only
+// in `en` is findable, with the track named — and a record whose tracks are
+// identical carries nothing extra; no English VTT is shipped again as a
+// subtitle track; and the one-shot pass (ALT_TRACKS_VERSION) re-reads exactly
+// the records that can hold an alternate.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexAltTracks.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-alts-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+const { hitsAcrossTracks } = await import("../lib/captionTracks");
+
+const paths = getPaths();
+const CHANNEL = "example-channel";
+const SITE = "testsite";
+const DIFFER = "DifferTrack1"; // en-orig + an en that says other words
+const SAME = "SameTracks01"; // en-orig + an identical en
+const WHISPER = "Transcribed1"; // transcript.json + captions that differ
+const SPANISH = "SpanishSub01"; // en-orig + a Spanish subtitle track
+
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8");
+const ROLLING = fixture("vtt-rolling.vtt");
+const vtt = (...lines: [string, string, string][]) =>
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ lines.map(([a, b, t]) => `${a} --> ${b}\n${t}\n`).join("\n");
+// What the served `en` says: the same opening, then a word en-orig never has,
+// far from anything en-orig matches.
+const SERVED_EN = vtt(
+ ["00:00:03.080", "00:00:05.670", "are talking about the harbor"],
+ ["00:01:40.000", "00:01:43.000", "the zeppelin landed in nineteen thirty"],
+);
+const WHISPER_JSON = JSON.stringify({
+ transcription: [
+ { offsets: { from: 0, to: 2000 }, text: " a local transcription says hello" },
+ { offsets: { from: 2000, to: 4000 }, text: " and nothing else" },
+ ],
+});
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id);
+
+function seed(): void {
+ rmSync(paths.transcriptsDir, { recursive: true, force: true });
+ rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true });
+ writeFileSync(paths.settingsFile, "{}");
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: CHANNEL,
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+ for (const [i, id] of [DIFFER, SAME, WHISPER, SPANISH].entries()) {
+ writeJson(path.join(dirOf(id), "metadata.info.json"), {
+ id,
+ title: `Video ${id}`,
+ upload_date: `2026060${i + 1}`,
+ duration: 200,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ }
+ writeFileSync(path.join(dirOf(DIFFER), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(DIFFER), "transcript.en.vtt"), SERVED_EN);
+ writeFileSync(path.join(dirOf(SAME), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(SAME), "transcript.en.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(WHISPER), "transcript.json"), WHISPER_JSON);
+ writeFileSync(path.join(dirOf(WHISPER), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(SPANISH), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(
+ path.join(dirOf(SPANISH), "transcript.es.vtt"),
+ vtt(["00:00:01.000", "00:00:02.000", "hola a todos"]),
+ );
+}
+
+type Cue = { start: number; end: number; text: string };
+type Rec = {
+ id: string;
+ cues?: Cue[];
+ track?: string;
+ altTracks?: { track: string; cues: Cue[] }[];
+};
+
+// Every record of the channel's shared transcript pages, by id.
+function pageRecords(): Map<string, Rec> {
+ const dir = path.join(paths.exportSharedTranscriptsDir, CHANNEL);
+ const out = new Map<string, Rec>();
+ for (const name of readdirSync(dir)) {
+ if (!/^page-\d+\.json$/.test(name)) continue;
+ for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as Rec[]) {
+ out.set(r.id, r);
+ }
+ }
+ return out;
+}
+function subsTracks(): Record<string, string[]> {
+ const dir = path.join(paths.exportSharedSubsDir, CHANNEL);
+ const out: Record<string, string[]> = {};
+ if (!existsSync(dir)) return out;
+ for (const name of readdirSync(dir)) {
+ if (!/^page-\d+\.json$/.test(name)) continue;
+ for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as {
+ id: string;
+ tracks: Record<string, unknown>;
+ }[]) {
+ out[r.id] = Object.keys(r.tracks).sort();
+ }
+ }
+ return out;
+}
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a differing served en rides on the page as an alternate; identical tracks add nothing", async () => {
+ seed();
+ await runIndex();
+ const recs = pageRecords();
+
+ const differ = recs.get(DIFFER)!;
+ assert.equal(differ.track, "en-orig");
+ assert.deepEqual(differ.altTracks?.map((t) => t.track), ["en"]);
+ // The word only `en` has is found — in `en`, named — and the opening both
+ // say is found once, in the primary.
+ const find = (q: string) =>
+ hitsAcrossTracks(differ, (cues) => cues.filter((c) => c.text.includes(q)));
+ assert.deepEqual(
+ find("zeppelin").map((h) => [h.track, Math.round(h.start)]),
+ [["en", 100]],
+ );
+ assert.deepEqual(find("harbor").map((h) => h.track), [undefined]);
+
+ // Identical tracks: no fields at all, so the record is what it always was.
+ const same = recs.get(SAME)!;
+ assert.equal("track" in same, false);
+ assert.equal("altTracks" in same, false);
+
+ // A transcription is the primary; the captions it replaced are an alternate.
+ const whisper = recs.get(WHISPER)!;
+ assert.equal(whisper.track, "transcription");
+ assert.equal(whisper.cues?.[0].text, "a local transcription says hello");
+ assert.deepEqual(whisper.altTracks?.map((t) => t.track), ["en-orig"]);
+
+ // No English VTT is a subtitle track any more; a Spanish one still is.
+ assert.deepEqual(subsTracks(), { [SPANISH]: ["es"] });
+});
+
+test("the one-shot pass re-reads exactly the records that can hold an alternate, once", async () => {
+ // The index as a build before alternate tracks left it: no `alts` records,
+ // no version.
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ await root.openDB({ name: "alts", encoding: "msgpack" }).clearAsync();
+ await root.openDB({ name: "meta", encoding: "msgpack" }).remove("altTracks");
+ } finally {
+ await root.close();
+ }
+ const log = await runIndex();
+ // DIFFER, SAME (two English VTTs) and WHISPER (a transcription beside
+ // captions); not SPANISH.
+ assert.ok(log.includes("Alternate tracks v1: 3 record(s) re-read."), log.join("\n"));
+ assert.deepEqual(pageRecords().get(DIFFER)?.altTracks?.map((t) => t.track), ["en"]);
+
+ const again = await runIndex();
+ assert.equal(again.some((l) => /Alternate tracks v1/.test(l)), false, again.join("\n"));
+});
diff --git a/common/controller/buildIndexCaptionTrack.test.ts b/common/controller/buildIndexCaptionTrack.test.ts
@@ -0,0 +1,156 @@
+// Integration: the caption-track pass (CAPTION_TRACK_RULE_VERSION) re-reads,
+// through the REAL buildIndex over a temp corpus, exactly the records an older
+// caption-track rule may have read differently — one with two English VTTs,
+// one whose stored cues are empty — and leaves every other record alone.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexCaptionTrack.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-captions-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const CHANNEL = "example-channel";
+const SITE = "testsite";
+const BOTH = "BothTracks01"; // served en (cue blocks) + en-orig (rolling)
+const BLOCKS = "CueBlocks001"; // a lone cue-block en
+const ROLL = "RollingOnly1"; // a lone rolling en — nothing for the pass to do
+
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8");
+const ROLLING = fixture("vtt-rolling.vtt");
+const CUE_BLOCKS = fixture("vtt-cue-blocks.vtt");
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id);
+
+function seed(): void {
+ rmSync(paths.transcriptsDir, { recursive: true, force: true });
+ rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true });
+ writeFileSync(paths.settingsFile, "{}");
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: CHANNEL,
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+ for (const [i, id] of [BOTH, BLOCKS, ROLL].entries()) {
+ writeJson(path.join(dirOf(id), "metadata.info.json"), {
+ id,
+ title: `Video ${id}`,
+ upload_date: `2026060${i + 1}`,
+ duration: 30,
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ }
+ writeFileSync(path.join(dirOf(BOTH), "transcript.en.vtt"), CUE_BLOCKS);
+ writeFileSync(path.join(dirOf(BOTH), "transcript.en-orig.vtt"), ROLLING);
+ writeFileSync(path.join(dirOf(BLOCKS), "transcript.en.vtt"), CUE_BLOCKS);
+ writeFileSync(path.join(dirOf(ROLL), "transcript.en.vtt"), ROLLING);
+}
+
+type Summary = { id: string };
+type Cue = { start: number; end: number; text: string };
+
+function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T {
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ return fn((name) => root.openDB({ name, encoding: "msgpack" }));
+ } finally {
+ root.close();
+ }
+}
+const keyOf = (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>, id: string) => {
+ for (const { key, value } of db("sums").getRange()) {
+ if ((value as Summary).id === id) return key;
+ }
+ throw new Error(`${id} is not indexed`);
+};
+const cuesOf = (id: string) => withIndex((db) => db("cues").get(keyOf(db, id)) as Cue[] | undefined);
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a fresh index reads en-orig beside a served en, and a lone cue-block en as text", async () => {
+ seed();
+ const log = await runIndex();
+ assert.equal(cuesOf(BOTH)?.[0].text, "are talking about the harbor");
+ assert.equal(cuesOf(BLOCKS)?.length, 7);
+ assert.equal(cuesOf(ROLL)?.length, 3);
+ // A first build has nothing indexed under an older rule: no pass.
+ assert.equal(log.some((l) => /Caption track/.test(l)), false, log.join("\n"));
+});
+
+test("an index built under the old rule is re-read once, for exactly the records the rule reaches", async () => {
+ // The index as an older build left it: the served en's text for BOTH, no
+ // cues for BLOCKS, and no caption-track version recorded.
+ const heldRoll = cuesOf(ROLL);
+ withIndex((db) => {
+ const cues = db("cues");
+ cues.putSync(keyOf(db, BOTH), [{ start: 0, end: 6, text: "served words" }]);
+ cues.putSync(keyOf(db, BLOCKS), []);
+ // ROLL's record is marked so a re-read would show.
+ cues.putSync(keyOf(db, ROLL), [{ start: 0, end: 1, text: "untouched" }]);
+ db("meta").removeSync("captionTrackRule");
+ });
+
+ const log = await runIndex();
+ assert.ok(log.includes("Caption track v1: 2 record(s) re-read."), log.join("\n"));
+ assert.ok(
+ log.includes(
+ `Caption track v1: ${CHANNEL}: 2 re-read, 2 now read different text, 1 had no text and now do.`,
+ ),
+ log.join("\n"),
+ );
+ assert.equal(cuesOf(BOTH)?.[0].text, "are talking about the harbor");
+ assert.equal(cuesOf(BLOCKS)?.length, 7);
+ assert.deepEqual(cuesOf(ROLL), [{ start: 0, end: 1, text: "untouched" }]);
+ assert.notDeepEqual(heldRoll, cuesOf(ROLL));
+
+ // Recorded: the next build does not look again.
+ const again = await runIndex();
+ assert.equal(again.some((l) => /Caption track/.test(l)), false, again.join("\n"));
+});
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -10,6 +10,7 @@ import {
isVideoTranscribed,
readVideoFiles,
LIVE_CHAT_FILENAME,
+ ORIG_VTT_FILENAME,
VTT_FILENAME,
type VideoFiles,
} from "../lib/videoStatus";
@@ -74,7 +75,7 @@ import { loadMaybeMissing } from "./quickAvailabilityCheck";
import { loadRoster } from "./rosterStore";
import { deriveChannelSets } from "./channelSets";
import { readTranscriptCoverage } from "./normalizeTranscript";
-import { resolveVttProvenance } from "../lib/subtitleProvenance";
+import { resolveCaptionsProvenance } from "../lib/subtitleProvenance";
import { isIncompleteTranscript } from "../lib/transcriptCoverage";
// THE PER-OPERATION WORK LISTS, keyed by operation id.
@@ -1007,8 +1008,10 @@ export async function generateChannelSnapshot(
// both the auto-subs work lane (no whisper yet) and the superseded
// backup inventory (whisper already won) need it. Same conditional
// per-video sidecar read pattern as the cues.json coverage read above.
+ // Over every English VTT (resolveCaptionsProvenance): the rule reads
+ // en-orig even beside a human `en`, which is not ASR-only.
const vttProvenance = files.ytVttFile
- ? await resolveVttProvenance(dir, files.ytVttFile)
+ ? await resolveCaptionsProvenance(dir, files.entries)
: null;
// Only transcribed videos can carry a digest, so everything else skips
// the sidecar read entirely — the same conditional per-video
@@ -1414,11 +1417,13 @@ export async function generateChannelSnapshot(
// A transcript that exists only under a non-standard VTT name — either a
// regional/auto English track (transcript.en-US.vtt) now picked up by the
// fallback, or a foreign-only transcript.<lang>.vtt that isn't recognized as
- // English. Whisper transcripts and the canonical transcript.en.vtt are fine.
+ // English. Whisper transcripts, the canonical transcript.en.vtt and the
+ // original-audio transcript.en-orig.vtt (the rule's first pick) are fine.
if (
!files.hasWhisper &&
files.hasNonCanonicalVtt &&
- files.ytVttFile !== VTT_FILENAME
+ files.ytVttFile !== VTT_FILENAME &&
+ files.ytVttFile !== ORIG_VTT_FILENAME
) {
nonStandardVtt.push(id);
}
diff --git a/common/controller/normalizeCaptionTrack.test.ts b/common/controller/normalizeCaptionTrack.test.ts
@@ -0,0 +1,160 @@
+// transcript.cues.json under the caption-track rule: normalize records the
+// track it read and the rule that chose it, and a cues.json made under an older
+// rule is stale exactly where the rule could now read something else.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/normalizeCaptionTrack.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, utimesSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { isCuesJsonFresh, normalizeTranscript, readNormalizedTranscript } from "./normalizeTranscript";
+import { CAPTION_TRACK_RULE_VERSION, TRANSCRIPT_PIN_FILENAME } from "../lib/videoStatus";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "normalize-caption-"));
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8");
+const ROLLING = fixture("vtt-rolling.vtt");
+const CUE_BLOCKS = fixture("vtt-cue-blocks.vtt");
+
+const META = {
+ id: "vid",
+ title: "A video",
+ upload_date: "20260601",
+ duration: 30,
+ webpage_url: "https://www.youtube.com/watch?v=vid",
+ extractor_key: "Youtube",
+};
+
+let n = 0;
+function videoDir(files: Record<string, string>): string {
+ const dir = path.join(ROOT, `c${n}`, "data", `vid${n++}`);
+ mkdirSync(dir, { recursive: true });
+ writeFileSync(path.join(dir, "metadata.info.json"), JSON.stringify(META));
+ for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body);
+ return dir;
+}
+
+// A cues.json as the app wrote it before the rule had a version: no track
+// recorded, newer than every input.
+function oldCuesJson(dir: string, cues: { start: number; end: number; text: string }[]): void {
+ const p = path.join(dir, "transcript.cues.json");
+ writeFileSync(
+ p,
+ JSON.stringify({ version: 2, source: "vtt", transcriptFormat: "vtt", id: "vid", slug: "c/vid", cues }),
+ );
+ const later = new Date(Date.now() + 10_000);
+ utimesSync(p, later, later);
+}
+
+test("normalize reads en-orig, and records the track and the rule in the file's first bytes", async () => {
+ const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-orig.vtt": ROLLING });
+ const out = await normalizeTranscript({ videoDir: dir, channelSlug: "c" });
+ assert.equal(out.status, "wrote");
+ const raw = readFileSync(path.join(dir, "transcript.cues.json"), "utf8");
+ assert.match(raw.slice(0, 200), /"vttFile":"transcript\.en-orig\.vtt","captionTrackRule":\d+/);
+ const n = await readNormalizedTranscript(path.join(dir, "transcript.cues.json"));
+ assert.equal(n?.vttFile, "transcript.en-orig.vtt");
+ assert.equal(n?.captionTrackRule, CAPTION_TRACK_RULE_VERSION);
+ assert.equal(n?.cues?.[0].text, "are talking about the harbor");
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+ assert.equal((await normalizeTranscript({ videoDir: dir, channelSlug: "c" })).status, "fresh");
+});
+
+test("a cues.json from before the rule is fresh beside a byte-identical en/en-orig pair: it reads the same", async () => {
+ const dir = videoDir({ "transcript.en.vtt": ROLLING, "transcript.en-orig.vtt": ROLLING });
+ oldCuesJson(dir, [{ start: 0, end: 1, text: "same words" }]);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+ assert.equal((await normalizeTranscript({ videoDir: dir, channelSlug: "c" })).status, "fresh");
+});
+
+test("a cues.json from before the rule is fresh beside two tracks the older rule ranked the same way", async () => {
+ const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-US.vtt": ROLLING });
+ oldCuesJson(dir, [{ start: 0, end: 1, text: "kept" }]);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+});
+
+test("a cues.json from before the rule with no cues is stale beside two tracks: the fallback may find text", async () => {
+ const dir = videoDir({ "transcript.en.vtt": ROLLING, "transcript.en-orig.vtt": ROLLING });
+ oldCuesJson(dir, []);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, false);
+});
+
+test("a cues.json from before the rule is stale beside an en and en-orig that differ, and normalize rewrites it", async () => {
+ const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-orig.vtt": ROLLING });
+ oldCuesJson(dir, [{ start: 0, end: 1, text: "served words" }]);
+ assert.deepEqual(await isCuesJsonFresh(dir), {
+ fresh: false,
+ reason: "stale",
+ cuesPath: path.join(dir, "transcript.cues.json"),
+ });
+ assert.equal((await normalizeTranscript({ videoDir: dir, channelSlug: "c" })).status, "wrote");
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+});
+
+test("a cues.json from before the rule is stale over a lone cue-block track (it parsed to nothing)", async () => {
+ const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS });
+ oldCuesJson(dir, []);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, false);
+ await normalizeTranscript({ videoDir: dir, channelSlug: "c" });
+ const n = await readNormalizedTranscript(path.join(dir, "transcript.cues.json"));
+ assert.equal(n?.cues?.length, 7);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+});
+
+test("a cues.json from before the rule that holds cues stays fresh over a lone cue-block track: the old parse made none", async () => {
+ const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS });
+ oldCuesJson(dir, [{ start: 0, end: 1, text: "written by hand" }]);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+});
+
+test("a cues.json from before the rule stays fresh over a lone rolling track: nothing to choose, same parse", async () => {
+ const dir = videoDir({ "transcript.en.vtt": ROLLING });
+ oldCuesJson(dir, [{ start: 0, end: 1, text: "kept" }]);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+ assert.equal((await normalizeTranscript({ videoDir: dir, channelSlug: "c" })).status, "fresh");
+});
+
+test("a newer secondary track, or a new pin, makes cues.json stale", async () => {
+ const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-orig.vtt": ROLLING });
+ const at = (name: string, s: number) => {
+ const t = new Date(Date.now() + s * 1000);
+ utimesSync(path.join(dir, name), t, t);
+ };
+ await normalizeTranscript({ videoDir: dir, channelSlug: "c" });
+ at("transcript.cues.json", 10);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+ // Not the track the rule reads — still an input (the fallback may read it).
+ at("transcript.en.vtt", 20);
+ assert.equal((await isCuesJsonFresh(dir)).reason, "stale");
+ await normalizeTranscript({ videoDir: dir, channelSlug: "c" });
+ at("transcript.cues.json", 30);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+
+ // The operator pins transcript.en.vtt: stale, and the rewrite reads en.
+ writeFileSync(
+ path.join(dir, TRANSCRIPT_PIN_FILENAME),
+ JSON.stringify({ from: "transcript.en.vtt", pinnedAt: "" }),
+ );
+ at(TRANSCRIPT_PIN_FILENAME, 40);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, false);
+ await normalizeTranscript({ videoDir: dir, channelSlug: "c" });
+ const n = await readNormalizedTranscript(path.join(dir, "transcript.cues.json"));
+ assert.equal(n?.vttFile, "transcript.en.vtt");
+});
+
+test("whisper wins: a whisper cues.json is never asked about caption tracks", async () => {
+ const dir = videoDir({
+ "transcript.en.vtt": CUE_BLOCKS,
+ "transcript.en-orig.vtt": ROLLING,
+ "transcript.json": JSON.stringify({ transcription: [{ offsets: { from: 0, to: 1000 }, text: "spoken" }] }),
+ });
+ const p = path.join(dir, "transcript.cues.json");
+ writeFileSync(p, JSON.stringify({ version: 2, source: "whisper", cues: [{ start: 0, end: 1, text: "spoken" }] }));
+ const later = new Date(Date.now() + 10_000);
+ utimesSync(p, later, later);
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+});
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -4,9 +4,9 @@
// shape that buildIndex would emit, plus a `source` marker.
import path from "node:path";
-import { readdir, readFile, stat } from "node:fs/promises";
+import { open, readdir, readFile, stat } from "node:fs/promises";
import { writeJsonAtomic } from "../lib/jsonFile-server";
-import { parseVtt, type Cue } from "../lib/vtt";
+import { hasWordTiming, type Cue } from "../lib/vtt";
import {
detectTranscriptFormat,
parseTranscriptJson,
@@ -19,13 +19,17 @@ import {
type TranscriptCoverage,
} from "../lib/transcriptCoverage";
import {
+ CAPTION_TRACK_RULE_VERSION,
CUES_JSON_FILENAME,
META_FILENAME,
+ ORIG_VTT_FILENAME,
VTT_FILENAME,
WHISPER_FILENAME,
+ captionInputs,
+ englishVttsByPreference,
pickIndexTranscript,
+ readEnglishVttCues,
readVideoFiles,
- resolvePrimaryVtt,
type IndexTranscript,
} from "../lib/videoStatus";
@@ -36,6 +40,12 @@ export const CUES_FILE_VERSION = 2;
export type NormalizedTranscript = TranscriptDetail & {
version: number;
source: IndexTranscript["kind"] | "live_chat";
+ // source "vtt" only: the English VTT the cues were read from, and the
+ // caption-track rule that chose it (CAPTION_TRACK_RULE_VERSION). Written
+ // right after `source`, so they sit in the file's first bytes and
+ // isCuesJsonFresh can read them without parsing the cues.
+ vttFile?: string;
+ captionTrackRule?: number;
// The raw transcript format this was parsed from. Durable per-video record so
// a later re-normalize knows how to read the raw file without re-sniffing.
transcriptFormat?: TranscriptOutputFormat;
@@ -76,24 +86,14 @@ export async function normalizeTranscript(
const picked = pickIndexTranscript(files);
if (!picked) return { status: "skipped", reason: "no-raw-transcript" };
- const cuesPath = path.join(opts.videoDir, CUES_JSON_FILENAME);
const metaPath = path.join(opts.videoDir, META_FILENAME);
const transcriptPath = path.join(opts.videoDir, picked.filename);
- const [cuesMs, metaStatMs, transcriptStatMs] = await Promise.all([
- mtimeMs(cuesPath),
- mtimeMs(metaPath),
- mtimeMs(transcriptPath),
- ]);
-
- if (
- !opts.force &&
- cuesMs !== null &&
- metaStatMs !== null &&
- transcriptStatMs !== null &&
- cuesMs >= metaStatMs &&
- cuesMs >= transcriptStatMs
- ) {
+ // The same question every reader of cues.json asks (isCuesJsonFresh), so a
+ // normalize pass rewrites exactly the files they refuse.
+ const freshness = await isCuesJsonFresh(opts.videoDir);
+ const cuesPath = freshness.cuesPath;
+ if (!opts.force && freshness.fresh) {
return { status: "fresh", cuesPath };
}
@@ -106,14 +106,20 @@ export async function normalizeTranscript(
opts.configName,
);
- const rawTranscript = await readFile(transcriptPath, "utf8");
let cues: Cue[];
let transcriptFormat: TranscriptOutputFormat;
+ let vttFile: string | undefined;
try {
if (picked.kind === "vtt") {
- cues = parseVtt(rawTranscript);
+ // The caption-track rule: the first English VTT, in preference order,
+ // with cues (lib/videoStatus.ts).
+ const read = await readEnglishVttCues(opts.videoDir, files.entries);
+ if (!read) throw new Error("no English VTT could be read");
+ cues = read.cues;
+ vttFile = read.filename;
transcriptFormat = "vtt";
} else {
+ const rawTranscript = await readFile(transcriptPath, "utf8");
// Resolve the JSON format: authoritative hint -> recorded per-video tag
// -> content sniff -> whisper fallback (the only app before chough).
transcriptFormat =
@@ -132,6 +138,9 @@ export async function normalizeTranscript(
const out: NormalizedTranscript = {
version: CUES_FILE_VERSION,
source: picked.kind,
+ ...(vttFile !== undefined
+ ? { vttFile, captionTrackRule: CAPTION_TRACK_RULE_VERSION }
+ : {}),
transcriptFormat,
...summary,
cues,
@@ -140,7 +149,7 @@ export async function normalizeTranscript(
// Compact, no trailing newline: transcript.cues.json's historical bytes.
await writeJsonAtomic(cuesPath, out, { indent: 0, newline: false });
opts.log?.(
- `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${transcriptFormat}, ${cues.length} cues)`,
+ `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${vttFile ?? transcriptFormat}, ${cues.length} cues)`,
);
return { status: "wrote", cuesPath };
}
@@ -214,22 +223,33 @@ export type CuesFreshReason =
// Helper: given a video dir, decide whether transcript.cues.json (if present)
// is at least as new as metadata.info.json and the raw transcript file. Used
// by buildIndex to know whether it can trust cues.json without re-parsing.
+//
+// CAPTIONS: the raw transcript is EVERY caption input (captionInputs — each
+// English VTT, since the content fallback may read any of them, and the
+// operator's pin), and a cues.json that is new enough must also have been made
+// under the current caption-track rule. Made under an older one, it is `stale`
+// where the rule could read it differently: its cues may come from a track the
+// rule no longer picks (a served `en` that differs from the `en-orig` beside
+// it), or be the zero cues a cue-block VTT used to parse to
+// (followsCaptionTrackRule). Telling costs a few small reads, never a parse.
export async function isCuesJsonFresh(
videoDir: string,
): Promise<{ fresh: boolean; reason: CuesFreshReason; cuesPath: string }> {
const cuesPath = path.join(videoDir, CUES_JSON_FILENAME);
const metaPath = path.join(videoDir, META_FILENAME);
- // Resolve the actual primary VTT (may be a regional/auto English track like
- // transcript.en-US.vtt) rather than assuming the literal transcript.en.vtt.
const entries = await readdir(videoDir).catch(() => [] as string[]);
- const vttPath = path.join(videoDir, resolvePrimaryVtt(entries) ?? VTT_FILENAME);
const whisperPath = path.join(videoDir, WHISPER_FILENAME);
- const [cuesMs, metaMs, vttMs, whisperMs] = await Promise.all([
+ const inputs = captionInputs(entries);
+ const [cuesMs, metaMs, whisperMs, ...inputMs] = await Promise.all([
mtimeMs(cuesPath),
mtimeMs(metaPath),
- mtimeMs(vttPath),
mtimeMs(whisperPath),
+ ...inputs.map((n) => mtimeMs(path.join(videoDir, n))),
]);
+ const vttMs = inputMs.reduce<number | null>(
+ (max, ms) => (ms !== null && (max === null || ms > max) ? ms : max),
+ null,
+ );
// Prefer whisper if present (matches pickIndexTranscript priority).
const rawMs = whisperMs ?? vttMs;
// Ordered to match normalizeTranscript's OWN precedence (metadata, then raw,
@@ -243,5 +263,76 @@ export async function isCuesJsonFresh(
if (cuesMs < metaMs || cuesMs < rawMs) {
return { fresh: false, reason: "stale", cuesPath };
}
+ if (whisperMs === null && !(await followsCaptionTrackRule(videoDir, cuesPath, entries))) {
+ return { fresh: false, reason: "stale", cuesPath };
+ }
return { fresh: true, reason: "fresh", cuesPath };
}
+
+const HEAD_BYTES = 2048;
+
+async function readHead(file: string, bytes = HEAD_BYTES): Promise<string> {
+ const fh = await open(file, "r");
+ try {
+ const buf = Buffer.alloc(bytes);
+ const { bytesRead } = await fh.read(buf, 0, bytes, 0);
+ return buf.subarray(0, bytesRead).toString("utf8");
+ } finally {
+ await fh.close();
+ }
+}
+
+async function readTail(file: string, bytes = 64): Promise<string> {
+ const fh = await open(file, "r");
+ try {
+ const { size } = await fh.stat();
+ const start = Math.max(0, size - bytes);
+ const buf = Buffer.alloc(size - start);
+ const { bytesRead } = await fh.read(buf, 0, buf.length, start);
+ return buf.subarray(0, bytesRead).toString("utf8");
+ } finally {
+ await fh.close();
+ }
+}
+
+// Whether a caption cues.json, already new enough by mtime, was made under the
+// current caption-track rule — normalize records it as `captionTrackRule` in
+// the file's first bytes — or, made before the rule had a version, holds what
+// the rule would read anyway. The older rule ranked transcript.en.vtt first
+// and parsed a cue-block VTT to no cues, so an unversioned file is stale only:
+//
+// when it holds NO cues (the cue-block parse, or the fallback to the next
+// track, may find text now) — the cues are the last key normalize writes,
+// so that is the file's tail; unless its one English VTT has word timing,
+// whose parse did not change and which had nothing to choose from;
+// when transcript.en.vtt and transcript.en-orig.vtt are both present and
+// differ — it was read from en, and en-orig is read now. A byte-identical
+// pair (equal size: on this corpus every one of 50,461 equal-size pairs was
+// byte-identical) reads the same either way.
+//
+// One to three small reads (a head, a tail, two stats), never a parse. A
+// mis-read only costs a rewrite, after which the record is there.
+async function followsCaptionTrackRule(
+ videoDir: string,
+ cuesPath: string,
+ entries: readonly string[],
+): Promise<boolean> {
+ const vtts = englishVttsByPreference(entries);
+ if (vtts.length === 0) return true;
+ try {
+ if (vtts.length === 1 && hasWordTiming(await readHead(path.join(videoDir, vtts[0])))) {
+ return true;
+ }
+ const m = (await readHead(cuesPath)).match(/"captionTrackRule"\s*:\s*(\d+)/);
+ if (m !== null && Number(m[1]) === CAPTION_TRACK_RULE_VERSION) return true;
+ if (/"cues"\s*:\s*\[\s*\]\s*\}\s*$/.test(await readTail(cuesPath))) return false;
+ if (!entries.includes(VTT_FILENAME) || !entries.includes(ORIG_VTT_FILENAME)) return true;
+ const [en, orig] = await Promise.all([
+ stat(path.join(videoDir, VTT_FILENAME)),
+ stat(path.join(videoDir, ORIG_VTT_FILENAME)),
+ ]);
+ return en.size === orig.size;
+ } catch {
+ return false;
+ }
+}
diff --git a/common/controller/reconcileVideoDirs.ts b/common/controller/reconcileVideoDirs.ts
@@ -28,6 +28,10 @@ export type ReconcileOpts = {
channelDir: string;
dryRun?: boolean;
onLog?: (s: string) => void;
+ // Reconcile only the dirs this accepts (default: every dir). A caller that
+ // knows which records it means to move (`archilyzer wayback refresh`,
+ // controller/waybackRefresh.ts) leaves the rest to the snapshot's pass.
+ only?: (dirName: string) => boolean;
};
// Files the canonical dir's copy should win on a name collision: these are
@@ -136,7 +140,10 @@ export async function reconcileVideoDirs(
const entries = await readdir(dataDir, { withFileTypes: true }).catch(
() => [],
);
- const dirNames = entries.filter((e) => e.isDirectory()).map((e) => e.name);
+ const dirNames = entries
+ .filter((e) => e.isDirectory())
+ .map((e) => e.name)
+ .filter((name) => !opts.only || opts.only(name));
for (const name of dirNames) {
try {
diff --git a/common/controller/rosterStore.ts b/common/controller/rosterStore.ts
@@ -141,6 +141,33 @@ export function mergeRoster(
return { ...roster, version: ROSTER_VERSION, updatedAt: now, entries };
}
+// A RENAMED RECORD keeps its roster entry under its new id. Not a removal:
+// the entry moves (url, firstSeenAt, source and all), which is what a dir
+// renamed to its canonical id (reconcileVideoDirs.ts) needs — left under the
+// old id it would read as a video the channel has and nobody downloaded.
+// When both ids have an entry the new one's stays and the earlier
+// firstSeenAt wins. Returns the SAME object when nothing moved.
+export function renameRosterEntries(
+ roster: Roster,
+ renames: ReadonlyArray<{ from: string; to: string }>,
+ now: string,
+): Roster {
+ const entries: Record<string, RosterEntry> = { ...roster.entries };
+ let changed = false;
+ for (const { from, to } of renames) {
+ const prev = entries[from];
+ if (!prev || from === to) continue;
+ const there = entries[to];
+ entries[to] = there
+ ? { ...there, firstSeenAt: prev.firstSeenAt && prev.firstSeenAt < there.firstSeenAt ? prev.firstSeenAt : there.firstSeenAt, url: there.url || prev.url }
+ : prev;
+ delete entries[from];
+ changed = true;
+ }
+ if (!changed) return roster;
+ return { ...roster, version: ROSTER_VERSION, updatedAt: now, entries };
+}
+
// Stamp the outcome of an enumeration. Kept separate from mergeRoster because a
// REJECTED enumeration still merges (additively, losing nothing) while recording
// that its listing was not trusted — that record is what the next enumeration
diff --git a/common/controller/waybackRefresh.test.ts b/common/controller/waybackRefresh.test.ts
@@ -0,0 +1,242 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { lstat, mkdir, mkdtemp, readFile, readdir, readlink, stat, symlink, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import { buildWaybackProvenance } from "../lib/wayback";
+import { loadMetadataHistory } from "../lib/metadataHistory-server";
+import { refreshWaybackRecords } from "./waybackRefresh";
+import { renameRosterEntries, type Roster } from "./rosterStore";
+import { main as refreshCli } from "../bin/wayback-refresh";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/waybackRefresh.test.ts
+//
+// A temp corpus of Wayback copies imported before the app knew what one was:
+// an archived YouTube page in `watch/`, two raw JW Player files in
+// `<jwId>-<rendition>.mp4/`, and a plain record. Every id here is invented.
+
+const SLUG = "demo-wayback";
+const YT = "Abc123def45";
+const PAGE = `https://web.archive.org/web/20220102030405/https://www.youtube.com/watch?v=${YT}`;
+const JW_A = "Qw3rTy12";
+const JW_B = "Zx9vBn34";
+const fileUrl = (jw: string, ts: string) =>
+ `https://web.archive.org/web/${ts}id_/https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${jw}-12345678.mp4?token=0_abc_0xdef`;
+const FILE_A = fileUrl(JW_A, "20200102030405");
+const FILE_B = fileUrl(JW_B, "20200103030405");
+
+async function withCorpus(
+ fn: (paths: Paths, dataDir: string, channelDir: string) => Promise<void>,
+): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-wayback-"));
+ const transcriptsDir = path.join(dir, "corpus");
+ const paths = {
+ transcriptsDir,
+ channelsDir: path.join(transcriptsDir, "channels"),
+ jobsDir: path.join(transcriptsDir, ".jobs"),
+ } as Paths;
+ const channelDir = path.join(paths.channelsDir, SLUG);
+ const dataDir = path.join(channelDir, "data");
+ await mkdir(dataDir, { recursive: true });
+ await mkdir(paths.jobsDir, { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ name: "Demo", platform: "archiveorg", handling: "transcribe" }),
+ );
+ const record = async (name: string, info: Record<string, unknown>) => {
+ await mkdir(path.join(dataDir, name), { recursive: true });
+ await writeFile(path.join(dataDir, name, "metadata.info.json"), JSON.stringify(info));
+ };
+ await record("watch", {
+ id: YT,
+ title: "An archived upload",
+ upload_date: "20090102",
+ extractor_key: "YoutubeWebArchive",
+ webpage_url: PAGE,
+ webpage_url_basename: "watch",
+ });
+ for (const [jw, url] of [[JW_A, FILE_A], [JW_B, FILE_B]] as const) {
+ await record(`${jw}-12345678.mp4`, {
+ id: `${jw}-12345678`,
+ title: `${jw}-12345678`,
+ extractor_key: "Generic",
+ webpage_url: url,
+ webpage_url_basename: `${jw}-12345678.mp4`,
+ });
+ }
+ await record("Plain12345a", { id: "Plain12345a", title: "Plain", upload_date: "20200101", extractor_key: "Youtube", webpage_url: "https://www.youtube.com/watch?v=Plain12345a" });
+ // A tiered file: a relative link into media/<old id>/ (release 17).
+ await mkdir(path.join(channelDir, "media", `${JW_A}-12345678.mp4`), { recursive: true });
+ await writeFile(path.join(channelDir, "media", `${JW_A}-12345678.mp4`, "audio.mp3"), "audio");
+ await symlink(
+ path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3"),
+ path.join(dataDir, `${JW_A}-12345678.mp4`, "audio.mp3"),
+ );
+ const entry = (url: string) => ({ url, firstSeenAt: "2026-01-01T00:00:00.000Z", lastListedAt: "2026-01-01T00:00:00.000Z", source: "import" });
+ await writeFile(
+ path.join(channelDir, "roster.json"),
+ JSON.stringify({
+ version: 1,
+ updatedAt: "2026-01-01T00:00:00.000Z",
+ lastSweep: null,
+ entries: {
+ watch: entry(PAGE),
+ [`${JW_A}-12345678.mp4`]: entry(FILE_A),
+ [`${JW_B}-12345678.mp4`]: entry(FILE_B),
+ Plain12345a: entry("https://www.youtube.com/watch?v=Plain12345a"),
+ },
+ }),
+ );
+ await fn(paths, dataDir, channelDir);
+}
+
+const titles = {
+ [JW_A]: { title: "Show: First Guest", upload_date: "2019-03-03" },
+ [`${JW_B}-12345678.mp4`]: { title: "Show: Second Guest", upload_date: "20190310" },
+ [YT]: { title: "Not applied: the page has a title", upload_date: "20000101" },
+ nobody: { title: "No such record" },
+};
+
+test("a dry run reports every rename and title and writes nothing", async () => {
+ await withCorpus(async (paths, dataDir, channelDir) => {
+ const rosterBefore = await readFile(path.join(channelDir, "roster.json"), "utf8");
+ const r = await refreshWaybackRecords({ slug: SLUG, paths, dryRun: true, titles });
+ assert.deepEqual(
+ r.records.map((x) => [x.from, x.to, x.rename, x.sidecar]),
+ [
+ [`${JW_A}-12345678.mp4`, JW_A, "renamed", true],
+ [`${JW_B}-12345678.mp4`, JW_B, "renamed", true],
+ ["watch", YT, "renamed", true],
+ ],
+ );
+ assert.deepEqual(r.records[0].changes, {
+ title: { from: `${JW_A}-12345678`, to: "Show: First Guest" },
+ upload_date: { from: null, to: "20190303" },
+ });
+ assert.deepEqual(r.records[2].changes, {});
+ assert.deepEqual(r.unmatchedTitles, ["nobody"]);
+ assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, `${JW_B}-12345678.mp4`, "Plain12345a", "watch"].sort());
+ assert.equal(await readFile(path.join(channelDir, "roster.json"), "utf8"), rosterBefore);
+ await assert.rejects(stat(path.join(dataDir, "watch", "wayback.json")));
+ });
+});
+
+test("a real run renames through the reconcile pass, moves the roster, writes sidecars and titles; a second run changes nothing", async () => {
+ await withCorpus(async (paths, dataDir, channelDir) => {
+ const r = await refreshWaybackRecords({ slug: SLUG, paths, titles, now: () => new Date("2026-02-02T00:00:00.000Z") });
+ assert.equal(r.failed.length, 0);
+ assert.deepEqual((await readdir(dataDir)).sort(), [JW_A, JW_B, "Plain12345a", YT].sort());
+
+ // The tier link moved with its dir and still resolves.
+ const link = path.join(dataDir, JW_A, "audio.mp3");
+ assert.ok((await lstat(link)).isSymbolicLink());
+ assert.equal(await readlink(link), path.join("..", "..", "media", `${JW_A}-12345678.mp4`, "audio.mp3"));
+ assert.equal(await readFile(link, "utf8"), "audio");
+
+ const roster = JSON.parse(await readFile(path.join(channelDir, "roster.json"), "utf8"));
+ assert.deepEqual(Object.keys(roster.entries).sort(), [JW_A, JW_B, "Plain12345a", YT].sort());
+ assert.equal(roster.entries[YT].url, PAGE);
+ assert.equal(roster.entries[YT].firstSeenAt, "2026-01-01T00:00:00.000Z");
+
+ assert.deepEqual(JSON.parse(await readFile(path.join(dataDir, YT, "wayback.json"), "utf8")), buildWaybackProvenance(PAGE));
+ assert.equal(JSON.parse(await readFile(path.join(dataDir, JW_A, "wayback.json"), "utf8")).originalUrl, FILE_A.replace(/^.*?id_\//, ""));
+ await assert.rejects(stat(path.join(dataDir, "Plain12345a", "wayback.json")));
+
+ const info = JSON.parse(await readFile(path.join(dataDir, JW_B, "metadata.info.json"), "utf8"));
+ assert.equal(info.title, "Show: Second Guest");
+ assert.equal(info.upload_date, "20190310");
+ const page = JSON.parse(await readFile(path.join(dataDir, YT, "metadata.info.json"), "utf8"));
+ assert.equal(page.title, "An archived upload");
+ assert.equal(page.upload_date, "20090102");
+ const history = await loadMetadataHistory(path.join(dataDir, JW_A));
+ assert.equal(history?.entries.at(-1)?.by, "wayback-provenance");
+
+ const again = await refreshWaybackRecords({ slug: SLUG, paths, titles });
+ assert.deepEqual(
+ again.records.map((x) => [x.from, x.to, x.rename ?? null, x.sidecar, Object.keys(x.changes).length]),
+ [
+ [YT, YT, null, false, 0],
+ [JW_A, JW_A, null, false, 0],
+ [JW_B, JW_B, null, false, 0],
+ ],
+ );
+ });
+});
+
+test("a record a live job names is held; a dead writer's job is not", async () => {
+ await withCorpus(async (paths, dataDir) => {
+ const meta = (id: string, videoId: string, pid: number) =>
+ writeFile(
+ path.join(paths.jobsDir, `${id}.meta.json`),
+ JSON.stringify({ id, kind: "transcribe-one", queueKey: "transcription", channelSlug: SLUG, videoId, status: "running", queuedAt: 1, pid }),
+ );
+ await meta("01AAAAAAAAAAAAAAAAAAAAAAAA", `${JW_A}-12345678.mp4`, process.ppid);
+ await meta("01BBBBBBBBBBBBBBBBBBBBBBBB", "watch", 2 ** 22 + 12345);
+ // An auto-queue lane's newest pick is in flight; an old one is not.
+ paths.autoQueueStateFile = path.join(paths.transcriptsDir, ".auto-queue", "state.json");
+ await mkdir(path.dirname(paths.autoQueueStateFile), { recursive: true });
+ const at = Date.parse("2026-02-02T00:00:00.000Z");
+ await writeFile(
+ paths.autoQueueStateFile,
+ JSON.stringify({
+ transcription: {
+ picks: [
+ { at, leafId: "x", videoId: `${JW_B}-12345678.mp4`, channelSlug: SLUG },
+ // Finished: its outcome sidecar is newer than the pick (below).
+ { at: at - 1000, leafId: "x", videoId: "watch", channelSlug: SLUG },
+ ],
+ },
+ // Too old to be in flight.
+ download: { picks: [{ at: at - 7 * 3600 * 1000, leafId: "x", videoId: "watch", channelSlug: SLUG }] },
+ }),
+ );
+ await writeFile(path.join(dataDir, "watch", "transcribe-outcome.json"), "{}");
+ const r = await refreshWaybackRecords({ slug: SLUG, paths, titles, now: () => new Date(at + 60_000) });
+ assert.equal(r.records.find((x) => x.from === `${JW_B}-12345678.mp4`)!.rename, "held");
+ const a = r.records.find((x) => x.from === `${JW_A}-12345678.mp4`)!;
+ assert.equal(a.rename, "held");
+ assert.equal(a.to, a.from);
+ assert.deepEqual(a.changes, {});
+ assert.equal(r.records.find((x) => x.from === "watch")!.rename, "renamed");
+ assert.deepEqual((await readdir(dataDir)).sort(), [`${JW_A}-12345678.mp4`, `${JW_B}-12345678.mp4`, "Plain12345a", YT].sort());
+ });
+});
+
+test("renameRosterEntries moves an entry and keeps the earlier sighting", () => {
+ const e = (url: string, firstSeenAt: string) => ({ url, firstSeenAt, lastListedAt: firstSeenAt, source: "import" as const });
+ const roster: Roster = {
+ version: 1,
+ updatedAt: "",
+ lastSweep: null,
+ entries: { old: e("u-old", "2026-01-01"), both: e("u-both", "2026-01-01"), keep: e("u-keep", "2026-03-03") },
+ };
+ const next = renameRosterEntries(roster, [{ from: "old", to: "new" }, { from: "both", to: "keep" }], "now");
+ assert.deepEqual(Object.keys(next.entries).sort(), ["keep", "new"]);
+ assert.equal(next.entries.new.url, "u-old");
+ assert.equal(next.entries.keep.firstSeenAt, "2026-01-01");
+ assert.equal(next.entries.keep.url, "u-keep");
+ assert.equal(renameRosterEntries(next, [{ from: "gone", to: "x" }], "now"), next);
+});
+
+test("the CLI: --titles from a file, old → new printed, exit 0", async () => {
+ await withCorpus(async (paths, dataDir) => {
+ const file = path.join(paths.transcriptsDir, "titles.json");
+ await writeFile(file, JSON.stringify(titles));
+ const lines: string[] = [];
+ const orig = console.log;
+ console.log = (...a: unknown[]) => void lines.push(a.join(" "));
+ try {
+ assert.equal(await refreshCli({ slug: SLUG, dryRun: false, titlesFile: file, paths }), 0);
+ } finally {
+ console.log = orig;
+ }
+ const out = lines.join("\n");
+ assert.match(out, new RegExp(`watch → ${YT}`));
+ assert.match(out, new RegExp(`${JW_A}-12345678\\.mp4 → ${JW_A}`));
+ assert.match(out, /3 Wayback records, 3 renamed, 3 wayback\.json written, 2 retitled\./);
+ assert.ok((await readdir(dataDir)).includes(YT));
+ assert.equal(await refreshCli({ slug: "Not A Slug", dryRun: true, paths }), 2);
+ });
+});
diff --git a/common/controller/waybackRefresh.ts b/common/controller/waybackRefresh.ts
@@ -0,0 +1,342 @@
+// WAYBACK MACHINE COPIES, BROUGHT UP TO THE WAYBACK RULES — offline.
+//
+// A record imported from a Wayback capture (lib/wayback.ts) before the app
+// knew what one was carries no `wayback.json`, and its dir is named by the last
+// segment of the capture URL: `watch` for an archived YouTube page (a name
+// every such capture shares), `<jwId>-<rendition>.mp4` for a raw JW Player
+// file. This brings every such record of a channel up to the rules:
+//
+// wayback.json written from the capture URL (lib/wayback-server.ts)
+// data/<id>/ renamed to its canonical id (lib/videoId.ts) by the
+// snapshot's own pass, reconcileVideoDirs — the media
+// tier's relative links move with the dir, and a dir
+// already there is merged, never overwritten
+// roster.json the entry moved to the new id (renameRosterEntries)
+// metadata.info.json with `titles`: the title and upload_date the operator
+// found for a record that has none (a raw file's title
+// is its file name), through `patchMetadataInfo`, so the
+// change is in metadata.history.json as
+// `wayback-provenance`
+//
+// A record a live job names (a `.jobs/` meta, queued or running, whose writer
+// is alive, or one of an auto-queue lane's newest picks) is skipped and
+// reported; a live job over the whole channel holds every record. Only what differs is written, so a second run writes nothing.
+// No network.
+
+import path from "node:path";
+import { readdir, readFile, stat } from "node:fs/promises";
+import { getPaths, type Paths } from "../lib/paths";
+import { readJsonFile } from "../lib/jsonFile-server";
+import { assertChannelTextReadable, readRelocationMarker } from "../lib/channelMedia";
+import { parseWaybackUrl } from "../lib/wayback";
+import { ensureWaybackProvenance } from "../lib/wayback-server";
+import { extractVideoId } from "../lib/videoId";
+import { patchMetadataInfo } from "../lib/metadataHistory-server";
+import { readChannelConfig } from "./channels";
+import { listChannelVideoIds } from "./keptVideos";
+import { reconcileVideoDirs } from "./reconcileVideoDirs";
+import { loadRoster, renameRosterEntries, writeRoster } from "./rosterStore";
+import { writerIsGone } from "../jobs/bootQueuedJobs";
+import type { JobMeta } from "../jobs/jobMeta";
+import type { AutoQueuePick } from "../jobs/autoQueueState";
+
+// What the operator found for a record: its title and the day it is of.
+export type WaybackTitle = { title?: string; upload_date?: string };
+
+export type WaybackRecordResult = {
+ // The dir's name before, and after (the same when it was not renamed).
+ from: string;
+ to: string;
+ // The capture the record was fetched from.
+ captureUrl: string;
+ rename?: "renamed" | "merged" | "conflict" | "held";
+ // Why a rename did not happen (a live job, a conflict, a capture that is
+ // not the record's webpage_url).
+ note?: string;
+ // wayback.json was (or would be) written.
+ sidecar: boolean;
+ // Per key: the value before and after.
+ changes: Partial<Record<"title" | "upload_date", { from: unknown; to: unknown }>>;
+};
+
+export type WaybackRefreshResult = {
+ records: WaybackRecordResult[];
+ // `titles` entries no Wayback record of the channel matched.
+ unmatchedTitles: string[];
+ failed: { id: string; error: string }[];
+};
+
+type Info = Record<string, unknown>;
+
+async function readInfo(videoDir: string): Promise<Info | null> {
+ const read = await readJsonFile(path.join(videoDir, "metadata.info.json"));
+ return read.ok && read.value && typeof read.value === "object" && !Array.isArray(read.value)
+ ? (read.value as Info)
+ : null;
+}
+
+// The URL the managed download ran yt-dlp on: the last argument of the
+// command line it logged (`$ yt-dlp … -- <url>`).
+async function urlFromDownloadLog(videoDir: string): Promise<string | null> {
+ let head: string;
+ try {
+ head = (await readFile(path.join(videoDir, "download.log"), "utf8")).slice(0, 16 * 1024);
+ } catch {
+ return null;
+ }
+ for (const line of head.split("\n")) {
+ const m = /^\$ yt-dlp .* -- (\S+)\s*$/.exec(line);
+ if (m && parseWaybackUrl(m[1])) return m[1];
+ }
+ return null;
+}
+
+// The capture a record was fetched from: its webpage_url, its original_url,
+// or the URL its download log ran on.
+async function captureUrlOf(videoDir: string, info: Info | null): Promise<string | null> {
+ for (const key of ["webpage_url", "original_url"]) {
+ const v = info?.[key];
+ if (typeof v === "string" && parseWaybackUrl(v)) return v;
+ }
+ return urlFromDownloadLog(videoDir);
+}
+
+// The live jobs on a channel: the record ids they name, and whether one of
+// them covers the whole channel.
+async function liveJobs(
+ paths: Paths,
+ slug: string,
+ now: number,
+): Promise<{ ids: Set<string>; channelWide: string[] }> {
+ const ids = new Set<string>();
+ const channelWide: string[] = [];
+ const names = paths.jobsDir ? await readdir(paths.jobsDir).catch(() => [] as string[]) : [];
+ for (const name of names) {
+ if (!name.endsWith(".meta.json")) continue;
+ const read = await readJsonFile(path.join(paths.jobsDir, name));
+ if (!read.ok || !read.value || typeof read.value !== "object") continue;
+ const meta = read.value as JobMeta;
+ if (meta.channelSlug !== slug) continue;
+ if (meta.status !== "queued" && meta.status !== "running") continue;
+ if (writerIsGone(meta)) continue;
+ if (meta.videoId) ids.add(meta.videoId);
+ else channelWide.push(`${meta.kind} ${meta.id}`);
+ }
+ // The auto-queue lanes are jobs of no channel, and the record each is on
+ // is its newest pick (jobs/autoQueueState.ts). A lane runs a few workers, so
+ // the newest few recent picks of this channel are held.
+ if (paths.autoQueueStateFile) {
+ const read = await readJsonFile(paths.autoQueueStateFile);
+ const state = read.ok && read.value && typeof read.value === "object" ? (read.value as Record<string, unknown>) : {};
+ const since = now - LANE_PICK_HOLD_MS;
+ for (const [kind, kindState] of Object.entries(state)) {
+ const picks = (kindState as { picks?: unknown } | null)?.picks;
+ if (!Array.isArray(picks)) continue;
+ for (const pick of picks.slice(0, LANE_PICKS_HELD) as Partial<AutoQueuePick>[]) {
+ if (pick?.channelSlug !== slug || typeof pick.videoId !== "string") continue;
+ const at = pick.at ?? 0;
+ if (at < since) continue;
+ // A pick whose outcome sidecar was written since is finished.
+ const outcome = LANE_OUTCOME[kind];
+ if (outcome) {
+ const done = await stat(path.join(paths.channelsDir, slug, "data", pick.videoId, outcome)).catch(() => null);
+ if (done && done.mtimeMs >= at) continue;
+ }
+ ids.add(pick.videoId);
+ }
+ }
+ }
+ return { ids, channelWide };
+}
+
+// How many of a lane's newest picks may still be in flight, and for how long.
+const LANE_PICKS_HELD = 4;
+const LANE_PICK_HOLD_MS = 6 * 60 * 60 * 1000;
+// The sidecar a lane's run writes when it ends, by lane kind.
+const LANE_OUTCOME: Record<string, string> = {
+ transcription: "transcribe-outcome.json",
+ download: "download-outcome.json",
+};
+
+// A title that is not one: absent, or the file's own name (what yt-dlp's
+// generic extractor titles a raw file with).
+function hasRealTitle(info: Info, names: string[]): boolean {
+ const t = typeof info.title === "string" ? info.title.trim() : "";
+ if (!t) return false;
+ const own = new Set<string>(names);
+ for (const key of ["id", "display_id", "webpage_url_basename"]) {
+ const v = info[key];
+ if (typeof v === "string") {
+ own.add(v);
+ own.add(v.replace(/\.[A-Za-z0-9]{2,4}$/, ""));
+ }
+ }
+ return !own.has(t);
+}
+
+// `YYYYMMDD` from `YYYYMMDD` or `YYYY-MM-DD`; null otherwise.
+export function normalizeUploadDate(v: unknown): string | null {
+ if (typeof v !== "string") return null;
+ const s = v.trim();
+ if (/^\d{8}$/.test(s)) return s;
+ const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(s);
+ return m ? `${m[1]}${m[2]}${m[3]}` : null;
+}
+
+export async function refreshWaybackRecords(opts: {
+ slug: string;
+ paths?: Paths;
+ dryRun?: boolean;
+ // Record id (its new id, or its old dir name) → its title and date.
+ titles?: Record<string, WaybackTitle>;
+ onLog?: (line: string) => void;
+ now?: () => Date;
+}): Promise<WaybackRefreshResult> {
+ const paths = opts.paths ?? getPaths();
+ const log = opts.onLog ?? (() => {});
+ const dryRun = opts.dryRun === true;
+ const config = await readChannelConfig(paths, opts.slug);
+ if (!config) throw new Error(`Channel "${opts.slug}" not found`);
+ // An unreadable text tier is not an empty channel (AGENTS.md).
+ await assertChannelTextReadable(paths, opts.slug, config);
+ if (await readRelocationMarker(paths, opts.slug)) {
+ throw new Error(`Channel "${opts.slug}" is relocating (.relocating.json) — run again when the move is done`);
+ }
+
+ const channelDir = path.join(paths.channelsDir, opts.slug);
+ const dataDir = path.join(channelDir, "data");
+ const result: WaybackRefreshResult = { records: [], unmatchedTitles: [], failed: [] };
+ const jobs = await liveJobs(paths, opts.slug, (opts.now?.() ?? new Date()).getTime());
+
+ // ─── Find the Wayback records ───
+ type Found = { rec: WaybackRecordResult; info: Info | null; canonical: string | null };
+ const found: Found[] = [];
+ for (const name of (await listChannelVideoIds(paths, opts.slug)).sort()) {
+ const videoDir = path.join(dataDir, name);
+ try {
+ const info = await readInfo(videoDir);
+ const captureUrl = await captureUrlOf(videoDir, info);
+ if (!captureUrl) continue;
+ // The snapshot names a dir by its webpage_url (reconcileVideoDirs.ts),
+ // so that is the only name a rename can give it that lasts.
+ const webpageUrl = typeof info?.webpage_url === "string" ? info.webpage_url : null;
+ const canonical = webpageUrl && parseWaybackUrl(webpageUrl) ? extractVideoId(webpageUrl) : null;
+ const rec: WaybackRecordResult = { from: name, to: name, captureUrl, sidecar: false, changes: {} };
+ if (canonical && canonical !== name) rec.to = canonical;
+ else if (!canonical && webpageUrl !== captureUrl) {
+ rec.note = "webpage_url is not the capture; the dir keeps its name";
+ }
+ found.push({ rec, info, canonical });
+ } catch (err) {
+ result.failed.push({ id: name, error: (err as Error).message });
+ }
+ }
+
+ const held = (rec: WaybackRecordResult): string | null => {
+ if (jobs.channelWide.length > 0) return `a live job holds the channel (${jobs.channelWide.join(", ")})`;
+ if (jobs.ids.has(rec.from) || jobs.ids.has(rec.to)) return "a live job names this record";
+ return null;
+ };
+
+ // ─── Rename, through the snapshot's own pass ───
+ const toRename = new Set<string>();
+ for (const { rec } of found) {
+ if (rec.to === rec.from) continue;
+ const why = held(rec);
+ if (why) {
+ rec.rename = "held";
+ rec.note = why;
+ rec.to = rec.from;
+ continue;
+ }
+ toRename.add(rec.from);
+ }
+ if (toRename.size > 0) {
+ const r = await reconcileVideoDirs({ channelDir, dryRun, only: (name) => toRename.has(name) });
+ const byFrom = new Map(found.map((f) => [f.rec.from, f.rec]));
+ for (const x of r.renamed) {
+ const rec = byFrom.get(x.from);
+ if (rec) rec.rename = "renamed";
+ }
+ for (const x of r.merged) {
+ const rec = byFrom.get(x.from);
+ if (rec) {
+ rec.rename = "merged";
+ rec.note = `merged into the existing ${x.to}/ (${x.movedFiles.length} files)`;
+ }
+ }
+ for (const x of r.conflicts) {
+ const rec = byFrom.get(x.from);
+ if (rec) {
+ rec.rename = "conflict";
+ rec.note = x.reason;
+ rec.to = rec.from;
+ }
+ }
+ const moved = [...r.renamed, ...r.merged].map((x) => ({ from: x.from, to: x.to }));
+ if (!dryRun && moved.length > 0) {
+ const now = (opts.now?.() ?? new Date()).toISOString();
+ const before = await loadRoster(paths, opts.slug);
+ const after = renameRosterEntries(before, moved, now);
+ if (after !== before) await writeRoster(paths, opts.slug, after);
+ }
+ }
+
+ // ─── The sidecar, and the titles ───
+ const titles = opts.titles ?? {};
+ const usedTitles = new Set<string>();
+ for (const { rec, info } of found) {
+ // A dry run renamed nothing: the record is still under its old name.
+ const videoDir = path.join(dataDir, dryRun ? rec.from : rec.to);
+ try {
+ const s = await ensureWaybackProvenance(videoDir, rec.captureUrl, { dryRun });
+ rec.sidecar = s.written;
+ const key = rec.to in titles ? rec.to : rec.from in titles ? rec.from : null;
+ if (key === null || !info) continue;
+ usedTitles.add(key);
+ if (held(rec)) {
+ rec.note = rec.note ?? held(rec)!;
+ continue;
+ }
+ const want = titles[key];
+ const patch: Record<string, unknown> = {};
+ const title = typeof want.title === "string" ? want.title.trim() : "";
+ if (title && !hasRealTitle(info, [rec.from, rec.to]) && info.title !== title) {
+ patch.title = title;
+ rec.changes.title = { from: info.title ?? null, to: title };
+ }
+ const date = normalizeUploadDate(want.upload_date);
+ if (want.upload_date !== undefined && !date) {
+ rec.note = `upload_date "${String(want.upload_date)}" is not YYYYMMDD or YYYY-MM-DD`;
+ } else if (date && normalizeUploadDate(info.upload_date) === null) {
+ patch.upload_date = date;
+ rec.changes.upload_date = { from: info.upload_date ?? null, to: date };
+ }
+ if (Object.keys(patch).length > 0 && !dryRun) {
+ await patchMetadataInfo(videoDir, patch, { by: "wayback-provenance", requestedBy: "cli", onLog: log });
+ }
+ } catch (err) {
+ result.failed.push({ id: rec.from, error: (err as Error).message });
+ }
+ }
+ result.unmatchedTitles = Object.keys(titles).filter((k) => !usedTitles.has(k)).sort();
+ result.records = found.map((f) => f.rec);
+ for (const rec of result.records) log(formatWaybackRecord(rec, dryRun));
+ return result;
+}
+
+// One record as the CLI prints it: `old → new`, then what was written.
+export function formatWaybackRecord(rec: WaybackRecordResult, dryRun: boolean): string {
+ const would = dryRun ? "would be " : "";
+ const head =
+ rec.from === rec.to
+ ? `${rec.from} (name kept)`
+ : `${rec.from} → ${rec.to}${rec.rename === "merged" ? " (merged)" : ""}`;
+ const lines = [head];
+ if (rec.note) lines.push(` ${rec.rename === "held" || rec.rename === "conflict" ? "NOT RENAMED: " : ""}${rec.note}`);
+ if (rec.sidecar) lines.push(` wayback.json ${would}written`);
+ for (const [k, c] of Object.entries(rec.changes)) {
+ lines.push(` ${k}: ${JSON.stringify(c!.from)} → ${JSON.stringify(c!.to)}`);
+ }
+ return lines.join("\n");
+}
diff --git a/common/jobs/platformGap.test.ts b/common/jobs/platformGap.test.ts
@@ -0,0 +1,72 @@
+import { test, beforeEach } from "node:test";
+import assert from "node:assert/strict";
+import {
+ clearPlatformGaps,
+ notePlatformGap,
+ platformGapRemainingMs,
+ waitForPlatformGap,
+} from "./platformGap";
+
+beforeEach(() => clearPlatformGaps());
+
+test("a platform with no import yet is asked at once", () => {
+ assert.equal(platformGapRemainingMs("odysee", 1_000), 0);
+});
+
+test("the next import waits the gap from when the last one settled, per platform", () => {
+ notePlatformGap("odysee", 60_000, 1_000);
+ assert.equal(platformGapRemainingMs("odysee", 1_000), 60_000);
+ assert.equal(platformGapRemainingMs("odysee", 31_000), 30_000);
+ assert.equal(platformGapRemainingMs("odysee", 61_000), 0);
+ assert.equal(platformGapRemainingMs("bitchute", 1_000), 0);
+});
+
+test("a later start is never shortened by a shorter gap", () => {
+ notePlatformGap("odysee", 90_000, 0);
+ notePlatformGap("odysee", 10_000, 5_000);
+ assert.equal(platformGapRemainingMs("odysee", 5_000), 85_000);
+});
+
+test("the gap lives on globalThis, so every bundle shares it", () => {
+ notePlatformGap("odysee", 1_000, 0);
+ assert.equal(globalThis.__yttPlatformGap__?.get("odysee"), 1_000);
+});
+
+test("waiting re-reads the remaining time after each sleep and logs each wait", async () => {
+ const left = [45_000, 2_000, 0];
+ const slept: number[] = [];
+ const lines: string[] = [];
+ const waited = await waitForPlatformGap({
+ label: "Odysee",
+ remainingMs: () => left.shift() ?? 0,
+ onLog: (l) => lines.push(l),
+ sleep: async (ms) => {
+ slept.push(ms);
+ },
+ });
+ assert.deepEqual(slept, [45_000, 2_000]);
+ assert.equal(waited, 47_000);
+ assert.deepEqual(lines, [
+ "Odysee asks for a gap between videos: waiting 45s.",
+ "Odysee asks for a gap between videos: waiting 2s.",
+ ]);
+});
+
+test("nothing to wait: no sleep, no line", async () => {
+ const lines: string[] = [];
+ const waited = await waitForPlatformGap({
+ label: "Odysee",
+ remainingMs: async () => 0,
+ onLog: (l) => lines.push(l),
+ sleep: async () => assert.fail("must not sleep"),
+ });
+ assert.equal(waited, 0);
+ assert.deepEqual(lines, []);
+});
+
+test("an aborted job stops waiting", async () => {
+ const ac = new AbortController();
+ const p = waitForPlatformGap({ label: "Odysee", remainingMs: () => 60_000, signal: ac.signal });
+ ac.abort();
+ await assert.rejects(p, { name: "AbortError" });
+});
diff --git a/common/jobs/platformGap.ts b/common/jobs/platformGap.ts
@@ -0,0 +1,82 @@
+// THE GAP BETWEEN TWO ONE-OFF IMPORTS ON ONE PLATFORM. A batch download, a
+// sync, a persist and the auto-download lane each wait their platform's floor
+// between two videos — but every import is its own job, so ten imports queued
+// on `platform:odysee` used to ask Odysee as fast as each job could finish
+// (Odysee answered 429 at 10–50 s apart, 2026-10-06). An import on a platform
+// with an import floor (platformImportMinGapSeconds, ytdlp/platformArgs.mjs)
+// waits here first, and sets the platform's next start when it settles.
+//
+// Process-wide (on globalThis, as the job registry is: a route bundle and the
+// job runner must share one view) and in memory: a restart is itself a gap.
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttPlatformGap__: Map<string, number> | undefined;
+}
+
+function nextStartAt(): Map<string, number> {
+ if (!globalThis.__yttPlatformGap__) globalThis.__yttPlatformGap__ = new Map();
+ return globalThis.__yttPlatformGap__;
+}
+
+// Milliseconds before `platform` may be asked again (0 = now).
+export function platformGapRemainingMs(platform: string, now: number = Date.now()): number {
+ const at = nextStartAt().get(platform);
+ return at !== undefined && at > now ? at - now : 0;
+}
+
+// A unit on `platform` settled at `now`: the next one starts no sooner than
+// `gapMs` later. A later start already set is kept (never shortened).
+export function notePlatformGap(platform: string, gapMs: number, now: number = Date.now()): void {
+ if (gapMs <= 0) return;
+ const at = now + gapMs;
+ const map = nextStartAt();
+ if ((map.get(platform) ?? 0) < at) map.set(platform, at);
+}
+
+export function clearPlatformGaps(): void {
+ nextStartAt().clear();
+}
+
+// Wait out `remainingMs()` — re-read after each wait, since another import or
+// a 429 may have moved it — logging once per wait. Rejects with an AbortError
+// when `signal` aborts.
+export async function waitForPlatformGap(opts: {
+ label: string;
+ remainingMs: () => Promise<number> | number;
+ signal?: AbortSignal;
+ onLog?: (line: string) => void;
+ sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
+}): Promise<number> {
+ const sleep = opts.sleep ?? abortableSleep;
+ let waited = 0;
+ for (;;) {
+ if (opts.signal?.aborted) throw abortError();
+ const ms = await opts.remainingMs();
+ if (ms <= 0) return waited;
+ opts.onLog?.(`${opts.label} asks for a gap between videos: waiting ${Math.ceil(ms / 1000)}s.`);
+ await sleep(ms, opts.signal);
+ waited += ms;
+ }
+}
+
+function abortError(): Error {
+ const err = new Error("aborted while waiting for the platform gap");
+ err.name = "AbortError";
+ return err;
+}
+
+function abortableSleep(ms: number, signal?: AbortSignal): Promise<void> {
+ return new Promise((resolve, reject) => {
+ if (signal?.aborted) return reject(abortError());
+ const t = setTimeout(() => {
+ signal?.removeEventListener("abort", onAbort);
+ resolve();
+ }, ms);
+ const onAbort = () => {
+ clearTimeout(t);
+ reject(abortError());
+ };
+ signal?.addEventListener("abort", onAbort, { once: true });
+ });
+}
diff --git a/common/lib/__fixtures__/vtt-cue-blocks.vtt b/common/lib/__fixtures__/vtt-cue-blocks.vtt
@@ -0,0 +1,29 @@
+WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.960 --> 00:00:06.000
+welcome back everyone today we are talking
+about the harbor bridge finally finally
+
+00:00:06.000 --> 00:00:09.800
+I have been teasing that all week and let me
+just say to every reader out there that
+
+00:00:09.800 --> 00:00:16.200
+no matter what you build in life you never
+ever skip the load test plus later on
+
+00:00:16.200 --> 00:00:20.383
+we will look at the tide tables for the
+north pier & the <old> lighthouse
+
+00:00:20.383 --> 00:00:20.393
+[Applause]
+
+00:00:20.393 --> 00:00:20.400
+[Music]
+
+00:00:20.400 --> 00:00:26.640
+<i>right</i> to begin let us start with the
+end the conclusion here I have said this
diff --git a/common/lib/__fixtures__/vtt-rolling.vtt b/common/lib/__fixtures__/vtt-rolling.vtt
@@ -0,0 +1,31 @@
+WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.960 --> 00:00:03.070 align:start position:0%
+
+welcome<00:00:01.160><c> back</c><00:00:01.319><c> everyone</c><00:00:01.520><c> today</c><00:00:01.760><c> we</c>
+
+00:00:03.070 --> 00:00:03.080 align:start position:0%
+welcome back everyone today we
+
+
+00:00:03.080 --> 00:00:05.670 align:start position:0%
+welcome back everyone today we
+are<00:00:03.240><c> talking</c><00:00:03.639><c> about</c><00:00:04.319><c> the</c><00:00:04.759><c> harbor</c>
+
+00:00:05.670 --> 00:00:05.680 align:start position:0%
+are talking about the harbor
+
+
+00:00:05.680 --> 00:00:06.869 align:start position:0%
+are talking about the harbor
+bridge<00:00:06.000><c> finally</c><00:00:06.120><c> finally</c>
+
+00:00:06.869 --> 00:00:06.879 align:start position:0%
+bridge finally finally
+
+
+00:00:06.879 --> 00:00:09.310 align:start position:0%
+bridge finally finally
+I<00:00:07.000><c> have</c><00:00:07.120><c> been</c><00:00:07.279><c> teasing</c><00:00:07.560><c> that</c>
diff --git a/common/lib/captionTrack.test.ts b/common/lib/captionTrack.test.ts
@@ -0,0 +1,117 @@
+// The caption-track rule (lib/videoStatus.ts): which English VTT a video's
+// transcript is read from, by name and then by content.
+//
+// Run with: node_modules/.bin/tsx --test common/lib/captionTrack.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ CAPTION_TRACK_RULE_VERSION,
+ TRANSCRIPT_PIN_FILENAME,
+ captionInputs,
+ englishVttsByPreference,
+ readEnglishVttCues,
+ readSubTracks,
+ resolvePrimaryVtt,
+} from "./videoStatus";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "caption-track-"));
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "__fixtures__", name), "utf8");
+
+const ROLLING = fixture("vtt-rolling.vtt");
+const CUE_BLOCKS = fixture("vtt-cue-blocks.vtt");
+const EMPTY = "WEBVTT\nKind: captions\nLanguage: en\n\n";
+
+let n = 0;
+function videoDir(files: Record<string, string>): string {
+ const dir = path.join(ROOT, `v${n++}`);
+ mkdirSync(dir, { recursive: true });
+ for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body);
+ return dir;
+}
+
+test("en-orig ranks above en; regional above auto-translated; translations are not English", () => {
+ const entries = [
+ "transcript.en-en-US.vtt",
+ "transcript.en.vtt",
+ "transcript.es-en-US.vtt",
+ "transcript.en-GB.vtt",
+ "transcript.en-orig.vtt",
+ "transcript.json",
+ ];
+ assert.deepEqual(englishVttsByPreference(entries), [
+ "transcript.en-orig.vtt",
+ "transcript.en.vtt",
+ "transcript.en-GB.vtt",
+ "transcript.en-en-US.vtt",
+ ]);
+ assert.equal(resolvePrimaryVtt(entries), "transcript.en-orig.vtt");
+ assert.equal(resolvePrimaryVtt(["transcript.en.vtt", "transcript.en-US.vtt"]), "transcript.en.vtt");
+ assert.equal(resolvePrimaryVtt(["transcript.es.vtt"]), null);
+});
+
+test("the operator's pin puts transcript.en.vtt first, and is a caption input", () => {
+ const entries = ["transcript.en-orig.vtt", "transcript.en.vtt", TRANSCRIPT_PIN_FILENAME];
+ assert.equal(resolvePrimaryVtt(entries), "transcript.en.vtt");
+ assert.deepEqual(captionInputs(entries), [
+ "transcript.en.vtt",
+ "transcript.en-orig.vtt",
+ TRANSCRIPT_PIN_FILENAME,
+ ]);
+ // A pin with no English VTT is no input.
+ assert.deepEqual(captionInputs([TRANSCRIPT_PIN_FILENAME]), []);
+});
+
+test("readEnglishVttCues reads en-orig when both tracks have text", async () => {
+ const dir = videoDir({
+ "transcript.en.vtt": CUE_BLOCKS,
+ "transcript.en-orig.vtt": ROLLING,
+ });
+ const got = await readEnglishVttCues(dir);
+ assert.equal(got?.filename, "transcript.en-orig.vtt");
+ assert.equal(got?.cues[0].text, "are talking about the harbor");
+});
+
+test("readEnglishVttCues falls back past a track with no cues", async () => {
+ const dir = videoDir({
+ "transcript.en-orig.vtt": EMPTY,
+ "transcript.en.vtt": CUE_BLOCKS,
+ });
+ const got = await readEnglishVttCues(dir);
+ assert.equal(got?.filename, "transcript.en.vtt");
+ assert.equal(got?.cues.length, 7);
+});
+
+test("readEnglishVttCues: every track empty is the first track with no cues; none is null", async () => {
+ const dir = videoDir({
+ "transcript.en-orig.vtt": EMPTY,
+ "transcript.en.vtt": EMPTY,
+ });
+ assert.deepEqual(await readEnglishVttCues(dir), { filename: "transcript.en-orig.vtt", cues: [] });
+ assert.equal(await readEnglishVttCues(videoDir({ "transcript.es.vtt": ROLLING })), null);
+});
+
+test("readSubTracks lists no English VTT: those are caption tracks (lib/captionTracks.ts)", async () => {
+ const dir = videoDir({
+ "transcript.en.vtt": CUE_BLOCKS,
+ "transcript.en-orig.vtt": ROLLING,
+ "transcript.es.vtt": ROLLING,
+ });
+ const tracks = (await readSubTracks(dir)).map((t) => t.track).sort();
+ assert.deepEqual(tracks, ["es"]);
+});
+
+test("umtool's copy of the caption-track rule version matches", () => {
+ const src = readFileSync(
+ path.join(import.meta.dirname, "..", "..", "umtool", "report-to-video", "cues.mjs"),
+ "utf8",
+ );
+ const m = src.match(/export const CAPTION_TRACK_RULE_VERSION = (\d+);/);
+ assert.equal(Number(m?.[1]), CAPTION_TRACK_RULE_VERSION);
+});
diff --git a/common/lib/captionTracks-server.ts b/common/lib/captionTracks-server.ts
@@ -0,0 +1,129 @@
+// The alternate tracks of one video dir, read from disk (lib/captionTracks.ts
+// says what an alternate is). SERVER-ONLY (node:fs).
+
+import path from "node:path";
+import { readdir, readFile } from "node:fs/promises";
+import { parseVtt, type Cue } from "./vtt";
+import { parseTranscriptJson } from "./whisper";
+import {
+ TRANSCRIPT_PIN_FILENAME,
+ VTT_FILENAME,
+ WHISPER_FILENAME,
+ englishVttsByPreference,
+ readEnglishVttCues,
+} from "./videoStatus";
+import {
+ PINNED_TRACK,
+ TRANSCRIPTION_TRACK,
+ distinctAltTracks,
+ trackOfVttFile,
+ type AltTrack,
+ type TrackFields,
+} from "./captionTracks";
+
+// The track id of an English caption file in a listing: its language code, or
+// `pinned` for transcript.en.vtt while the operator's pin stands.
+export function trackIdOfVtt(filename: string, entries: readonly string[]): string {
+ if (filename === VTT_FILENAME && entries.includes(TRANSCRIPT_PIN_FILENAME)) {
+ return PINNED_TRACK;
+ }
+ return trackOfVttFile(filename) ?? filename;
+}
+
+// Whether a listing can hold an alternate at all — no file is read. A caption
+// record needs two English VTTs; a transcribed one, one.
+export function mayHaveAltTracks(
+ primaryKind: "vtt" | "whisper",
+ entries: readonly string[],
+): boolean {
+ const n = englishVttsByPreference(entries).length;
+ return primaryKind === "whisper" ? n >= 1 : n >= 2;
+}
+
+// Every English VTT of a listing, parsed, in preference order. One that cannot
+// be read is left out.
+export async function readEnglishVttTracks(
+ videoDir: string,
+ entries: readonly string[],
+): Promise<{ filename: string; cues: Cue[] }[]> {
+ const out: { filename: string; cues: Cue[] }[] = [];
+ for (const filename of englishVttsByPreference(entries)) {
+ try {
+ out.push({
+ filename,
+ cues: parseVtt(await readFile(path.join(videoDir, filename), "utf8")),
+ });
+ } catch {
+ // unreadable: not a track
+ }
+ }
+ return out;
+}
+
+// A record's track fields: the primary's id and the English tracks whose words
+// differ from it. `primaryCues` is what the record's transcript holds (the
+// dedupe compares against it); for a caption record, the primary is the first
+// track in preference order that has a cue — the caption-track rule's content
+// fallback, so the id names the track the words really came from. Empty
+// (no fields) when nothing differs.
+export async function readTrackFields(
+ videoDir: string,
+ primaryKind: "vtt" | "whisper",
+ primaryCues: readonly Cue[] | undefined,
+ entries?: readonly string[],
+): Promise<TrackFields> {
+ const listing = entries ?? (await readdir(videoDir).catch(() => [] as string[]));
+ if (!mayHaveAltTracks(primaryKind, listing)) return {};
+ const vtts = await readEnglishVttTracks(videoDir, listing);
+ let primaryTrack: string;
+ let primary: readonly Cue[] | undefined = primaryCues;
+ let candidates: { filename: string; cues: Cue[] }[];
+ if (primaryKind === "whisper") {
+ primaryTrack = TRANSCRIPTION_TRACK;
+ candidates = vtts;
+ } else {
+ if (vtts.length === 0) return {};
+ const idx = Math.max(0, vtts.findIndex((t) => t.cues.length > 0));
+ primaryTrack = trackIdOfVtt(vtts[idx].filename, listing);
+ primary ??= vtts[idx].cues;
+ candidates = vtts.filter((_, i) => i !== idx);
+ }
+ const alts: AltTrack[] = distinctAltTracks(
+ primary,
+ candidates.map((t) => ({ track: trackIdOfVtt(t.filename, listing), cues: t.cues })),
+ );
+ if (alts.length === 0) return {};
+ return { track: primaryTrack, altTracks: alts };
+}
+
+// Every track of a video dir, read from disk, primary first: what the editor's
+// transcript reader shows and switches between. The primary is the record's
+// transcript by the same rules the index reads it with — a local
+// transcription (transcript.json) over captions, captions by the caption-track
+// rule — and the rest are the alternates readTrackFields keeps. Null when the
+// dir holds no transcript.
+export async function readVideoTracks(
+ videoDir: string,
+): Promise<{ tracks: AltTrack[] } | null> {
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ let primary: AltTrack | null = null;
+ let kind: "vtt" | "whisper" = "vtt";
+ if (entries.includes(WHISPER_FILENAME)) {
+ try {
+ primary = {
+ track: TRANSCRIPTION_TRACK,
+ cues: parseTranscriptJson(await readFile(path.join(videoDir, WHISPER_FILENAME), "utf8")),
+ };
+ kind = "whisper";
+ } catch {
+ primary = null;
+ }
+ }
+ if (!primary) {
+ const read = await readEnglishVttCues(videoDir, entries);
+ if (!read) return null;
+ primary = { track: trackIdOfVtt(read.filename, entries), cues: read.cues };
+ }
+ const fields = await readTrackFields(videoDir, kind, primary.cues, entries);
+ return { tracks: [primary, ...(fields.altTracks ?? [])] };
+}
diff --git a/common/lib/captionTracks.test.ts b/common/lib/captionTracks.test.ts
@@ -0,0 +1,189 @@
+// The tracks of a transcript (lib/captionTracks.ts + captionTracks-server.ts):
+// labels from ids, which alternates are kept, and how a search finds a word
+// across them.
+//
+// Run with: node_modules/.bin/tsx --test common/lib/captionTracks.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ cuesOfTrack,
+ distinctAltTracks,
+ hitsAcrossTracks,
+ inTrackLabel,
+ recordTracks,
+ trackKind,
+ trackLabel,
+ trackLabels,
+ uncoveredAltHits,
+} from "./captionTracks";
+import { readTrackFields, readVideoTracks } from "./captionTracks-server";
+import { TRANSCRIPT_PIN_FILENAME } from "./videoStatus";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "caption-tracks-"));
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const cue = (start: number, text: string) => ({ start, end: start + 2, text });
+const vtt = (...cues: [number, string][]) =>
+ "WEBVTT\n\n" +
+ cues
+ .map(([s, t]) => {
+ const ts = (x: number) => `00:${String(Math.floor(x / 60)).padStart(2, "0")}:${String(x % 60).padStart(2, "0")}.000`;
+ return `${ts(s)} --> ${ts(s + 2)}\n${t}\n`;
+ })
+ .join("\n");
+
+let n = 0;
+function videoDir(files: Record<string, string>): string {
+ const dir = path.join(ROOT, `v${n++}`);
+ mkdirSync(dir, { recursive: true });
+ for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body);
+ return dir;
+}
+
+test("labels are plain words derived from the track id", () => {
+ assert.equal(trackLabel("en-orig"), "original audio captions");
+ assert.equal(trackLabel("en"), "uploaded captions");
+ assert.equal(trackLabel("en-en-US"), "auto-translated captions");
+ assert.equal(trackLabel("en-GB"), "UK English captions");
+ assert.equal(trackLabel("en-NG"), "regional captions (en-NG)");
+ assert.equal(trackLabel("en-US-orig"), "original audio captions (US)");
+ // A track YouTube named itself: another uploaded one.
+ assert.equal(trackLabel("en-uYU-mmqFLq8"), "other uploaded captions");
+ assert.equal(trackKind("en-JkeT_87f4cc"), "uploaded");
+ assert.deepEqual(trackLabels(["en-orig", "en-aa-1", "en-bb-2"]), [
+ "original audio captions",
+ "other uploaded captions (en-aa-1)",
+ "other uploaded captions (en-bb-2)",
+ ]);
+ assert.equal(trackLabel("transcription"), "transcription");
+ assert.equal(trackLabel("pinned"), "chosen captions");
+ assert.equal(inTrackLabel("en"), "in uploaded captions");
+ assert.equal(trackKind("en-US"), "regional");
+ assert.equal(trackKind("live_chat"), "other");
+});
+
+test("an alternate is kept only where its words differ from the primary and every kept one", () => {
+ const primary = [cue(0, "hello there")];
+ const kept = distinctAltTracks(primary, [
+ { track: "en", cues: [cue(0, "hello there")] }, // identical words
+ { track: "en-GB", cues: [cue(5, "hello there")] }, // timing alone differs
+ { track: "en-US", cues: [cue(0, "hello their")] }, // other words: kept
+ { track: "en-en-US", cues: [cue(0, "hello their")] }, // same as en-US
+ { track: "en-CA", cues: [] }, // empty
+ ]);
+ assert.deepEqual(kept.map((t) => t.track), ["en-US"]);
+});
+
+test("recordTracks and cuesOfTrack: primary first; an unknown track is undefined", () => {
+ const rec = {
+ cues: [cue(0, "a")],
+ track: "en-orig",
+ altTracks: [{ track: "en", cues: [cue(0, "b")] }],
+ };
+ assert.deepEqual(recordTracks(rec), ["en-orig", "en"]);
+ assert.equal(cuesOfTrack(rec, null)?.[0].text, "a");
+ assert.equal(cuesOfTrack(rec, "en-orig")?.[0].text, "a");
+ assert.equal(cuesOfTrack(rec, "en")?.[0].text, "b");
+ assert.equal(cuesOfTrack(rec, "en-GB"), undefined);
+ assert.deepEqual(recordTracks({}), []);
+});
+
+test("a search finds a word every track says once, in the primary, and an alternate's own word there", () => {
+ const rec = {
+ cues: [cue(10, "the bridge opened"), cue(300, "and then we left")],
+ track: "en-orig",
+ altTracks: [
+ {
+ track: "en",
+ cues: [cue(11, "the bridge opened in 1932"), cue(200, "a zeppelin flew over the bridge")],
+ },
+ ],
+ };
+ const find = (q: string) =>
+ hitsAcrossTracks(rec, (cues) => cues.filter((c) => c.text.includes(q)));
+ assert.deepEqual(
+ find("bridge").map((h) => [h.start, h.track]),
+ [
+ [10, undefined],
+ [200, "en"],
+ ],
+ );
+ assert.deepEqual(find("1932"), [{ ...cue(11, "the bridge opened in 1932"), track: "en" }]);
+ assert.deepEqual(find("nothing"), []);
+ // A record with no alternates is its primary alone.
+ assert.deepEqual(
+ hitsAcrossTracks({ cues: rec.cues }, (cues) => cues.filter((c) => c.text.includes("left"))),
+ [cue(300, "and then we left")],
+ );
+});
+
+test("uncoveredAltHits drops an alternate hit within the window of a primary one", () => {
+ assert.deepEqual(
+ uncoveredAltHits([{ start: 100 }, { start: 500 }], [{ start: 90 }, { start: 130 }, { start: 515 }, { start: 900 }]),
+ [{ start: 130 }, { start: 900 }],
+ );
+ assert.deepEqual(uncoveredAltHits([], [{ start: 1 }]), [{ start: 1 }]);
+});
+
+test("readTrackFields: a differing en beside en-orig is an alternate; identical tracks are none", async () => {
+ const differ = videoDir({
+ "transcript.en-orig.vtt": vtt([1, "said words"]),
+ "transcript.en.vtt": vtt([1, "uploaded words"]),
+ });
+ assert.deepEqual(await readTrackFields(differ, "vtt", undefined), {
+ track: "en-orig",
+ altTracks: [{ track: "en", cues: [{ start: 1, end: 3, text: "uploaded words" }] }],
+ });
+ const same = videoDir({
+ "transcript.en-orig.vtt": vtt([1, "said words"]),
+ "transcript.en.vtt": vtt([1, "said words"]),
+ });
+ assert.deepEqual(await readTrackFields(same, "vtt", undefined), {});
+ // One English VTT: nothing to read.
+ const lone = videoDir({ "transcript.en.vtt": vtt([1, "x"]) });
+ assert.deepEqual(await readTrackFields(lone, "vtt", undefined), {});
+});
+
+test("readTrackFields: an empty en-orig falls through to en as the primary, as the rule reads it", async () => {
+ const dir = videoDir({
+ "transcript.en-orig.vtt": "WEBVTT\n\n",
+ "transcript.en.vtt": vtt([1, "served words"]),
+ "transcript.en-GB.vtt": vtt([1, "british words"]),
+ });
+ assert.deepEqual(await readTrackFields(dir, "vtt", undefined), {
+ track: "en",
+ altTracks: [{ track: "en-GB", cues: [{ start: 1, end: 3, text: "british words" }] }],
+ });
+});
+
+test("readTrackFields: the operator's pin names the primary `pinned`; its source is not repeated", async () => {
+ const dir = videoDir({
+ "transcript.en-orig.vtt": vtt([1, "said words"]),
+ "transcript.en.vtt": vtt([1, "said words"]), // the copy of en-orig the pin made
+ "transcript.en-US.vtt": vtt([1, "other words"]),
+ [TRANSCRIPT_PIN_FILENAME]: JSON.stringify({ from: "transcript.en-orig.vtt", pinnedAt: "" }),
+ });
+ assert.deepEqual(await readTrackFields(dir, "vtt", undefined), {
+ track: "pinned",
+ altTracks: [{ track: "en-US", cues: [{ start: 1, end: 3, text: "other words" }] }],
+ });
+});
+
+test("readVideoTracks: a transcription is the primary and its captions the alternate", async () => {
+ const dir = videoDir({
+ "transcript.json": JSON.stringify({
+ transcription: [{ offsets: { from: 0, to: 1000 }, text: " machine words" }],
+ }),
+ "transcript.en-orig.vtt": vtt([1, "caption words"]),
+ });
+ const read = await readVideoTracks(dir);
+ assert.deepEqual(read?.tracks.map((t) => [t.track, t.cues[0].text]), [
+ ["transcription", "machine words"],
+ ["en-orig", "caption words"],
+ ]);
+ assert.equal(await readVideoTracks(videoDir({ "metadata.info.json": "{}" })), null);
+});
diff --git a/common/lib/captionTracks.ts b/common/lib/captionTracks.ts
@@ -0,0 +1,225 @@
+// THE TRACKS OF A TRANSCRIPT — one notion of "track" for every reader.
+//
+// A record's transcript is read from ONE track, its primary, chosen by the
+// caption-track rule (lib/videoStatus.ts: en-orig first, the operator's pin
+// above all, a local transcription above captions). A record may hold other
+// English tracks beside it: the served `en`, a regional en-GB, an en→en
+// auto-translation, or the captions a local transcription replaced. Human
+// captions are not always a transcript of what was said, so those stay
+// readable and searchable as ALTERNATES — a viewer's choice, never a change to
+// the primary (the pin, transcript-pin.json, is how the primary changes).
+//
+// An alternate is kept only where its words differ from the primary's and from
+// every alternate kept before it: most served `en` tracks are byte-identical to
+// en-orig, and shipping or indexing those would double the corpus for nothing.
+//
+// Track ids are the caption's language code as its file names it
+// (transcript.<id>.vtt), plus two that no file names: `transcription` (a local
+// whisper/parakeet transcript) and `pinned` (transcript.en.vtt while the
+// operator's pin stands — a copy of whichever track was picked). Labels are
+// derived from the id, here and nowhere else, so the editor, the export site,
+// a search hit and the MCP all say the same words.
+//
+// Pure: no node built-ins, safe in a client bundle.
+
+import type { Cue } from "./vtt";
+
+export const TRANSCRIPTION_TRACK = "transcription";
+export const PINNED_TRACK = "pinned";
+
+export type AltTrack = { track: string; cues: Cue[] };
+
+// What a transcript record carries about its tracks. Both fields are OMITTED
+// when the record has no alternate that differs (the common case), so those
+// records' pages stay byte-identical to the ones already on disk.
+export type TrackFields = {
+ // The primary's track id.
+ track?: string;
+ // The English tracks whose words differ from the primary's, in preference
+ // order.
+ altTracks?: AltTrack[];
+};
+
+const REGIONS: Record<string, string> = {
+ US: "US",
+ GB: "UK",
+ UK: "UK",
+ CA: "Canadian",
+ AU: "Australian",
+ IE: "Irish",
+ IN: "Indian",
+ NZ: "New Zealand",
+ ZA: "South African",
+};
+
+// The kind of a track, from its id.
+export type TrackKind =
+ | "original"
+ | "uploaded"
+ | "regional"
+ | "translated"
+ | "transcription"
+ | "pinned"
+ | "other";
+
+// A region subtag as YouTube writes one: two letters (en-US) or three digits
+// (en-419). Anything else after `en-` is a track YouTube named itself — an
+// uploader's extra caption track (en-uYU-mmqFLq8).
+const REGION_RE = /^(?:[A-Za-z]{2}|\d{3})$/;
+
+export function trackKind(track: string): TrackKind {
+ if (track === TRANSCRIPTION_TRACK) return "transcription";
+ if (track === PINNED_TRACK) return "pinned";
+ if (track === "en-orig" || /^en-[^-]+-orig$/.test(track)) return "original";
+ if (track === "en") return "uploaded";
+ if (/^en-en(?:-|$)/.test(track)) return "translated";
+ const m = track.match(/^en-(.+)$/);
+ if (m) return REGION_RE.test(m[1]) ? "regional" : "uploaded";
+ return "other";
+}
+
+// "UK" for en-GB, the code itself for a region this table does not name.
+function regionName(code: string): string {
+ return REGIONS[code.toUpperCase()] ?? code;
+}
+
+// A plain label for a track — what a person reads in the switcher and on a
+// search hit ("in uploaded captions").
+export function trackLabel(track: string): string {
+ switch (trackKind(track)) {
+ case "original": {
+ // en-US-orig: the original audio's captions, under a region.
+ const m = track.match(/^en-([^-]+)-orig$/);
+ return m ? `original audio captions (${regionName(m[1])})` : "original audio captions";
+ }
+ case "uploaded":
+ // A track YouTube named (en-uYU-mmqFLq8) is another uploaded one.
+ return track === "en" ? "uploaded captions" : "other uploaded captions";
+ case "translated":
+ return "auto-translated captions";
+ case "transcription":
+ return "transcription";
+ case "pinned":
+ return "chosen captions";
+ case "regional": {
+ const name = REGIONS[track.slice(3).toUpperCase()];
+ return name ? `${name} English captions` : `regional captions (${track})`;
+ }
+ default:
+ return `captions (${track})`;
+ }
+}
+
+// The labels of a switcher's tracks, in order: trackLabel, with the id added
+// where two tracks would otherwise read the same (two tracks YouTube named).
+export function trackLabels(tracks: readonly string[]): string[] {
+ const labels = tracks.map(trackLabel);
+ return labels.map((l, i) =>
+ labels.indexOf(l) !== labels.lastIndexOf(l) ? `${l} (${tracks[i]})` : l,
+ );
+}
+
+// The track id of a caption file name (transcript.<id>.vtt), or null.
+export function trackOfVttFile(filename: string): string | null {
+ const m = filename.match(/^transcript\.([^.]+)\.vtt$/);
+ return m ? m[1] : null;
+}
+
+// A cue list's words, for "do these two tracks say the same thing" — timing
+// alone is not a different transcript.
+export function cueWords(cues: readonly Cue[]): string {
+ return cues.map((c) => c.text).join("\n");
+}
+
+// The alternates worth keeping: tracks with at least one cue whose words differ
+// from the primary's and from every track kept before them. `candidates` in
+// preference order; the primary is not among them.
+export function distinctAltTracks(
+ primary: readonly Cue[] | undefined,
+ candidates: readonly AltTrack[],
+): AltTrack[] {
+ const seen = new Set<string>([cueWords(primary ?? [])]);
+ const out: AltTrack[] = [];
+ for (const c of candidates) {
+ if (c.cues.length === 0) continue;
+ const words = cueWords(c.cues);
+ if (seen.has(words)) continue;
+ seen.add(words);
+ out.push(c);
+ }
+ return out;
+}
+
+// The tracks a record offers, primary first: what a switcher lists. A record
+// with no alternates offers just its primary (or nothing, with no track id).
+export function recordTracks(rec: TrackFields): string[] {
+ if (!rec.altTracks || rec.altTracks.length === 0) return rec.track ? [rec.track] : [];
+ return [rec.track ?? "", ...rec.altTracks.map((t) => t.track)].filter((t) => t !== "");
+}
+
+// The cues of one track of a record: the primary when `track` is absent or
+// names it, an alternate when it names one, undefined when the record has no
+// such track.
+export function cuesOfTrack<R extends TrackFields & { cues?: Cue[] | undefined }>(
+ rec: R,
+ track: string | null | undefined,
+): Cue[] | undefined {
+ if (!track || track === rec.track) return rec.cues;
+ return rec.altTracks?.find((t) => t.track === track)?.cues;
+}
+
+// How far (seconds) a primary hit covers an alternate's. An alternate hit with
+// a primary hit this close is the same moment found twice — the primary's is
+// the one shown. Only a match the primary has nowhere near is the alternate's
+// to report.
+export const ALT_HIT_COVERED_SEC = 20;
+
+// Drop the alternate hits a primary hit already covers (ALT_HIT_COVERED_SEC).
+// Both lists carry `start` in seconds; the primary's needs no order.
+export function uncoveredAltHits<H extends { start: number }>(
+ primary: readonly { start: number }[],
+ alt: readonly H[],
+): H[] {
+ if (primary.length === 0) return [...alt];
+ const starts = primary.map((h) => h.start).sort((a, b) => a - b);
+ return alt.filter((h) => {
+ // Binary search for the nearest primary start.
+ let lo = 0;
+ let hi = starts.length - 1;
+ while (lo < hi) {
+ const mid = (lo + hi) >> 1;
+ if (starts[mid] < h.start) lo = mid + 1;
+ else hi = mid;
+ }
+ const near = [starts[lo], starts[lo - 1]].filter((s) => s !== undefined);
+ return !near.some((s) => Math.abs(s - h.start) <= ALT_HIT_COVERED_SEC);
+ });
+}
+
+// THE ONE RULE for matching a record's words across its tracks, used by every
+// search (the viewer's leaf pipeline, the MCP, the query-tree evaluator):
+// `find` is run over the primary's cues, then over each alternate's, and an
+// alternate's hit is kept only where no hit already kept is within
+// ALT_HIT_COVERED_SEC — so a word every track says is found once, in the
+// primary, and a word only an alternate says is found there, wearing that
+// alternate's track id. Ordered by start when an alternate adds anything.
+export function hitsAcrossTracks<H extends { start: number }>(
+ rec: TrackFields & { cues?: readonly Cue[] | undefined },
+ find: (cues: readonly Cue[]) => H[],
+): (H & { track?: string })[] {
+ const out: (H & { track?: string })[] = rec.cues ? find(rec.cues) : [];
+ let added = false;
+ for (const alt of rec.altTracks ?? []) {
+ for (const h of uncoveredAltHits(out, find(alt.cues))) {
+ out.push({ ...h, track: alt.track });
+ added = true;
+ }
+ }
+ if (added) out.sort((a, b) => a.start - b.start);
+ return out;
+}
+
+// "in uploaded captions" — what a hit from an alternate says about itself.
+export function inTrackLabel(track: string): string {
+ return `in ${trackLabel(track)}`;
+}
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -80,7 +80,10 @@ const SHARD_SCHEME = {
"<channel.manifests.transcripts> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }",
transcriptPage:
"/transcripts/<slug>/page-<NNNN>.json -> array of { id, title, uploadDate, " +
- "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }",
+ "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }; " +
+ "a record with other English caption tracks whose words differ from its transcript " +
+ "also carries `track` (the transcript's track id, e.g. \"en-orig\") and " +
+ "`altTracks: [{ track, cues }]` (e.g. \"en\", the uploaded captions) — absent otherwise",
subsManifest:
"<channel.manifests.subs> -> lighter list-view records under the same slugToPage scheme",
summariesIndex:
diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts
@@ -68,6 +68,10 @@ export const METADATA_HISTORY_WRITERS = [
// metadata API (controller/archiveOrgDownload.ts) — archive.org records are
// not fetched by yt-dlp.
"archiveorg-import",
+ // A Wayback Machine copy's title and date, set from what the operator found
+ // for it (`archilyzer wayback refresh --titles`, controller/waybackRefresh.ts)
+ // — a raw media file captured by the Wayback Machine carries no title.
+ "wayback-provenance",
] as const;
export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -19,6 +19,7 @@
// report-citation UI, and build tools alike.
import { detectPlatform, type Platform } from "./platform";
+import { parseWaybackUrl } from "./wayback";
export type MomentUrlInput = {
// Public origin of the archilyzer viewer that owns this video (a RemoteSource
@@ -55,6 +56,9 @@ export function platformMomentUrl(
if (!webpageUrl) return null;
const secs = Math.max(0, Math.floor(seconds || 0));
if (secs <= 0) return webpageUrl;
+ // A Wayback capture (lib/wayback.ts) is the page that plays, and a time
+ // param would name a different URL — one the Wayback Machine never captured.
+ if (parseWaybackUrl(webpageUrl)) return webpageUrl;
const plat = platform ?? detectPlatform(webpageUrl);
let u: URL;
try {
diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts
@@ -140,11 +140,13 @@ export type RecordView = {
originalUrl?: string;
// What `originalUrl` is, when "Original" would not say: "archive.org" for an
// archive.org record, "YouTube" for an archive.org mirror of a YouTube
- // upload (whose originalUrl is the upload at the cited second).
+ // upload (whose originalUrl is the upload at the cited second), "Original
+ // (may be gone)" for a Wayback Machine copy (lib/wayback.ts).
originalLabel?: string;
// Where a reader can fetch the recording itself to check it, derived from
// the record's provenance (lib/archiveOrg.ts archiveOrgCitationLinks): the
- // archive.org page and the item's torrent. Absent for a record with none.
+ // archive.org page and the item's torrent; a Wayback copy's capture page
+ // (lib/wayback.ts waybackCitationLinks). Absent for a record with none.
downloads?: { label: string; url: string }[];
// The record in this site's corpus (`/?v=<channel>/<id>&t=<s>`): a FULL site
// only — a cited site has no corpus to open.
diff --git a/common/lib/search/evalTree.test.ts b/common/lib/search/evalTree.test.ts
@@ -374,3 +374,22 @@ test("a non-empty curated-tag selection makes a filter selective", () => {
// a tag-only query would plan a full scan.
assert.equal(filterIsSelective(openFilters({ curatedTags: ["eva-collab"] })), true);
});
+
+test("a transcripts leaf reads the record's alternate tracks; a hit only one holds wears its track", () => {
+ const r = evalLeaf(
+ newLeaf({ id: "l", query: "zeppelin" }),
+ { scope: "transcripts", test: (t) => t.includes("zeppelin") || t.includes("bridge") },
+ ctx({
+ cues: [cue(3, "the bridge")],
+ altTracks: [{ track: "en", cues: [cue(4, "the bridge"), cue(100, "a zeppelin")] }],
+ }),
+ );
+ assert.equal(r.count, 2, "the bridge once (primary), the zeppelin once (en)");
+ assert.deepEqual(
+ r.hits.map((h) => [h.seconds, h.track]),
+ [
+ [3, undefined],
+ [100, "en"],
+ ],
+ );
+});
diff --git a/common/lib/search/evalTree.ts b/common/lib/search/evalTree.ts
@@ -32,6 +32,7 @@ import {
import { matchAliases, type SearchAlias } from "../searchAliases";
import { VIDEO_STATES, type VideoState } from "../availability";
import type { Cue } from "../vtt";
+import { hitsAcrossTracks, type AltTrack } from "../captionTracks";
import { clock, truncate, type Matcher } from "./window";
export type { Matcher };
@@ -131,6 +132,9 @@ export type RecordCtx = {
description: string;
tags: string;
cues: Cue[];
+ // The record's alternate English tracks (lib/captionTracks.ts), when it has
+ // any: the transcripts scope reads them too.
+ altTracks?: AltTrack[];
chatCues: Cue[];
// The post body, when this record IS a post rather than a video. Empty for a
// video record, so a posts-scope leaf never matches one.
@@ -159,11 +163,16 @@ export function evalLeaf(
let count = 0;
switch (m.scope) {
case "transcripts":
- for (const cue of ctx.cues) {
- if (!m.test(cue.text)) continue;
+ // Every English track of the record (lib/captionTracks.ts): a match only
+ // an alternate holds wears that alternate's track id.
+ for (const cue of hitsAcrossTracks(
+ { cues: ctx.cues, altTracks: ctx.altTracks },
+ (cues) => cues.filter((c) => m.test(c.text)),
+ )) {
count++;
push({
scope: "transcripts",
+ ...(cue.track ? { track: cue.track } : {}),
clock: clock(cue.start),
seconds: cue.start,
text: truncate(cue.text, max),
diff --git a/common/lib/search/leafPipeline.test.ts b/common/lib/search/leafPipeline.test.ts
@@ -284,3 +284,28 @@ test("a regex leaf compiles once; a malformed pattern matches nothing", async ()
const bad = await run(newLeaf({ query: "n([edle", useRegex: true }), ["c/v1"], f);
assert.deepEqual(bad.slugs, []);
});
+
+test("a transcripts leaf reads a record's alternate tracks too, and names the track of a hit only one holds", async () => {
+ const rec = {
+ ...record("v1", [{ start: 3, text: "the harbor bridge" }]),
+ track: "en-orig",
+ altTracks: [
+ {
+ track: "en",
+ cues: [
+ { start: 4, end: 6, text: "the harbor bridge" },
+ { start: 100, end: 102, text: "a zeppelin overhead" },
+ ],
+ },
+ ],
+ } as TranscriptDetail;
+ const a = archive({ pages: [[rec, record("v2", [{ start: 0, text: "nothing" }])]] });
+ const only = await run(newLeaf({ query: "zeppelin" }), ["c/v1", "c/v2"], fetchersFor(a));
+ assert.deepEqual(only.slugs, ["c/v1"]);
+ const hits = only.hits.get("c/v1") as { start: number; track?: string; scope: string }[];
+ assert.deepEqual(hits.map((h) => [h.start, h.track, h.scope]), [[100, "en", "transcripts"]]);
+ // A word both tracks say near the same moment: one hit, the primary's.
+ const both = await run(newLeaf({ query: "harbor" }), ["c/v1"], fetchersFor(a));
+ const bh = both.hits.get("c/v1") as { track?: string }[];
+ assert.deepEqual(bh.map((h) => h.track), [undefined]);
+});
diff --git a/common/lib/search/leafPipeline.ts b/common/lib/search/leafPipeline.ts
@@ -25,6 +25,7 @@ import type { SubsDetail } from "../subs";
import type { Post } from "../posts";
import type { LayerScope, LeafNode } from "../searchQuery";
import { findHitsInCues, findHitsInText, type Hit, type SubsHit } from "./window";
+import { hitsAcrossTracks } from "../captionTracks";
export type { Hit, SubsHit };
@@ -163,9 +164,10 @@ export function createSearchPipeline(
useRegex,
regex,
)
- : full.cues
- ? findHitsInCues(full.cues, query, useRegex, regex, remaining)
- : [];
+ : // Every English track of the record (lib/captionTracks.ts).
+ hitsAcrossTracks(full, (cues) =>
+ findHitsInCues(cues, query, useRegex, regex, remaining),
+ );
if (hits.length > 0) {
localHits[slug] = hits;
totalSoFar += hits.length;
@@ -665,6 +667,7 @@ export function runLeafPipeline(opts: LeafPipelineOptions): LeafController {
list.map((h) => ({
leafId: leaf.id,
scope: leaf.scope,
+ ...(h.track ? { track: h.track } : {}),
start: h.start,
text: h.text,
})),
diff --git a/common/lib/search/window.ts b/common/lib/search/window.ts
@@ -110,7 +110,9 @@ export function windowedTranscript(
// — on the cue it starts in — and the emitted text widens to the window only
// in that case. Everything else emits the matched cue verbatim.
-export type Hit = { start: number; text: string };
+// `track`: a transcript hit from one of the record's ALTERNATE English tracks
+// (lib/captionTracks.ts, hitsAcrossTracks). Absent for a hit in the primary.
+export type Hit = { start: number; text: string; track?: string };
export type SubsHit = { track: string; start: number; text: string };
diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts
@@ -11,6 +11,8 @@ import "./attribution-server";
import "./diarization-server";
import "./digest-server";
import "./metadataHistory-server";
+import "./transcriptPin-server";
+import "./wayback-server";
import {
availabilitySidecar,
loadAvailability,
@@ -42,7 +44,7 @@ async function scratch(): Promise<string> {
return mkdtemp(path.join(os.tmpdir(), "sidecar-"));
}
-test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are declared", () => {
+test("every declared sidecar filename escapes SUB_FILE_RE, and all twelve are declared", () => {
assert.deepEqual([...SIDECAR_FILENAMES].sort(), [
"ai-digest.json",
"ai-digest.overrides.json",
@@ -55,6 +57,8 @@ test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are de
"exclude-truncated-check.json",
"metadata.history.json",
"transcribe-outcome.json",
+ "transcript-pin.json",
+ "wayback.json",
]);
for (const name of SIDECAR_FILENAMES) {
assert.ok(!SUB_FILE_RE.test(name), name);
diff --git a/common/lib/subtitleProvenance.ts b/common/lib/subtitleProvenance.ts
@@ -1,7 +1,7 @@
import path from "node:path";
import { createReadStream } from "node:fs";
import { readFile } from "node:fs/promises";
-import type { VideoFiles } from "./videoStatus";
+import { englishVttsByPreference, type VideoFiles } from "./videoStatus";
// Where a VTT transcript came from: YouTube's speech recognition ("asr", what
// yt-dlp downloads under --write-auto-subs) or a human-authored/uploaded track
@@ -140,6 +140,27 @@ export async function resolveVttProvenance(
}
}
+// Where a video's English CAPTIONS came from, taken over every English VTT and
+// not just the one the caption-track rule reads: "asr" only when each of them
+// is, "manual" when any is, else "unknown". The rule prefers the original-audio
+// ASR track (en-orig) even beside a human `en`, so asking only the track it
+// reads would call a video with human captions ASR-only and schedule them for
+// replacement. Single-track dirs pay the one sniff they always did.
+export async function resolveCaptionsProvenance(
+ videoDir: string,
+ entries: readonly string[],
+): Promise<SubtitleProvenance | null> {
+ const vtts = englishVttsByPreference(entries);
+ if (vtts.length === 0) return null;
+ let unknown = false;
+ for (const name of vtts) {
+ const p = await resolveVttProvenance(videoDir, name);
+ if (p === "manual") return "manual";
+ if (p === "unknown") unknown = true;
+ }
+ return unknown ? "unknown" : "asr";
+}
+
// True when this video's ONLY transcript is YouTube ASR — the work-lane
// candidate rule, shared by the snapshot buckets, the whisper gate
// (transcribeOneFromQueue) and the auto-runner's download override so all four
@@ -152,5 +173,5 @@ export async function isAutoSubsOnly(
if (!files.ytVttFile || files.hasWhisper || files.isUntranscribable) {
return false;
}
- return (await resolveVttProvenance(videoDir, files.ytVttFile)) === "asr";
+ return (await resolveCaptionsProvenance(videoDir, files.entries)) === "asr";
}
diff --git a/common/lib/transcriptPin-server.ts b/common/lib/transcriptPin-server.ts
@@ -0,0 +1,39 @@
+// THE OPERATOR'S CAPTION-TRACK PICK. The editor's "Set as transcript" copies
+// the chosen track to transcript.en.vtt and writes this sidecar beside it;
+// while it is present, the caption-track rule ranks transcript.en.vtt first
+// (lib/videoStatus.ts, englishVttsByPreference) — above en-orig, which the rule
+// otherwise prefers. Only the file's PRESENCE is read by the rule; the record
+// says which track was copied and when, for a person reading the dir.
+//
+// SERVER-ONLY (node:fs).
+
+import { TRANSCRIPT_PIN_FILENAME } from "./videoStatus";
+import { sidecar, sidecarField } from "./sidecar-server";
+
+export type TranscriptPinRecord = {
+ // The track the operator picked (copied to transcript.en.vtt).
+ from: string;
+ pinnedAt: string;
+};
+
+// Any parseable record still pins: the rule reads presence, so the record's
+// shape must never make a pin read as absent here and present there.
+export function coerceTranscriptPin(value: unknown): TranscriptPinRecord {
+ const v = value as Partial<TranscriptPinRecord> | null;
+ return {
+ from: typeof v?.from === "string" ? v.from : "",
+ pinnedAt: typeof v?.pinnedAt === "string" ? v.pinnedAt : "",
+ };
+}
+
+export const transcriptPinSidecar = sidecar(
+ TRANSCRIPT_PIN_FILENAME,
+ sidecarField(coerceTranscriptPin),
+);
+
+export async function pinTranscript(videoDir: string, from: string): Promise<void> {
+ await transcriptPinSidecar.write(videoDir, {
+ from,
+ pinnedAt: new Date().toISOString(),
+ });
+}
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -5,6 +5,8 @@ import { defaultWebpageUrl, detectPlatform } from "./platform";
import { archiveOrgPlayableUrl } from "./archiveOrg";
import { bitchutePlayableUrl } from "./bitchute";
import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId";
+import { parseWaybackUrl } from "./wayback";
+import { extractVideoId } from "./videoId";
import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
import type { MediaType, VideoStat, VideoStatus } from "./stats";
import type { VideoState } from "./availability";
@@ -77,7 +79,8 @@ export function loadRawMetadataFromDir(
// The platform a record is from, by yt-dlp's extractor. archive.org's
// extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's
-// `YoutubeWebArchive`, which is a YouTube video.
+// `YoutubeWebArchive`, which is a YouTube video: the original's platform (the
+// record's `wayback.json` says it is a copy, lib/wayback-server.ts).
//
// AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the
// app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has
@@ -144,12 +147,17 @@ export function summarize(
const platform = platformFromMetadata(meta);
// archive.org: yt-dlp's id for one file of an item is `<identifier>/<path>`,
// which is not a slug; the canonical id (lib/archiveOrgId.ts) is.
+ // A Wayback capture (lib/wayback.ts): the id its dir is named by — what
+ // the capture is of (lib/videoId.ts) — not yt-dlp's, which for a raw media
+ // file is the file's name (`<jwId>-<rendition>`).
+ const waybackId = parseWaybackUrl(meta.webpage_url) ? extractVideoId(meta.webpage_url!) : null;
const id =
- platform === "odysee"
+ waybackId ??
+ (platform === "odysee"
? (meta.webpage_url_basename ?? meta.id ?? videoDir)
: platform === "archiveorg"
? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir)
- : (meta.id ?? videoDir);
+ : (meta.id ?? videoDir));
const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1];
return {
slug: `${channelSlug}/${id}`,
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -1,5 +1,6 @@
import path from "node:path";
import type { Cue } from "./vtt";
+import type { TrackFields } from "./captionTracks";
import { readFile } from "fs-extra";
import { getPaths } from "./paths";
@@ -69,9 +70,12 @@ export type DisplaySummary = {
curatedTags?: string[];
};
+// `track` / `altTracks` (lib/captionTracks.ts): the primary's track id and the
+// other English tracks whose words differ from it — present only on a record
+// that has such a track.
export type TranscriptDetail = TranscriptSummary & {
cues: Cue[] | undefined;
-};
+} & TrackFields;
export type TranscriptPage = TranscriptDetail[];
diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts
@@ -10,11 +10,31 @@
// for the native-id resolution that reads metadata.info.json.
import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId";
+import { isJwPlayerHost, isWaybackHost, jwPlayerMediaId, parseWaybackUrl } from "./wayback";
export function extractVideoId(url: string): string | null {
+ return extractVideoIdAt(url, 0);
+}
+
+function extractVideoIdAt(url: string, depth: number): string | null {
try {
const u = new URL(url);
const host = u.hostname.toLowerCase();
+ if (isWaybackHost(host)) {
+ // A Wayback capture is named by what it is a capture OF
+ // (lib/wayback.ts): an archived YouTube page by its YouTube id, a JW
+ // Player file by its media id. Its own path's last segment is the
+ // original's (`watch`, `<id>-<rendition>.mp4`) — a name two captures
+ // share. A capture of a capture is not unwrapped twice.
+ const ref = depth === 0 ? parseWaybackUrl(url) : null;
+ return ref ? extractVideoIdAt(ref.originalUrl, depth + 1) : null;
+ }
+ if (isJwPlayerHost(host)) {
+ // One media id across every rendition and host; a JW URL that names
+ // none falls through to the last segment below.
+ const jw = jwPlayerMediaId(u);
+ if (jw) return jw;
+ }
if (isArchiveOrgItemHost(host)) {
// A whole item → its identifier; one file inside an item → a stable
// `<identifier>__<slug>-<hash>` (lib/archiveOrgId.ts). A URL that names
diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts
@@ -1,12 +1,14 @@
import path from "node:path";
import { readdir, readFile, stat } from "node:fs/promises";
import { isPartAudioFile, isRealAudioFile } from "./mediaFiles";
+import { parseVtt, type Cue } from "./vtt";
export type VideoFiles = {
hasMeta: boolean;
hasYtVtt: boolean;
- // The resolved primary English VTT filename (transcript.en.vtt when present,
- // otherwise the best regional/auto English track — see resolvePrimaryVtt).
+ // The resolved primary English VTT filename by name (transcript.en-orig.vtt,
+ // else transcript.en.vtt, else the best regional/auto English track — see
+ // resolvePrimaryVtt; the cues may come from a later one, readEnglishVttCues).
// Null when no English VTT exists. hasYtVtt === (ytVttFile !== null).
ytVttFile: string | null;
// True when a transcript.<lang>.vtt exists under a non-canonical name (i.e.
@@ -40,8 +42,9 @@ export type VideoFiles = {
export type IndexTranscript =
| { kind: "whisper"; filename: "transcript.json" }
- // filename is usually "transcript.en.vtt" but may be a regional/auto English
- // track (e.g. "transcript.en-US.vtt") when YouTube served no plain `en` track.
+ // filename is the most preferred English track by name (resolvePrimaryVtt):
+ // transcript.en-orig.vtt, transcript.en.vtt, or a regional/auto one such as
+ // transcript.en-US.vtt. The cues are read with readEnglishVttCues.
| { kind: "vtt"; filename: string };
export const VTT_FILENAME = "transcript.en.vtt";
@@ -79,9 +82,9 @@ export type SubTrack = {
ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3";
};
-// Treat any transcript.<x>.<y> file as a sub track when it isn't one of the
-// primary transcript outputs (transcript.en.vtt, transcript.json) or a
-// derived/auxiliary file (transcript.cues.json). Live chat lands as
+// Treat any transcript.<x>.<y> file as a sub track when it isn't a transcript
+// (transcript.json, or ANY English VTT — the primary and its alternate tracks,
+// lib/captionTracks.ts) or a derived/auxiliary file (transcript.cues.json). Live chat lands as
// transcript.live_chat.json; non-en languages as transcript.<lang>.vtt.
// Exported for lib/sidecar-server.ts, which refuses to declare a sidecar this
// matches (a sidecar so named would be read as a subtitle track).
@@ -92,18 +95,50 @@ function isSubExt(value: string): value is SubExt {
return (SUB_EXT_VALUES as readonly string[]).includes(value);
}
-// Resolve the primary English transcript VTT in a video dir. Normally this is
-// the canonical transcript.en.vtt, but YouTube sometimes serves a video's
-// English captions only under regional/auto codes (transcript.en-US.vtt,
-// transcript.en-en-US.vtt, transcript.en-orig.vtt) with no plain `en` track. A
-// file counts as English iff the FIRST segment of its language code is `en`, so
-// translations like transcript.ab-en-US.vtt / transcript.es-en-US.vtt are
-// excluded. Preference: en (canonical) > en-orig (original audio) >
-// regional/manual en-US,en-GB,… > auto-translated en-en-* variants.
+// THE CAPTION-TRACK RULE — which English VTT a video's transcript is read from.
+// It lives here and nowhere else; every reader that turns captions into cues
+// (the index, normalize, report compose) goes through englishVttsByPreference /
+// readEnglishVttCues, and the MCP, the export and report-to-video read what
+// those wrote.
+//
+// A file counts as English iff the FIRST segment of its language code is `en`,
+// so translations like transcript.ab-en-US.vtt / transcript.es-en-US.vtt are
+// excluded. Preference:
+//
+// en-orig the captions of the ORIGINAL audio — the speaker's words.
+// en canonical. Usually the same text as en-orig, but for some videos
+// YouTube serves a rewritten/translated `en` that changes facts
+// (a date, a word), and for some livestream VODs one in a cue-block
+// shape with no word timing.
+// en-US, en-GB, … regional/manual.
+// en-en-* auto-translated en→en variants.
+//
+// AN OPERATOR'S PICK BEATS ALL OF IT. The editor's "Set as transcript" copies
+// the chosen track to transcript.en.vtt and writes TRANSCRIPT_PIN_FILENAME
+// beside it (lib/transcriptPin-server.ts); while that file is present,
+// transcript.en.vtt ranks first. Only its presence is read — the rule stays a
+// function of the listing.
+//
+// Name order is half the rule. The other half is CONTENT: the transcript is the
+// first track in this order that parses to at least one cue (readEnglishVttCues),
+// so a track that is present but empty never hides one that has text.
+// CAPTION_TRACK_RULE_VERSION names this rule; bump it when the order or the
+// fallback changes, and the next index build re-reads the records the change
+// can reach (buildIndex.ts, "Caption track"), and a cues.json normalized under
+// an older one reads as stale where it could differ (normalizeTranscript.ts).
+// umtool/report-to-video/cues.mjs carries a copy (plain node, no tsx); change
+// one, change the other — captionTrack.test.ts holds them equal.
+export const CAPTION_TRACK_RULE_VERSION = 1;
+
+export const ORIG_VTT_FILENAME = "transcript.en-orig.vtt";
+// Not transcript.<x>.<y>: SUB_FILE_RE would read it as a subtitle track.
+export const TRANSCRIPT_PIN_FILENAME = "transcript-pin.json";
+
const EN_VTT_RE = /^transcript\.(en(?:-[^.]+)?)\.vtt$/;
-function englishVttRank(track: string): number {
- if (track === "en") return 0;
- if (track === "en-orig") return 1;
+function englishVttRank(track: string, pinned: boolean): number {
+ if (track === "en" && pinned) return -1;
+ if (track === "en-orig") return 0;
+ if (track === "en") return 1;
if (/^en-en(?:-|$)/.test(track)) return 3; // auto-translated en→en variants
return 2; // regional/manual en-US, en-GB, …
}
@@ -115,15 +150,63 @@ export function isEnglishVtt(name: string): boolean {
return EN_VTT_RE.test(name);
}
-export function resolvePrimaryVtt(entries: string[]): string | null {
- let best: { name: string; rank: number } | null = null;
+// Every English VTT in a listing, most preferred first (ties by name, so the
+// order never depends on readdir's).
+export function englishVttsByPreference(entries: readonly string[]): string[] {
+ const pinned = entries.includes(TRANSCRIPT_PIN_FILENAME);
+ const ranked: { name: string; rank: number }[] = [];
for (const e of entries) {
const m = e.match(EN_VTT_RE);
- if (!m) continue;
- const rank = englishVttRank(m[1]);
- if (!best || rank < best.rank) best = { name: e, rank };
+ if (m) ranked.push({ name: e, rank: englishVttRank(m[1], pinned) });
+ }
+ ranked.sort((a, b) => a.rank - b.rank || (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
+ return ranked.map((r) => r.name);
+}
+
+// The most preferred English VTT BY NAME — no file is read. This is the
+// track the index stats for change detection and the one the editor labels the
+// primary; the cues themselves come from readEnglishVttCues, which moves past
+// it when it parses to nothing.
+export function resolvePrimaryVtt(entries: readonly string[]): string | null {
+ return englishVttsByPreference(entries)[0] ?? null;
+}
+
+// The files a caption transcript is derived from: every English VTT (the
+// content fallback can reach any of them) and the operator's pin. Their newest
+// mtime is what a cues.json, or an index record, is compared against.
+export function captionInputs(entries: readonly string[]): string[] {
+ const out = englishVttsByPreference(entries);
+ if (out.length > 0 && entries.includes(TRANSCRIPT_PIN_FILENAME)) {
+ out.push(TRANSCRIPT_PIN_FILENAME);
+ }
+ return out;
+}
+
+// The cues of a video's caption transcript: the first English VTT, in
+// preference order, that parses to at least one cue. When every track parses
+// to nothing, the most preferred one is returned with its empty list (the
+// video has captions, they say nothing). Null when there is no English VTT, or
+// none can be read. `entries` is the caller's readdir of `videoDir`, when it
+// has one.
+export async function readEnglishVttCues(
+ videoDir: string,
+ entries?: readonly string[],
+): Promise<{ filename: string; cues: Cue[] } | null> {
+ const names = englishVttsByPreference(
+ entries ?? (await readdir(videoDir).catch(() => [] as string[])),
+ );
+ let first: { filename: string; cues: Cue[] } | null = null;
+ for (const filename of names) {
+ let cues: Cue[];
+ try {
+ cues = parseVtt(await readFile(path.join(videoDir, filename), "utf8"));
+ } catch {
+ continue;
+ }
+ if (cues.length > 0) return { filename, cues };
+ first ??= { filename, cues };
}
- return best?.name ?? null;
+ return first;
}
// Any transcript.<lang>.vtt file (any language code). These are the candidate
@@ -191,13 +274,14 @@ export async function readVideoFiles(
export async function readSubTracks(videoDir: string): Promise<SubTrack[]> {
const entries = await readdir(videoDir).catch(() => [] as string[]);
- const primaryVtt = resolvePrimaryVtt(entries);
const tracks: SubTrack[] = [];
for (const entry of entries) {
- if (entry === VTT_FILENAME || entry === WHISPER_FILENAME) continue;
- // The resolved primary English VTT (e.g. transcript.en-US.vtt when there's
- // no transcript.en.vtt) is the main transcript, not an alternate sub-track.
- if (entry === primaryVtt) continue;
+ if (entry === WHISPER_FILENAME) continue;
+ // Every English VTT is a CAPTION track, not a sub track: the primary is
+ // the transcript, and the others are its alternate tracks, kept only
+ // where their words differ (lib/captionTracks.ts) — not shipped again,
+ // identical or not, as subtitles.
+ if (isEnglishVtt(entry)) continue;
if (entry === CUES_JSON_FILENAME) continue;
if (entry === LIVE_CHAT_CUES_FILENAME) continue;
const m = entry.match(SUB_FILE_RE);
diff --git a/common/lib/vtt.test.ts b/common/lib/vtt.test.ts
@@ -1,6 +1,14 @@
import { test } from "node:test";
import assert from "node:assert/strict";
-import { cuesToText, cuesToSrt, type Cue } from "./vtt";
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { cuesToText, cuesToSrt, hasWordTiming, parseVtt, type Cue } from "./vtt";
+
+// Both fixtures keep the structure of real YouTube downloads (one rolling
+// auto-caption track, one served `en` track of a livestream VOD) with the
+// words replaced.
+const fixture = (name: string) =>
+ readFileSync(path.join(import.meta.dirname, "__fixtures__", name), "utf8");
// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/vtt.test.ts
@@ -37,3 +45,53 @@ test("cuesToSrt clamps negative times to zero", () => {
const neg = [{ start: -3, end: -1, text: "neg" }];
assert.equal(cuesToSrt(neg), "1\n00:00:00,000 --> 00:00:00,000\nneg\n");
});
+
+test("parseVtt reads a rolling auto-caption track one new line per cue", () => {
+ const cues = parseVtt(fixture("vtt-rolling.vtt"));
+ // Not asserted: the FIRST cue, whose carried-over line is a lone space —
+ // the body read stops at a whitespace-only line, so that cue reads as empty.
+ // That is the rolling parse as it stands, unchanged here.
+ assert.deepEqual(
+ cues.slice(-3).map((c) => c.text),
+ [
+ "are talking about the harbor",
+ "bridge finally finally",
+ "I have been teasing that",
+ ],
+ );
+ assert.equal(cues.at(-3)!.start, 3.08);
+});
+
+test("parseVtt reads a cue-block track (no word timing, line ends) — every line of every cue", () => {
+ const src = fixture("vtt-cue-blocks.vtt");
+ assert.equal(hasWordTiming(src), false);
+ const cues = parseVtt(src);
+ assert.equal(cues.length, 7);
+ assert.deepEqual(cues[0], {
+ start: 0.96,
+ end: 6,
+ text: "welcome back everyone today we are talking about the harbor bridge finally finally",
+ });
+ // Entities decoded after the tags are stripped: an escaped "<old>" is text.
+ assert.equal(cues[3].text, "we will look at the tide tables for the north pier & the <old> lighthouse");
+ assert.equal(cues[4].text, "[Applause]");
+ // A real tag is stripped.
+ assert.equal(cues[6].text, "right to begin let us start with the end the conclusion here I have said this");
+});
+
+test("parseVtt: the shape is decided per document — a rolling track's untagged repeat cues are not read as text", () => {
+ const src = fixture("vtt-rolling.vtt");
+ assert.equal(hasWordTiming(src), true);
+ // Seven cues in the file; the repeats add nothing.
+ assert.equal(parseVtt(src).length, 3);
+});
+
+test("parseVtt reads cue identifiers and settings as structure, not text", () => {
+ const src =
+ "WEBVTT\n\nNOTE a comment\n\n1\n00:00:01.000 --> 00:00:02.500 line:90%\n<v Host>First line\n\n" +
+ "2\n00:00:03.000 --> 00:00:04.000\nSecond line\n";
+ assert.deepEqual(parseVtt(src), [
+ { start: 1, end: 2.5, text: "First line" },
+ { start: 3, end: 4, text: "Second line" },
+ ]);
+});
diff --git a/common/lib/vtt.ts b/common/lib/vtt.ts
@@ -2,6 +2,13 @@ export type Cue = { start: number; end: number; text: string };
const TIMING_TAG_RE = /<\d{2}:\d{2}:\d{2}\.\d{3}>/;
+// Whether VTT text carries inline word timing (<hh:mm:ss.mmm>) — the mark of
+// YouTube's rolling auto-caption shape, which parseVtt reads differently from
+// plain cue blocks.
+export function hasWordTiming(src: string): boolean {
+ return TIMING_TAG_RE.test(src);
+}
+
function parseTimestamp(ts: string): number {
const m = ts.match(/(\d+):(\d+):(\d+)\.(\d+)/);
if (!m) return 0;
@@ -27,9 +34,32 @@ function stripTags(s: string): string {
.trim();
}
+// Strip EVERY tag from a plain cue-block line (<i>, <b>, <v Speaker>, a stray
+// <c>), then decode entities — in that order, so an escaped "<3" survives as
+// text instead of being read as the start of a tag.
+function stripPlainLine(s: string): string {
+ return decodeEntities(s.replace(/<[^>]*>/g, ""))
+ .replace(/\s+/g, " ")
+ .trim();
+}
+
+// Two VTT shapes, decided once per DOCUMENT:
+//
+// ROLLING (YouTube auto-captions, word timing). Each cue repeats the line
+// before it and adds one new line carrying inline <hh:mm:ss.mmm><c>…</c> word
+// timing; between them sits a ~10 ms cue that repeats the text with no tags.
+// Only the tagged line is new, so only it is kept — reading every line would
+// say each sentence two or three times.
+//
+// CUE BLOCKS (uploaded captions, and the `en` track YouTube serves for some
+// livestream VODs: two lines a cue, ` ` at each line end, no word
+// timing). Every line of a cue is its text. A document of this shape has no
+// timing tag anywhere, which is what tells the two apart — read as ROLLING it
+// came out as zero cues, and the video as textless.
export function parseVtt(src: string): Cue[] {
const lines = src.replace(/\r\n/g, "\n").split("\n");
const cues: Cue[] = [];
+ const rolling = hasWordTiming(src);
let i = 0;
while (i < lines.length) {
@@ -50,12 +80,13 @@ export function parseVtt(src: string): Cue[] {
i++;
}
- const taggedLines = body.filter((l) => TIMING_TAG_RE.test(l));
let text: string;
- if (taggedLines.length > 0) {
- text = stripTags(taggedLines[taggedLines.length - 1]);
+ if (!rolling) {
+ text = body.map(stripPlainLine).filter(Boolean).join(" ");
} else {
- continue;
+ const taggedLines = body.filter((l) => TIMING_TAG_RE.test(l));
+ if (taggedLines.length === 0) continue;
+ text = stripTags(taggedLines[taggedLines.length - 1]);
}
if (text) cues.push({ start, end, text });
diff --git a/common/lib/wayback-server.ts b/common/lib/wayback-server.ts
@@ -0,0 +1,43 @@
+// WAYBACK PROVENANCE ON DISK — the `wayback.json` sidecar.
+//
+// A record downloaded from a Wayback Machine capture (lib/wayback.ts) carries
+// what it is a copy of: the original URL, the capture's timestamp, the capture
+// as a page that plays and as its raw bytes. Built from the capture URL alone
+// — no request — so it is written on every download of one
+// (ytdlp/downloadOneManaged.ts) and by `archilyzer wayback refresh` for a
+// record imported before this existed (controller/waybackRefresh.ts).
+
+import {
+ WAYBACK_PROVENANCE_FILENAME,
+ buildWaybackProvenance,
+ coerceWaybackProvenance,
+ sameWaybackProvenance,
+ type WaybackProvenance,
+} from "./wayback";
+import { sidecar, sidecarField } from "./sidecar-server";
+
+export const waybackProvenanceSidecar = sidecar(
+ WAYBACK_PROVENANCE_FILENAME,
+ sidecarField(coerceWaybackProvenance),
+);
+
+export const { load: loadWaybackProvenance, write: writeWaybackProvenance } =
+ waybackProvenanceSidecar;
+
+// The sidecar for a record fetched by `url`: written when the URL is a
+// capture and the one on disk is absent or says otherwise. Returns what the
+// record now carries (null for a URL that is not a capture) and whether it was
+// (or, with `dryRun`, would be) written.
+export async function ensureWaybackProvenance(
+ videoDir: string,
+ url: string,
+ opts: { dryRun?: boolean; onLog?: (line: string) => void } = {},
+): Promise<{ provenance: WaybackProvenance | null; written: boolean }> {
+ const next = buildWaybackProvenance(url);
+ if (!next) return { provenance: null, written: false };
+ const prev = await loadWaybackProvenance(videoDir);
+ if (prev && sameWaybackProvenance(prev, next)) return { provenance: prev, written: false };
+ if (!opts.dryRun) await writeWaybackProvenance(videoDir, next);
+ opts.onLog?.(`Wayback provenance: a capture of ${next.originalUrl} (${next.captureTs}).\n`);
+ return { provenance: next, written: true };
+}
diff --git a/common/lib/wayback.test.ts b/common/lib/wayback.test.ts
@@ -0,0 +1,162 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ buildWaybackProvenance,
+ coerceWaybackProvenance,
+ jwPlayerMediaId,
+ parseWaybackUrl,
+ waybackCaptureDate,
+ waybackCitationLinks,
+} from "./wayback";
+import { ensureWaybackProvenance, loadWaybackProvenance } from "./wayback-server";
+import { extractVideoId } from "./videoId";
+import { platformMomentUrl } from "./momentUrl";
+import { platformFromMetadata, summarize } from "./transcripts-server";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/wayback.test.ts
+//
+// Every id, account and token here is invented.
+
+const YT = "Abc123def45";
+const JW = "Qw3rTy12";
+const ARCHIVED_PAGE = `https://web.archive.org/web/20210102030405/https://www.youtube.com/watch?v=${YT}`;
+const JW_FILE = `https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${JW}-12345678.mp4?token=0_abc_0xdef`;
+const ARCHIVED_FILE = `https://web.archive.org/web/20200102030405id_/${JW_FILE}`;
+
+test("parseWaybackUrl: timestamp, modifier and the original with its own query", () => {
+ assert.deepEqual(parseWaybackUrl(ARCHIVED_PAGE), {
+ captureTs: "20210102030405",
+ modifier: "",
+ originalUrl: `https://www.youtube.com/watch?v=${YT}`,
+ });
+ assert.deepEqual(parseWaybackUrl(ARCHIVED_FILE), {
+ captureTs: "20200102030405",
+ modifier: "id_",
+ originalUrl: JW_FILE,
+ });
+ // A short timestamp, another modifier, no scheme, a collapsed scheme, an
+ // encoded original, the older path without /web/.
+ assert.equal(parseWaybackUrl("https://web.archive.org/web/2019im_/example.com/a.png")?.originalUrl, "http://example.com/a.png");
+ assert.equal(parseWaybackUrl("https://web.archive.org/web/2019/https:/example.com/x")?.originalUrl, "https://example.com/x");
+ assert.equal(
+ parseWaybackUrl("https://web.archive.org/web/2019/https%3A%2F%2Fexample.com%2Fx")?.originalUrl,
+ "https://example.com/x",
+ );
+ assert.equal(parseWaybackUrl("http://wayback.archive.org/20190101000000/http://example.com/")?.captureTs, "20190101000000");
+ // Not captures.
+ assert.equal(parseWaybackUrl("https://web.archive.org/web/*/example.com"), null);
+ assert.equal(parseWaybackUrl("https://archive.org/details/some-item"), null);
+ assert.equal(parseWaybackUrl(`https://www.youtube.com/watch?v=${YT}`), null);
+ assert.equal(parseWaybackUrl("not a url"), null);
+});
+
+test("jwPlayerMediaId: the media id across hosts and renditions", () => {
+ assert.equal(jwPlayerMediaId(JW_FILE), JW);
+ assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/videos/${JW}-AbCdEf12.mp4`), JW);
+ assert.equal(jwPlayerMediaId(`https://content.jwplatform.com/videos/${JW}.mp4`), JW);
+ assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/manifests/${JW}.m3u8`), JW);
+ assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/v2/media/${JW}`), JW);
+ assert.equal(jwPlayerMediaId("https://cdn.jwplayer.com/libraries/AbCd1234.js"), null);
+ assert.equal(jwPlayerMediaId(`https://example.com/videos/${JW}-1.mp4`), null);
+});
+
+test("extractVideoId unwraps a capture into what it is a capture of", () => {
+ assert.equal(extractVideoId(ARCHIVED_PAGE), YT);
+ assert.equal(extractVideoId(`https://web.archive.org/web/2021if_/https://youtu.be/${YT}`), YT);
+ assert.equal(extractVideoId(ARCHIVED_FILE), JW);
+ assert.equal(extractVideoId(JW_FILE), JW);
+ // Unwrapped once: a capture of a capture names no id.
+ assert.equal(extractVideoId(`https://web.archive.org/web/2021/${ARCHIVED_PAGE}`), null);
+ // A Wayback page that is not a capture names no record.
+ assert.equal(extractVideoId("https://web.archive.org/web/*/example.com"), null);
+ // A capture of a page the app does not know: the original's last segment.
+ assert.equal(extractVideoId("https://web.archive.org/web/2021/https://example.com/media/clip-7.mp4"), "clip-7.mp4");
+ // Nothing else moved.
+ assert.equal(extractVideoId(`https://www.youtube.com/watch?v=${YT}`), YT);
+ assert.equal(extractVideoId("https://example.com/feed/episode-1.mp3"), "episode-1.mp3");
+});
+
+test("buildWaybackProvenance: an archived YouTube page's original is its watch URL", () => {
+ assert.deepEqual(buildWaybackProvenance(`https://web.archive.org/web/20210102id_/https://m.youtube.com/watch?v=${YT}&feature=share`), {
+ originalUrl: `https://www.youtube.com/watch?v=${YT}`,
+ captureTs: "20210102",
+ waybackUrl: `https://web.archive.org/web/20210102/https://m.youtube.com/watch?v=${YT}&feature=share`,
+ rawUrl: `https://web.archive.org/web/20210102id_/https://m.youtube.com/watch?v=${YT}&feature=share`,
+ });
+ const file = buildWaybackProvenance(ARCHIVED_FILE)!;
+ assert.equal(file.originalUrl, JW_FILE);
+ assert.equal(file.waybackUrl, `https://web.archive.org/web/20200102030405/${JW_FILE}`);
+ assert.equal(file.rawUrl, ARCHIVED_FILE);
+ assert.equal(buildWaybackProvenance(JW_FILE), null);
+ assert.deepEqual(coerceWaybackProvenance(JSON.parse(JSON.stringify(file))), file);
+ assert.equal(coerceWaybackProvenance({ ...file, captureTs: "yesterday" }), null);
+ assert.equal(coerceWaybackProvenance([]), null);
+});
+
+test("waybackCaptureDate and the citation links", () => {
+ assert.equal(waybackCaptureDate("20210102030405"), "2021-01-02");
+ assert.equal(waybackCaptureDate("2021"), "2021");
+ assert.equal(waybackCaptureDate("202101"), "2021-01");
+ const prov = buildWaybackProvenance(ARCHIVED_PAGE)!;
+ const links = waybackCitationLinks(prov, {
+ originalMomentUrl: platformMomentUrl(prov.originalUrl, null, 90),
+ });
+ assert.deepEqual(links.original, {
+ label: "Original (may be gone)",
+ url: `https://www.youtube.com/watch?v=${YT}&t=90s`,
+ });
+ assert.deepEqual(links.copy, {
+ label: "Wayback Machine copy, 2021-01-02",
+ url: `https://web.archive.org/web/20210102030405/https://www.youtube.com/watch?v=${YT}`,
+ });
+});
+
+test("platformMomentUrl: a capture is the page that plays, never given a time param", () => {
+ assert.equal(platformMomentUrl(ARCHIVED_PAGE, "youtube", 125), ARCHIVED_PAGE);
+ assert.equal(platformMomentUrl(`https://www.youtube.com/watch?v=${YT}`, "youtube", 125), `https://www.youtube.com/watch?v=${YT}&t=125s`);
+});
+
+test("platformFromMetadata: YoutubeWebArchive is the original's platform, YouTube", () => {
+ assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive", extractor: "web.archive:youtube", webpage_url: ARCHIVED_PAGE }), "youtube");
+ assert.equal(platformFromMetadata({ extractor: "web.archive:youtube", webpage_url: ARCHIVED_PAGE }), "youtube");
+});
+
+test("summarize: a capture's id is its dir's name, not yt-dlp's file-name id", () => {
+ const s = summarize("demo", `${JW}`, {
+ id: `${JW}-12345678`,
+ title: `${JW}-12345678`,
+ extractor_key: "Generic",
+ webpage_url: ARCHIVED_FILE,
+ });
+ assert.equal(s.id, JW);
+ assert.equal(s.slug, `demo/${JW}`);
+ const page = summarize("demo", YT, { id: YT, title: "A title", extractor_key: "YoutubeWebArchive", webpage_url: ARCHIVED_PAGE });
+ assert.equal(page.id, YT);
+ assert.equal(page.platform, "youtube");
+ assert.equal(page.webpageUrl, ARCHIVED_PAGE);
+});
+
+test("ensureWaybackProvenance writes the sidecar once, and nothing for a non-capture", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "wayback-sidecar-"));
+ assert.deepEqual(await ensureWaybackProvenance(dir, JW_FILE), { provenance: null, written: false });
+ assert.equal(await loadWaybackProvenance(dir), null);
+
+ const dry = await ensureWaybackProvenance(dir, ARCHIVED_PAGE, { dryRun: true });
+ assert.equal(dry.written, true);
+ assert.equal(await loadWaybackProvenance(dir), null);
+
+ const first = await ensureWaybackProvenance(dir, ARCHIVED_PAGE);
+ assert.equal(first.written, true);
+ const onDisk = JSON.parse(await readFile(path.join(dir, "wayback.json"), "utf8"));
+ assert.deepEqual(onDisk, buildWaybackProvenance(ARCHIVED_PAGE));
+ assert.equal((await ensureWaybackProvenance(dir, ARCHIVED_PAGE)).written, false);
+
+ // A sidecar for another capture is rewritten.
+ await writeFile(path.join(dir, "wayback.json"), JSON.stringify(buildWaybackProvenance(ARCHIVED_FILE)));
+ assert.equal((await ensureWaybackProvenance(dir, ARCHIVED_PAGE)).written, true);
+ assert.deepEqual(await loadWaybackProvenance(dir), buildWaybackProvenance(ARCHIVED_PAGE));
+});
diff --git a/common/lib/wayback.ts b/common/lib/wayback.ts
@@ -0,0 +1,211 @@
+// THE WAYBACK MACHINE — a capture of some other URL, and what it was a copy of.
+//
+// A Wayback capture URL wraps the original: `https://web.archive.org/web/
+// <timestamp>[<modifier>]/<original>`, where the timestamp is 4–14 digits
+// (yyyy[MM[dd[hh[mm[ss]]]]]) and the modifier is the replay mode — none (the
+// page in the Wayback frame, which is what plays), `id_` (the raw bytes as
+// captured), `im_`, `if_`, `js_`, `cs_`, `oe_`, … The original keeps its own
+// query string, so it is the rest of the path PLUS the capture URL's search.
+//
+// yt-dlp downloads either kind through the editor's import: an archived
+// YouTube page through its `YoutubeWebArchive` extractor (the record's id is
+// the YouTube id), a raw media file through `generic` (an id that is the file
+// name). Neither knows it is a copy; the `wayback.json` sidecar
+// (lib/wayback-server.ts) records that, and lib/videoId.ts names the record by
+// what the capture is OF.
+//
+// A LEAF: no imports, so lib/videoId.ts (itself a leaf apart from
+// archiveOrgId.ts) can unwrap a capture without a cycle.
+
+export const WAYBACK_PROVENANCE_FILENAME = "wayback.json";
+
+// Hosts that serve Wayback captures. archive.org's own hosts serve items
+// (lib/archiveOrgId.ts), never captures.
+const WAYBACK_HOSTS = new Set(["web.archive.org", "wayback.archive.org"]);
+
+export function isWaybackHost(host: string): boolean {
+ return WAYBACK_HOSTS.has(host.toLowerCase());
+}
+
+export type WaybackRef = {
+ // The capture's timestamp as the URL gives it (4–14 digits).
+ captureTs: string;
+ // The replay modifier (`id_`, `im_`, …), "" for the framed page.
+ modifier: string;
+ // The URL the capture is of, as the capture URL names it (scheme added
+ // when the capture URL left it off).
+ originalUrl: string;
+};
+
+// `/web/<ts><mod>/<original>`, or the older `/<ts><mod>/<original>`.
+const CAPTURE_PATH_RE = /^\/(?:web\/)?(\d{4,14})([a-z]{2}_)?\/(.+)$/s;
+
+// A Wayback capture URL's parts, or null for anything else (a calendar page
+// `/web/*/<url>`, a search, another host).
+export function parseWaybackUrl(url: string | null | undefined): WaybackRef | null {
+ if (!url) return null;
+ let u: URL;
+ try {
+ u = new URL(url);
+ } catch {
+ return null;
+ }
+ if (!isWaybackHost(u.hostname)) return null;
+ const m = CAPTURE_PATH_RE.exec(u.pathname);
+ if (!m) return null;
+ let rest = m[3];
+ // A percent-encoded original (`https%3A%2F%2F…`) is the same URL.
+ if (/^https?%3a/i.test(rest)) {
+ try {
+ rest = decodeURIComponent(rest);
+ } catch {
+ return null;
+ }
+ }
+ // The WHATWG parser keeps `https://` inside a path as written, but a capture
+ // URL that went through a path normaliser arrives as `https:/host`.
+ rest = rest.replace(/^(https?):\/(?!\/)/i, "$1://");
+ if (!/^https?:\/\//i.test(rest)) rest = `http://${rest}`;
+ const original = `${rest}${u.search}${u.hash}`;
+ try {
+ new URL(original);
+ } catch {
+ return null;
+ }
+ return { captureTs: m[1], modifier: m[2] ?? "", originalUrl: original };
+}
+
+// The capture as a page that plays (the Wayback frame, no modifier).
+export function waybackPageUrl(ref: Pick<WaybackRef, "captureTs" | "originalUrl">): string {
+ return `https://web.archive.org/web/${ref.captureTs}/${ref.originalUrl}`;
+}
+
+// The capture's raw bytes (`id_`), the form yt-dlp fetches a media file by.
+export function waybackRawUrl(ref: Pick<WaybackRef, "captureTs" | "originalUrl">): string {
+ return `https://web.archive.org/web/${ref.captureTs}id_/${ref.originalUrl}`;
+}
+
+// `YYYY-MM-DD` of a capture timestamp, or the year (/month) when that is all
+// it names.
+export function waybackCaptureDate(captureTs: string): string {
+ const y = captureTs.slice(0, 4);
+ const mo = captureTs.slice(4, 6);
+ const d = captureTs.slice(6, 8);
+ return [y, mo, d].filter((p) => p.length === 2 || p.length === 4).join("-");
+}
+
+// ─── JW Player ───
+
+// The hosts JW Player serves a media file from. A file there is
+// `…/videos/<mediaId>-<rendition>.<ext>` (cdn.jwplayer.com/videos/…,
+// content.jwplatform.com/videos/…, videos-fms.jwpsrv.com/content/conversions/
+// <account>/videos/…); the media id is eight alphanumerics and names the video
+// across every rendition.
+const JW_HOSTS = ["jwplayer.com", "jwplatform.com", "jwpsrv.com"];
+
+export function isJwPlayerHost(host: string): boolean {
+ const h = host.toLowerCase();
+ return JW_HOSTS.some((d) => h === d || h.endsWith(`.${d}`));
+}
+
+const JW_FILE_RE = /\/videos\/([A-Za-z0-9]{8})(?:-[A-Za-z0-9]+)?\.[A-Za-z0-9]+$/;
+const JW_MEDIA_RE = /\/(?:manifests|v2\/media|previews)\/([A-Za-z0-9]{8})(?:[-./]|$)/;
+
+// The JW media id a JW Player file or manifest URL names, or null.
+export function jwPlayerMediaId(url: string | URL): string | null {
+ let u: URL;
+ try {
+ u = typeof url === "string" ? new URL(url) : url;
+ } catch {
+ return null;
+ }
+ if (!isJwPlayerHost(u.hostname)) return null;
+ const m = JW_FILE_RE.exec(u.pathname) ?? JW_MEDIA_RE.exec(u.pathname);
+ return m ? m[1] : null;
+}
+
+// ─── The sidecar's record ───
+
+export type WaybackProvenance = {
+ // What the capture is a copy of. For an archived YouTube page, its watch
+ // URL (`https://www.youtube.com/watch?v=<id>`), whatever form was captured.
+ originalUrl: string;
+ // The capture's timestamp (4–14 digits).
+ captureTs: string;
+ // The capture as a page that plays.
+ waybackUrl: string;
+ // The capture's raw bytes.
+ rawUrl: string;
+};
+
+// A YouTube URL as its watch page, or null for any other URL.
+function youtubeWatchUrl(url: string): string | null {
+ let u: URL;
+ try {
+ u = new URL(url);
+ } catch {
+ return null;
+ }
+ const host = u.hostname.toLowerCase();
+ let id: string | null = null;
+ if (host === "youtu.be") id = u.pathname.split("/").filter(Boolean)[0] ?? null;
+ else if (host === "youtube.com" || host.endsWith(".youtube.com")) {
+ id = u.searchParams.get("v");
+ if (!id) {
+ const segs = u.pathname.split("/").filter(Boolean);
+ if ((segs[0] === "embed" || segs[0] === "shorts" || segs[0] === "v" || segs[0] === "live") && segs[1]) id = segs[1];
+ }
+ } else return null;
+ return id ? `https://www.youtube.com/watch?v=${id}` : null;
+}
+
+// The sidecar for a capture URL, or null when the URL is not one.
+export function buildWaybackProvenance(url: string): WaybackProvenance | null {
+ const ref = parseWaybackUrl(url);
+ if (!ref) return null;
+ return {
+ originalUrl: youtubeWatchUrl(ref.originalUrl) ?? ref.originalUrl,
+ captureTs: ref.captureTs,
+ waybackUrl: waybackPageUrl(ref),
+ rawUrl: waybackRawUrl(ref),
+ };
+}
+
+export function coerceWaybackProvenance(value: unknown): WaybackProvenance | null {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return null;
+ const v = value as Record<string, unknown>;
+ const str = (k: string) => (typeof v[k] === "string" && v[k] ? (v[k] as string) : null);
+ const originalUrl = str("originalUrl");
+ const captureTs = str("captureTs");
+ const waybackUrl = str("waybackUrl");
+ const rawUrl = str("rawUrl");
+ if (!originalUrl || !captureTs || !/^\d{4,14}$/.test(captureTs) || !waybackUrl || !rawUrl) return null;
+ return { originalUrl, captureTs, waybackUrl, rawUrl };
+}
+
+export function sameWaybackProvenance(a: WaybackProvenance, b: WaybackProvenance): boolean {
+ return (
+ a.originalUrl === b.originalUrl &&
+ a.captureTs === b.captureTs &&
+ a.waybackUrl === b.waybackUrl &&
+ a.rawUrl === b.rawUrl
+ );
+}
+
+// ─── Citing it ───
+
+export type WaybackLink = { label: string; url: string };
+
+// What a citation of an archived copy links: the original, named as the
+// original and as possibly gone (a capture exists because it may be), and the
+// Wayback copy, which plays. `originalMomentUrl` is the original at the cited
+// second when its platform takes one (lib/momentUrl.ts).
+export function waybackCitationLinks(
+ prov: WaybackProvenance,
+ opts: { originalMomentUrl?: string | null } = {},
+): { original: WaybackLink; copy: WaybackLink } {
+ return {
+ original: { label: "Original (may be gone)", url: opts.originalMomentUrl || prov.originalUrl },
+ copy: { label: `Wayback Machine copy, ${waybackCaptureDate(prov.captureTs)}`, url: prov.waybackUrl },
+ };
+}
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -74,10 +74,12 @@ import {
import type { TranscriptSummary } from "../lib/transcripts";
import { parseVtt, type Cue } from "../lib/vtt";
import { parseTranscriptJson } from "../lib/whisper";
-import { WHISPER_FILENAME, isEnglishVtt, resolvePrimaryVtt } from "../lib/videoStatus";
+import { WHISPER_FILENAME, englishVttsByPreference, readEnglishVttCues } from "../lib/videoStatus";
import { platformMomentUrl } from "../lib/momentUrl";
import { archiveOrgCitationLinks, type ArchiveOrgProvenance } from "../lib/archiveOrg";
import { loadArchiveOrgProvenance } from "../lib/archiveOrg-server";
+import { WAYBACK_PROVENANCE_FILENAME, waybackCitationLinks, type WaybackProvenance } from "../lib/wayback";
+import { loadWaybackProvenance } from "../lib/wayback-server";
import type { Platform } from "../lib/platform";
import { readAllPosts } from "../lib/posts-server";
import type { Post } from "../lib/posts";
@@ -236,24 +238,11 @@ type CitedRecord = {
// An archive.org record's provenance (its torrent, a mirror's original);
// null for every other record.
archiveOrg: ArchiveOrgProvenance | null;
+ // A Wayback Machine capture's provenance (lib/wayback.ts): what the record
+ // is an archived copy of. Null for every other record.
+ wayback: WaybackProvenance | null;
};
-// The English VTT tracks of a video dir, `en-orig` first, then the order
-// resolvePrimaryVtt prefers.
-function englishVttsByPreference(entries: readonly string[]): string[] {
- const vtts = entries.filter(isEnglishVtt);
- const ordered: string[] = [];
- const orig = vtts.find((n) => n === "transcript.en-orig.vtt");
- if (orig) ordered.push(orig);
- const rest = vtts.filter((n) => n !== orig);
- while (rest.length > 0) {
- const best = resolvePrimaryVtt(rest)!;
- ordered.push(best);
- rest.splice(rest.indexOf(best), 1);
- }
- return ordered;
-}
-
async function readCues(file: string, kind: "vtt" | "whisper"): Promise<Cue[]> {
try {
const raw = await readFile(file, "utf8");
@@ -290,17 +279,10 @@ export async function readCitedRecord(
if (!meta) return null;
summary = summarize(slug, id, meta, channelName);
if (entries.includes(WHISPER_FILENAME)) cues = await readCues(path.join(dir, WHISPER_FILENAME), "whisper");
- if (cues.length === 0) {
- const primary = resolvePrimaryVtt(entries);
- if (primary) cues = await readCues(path.join(dir, primary), "vtt");
- }
- }
- if (cues.length === 0) {
- for (const name of englishVttsByPreference(entries)) {
- cues = await readCues(path.join(dir, name), "vtt");
- if (cues.length > 0) break;
- }
}
+ // The caption-track rule (videoStatus.ts): the first English VTT, `en-orig`
+ // first, that has cues.
+ if (cues.length === 0) cues = (await readEnglishVttCues(dir, entries))?.cues ?? [];
const tracks: CitedRecord["tracks"] = [];
if (fresh.fresh) {
const n = await readNormalizedTranscript(fresh.cuesPath);
@@ -316,7 +298,8 @@ export async function readCitedRecord(
}
if (tracks.length === 0 && cues.length > 0) tracks.push({ name: "cues", cues });
const archiveOrg = summary.platform === "archiveorg" ? await loadArchiveOrgProvenance(dir) : null;
- return { summary, cues, tracks, archiveOrg };
+ const wayback = entries.includes(WAYBACK_PROVENANCE_FILENAME) ? await loadWaybackProvenance(dir) : null;
+ return { summary, cues, tracks, archiveOrg, wayback };
}
const isoDay = (uploadDate: string | undefined): string | undefined =>
@@ -629,9 +612,28 @@ export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promi
corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }),
});
}
- const { summary, archiveOrg } = (await recordOf(c.channel, c.id))!;
+ const { summary, archiveOrg, wayback } = (await recordOf(c.channel, c.id))!;
const audioOnly = isAudioOnlyPlatform(config?.platform);
const seconds = Math.max(0, Math.floor(c.start));
+ if (wayback) {
+ // An archived copy (Wayback Machine): the original, named as the
+ // original and as possibly gone, then the copy, which plays.
+ const links = waybackCitationLinks(wayback, {
+ originalMomentUrl: platformMomentUrl(wayback.originalUrl, null, c.start),
+ });
+ return defined({
+ channel: c.channel,
+ channelTitle: config?.name ?? (summary.channel || undefined),
+ id: c.id,
+ title: summary.title,
+ date: isoDay(summary.uploadDate),
+ platform: summary.platform,
+ originalUrl: links.original.url,
+ originalLabel: links.original.label,
+ downloads: [links.copy],
+ corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}),
+ });
+ }
if (summary.platform === "archiveorg") {
// archive.org: the original (YouTube at the second, for a mirror; else
// the archive.org page) plus the downloads a reader can check it from.
diff --git a/common/publish/composeReportsWayback.test.ts b/common/publish/composeReportsWayback.test.ts
@@ -0,0 +1,54 @@
+// A cited Wayback Machine copy carries its provenance into compose, and the
+// links a citation of it shows are derived from that (lib/wayback.ts
+// waybackCitationLinks): the original at the second, named as the original and
+// as possibly gone, then the Wayback copy. Every id here is invented.
+//
+// Run with: node_modules/.bin/tsx --test publish/composeReportsWayback.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { readCitedRecord } from "./composeReports";
+import { buildWaybackProvenance, waybackCitationLinks } from "../lib/wayback";
+import { platformMomentUrl } from "../lib/momentUrl";
+
+const YT = "Xyz987abc65";
+const CAPTURE = `https://web.archive.org/web/20190807060504/https://www.youtube.com/watch?v=${YT}`;
+
+test("readCitedRecord loads a Wayback copy's provenance; others get null", async () => {
+ const channels = mkdtempSync(path.join(tmpdir(), "compose-wayback-"));
+ const dir = path.join(channels, "demo-wayback", "data", YT);
+ mkdirSync(dir, { recursive: true });
+ writeFileSync(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({
+ id: YT,
+ extractor_key: "YoutubeWebArchive",
+ title: "An archived upload",
+ upload_date: "20090102",
+ webpage_url: CAPTURE,
+ }),
+ );
+ const prov = buildWaybackProvenance(CAPTURE)!;
+ writeFileSync(path.join(dir, "wayback.json"), JSON.stringify(prov));
+
+ const rec = await readCitedRecord(channels, "demo-wayback", YT);
+ assert.ok(rec);
+ assert.equal(rec.summary.platform, "youtube");
+ assert.equal(rec.summary.id, YT);
+ assert.deepEqual(rec.wayback, prov);
+ // The record's own moment link is the capture, which plays.
+ assert.equal(platformMomentUrl(rec.summary.webpageUrl, "youtube", 42), CAPTURE);
+ const links = waybackCitationLinks(rec.wayback!, {
+ originalMomentUrl: platformMomentUrl(rec.wayback!.originalUrl, null, 42),
+ });
+ assert.deepEqual(links.original, { label: "Original (may be gone)", url: `https://www.youtube.com/watch?v=${YT}&t=42s` });
+ assert.deepEqual(links.copy, { label: "Wayback Machine copy, 2019-08-07", url: CAPTURE });
+
+ const plain = path.join(channels, "demo-wayback", "data", "Plain12345a");
+ mkdirSync(plain, { recursive: true });
+ writeFileSync(path.join(plain, "metadata.info.json"), JSON.stringify({ id: "Plain12345a", extractor_key: "Youtube" }));
+ assert.equal((await readCitedRecord(channels, "demo-wayback", "Plain12345a"))?.wayback, null);
+});
diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts
@@ -6,6 +6,8 @@ import {
pacedPlatformArgs,
platformArgs,
platformArgsForUrl,
+ platformImportMinGapSeconds,
+ PLATFORM_IMPORT_MIN_GAP_SECONDS,
PLATFORM_ARGS,
PLATFORM_MIN_GAP_SECONDS,
platformMinGapSeconds,
@@ -185,3 +187,14 @@ test("BitChute is paced hardest: 3 s between requests, exponential retry sleeps,
assert.equal(platformMinGapSeconds(p), 0, String(p));
}
});
+
+test("one-off imports have their own floor: Odysee's 60 s, BitChute's batch floor, none elsewhere", () => {
+ assert.equal(PLATFORM_IMPORT_MIN_GAP_SECONDS.odysee, 60);
+ assert.equal(platformImportMinGapSeconds("odysee"), 60);
+ assert.equal(platformImportMinGapSeconds("bitchute"), 60);
+ // Odysee's batch gap is untouched: the import floor is imports only.
+ assert.equal(platformMinGapSeconds("odysee"), 0);
+ for (const p of ["youtube", "rumble", "archiveorg", "unknown", null]) {
+ assert.equal(platformImportMinGapSeconds(p), 0, String(p));
+ }
+});
diff --git a/common/ytdlp/channelArgs.ts b/common/ytdlp/channelArgs.ts
@@ -41,6 +41,8 @@ export {
platformArgsForUrl,
PLATFORM_MIN_GAP_SECONDS,
platformMinGapSeconds,
+ PLATFORM_IMPORT_MIN_GAP_SECONDS,
+ platformImportMinGapSeconds,
staticSleepRequestsSeconds,
withSleepRequests,
} from "./platformArgs.mjs";
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -1,7 +1,7 @@
import { removeMediaFile } from "../lib/mediaTier-server";
import { tierVideoDir } from "../lib/mediaTier-server";
import path from "node:path";
-import { appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises";
+import { access, appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises";
import { createWriteStream, type Dirent, type WriteStream } from "node:fs";
import { execa } from "execa";
import {
@@ -50,6 +50,8 @@ import { writeDownloadOutcome } from "../lib/downloadOutcome-server";
import { formatBytes } from "../lib/format";
import { recordAvailability } from "../lib/availability-server";
import { withMetadataHistory } from "../lib/metadataHistory-server";
+import { ensureWaybackProvenance } from "../lib/wayback-server";
+import { parseWaybackUrl } from "../lib/wayback";
import {
loadRawMetadata,
loadRawMetadataFromDir,
@@ -656,12 +658,37 @@ export async function downloadOneManaged(
if (detectPlatform(opts.videoUrl) === "archiveorg") {
return await downloadArchiveOrgManaged(opts, opts.archiveOrgDeps);
}
- return await runManagedDownload(opts, channelDir, startedAt, canonicalId);
+ const outcome = await runManagedDownload(opts, channelDir, startedAt, canonicalId);
+ await recordWaybackProvenance(opts, channelDir, canonicalId);
+ return outcome;
} finally {
logStream?.end();
}
}
+// A WAYBACK CAPTURE IS A COPY (lib/wayback.ts): once the record exists — its
+// metadata.info.json written by the prefetch or the download — the
+// `wayback.json` sidecar says of what. Built from the URL alone, so it costs no
+// request; never fails the download.
+async function recordWaybackProvenance(
+ opts: ManagedDownloadOpts,
+ channelDir: string,
+ canonicalId: string | null,
+): Promise<void> {
+ if (!canonicalId || !parseWaybackUrl(opts.videoUrl)) return;
+ const videoDir = path.join(channelDir, "data", canonicalId);
+ try {
+ await access(path.join(videoDir, "metadata.info.json"));
+ } catch {
+ return;
+ }
+ try {
+ await ensureWaybackProvenance(videoDir, opts.videoUrl, { onLog: opts.onLog });
+ } catch (err) {
+ opts.onLog(`Wayback provenance not written: ${(err as Error).message}\n`);
+ }
+}
+
async function runManagedDownload(
opts: ManagedDownloadOpts,
channelDir: string,
diff --git a/common/ytdlp/platformArgs.mjs b/common/ytdlp/platformArgs.mjs
@@ -94,6 +94,18 @@ export const PLATFORM_MIN_GAP_SECONDS = Object.freeze({
bitchute: 60,
});
+// THE FLOOR UNDER THE GAP BETWEEN TWO ONE-OFF IMPORTS on a platform, in
+// seconds, where it must be higher than the batch floor above. Each import is
+// its own job, so nothing else spaces them: Odysee's API answered HTTP 429 to
+// imports queued 10–50 s apart (2026-10-06). An import waits the larger of
+// this and PLATFORM_MIN_GAP_SECONDS after the last import there, plus up to
+// half again at random (jobs/platformGap.ts). Batch downloads and syncs of an
+// Odysee channel keep the operator's gap.
+/** @type {Readonly<Partial<Record<Platform, number>>>} */
+export const PLATFORM_IMPORT_MIN_GAP_SECONDS = Object.freeze({
+ odysee: 60,
+});
+
/**
* @param {string | null | undefined} platform
* @returns {number}
@@ -106,6 +118,17 @@ export function platformMinGapSeconds(platform) {
}
/**
+ * @param {string | null | undefined} platform
+ * @returns {number}
+ */
+export function platformImportMinGapSeconds(platform) {
+ if (!platform) return 0;
+ /** @type {Record<string, number | undefined>} */
+ const table = PLATFORM_IMPORT_MIN_GAP_SECONDS;
+ return Math.max(platformMinGapSeconds(platform), table[platform] ?? 0);
+}
+
+/**
* @param {Platform | null | undefined} platform
* @returns {string[]}
*/
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -12,6 +12,15 @@
- **Substitute your own yt-dlp in Docker.** Point `YTDLP_BIN` at a zipapp you built, or set `YTDLP_SOURCE_HOST_DIR` to a yt-dlp checkout and start with `docker-compose.ytdlp.yml`: the image runs it with its own python, and nothing is rebuilt. Every editor boot logs `yt-dlp: <path> <version> (image|override)` (`MISSING` when it does not run; the editor still starts), and `YTDLP_AUTO_UPDATE` updates the image's yt-dlp only, warning instead of touching yours.
- **The Docker image can publish.** It carries python, `pipx` and a pinned `git-filter-repo`, so the homepage's `/source` mirror builds in the container; `docker-compose.source.yml` mounts your repository read-only for it, and the scrub rules and denylist live in the config volume (`/data/config/archilyzer`). Cloudflare and R2 credentials come from `.env`. Run publish commands with `docker compose exec editor pnpm archilyzer …`, not `run --rm`. The `homepage` service serves a local deploy from the builds volume once there is one. RUNNING_IN_DOCKER.md has a Windows checklist.
- **`archilyzer doctor` checks what a publish needs.** Which yt-dlp runs (the image's, the host's or an override, and whether it runs), whether the Cloudflare token and the R2 keys are set (never their values; R2 only when a bucket is configured) — judged exactly as a deploy judges them —, the wrangler a deploy runs (the pinned one or your `WRANGLER_BIN`, and that it starts and is the expected major), free space for the site bundles, the publish lock (free, held by a running stage, or left by one that is gone — with the command to clear it; never cleared for you), the index stamp's age and which sites were built from an older one, the repository the source mirror reads, the private config dir, and whether this Node is new enough for the pinned wrangler (deploys need 22).
+- **A video's other English tracks are readable and searchable where their words differ.** Uploaded captions are not always a transcript of what was said, so the tracks beside the transcript stay: the served `en` beside `en-orig`, a regional or auto-translated track, and the captions a local transcription replaced. One is kept where its words differ from the transcript's and from every track kept before it; identical tracks, most of them, add nothing. The index keeps them in an `alts` sub-DB and writes `track` and `altTracks` onto the transcript record only then, so every other record's page is what it was. A search hit in a word only an alternate holds names the track; one every track says is found once, in the transcript. The video page's **Transcript** card reads the transcript and switches tracks ("Track: original audio captions ▾"); switching changes nothing on disk, and **Set as transcript** stays the way the transcript itself changes. English VTTs are no longer shipped as subtitle tracks. One notion of a track — ids, plain labels, which are kept, how a hit across them is found — lives in `common/lib/captionTracks.ts`.
+- **The next index build reads the alternate tracks once.** Every record that can hold one — two or more English VTTs, or a transcription beside captions — is re-read from disk, and nothing else; the log says `Alternate tracks v1: N record(s) re-read.` and how many hold a track whose words differ. The version is recorded only when no channel is held. A transcribed video's captions now count toward its change time, so a later caption fetch reaches the index.
+- **The MCP reads every English track.** `search_transcripts` and the query-tree tools match a record's alternate tracks and tag a snippet from one (`[in uploaded captions 1:30]`); `get_transcript` names a video's tracks in its header and reads another with `track`; `get_transcripts` windows a match only an alternate holds, under its name; `get_video_metadata` lists the other tracks without their cues. The sweep plan says what such a hit is before it is quoted.
+- **One-off imports from Odysee and BitChute are spaced, as a batch download is.** Each import is its own job, so nothing used to space them: imports queued on Odysee went out as fast as each finished, and Odysee answered 429. An imported Odysee or BitChute URL now waits at least 60 s after the last import from the same platform, plus up to half again at random; the job's log says "Odysee asks for a gap between videos: waiting Ns." An Odysee import, as a BitChute one already did, runs on that platform's own queue whatever the channel's platform, is refused while the platform is held or in a rate-limit cooldown, waits out a cooldown that began after it was queued, and records what the platform answered, so a 429 backs every path on that platform off. Batch downloads and syncs of an Odysee channel keep the operator's gap. The wait is held in memory, and a restart clears it. Needs a restart of the editor.
+- **A Wayback Machine capture is a copy, and says of what.** A capture URL (`web.archive.org/web/<timestamp>[id_|im_|…]/<original>`) names its record by what it is a capture of: an archived YouTube page by its YouTube id (no longer `watch`), a JW Player file by its media id (no longer `<id>-<rendition>.mp4`). Every download of a capture writes `wayback.json` (the original URL, the capture's timestamp, the capture page and its raw bytes); the video page says "Archived copy (Wayback Machine, <date>) of <original>"; a citation links the original, marked as possibly gone, and the Wayback copy, and its moment link is the capture, which plays (a capture URL never takes a time param). An existing record whose page is a capture is renamed to its id by the next snapshot.
+- **`archilyzer wayback refresh <slug> [--titles <file>] [--dry-run]`** brings a channel's Wayback copies up to that offline: `wayback.json`, the dir renamed through the snapshot's own reconcile pass with its roster entry moved, and with `--titles` (`id → {title, upload_date}`) the title and date of a raw file that has none, recorded in the metadata history as `wayback-provenance`. A record a live job holds is skipped and named; a second run changes nothing.
+- **A video's captions are read from its original-audio track first, and a track with no text never hides one that has it.** Where YouTube serves both, `transcript.en-orig.vtt` (the captions of the original audio) is read before `transcript.en.vtt`, whose text can be a rewrite of what was said; then regional tracks (`en-US`, `en-GB`, …), then auto-translated `en-en-*` ones. The transcript is the first track in that order that has cues. One rule (`englishVttsByPreference` / `readEnglishVttCues` in `common/lib/videoStatus.ts`) serves the index, normalize and report compose, so search, the export, the MCP and report videos read the same words. **Set as transcript** on a video's page copies the chosen track to `transcript.en.vtt` and pins it there with `transcript-pin.json`, which ranks it first; deleting `transcript-pin.json` returns the video to the automatic pick. The served `en` track stays readable and searchable as an alternate track where its words differ (below). Videos with a human-made `en` track are not counted as auto-captions-only, so the replace-auto-captions lane still leaves them alone.
+- **Captions in cue blocks are read.** A VTT with no inline word timing (uploaded captions, and the `en` track YouTube serves for some livestream recordings: two lines a cue, ` ` at each line end) is read cue by cue; it used to parse to no cues, which left those videos with no text in the index.
+- **The next index build re-reads the caption records the new rule reaches, once.** A record with more than one English track, or with no cues stored, is re-read from disk; nothing else is. The build log says `Caption track v1: N record(s) re-read.` and, per channel, how many now read different text and how many had none and now do. The version is recorded only when no channel is held. A `transcript.cues.json` written before the rule is stale only where the rule reads something else — its `transcript.en.vtt` and `transcript.en-orig.vtt` differ, or it holds no cues and a cue-block parse or the next track may have them — so a **Normalize** run rewrites exactly those, and the digest and attribution lanes hold them until it does; a cues file now records the track it was read from (`vttFile`) and the rule (`captionTrackRule`). umtool's report-to-video refuses such a stale local record by name rather than cut from it.
- **A cited moment at the very end of a recording prepares.** Prepare evidence media cuts a clip whose padding runs past the recording's end at the end (the recording's duration from its metadata), where it found no media for the padded span; a span that starts past the end is still refused. report-to-video keeps its strict rule.
- **Exporting a changed report records a new revision of it.** `reports export` (and **Export reports** on a site's Reports tab, and the end of a prepare) commits a revision to the report's own git history, `sites/<site>/reports/<id>/history-git/`, whenever its `report.json` changed since the last one: the `report.json`, its Markdown export and the checksums of every export file, with a message of `Revision N` and a summary of the change. A re-export of an unchanged report records nothing. The commits carry the site's name and a `noreply@<site>.invalid` address with dates in UTC, never your git name, email or time zone. The Reports tab shows each report's revision, its commit and the last change under **Exports**, and the site's next build publishes the history. Add `history-git/` to the corpus repository's `.gitignore`.
- **archive.org files come over BitTorrent when possible, else straight from archive.org — never through yt-dlp.** The chosen file of an archive.org import is fetched from the item's own torrent (`<identifier>_archive.torrent`, which lists archive.org as a web seed, so other peers take load off archive.org) with aria2c, only that file of the item, and seeded afterwards for 10 minutes or to a ratio of 1, whichever comes first; the log shows "torrent: <file> (n of m pieces, peers p, web seed yes)" and "seeding 10 min…". With no aria2c, a torrent that does not carry the file, or no progress for 5 minutes, it is downloaded directly from `archive.org/download/…` instead (resumable, backing off on 429/503), and the log says "fell back to direct download: <reason>". Every file is checked against archive.org's sha1/md5: a mismatch is downloaded once more directly, a second one fails the record. The record is written from the item's metadata: `metadata.info.json` with the file's page, the canonical id, the duration ffprobe measures and archive.org's playable copies of the file, the `archiveorg.json` provenance (a mirror's original title, date and uploader), and `audio.<fmt>` — an audio file already in the channel's format is used as is, anything else goes through the app's audio extraction, a video kept in the saved-video store when the channel keeps sources. An .avi/.mpeg/.flac/.wav original is fetched as archive.org's mp4 or mp3 of it. aria2c runs in its own process group: cancelling the job stops it and everything it started, and it stops itself if the editor exits. New settings block `archiveOrg` (`torrent`, `seedMinutes`, `seedRatio`, `stallMinutes`, `maxPeers`, `maxDownloadKiBps`, `maxUploadKiBps`), `ARIA2C_BIN`, an aria2c row in `archilyzer doctor`, and `aria2` in the runtime Docker images.
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -30,6 +30,13 @@ import {
recordDownloadBackoff,
recordPlatformClean,
} from "yt-dlp-transcript-common/jobs/downloadBackoff";
+import { downloadGapMs } from "yt-dlp-transcript-common/jobs/platformBackoff";
+import {
+ notePlatformGap,
+ platformGapRemainingMs,
+ waitForPlatformGap,
+} from "yt-dlp-transcript-common/jobs/platformGap";
+import { platformImportMinGapSeconds } from "yt-dlp-transcript-common/ytdlp/channelArgs";
import {
countNotYetDownloaded,
readChannelConfig,
@@ -531,6 +538,15 @@ export async function importVideoAction(
// again when on disk — and what BitChute answers is recorded on its pacing
// state below.
const bitchute = urlPlatform === "bitchute";
+ // ANY URL ON A PLATFORM WITH AN IMPORT FLOOR (platformImportMinGapSeconds:
+ // BitChute, Odysee) is paced though each import is its own job: it runs
+ // on the platform's own queue with the platform's args, is refused while the
+ // platform is held or cooling down, waits out the floor after the last import
+ // there (jobs/platformGap.ts) and any cooldown that began since it was
+ // queued, and records what the platform answered.
+ const pacedPlatform =
+ urlPlatform && platformImportMinGapSeconds(urlPlatform) > 0 ? urlPlatform : null;
+ const pacedLabel = pacedPlatform ? (PACED_LABELS[pacedPlatform] ?? pacedPlatform) : "";
let downloadConfig = channelConfig;
if (archiveOrg) {
const refused = await archiveOrgRefusal(paths, "The import");
@@ -555,9 +571,14 @@ export async function importVideoAction(
downloadConfig = { ...channelConfig, platform: "archiveorg" };
}
}
- if (bitchute) {
- const refused = await platformRefusal(paths, "bitchute", "BitChute", "The import");
+ if (pacedPlatform) {
+ const refused = await platformRefusal(paths, pacedPlatform, pacedLabel, "The import");
if (refused) return { ok: false, info: true, error: refused };
+ if (channelConfig.platform !== pacedPlatform) {
+ downloadConfig = { ...channelConfig, platform: pacedPlatform };
+ }
+ }
+ if (bitchute) {
const resolved = resolveBitchuteImportUrl(videoUrl);
if (!resolved.ok) return { ok: false, error: resolved.error };
videoUrl = resolved.url;
@@ -574,9 +595,6 @@ export async function importVideoAction(
error: `Already downloaded: data/${resolved.id}/ — BitChute is not asked for it again.`,
};
}
- if (channelConfig.platform !== "bitchute") {
- downloadConfig = { ...channelConfig, platform: "bitchute" };
- }
}
// Best-effort canonical id: used only for revalidation/labels. When null,
// downloadOneManaged falls back to %(id)s and the reconcile pass repairs the
@@ -589,7 +607,7 @@ export async function importVideoAction(
// queue control starts at the CHANNEL's queue, which is therefore read as
// "no choice made"; any other queue the operator picks still wins.
const channelQueue = downloadQueueKey(channelConfig);
- const ownQueue = archiveOrg || bitchute ? platformQueueKey(urlPlatform) : null;
+ const ownQueue = archiveOrg || pacedPlatform ? platformQueueKey(urlPlatform) : null;
const override =
ownQueue && queueKey !== undefined && queueKey.trim() === channelQueue
? undefined
@@ -607,6 +625,18 @@ export async function importVideoAction(
kind: "download",
});
try {
+ if (pacedPlatform) {
+ await waitForPlatformGap({
+ label: pacedLabel,
+ remainingMs: async () =>
+ Math.max(
+ platformGapRemainingMs(pacedPlatform),
+ await platformCooldownRemainingMs(pacedPlatform, paths).catch(() => 0),
+ ),
+ signal,
+ onLog: task.onLog,
+ });
+ }
const outcome = await downloadOneManaged({
channelSlug: slug,
channelConfig: downloadConfig,
@@ -619,16 +649,23 @@ export async function importVideoAction(
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
});
- if (bitchute) {
- // BitChute's shared pacing state learns what it said: a 429 backs
- // every BitChute path off (the lane, a sync, the next import).
- // Best-effort, as the batch downloads' bookkeeping is.
+ if (pacedPlatform) {
+ // The platform's shared pacing state learns what it said: a 429
+ // backs every path on it off (the lane, a sync, the next import).
+ // Best-effort, as the batch downloads' bookkeeping is. The next
+ // import there waits the floor from now, whatever the answer.
+ notePlatformGap(
+ pacedPlatform,
+ downloadGapMs(settings.sleepBetweenDownloadsSeconds, 0, 0, {
+ minSeconds: platformImportMinGapSeconds(pacedPlatform),
+ }),
+ );
const answer = importPlatformSignal(outcome);
try {
if (answer === "rate_limit" || answer === "network") {
- await recordDownloadBackoff("bitchute", paths, answer);
+ await recordDownloadBackoff(pacedPlatform, paths, answer);
} else if (answer === "clean") {
- const line = await recordPlatformClean("bitchute", paths);
+ const line = await recordPlatformClean(pacedPlatform, paths);
if (line) task.onLog(line);
}
} catch {
@@ -660,6 +697,9 @@ export async function importVideoAction(
});
}
+// How an import names a paced platform in what it says.
+const PACED_LABELS: Partial<Record<string, string>> = { bitchute: "BitChute", odysee: "Odysee" };
+
// A platform held, or in a rate-limit cooldown: a sentence, else null. An
// import asks a platform nothing while it has asked us to wait.
async function platformRefusal(
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -19,6 +19,7 @@ import type { AvailabilityHistoryEntry } from "yt-dlp-transcript-common/lib/avai
import type { SubtitleProvenance } from "yt-dlp-transcript-common/lib/subtitleProvenance";
import { formatBytes } from "yt-dlp-transcript-common/lib/format";
import { PipelineStageCard } from "../../../components/PipelineStageCard";
+import { TranscriptTracksReader } from "./cards/TranscriptTracksReader";
import { VideoNavStrip } from "./VideoNavStrip";
import { PipelineStatusStrip } from "./PipelineStatusStrip";
import { AvailabilityHistoryList } from "./cards/AvailabilityHistoryList";
@@ -322,6 +323,22 @@ export function VideoPanel({
<SubtitleDeferralLine slug={slug} videoId={videoId} />
+ {hasTranscript && (
+ <PipelineStageCard
+ id="transcript-read"
+ title="Transcript"
+ summary={
+ vttTracks.length + (hasTranscriptJson ? 1 : 0) > 1
+ ? "Read it; switch to another track where one says something else."
+ : "Read it."
+ }
+ defaultOpen={false}
+ tone="neutral"
+ >
+ <TranscriptTracksReader slug={slug} videoId={videoId} />
+ </PipelineStageCard>
+ )}
+
{vttTracks.length > 0 && (
<PipelineStageCard
id="transcript-source"
diff --git a/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptSourceSection.tsx b/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptSourceSection.tsx
@@ -5,7 +5,7 @@ import type { SubtitleProvenance } from "yt-dlp-transcript-common/lib/subtitlePr
import {
setPrimaryTranscriptAction,
} from "../../videoActions";
-import { CANONICAL_VTT, WHISPER_FILENAME } from "./videoFiles";
+import { CANONICAL_VTT, TRANSCRIPT_PIN_FILENAME, WHISPER_FILENAME } from "./videoFiles";
const PROVENANCE_LABEL: Record<SubtitleProvenance, string> = {
asr: "YouTube auto-captions",
@@ -46,10 +46,11 @@ export function TranscriptSourceSection({
<div className="flex flex-col gap-3">
<p className="text-sm text-muted-foreground">
Pick which subtitle track is this video's primary transcript. The
- chosen track is copied to <code>{CANONICAL_VTT}</code> — the canonical
- name the index and viewer read. Reversible: delete{" "}
- <code>{CANONICAL_VTT}</code> in Files to fall back to the automatic
- English pick, or choose another track to switch.
+ chosen track is copied to <code>{CANONICAL_VTT}</code> and pinned there
+ with <code>{TRANSCRIPT_PIN_FILENAME}</code>. Unpinned, the transcript
+ is the first English track with text, <code>transcript.en-orig.vtt</code>{" "}
+ first. Reversible: delete <code>{TRANSCRIPT_PIN_FILENAME}</code> in
+ Files to fall back to that pick, or choose another track to switch.
</p>
{hasWhisper && (
<p
diff --git a/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx b/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx
@@ -0,0 +1,100 @@
+"use client";
+
+// The video's transcript, read on demand, with the small track switcher when it
+// has more than one English track whose words differ (lib/captionTracks.ts).
+// Switching is a reader's choice and changes nothing on disk — "Set as
+// transcript" in the Transcript source card is how the primary changes.
+
+import { useState, useTransition } from "react";
+import type { AltTrack } from "yt-dlp-transcript-common/lib/captionTracks";
+import { trackLabels } from "yt-dlp-transcript-common/lib/captionTracks";
+import { formatTimestamp } from "yt-dlp-transcript-common/lib/vtt";
+import { readTranscriptTracksAction } from "../../videoActions";
+
+export function TranscriptTracksReader({
+ slug,
+ videoId,
+}: {
+ slug: string;
+ videoId: string;
+}) {
+ const [pending, startTransition] = useTransition();
+ const [tracks, setTracks] = useState<AltTrack[] | null>(null);
+ const [shown, setShown] = useState<string | null>(null);
+ const [error, setError] = useState<string | null>(null);
+
+ const load = () => {
+ setError(null);
+ startTransition(async () => {
+ const res = await readTranscriptTracksAction(slug, videoId);
+ if (!res.ok) {
+ setError(res.error);
+ return;
+ }
+ setTracks(res.tracks);
+ setShown(res.tracks[0]?.track ?? null);
+ });
+ };
+
+ if (!tracks) {
+ return (
+ <div className="flex flex-col gap-2">
+ <button
+ type="button"
+ onClick={load}
+ disabled={pending}
+ aria-label="read transcript"
+ className="self-start px-2 py-1 rounded border border-border text-xs hover:bg-muted disabled:opacity-50"
+ >
+ {pending ? "Reading…" : "Read transcript"}
+ </button>
+ {error && (
+ <span className="text-sm text-destructive" aria-label="read transcript error">
+ {error}
+ </span>
+ )}
+ </div>
+ );
+ }
+
+ const current = tracks.find((t) => t.track === shown) ?? tracks[0];
+ return (
+ <div className="flex flex-col gap-2">
+ {tracks.length > 1 && (
+ <label className="inline-flex items-center gap-1.5 self-start text-xs text-muted-foreground">
+ <span>Track:</span>
+ <select
+ aria-label="transcript track"
+ value={current.track}
+ onChange={(e) => setShown(e.target.value)}
+ className="rounded border border-border bg-background px-1 py-0.5 text-xs text-foreground"
+ >
+ {trackLabels(tracks.map((t) => t.track)).map((label, i) => (
+ <option key={tracks[i].track} value={tracks[i].track}>
+ {label}
+ {i === 0 ? " (default)" : ""}
+ </option>
+ ))}
+ </select>
+ </label>
+ )}
+ {current.cues.length === 0 ? (
+ <p className="text-sm text-muted-foreground">No cues in this track.</p>
+ ) : (
+ <ol
+ aria-label="transcript cues"
+ className="max-h-96 overflow-y-auto rounded border border-border divide-y divide-border text-sm"
+ >
+ {current.cues.map((c, i) => (
+ <li key={i} className="flex gap-3 px-3 py-1">
+ <span className="shrink-0 w-16 font-mono text-xs text-muted-foreground pt-0.5">
+ {formatTimestamp(c.start)}
+ </span>
+ <span className="min-w-0">{c.text}</span>
+ </li>
+ ))}
+ </ol>
+ )}
+ </div>
+ );
+}
diff --git a/editor/app/channels/[slug]/videos/[id]/components/cards/videoFiles.ts b/editor/app/channels/[slug]/videos/[id]/components/cards/videoFiles.ts
@@ -14,6 +14,8 @@ export type VideoFile = {
export const WHISPER_FILENAME = "transcript.json";
export const CANONICAL_VTT = "transcript.en.vtt";
+// lib/videoStatus.ts TRANSCRIPT_PIN_FILENAME (this module is client-side).
+export const TRANSCRIPT_PIN_FILENAME = "transcript-pin.json";
export function isTranscriptVttName(name: string): boolean {
return /^transcript\.[^.]+\.vtt$/.test(name);
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -13,6 +13,8 @@ import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/exclu
import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server";
import { loadArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg-server";
import type { ArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg";
+import { loadWaybackProvenance } from "yt-dlp-transcript-common/lib/wayback-server";
+import { waybackCaptureDate, type WaybackProvenance } from "yt-dlp-transcript-common/lib/wayback";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
readVideoMetadataForDisplay,
@@ -139,6 +141,8 @@ export default async function VideoDetailPage({
// Where an archive.org record came from (lib/archiveOrg-server.ts): the
// item and its torrent, and a mirror's original. Absent everywhere else.
const archiveOrg = await loadArchiveOrgProvenance(videoDir);
+ // What a Wayback Machine copy is a copy of (lib/wayback-server.ts).
+ const wayback = await loadWaybackProvenance(videoDir);
// The windows another tool asked this editor to fetch. One readdir of
// data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere,
// because that would be a walk of every video dir to draw one number.
@@ -191,6 +195,7 @@ export default async function VideoDetailPage({
excludedFromTruncatedCheck,
savedVideo,
archiveOrg,
+ wayback,
clipWindows,
vttProvenance,
coverage,
@@ -219,6 +224,7 @@ export default async function VideoDetailPage({
excludedFromTruncatedCheck,
savedVideo,
archiveOrg,
+ wayback,
clipWindows,
vttProvenance,
coverage,
@@ -285,6 +291,7 @@ export default async function VideoDetailPage({
)}
</div>
{archiveOrg && <ArchiveOrgProvenanceLine prov={archiveOrg} />}
+ {wayback && <WaybackProvenanceLine prov={wayback} />}
{meta.description && (
<details className="text-sm">
<summary className="cursor-pointer text-muted-foreground hover:text-foreground">
@@ -385,6 +392,23 @@ function ArchiveOrgProvenanceLine({ prov }: { prov: ArchiveOrgProvenance }) {
);
}
+// "Archived copy (Wayback Machine, <capture date>) of <original>".
+function WaybackProvenanceLine({ prov }: { prov: WaybackProvenance }) {
+ const link = "underline hover:text-foreground";
+ return (
+ <div aria-label="Wayback provenance" className="text-sm text-muted-foreground">
+ Archived copy (
+ <a href={prov.waybackUrl} target="_blank" rel="noreferrer" className={link}>
+ Wayback Machine, {waybackCaptureDate(prov.captureTs)}
+ </a>
+ ) of{" "}
+ <a href={prov.originalUrl} target="_blank" rel="noreferrer" className={`${link} break-all`}>
+ {prov.originalUrl}
+ </a>
+ </div>
+ );
+}
+
function formatUploadDate(s: string): string {
// yt-dlp emits YYYYMMDD. Render as YYYY-MM-DD; pass through anything else.
if (/^\d{8}$/.test(s)) {
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -46,6 +46,9 @@ import {
import { onDrive } from "yt-dlp-transcript-common/lib/storageHealth";
import { isTierable } from "yt-dlp-transcript-common/lib/mediaTier";
import { setExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server";
+import { pinTranscript } from "yt-dlp-transcript-common/lib/transcriptPin-server";
+import { readVideoTracks } from "yt-dlp-transcript-common/lib/captionTracks-server";
+import type { AltTrack } from "yt-dlp-transcript-common/lib/captionTracks";
import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions";
import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode";
import {
@@ -583,9 +586,10 @@ export async function deleteVideoFileAction(
// Promote a transcript.<lang>.vtt track to the canonical transcript.en.vtt so
// the index, snapshot, and viewer all treat it as the primary transcript. The
// chosen file is copied (not moved) so the original language-coded track is kept
-// and the choice stays reversible/repeatable — delete transcript.en.vtt to fall
-// back to the automatic regional-English pick, or pick a different track to
-// switch again.
+// and the choice stays reversible/repeatable — delete transcript-pin.json to
+// fall back to the automatic pick, or pick a different track to switch again.
+// The pin is what makes the copy win: the caption-track rule otherwise ranks
+// transcript.en-orig.vtt above transcript.en.vtt (lib/videoStatus.ts).
export async function setPrimaryTranscriptAction(
slug: string,
videoId: string,
@@ -611,6 +615,7 @@ export async function setPrimaryTranscriptAction(
if (filename !== VTT_FILENAME) {
await writeFileAtomic(path.join(videoDir, VTT_FILENAME), raw);
}
+ await pinTranscript(videoDir, filename);
// The video page reads the dir directly, so it reflects the new primary right
// away. The channel list + diagnostics bucket read the cached snapshot and
// refresh on the next snapshot regeneration (same as the other video actions).
@@ -620,6 +625,20 @@ export async function setPrimaryTranscriptAction(
return { ok: true };
}
+// The video's transcript tracks, primary first (lib/captionTracks.ts): the
+// primary and every other English track whose words differ from it. Read-only —
+// what the page's transcript reader shows and switches between. Choosing a
+// track there changes nothing on disk; "Set as transcript" above is how the
+// primary changes.
+export async function readTranscriptTracksAction(
+ slug: string,
+ videoId: string,
+): Promise<{ ok: true; tracks: AltTrack[] } | { ok: false; error: string }> {
+ const read = await readVideoTracks(videoDirOf(slug, videoId));
+ if (!read) return { ok: false, error: "This video has no transcript to read." };
+ return { ok: true, tracks: read.tracks };
+}
+
// A refusal carries what was submitted (lib/formState.ts).
export type DeleteDirActionResult = FormErrorState;
diff --git a/editor/e2e/import-video.spec.ts b/editor/e2e/import-video.spec.ts
@@ -113,11 +113,24 @@ test("a BitChute import runs on BitChute's queue and pace, and is never fetched
expect(queues).toEqual(["platform:bitchute"]);
}).toPass({ timeout: 15_000 });
- // Imported again: refused as already downloaded, nothing fetched.
+ // Imported again: refused as already downloaded, nothing fetched. The first
+ // run's end refreshes the page, and a click that lands during that refresh
+ // never reaches the button's handler — so click until the notice shows. An
+ // import that was NOT refused never shows it, and still fails here.
await expect(importBtn).toBeEnabled({ timeout: 30_000 });
- await importBtn.click();
- await expect(page.getByLabel("Import video notice")).toContainText(
- "Already downloaded",
- { timeout: 10_000 },
- );
+ await expect(async () => {
+ await importBtn.click();
+ await expect(page.getByLabel("Import video notice")).toContainText(
+ "Already downloaded",
+ { timeout: 3_000 },
+ );
+ }).toPass({ timeout: 30_000 });
+ // However many clicks that took, BitChute was asked once.
+ const imports: string[] = [];
+ for (const name of await readdir(resolvePath("test-transcripts/.jobs"))) {
+ if (!name.endsWith(".meta.json")) continue;
+ const m = JSON.parse(await readFile(resolvePath(`test-transcripts/.jobs/${name}`), "utf8"));
+ if (m.kind === "import-one" && m.videoId === id) imports.push(m.id);
+ }
+ expect(imports).toHaveLength(1);
});
diff --git a/editor/e2e/transcript-source.spec.ts b/editor/e2e/transcript-source.spec.ts
@@ -72,6 +72,40 @@ test("switching the transcript source promotes a track to transcript.en.vtt", as
).toBeVisible();
});
+// The page's transcript reader shows the primary and switches to another
+// English track only where its words differ — a reader's choice, nothing on
+// disk changes.
+test("the transcript reader switches to a track whose words differ", async ({
+ page,
+}) => {
+ await resetData("one-youtube-channel-with-data");
+ await writeFile(
+ resolvePath(`${DATA}/transcript.en-orig.vtt`),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nwords as spoken\n",
+ );
+ await writeFile(
+ resolvePath(`${DATA}/transcript.en.vtt`),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nwords as uploaded\n",
+ );
+
+ await page.goto(VIDEO_URL);
+ await page.getByLabel("Transcript stage summary", { exact: true }).click();
+ await page.getByRole("button", { name: "read transcript" }).click();
+
+ const cuesList = page.getByLabel("transcript cues", { exact: true });
+ await expect(cuesList).toContainText("words as spoken");
+ const switcher = page.getByLabel("transcript track", { exact: true });
+ await expect(switcher).toHaveValue("en-orig");
+ await expect(switcher.locator("option")).toHaveText([
+ "original audio captions (default)",
+ "uploaded captions",
+ ]);
+ await switcher.selectOption("en");
+ await expect(cuesList).toContainText("words as uploaded");
+ // Nothing was pinned.
+ expect(await pathExists(`${DATA}/transcript-pin.json`)).toBe(false);
+});
+
// The channel diagnostics surface a "non-standard transcript VTT name" bucket
// for videos whose transcript rides on a non-canonical VTT name.
test("diagnostics list a video with only transcript.en-US.vtt", async ({
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
+- **Search reads every English track of a video, and the transcript switches tracks.** Where a video has another English caption track whose words differ from its transcript — the uploaded captions beside the original audio's, a regional or auto-translated track — a query matches it too: a hit only that track holds says so ("in uploaded captions") and opens the transcript on that track at that moment, and a word both say is found once, in the transcript. The transcript reader shows a small "Track:" switcher beside the mode buttons on such a video; the transcript stays the default, and the choice rides on the share link (`vt`). Downloads and Copy MD take the track on show. Needs an index build and a rebuild and deploy of each site.
+- **A citation of a Wayback Machine copy links its original and the copy.** A cited record downloaded from a Wayback capture shows "Original (may be gone)", the original at the cited second where its platform takes one, and "Wayback Machine copy, <capture date>", the capture page, which plays. Its moment link is the capture: a capture URL never takes a time param.
+- **Transcripts read the original-audio captions.** Where a video has both, its transcript is YouTube's `en-orig` track (the captions of what was said) rather than the served `en`, which can reword it; a track with no text falls through to the next. Videos whose only captions are in cue blocks (some livestream recordings) have their text.
- **A report shows its revision, and every edit to it can be checked.** A report's date line ends with "revision N", linking to its history ("edited since revision N" when the report has changed since). The history page, `/reports/<id>/history/`, lists every revision, newest first: its number, date (UTC), commit hash and the sha256 of its `report.json`, what changed (claims added or removed, verdicts changed, claims edited, citations added or removed, quotes edited, title, series or subtitle changed), and each changed claim's title, text, verdict and findings with the words removed struck through and the words added marked. The same data is in `history.json` beside the page. Each report's history is its own git repository, published for cloning: `git clone <site>/reports/<id>/history/repo`. Each commit names the site as its author, with its date in UTC. The footer of the report's HTML, PDF and Markdown downloads begins with the revision number, and its sha256 can be looked up on the history page. Needs `reports export` and a rebuild and deploy of each site with reports.
- **A report can be saved whole: as one HTML page, a PDF, Markdown, or an evidence pack.** A report page's download line reads HTML · PDF · Markdown · Evidence pack · Citations JSON · CSV, each listed only when the site publishes it. The HTML is one file that opens with no network: the report with its verdicts, the document's sentences and the post screenshots inside it, numbered citations, and a reference list giving each quote's speaker, date, record, the original at its time and the moment page on the site. The PDF is that page printed. The Markdown is the same report as plain text with numbered references. The evidence pack is a zip of the page with its clips, stills and screenshots beside it, so the clips play offline. Each ends with a line naming the report's revision, its date and the start of its checksum. Needs `reports export` (or prepare) and a rebuild and deploy of each site with reports.
- **A report's claim can carry a flag, its header names the document under review, and a site with one report names it in the browser tab.** `report.json` claim `flag` (one line, at most 60 characters) shows as a small pill in the accent colour beside the claim's verdict, e.g. "No source given". On a report-only site with one report, the home page's tab title is the report's, as on the report's own page. A report's page header is its name — with a `series`, the series on one line in the accent colour and the title on the line below; without one, the title — then one small line of dates and the revision ("2026-10-04 · updated 2026-10-05 · revision 1"; "updated" only when it differs), then a card for the document under review: its title linking to the document, "<author> · <publisher> · <date>", and its archive links folded away, on a left rail in the document's colour (a source's `accent`, `"#rrggbb"`; without one, the border colour), then the subtitle. The page names no byline or site of its own. A claim that cites the document's own sentence shows that sentence — its still, else its words — on the same rail, with no link up to the card and no paraphrase beside it; a sentence of another document links "from <title>" to that document's box; a titled claim with no such sentence shows its text under the title, plain. A report's citation can say where its evidence came from (`origin`: `"subject"`, the document under review gave it; `"added"`, the report's author found it). A claim lists what the report added first, each card marked with the Archilyzer mark under its number ("Not in the article", or "Not in the source", is the mark's tooltip and what a screen reader says), then evidence of unknown origin, then what the document gave itself folded under "In the article (n)"; the reference list marks an added citation with the mark too, and a claim's flag pill wears the same mark. A fact-check's page reads in three tiers, each opened by a hairline with one, two or three dots: the quick take (the tally, the summary, and links to what the check found, every claim and the downloads); **What the check found**, every ruled claim grouped by verdict (contradicted, not found, partly, untestable, corroborated), one line each linking to the claim, with its `gist` (a new optional claim field, one line, at most 240 characters) and its flag; and **Every claim, with its evidence**, which opens with **How it was checked** (`method`, a new optional report field in markdown). A report of kind `sweep` has the first and last tiers only. Needs a rebuild and deploy of the site.
diff --git a/export/app/components/OfflineManager.tsx b/export/app/components/OfflineManager.tsx
@@ -15,6 +15,7 @@ import {
workerSupported,
type IndexHit,
} from "yt-dlp-transcript-common/components/searchIndexWorkerClient";
+import { inTrackLabel } from "yt-dlp-transcript-common/lib/captionTracks";
export type OfflineChannel = { slug: string; name: string };
@@ -336,6 +337,7 @@ function OfflineSearch({
{typeof hit.start === "number"
? ` · ${formatTime(hit.start)}`
: ""}
+ {hit.track ? ` · ${inTrackLabel(hit.track)}` : ""}
</span>
</li>
))}
diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts
@@ -292,6 +292,16 @@ export function transcriptPage() {
description: "Filmed on location with a zebra in the background.",
tags: ["news"],
cues: transcriptCues("transcript-only video"),
+ // An ALTERNATE English track (lib/captionTracks.ts): the uploaded
+ // captions, whose words differ from the primary's. "zeppelin" is said
+ // only here, at 200 s — what transcript-tracks.spec.ts searches for.
+ track: "en-orig",
+ altTracks: [
+ {
+ track: "en",
+ cues: [{ start: 200, end: 204, text: "uploaded words about a zeppelin" }],
+ },
+ ],
},
{
...makeSummary(VIDEO_CHAT_SMALL, "Small live chat"),
diff --git a/export/e2e/transcript-tracks.spec.ts b/export/e2e/transcript-tracks.spec.ts
@@ -0,0 +1,72 @@
+import { expect, test, type Page } from "@playwright/test";
+import { CHANNEL_SLUG, VIDEO_TRANSCRIPT_ONLY } from "./fixtures/data";
+import { expectModalOpen, installRoutes } from "./helpers";
+
+// A record's other English tracks (lib/captionTracks.ts). VIDEO_TRANSCRIPT_ONLY
+// carries an en-orig primary and an uploaded `en` that says "zeppelin" at
+// 200 s, which its primary never does; the other fixture videos have no
+// alternate.
+
+test.use({
+ permissions: ["clipboard-read", "clipboard-write"],
+});
+
+const SLUG = `${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}`;
+const leafInput = (page: Page) =>
+ page.locator('input[data-testid^="leaf-query-"]').first();
+
+test.describe("transcript tracks", () => {
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ });
+
+ test("the reader shows the primary and switches to the uploaded captions", async ({ page }) => {
+ await page.goto(`/?v=${SLUG}`);
+ await expectModalOpen(page);
+ const cues = page.getByTestId("cue-list");
+ await expect(cues).toContainText("transcript-only video — alpha line");
+
+ const switcher = page.getByTestId("track-switcher");
+ await expect(switcher).toHaveValue("en-orig");
+ await expect(switcher.locator("option")).toHaveText([
+ "original audio captions (default)",
+ "uploaded captions",
+ ]);
+ await switcher.selectOption("en");
+ await expect(cues).toContainText("uploaded words about a zeppelin");
+ await expect(cues).not.toContainText("alpha line");
+ expect(new URL(page.url()).searchParams.get("vt")).toBe("en");
+
+ // The share link reopens this track; back on the primary, it is off the URL.
+ await page.getByRole("button", { name: "Copy share link at current time" }).click();
+ const clip = new URL(await page.evaluate(() => navigator.clipboard.readText()));
+ expect(clip.searchParams.get("vt")).toBe("en");
+ await switcher.selectOption("en-orig");
+ await expect(cues).toContainText("alpha line");
+ expect(new URL(page.url()).searchParams.get("vt")).toBeNull();
+ });
+
+ test("a record with no alternate shows no switcher", async ({ page }) => {
+ await page.goto(`/?v=${CHANNEL_SLUG}/vid-chat-small`);
+ await expectModalOpen(page);
+ await expect(page.getByTestId("cue-list")).toContainText("small chat video");
+ await expect(page.getByTestId("track-switcher")).toHaveCount(0);
+ });
+
+ test("search finds a word only the uploaded captions hold, says so, and opens that track", async ({ page }) => {
+ await page.goto("/");
+ await leafInput(page).fill("zeppelin");
+ await page.getByTestId("search-submit").click();
+
+ const card = page.locator(`[data-result-slug="${SLUG}"]`);
+ await expect(card).toBeVisible({ timeout: 15_000 });
+ await expect(page.locator("[data-card-header]")).toHaveCount(1);
+ const hit = page.getByRole("button", { name: /in uploaded captions/i }).first();
+ await expect(hit).toContainText("zeppelin");
+ await hit.click();
+
+ await expectModalOpen(page);
+ await expect(page.getByTestId("track-switcher")).toHaveValue("en");
+ await expect(page.getByTestId("cue-list")).toContainText("uploaded words about a zeppelin");
+ });
+});
diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts
@@ -232,6 +232,16 @@ export function buildSweepInstructions(
);
steps.push(
+ `**A hit can come from another caption track.** Where a video has more ` +
+ `than one English track whose words differ, search reads them all; a ` +
+ `snippet tagged \`in uploaded captions\` (or another track) matched ` +
+ `words the primary transcript — the original audio's captions — does ` +
+ `not have there. Uploaded captions are not always what was said: read ` +
+ `the primary around that moment (\`get_transcript\`; \`track\` reads ` +
+ `the other one) before quoting, and say which track the words are from.`,
+ );
+
+ steps.push(
`**State the plan.** Report N (the enumerated total) and ` +
`\`ceil(N / ${req.batchSize})\` batches before you start. The report may ` +
`only ever claim the coverage this number justifies: N videos ` +
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -237,7 +237,7 @@ class StubSource implements ShardSource {
];
}
- private pages(ch: ChannelRef): TranscriptDetail[][] {
+ protected pages(ch: ChannelRef): TranscriptDetail[][] {
// One record per page so paging exercises multiple shard pages.
const recs = ch.slug === "chan-a" ? CHAN_A : CHAN_B;
return recs.map((r) => [r]);
@@ -1685,3 +1685,80 @@ test("server: the unadvertised channel/group singulars are still parsed", async
assert.match(out, /scope: 1 channel/);
await client.close();
});
+
+// ─── Alternate tracks (lib/captionTracks.ts) ───
+
+// chan-b plus b2: an en-orig primary and an uploaded `en` that says a word the
+// primary never does.
+const ALT_REC = vid(
+ "b2",
+ "Two tracks",
+ "chan-b",
+ cues([5, "the harbor bridge opened"]),
+ {
+ track: "en-orig",
+ altTracks: [{ track: "en", cues: cues([6, "the harbor bridge opened"], [90, "a zeppelin flew over"]) }],
+ },
+);
+class AltTrackSource extends StubSource {
+ protected override pages(ch: ChannelRef): TranscriptDetail[][] {
+ const base = super.pages(ch);
+ return ch.slug === "chan-b" ? [...base, [ALT_REC]] : base;
+ }
+}
+
+test("searchTranscripts: a word only an alternate track holds is found there, the track named", async () => {
+ const src = new AltTrackSource();
+ const res = await searchTranscripts(src, { query: "zeppelin" });
+ assert.equal(res.hits.length, 1);
+ assert.equal(res.hits[0].videoId, "b2");
+ assert.deepEqual(
+ res.hits[0].snippets.map((s) => [s.seconds, s.track]),
+ [[90, "en"]],
+ );
+ // Said by both tracks at the same moment: once, from the primary.
+ const both = await searchTranscripts(src, { query: "harbor bridge" });
+ assert.deepEqual(both.hits[0].snippets.map((s) => s.track), [undefined]);
+});
+
+test("server: a hit from an alternate track says which one", async () => {
+ const client = await connectClient(new AltTrackSource());
+ const out = firstText(
+ await client.callTool({ name: "search_transcripts", arguments: { query: "zeppelin" } }),
+ );
+ assert.match(out, /\[in uploaded captions \[1:30\]\(https:\/\/example.test\/b2\?t=90s\)\] a zeppelin flew over/);
+ await client.close();
+});
+
+test("server: get_transcript reads the primary by default and another track by `track`", async () => {
+ const client = await connectClient(new AltTrackSource());
+ const primary = firstText(
+ await client.callTool({ name: "get_transcript", arguments: { video_id: "b2" } }),
+ );
+ assert.doesNotMatch(primary, /zeppelin/);
+ assert.match(primary, /track: en-orig \(original audio captions\) — the primary/);
+ assert.match(primary, /other tracks: en \(uploaded captions\) — pass track to read one/);
+
+ const en = firstText(
+ await client.callTool({ name: "get_transcript", arguments: { video_id: "b2", track: "en" } }),
+ );
+ assert.match(en, /a zeppelin flew over/);
+ assert.match(en, /track: en \(uploaded captions\)/);
+
+ const bad = await client.callTool({
+ name: "get_transcript",
+ arguments: { video_id: "b2", track: "en-GB" },
+ });
+ assert.match(firstText(bad), /has no track "en-GB"; its tracks are: en-orig \(original audio captions\), en \(uploaded captions\)/);
+
+ // get_transcripts with a query windows a match only the alternate holds.
+ const batch = firstText(
+ await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b2"], query: "zeppelin" },
+ }),
+ );
+ assert.match(batch, /1 matching line\(s\) only in uploaded captions \(track en\), windowed/);
+ assert.match(batch, /a zeppelin flew over/);
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -10,6 +10,7 @@
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
import { postConversation, type Post } from "yt-dlp-transcript-common/lib/posts";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import { hitsAcrossTracks } from "yt-dlp-transcript-common/lib/captionTracks";
import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import {
type GroupNode,
@@ -65,6 +66,10 @@ export type Snippet = {
// reader has to be able to tell "the word appears in the description" from
// "the word was said at 0:00" — they license completely different citations.
scope?: LayerScope;
+ // A transcript hit from one of the record's ALTERNATE English tracks
+ // (lib/captionTracks.ts) — words its primary does not have there. Absent for
+ // a hit in the primary.
+ track?: string;
};
// A scope selector for a search/sweep: any mix of channel handles (slug / key /
@@ -614,13 +619,17 @@ export async function searchTranscripts(
let otherHit = false;
if (wantCues) {
- for (const cue of rec.cues ?? []) {
- if (!match(cue.text)) continue;
+ // Every English track of the record (lib/captionTracks.ts): a
+ // match only an alternate holds names that alternate.
+ for (const cue of hitsAcrossTracks(rec, (cues) =>
+ cues.filter((c) => match(c.text)),
+ )) {
matches++;
push({
clock: clock(cue.start),
seconds: cue.start,
text: truncate(cue.text, policy.snippetChars),
+ ...(cue.track ? { track: cue.track } : {}),
});
}
}
@@ -1124,6 +1133,7 @@ export async function runSearchSpec(
description: rec.description ?? "",
tags: (rec.tags ?? []).join(", "),
cues: rec.cues ?? [],
+ ...(rec.altTracks ? { altTracks: rec.altTracks } : {}),
chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [],
snippetsPerVideo,
includeSnippets,
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -15,6 +15,14 @@ import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import { isTagId, type PublishedTag } from "yt-dlp-transcript-common/lib/curatedTags";
import { groupPublishedTags } from "yt-dlp-transcript-common/lib/publishedTags";
import {
+ cuesOfTrack,
+ hitsAcrossTracks,
+ inTrackLabel,
+ recordTracks,
+ trackLabel,
+} from "yt-dlp-transcript-common/lib/captionTracks";
+import { windowedTranscript } from "yt-dlp-transcript-common/lib/search/window";
+import {
VIDEO_STATES,
isVideoState,
type VideoState,
@@ -44,6 +52,7 @@ import {
findVideo,
buildMatcher,
getWindowedTranscript,
+ MCP_POLICY,
runSearchSpec,
type SearchFilters,
type SearchResult,
@@ -582,6 +591,14 @@ export const TOOLS: Tool[] = [
type: "boolean",
description: "Prefix each caption line with a timestamp (default true).",
},
+ track: {
+ type: "string",
+ description:
+ "Optional: read one of the video's other English tracks instead of its " +
+ "primary transcript (e.g. \"en\" for the uploaded captions beside the " +
+ "original-audio \"en-orig\"). The header lists the tracks a video has; " +
+ "only tracks whose words differ from the primary are kept. Omit for the primary.",
+ },
},
required: ["video_id"],
additionalProperties: false,
@@ -1635,7 +1652,11 @@ async function handleSearch(
const stamp = base
? baseStamp(s.clock, s.seconds)
: stampMarkup(source, h, s.clock, s.seconds);
- const tag = s.scope && s.scope !== "transcripts" ? `${s.scope} ` : "";
+ const tag = s.scope && s.scope !== "transcripts"
+ ? `${s.scope} `
+ : s.track
+ ? `${inTrackLabel(s.track)} `
+ : "";
return ` - [${tag}${stamp}] ${s.text}`;
})
.join("\n");
@@ -2207,12 +2228,37 @@ async function handleGetTranscripts(
counts.length > 1
? counts.map((c) => `"${c.query}": ${c.n}`).join(", ")
: `${matchCount} matching line(s)`;
+ // The record's alternate tracks (lib/captionTracks.ts): a match only
+ // an alternate holds is windowed from that track, under its name.
+ const altBlocks: string[] = [];
+ let altMatches = 0;
+ const across = hitsAcrossTracks(record, (list) =>
+ list.filter((c) => matcher.match(c.text)),
+ );
+ for (const alt of record.altTracks ?? []) {
+ const own = new Set(across.filter((h) => h.track === alt.track).map((h) => h.text));
+ if (own.size === 0) continue;
+ const w = windowedTranscript(alt.cues, (t) => own.has(t), {
+ before,
+ after,
+ timestamps,
+ stamp,
+ maxLines: maxLines ?? MCP_POLICY.windowLineCap,
+ });
+ altMatches += w.matchCount;
+ altBlocks.push(
+ `_(${w.matchCount} matching line(s) only ${inTrackLabel(alt.track)} (track ${alt.track}), windowed)_\n${w.lines.join("\n")}`,
+ );
+ }
const body =
- matchCount === 0
+ matchCount === 0 && altMatches === 0
? `_(no lines matched ${
counts.length > 1 ? "any query" : "the query"
} in this transcript${counts.length > 1 ? ` — ${countNote}` : ""})_`
- : `_(${countNote}, windowed)_\n${lines.join("\n")}`;
+ : [
+ ...(matchCount > 0 ? [`_(${countNote}, windowed)_\n${lines.join("\n")}`] : []),
+ ...altBlocks,
+ ].join("\n\n");
blocks.push(`${head}\n\n${body}`);
} else {
const md = transcriptToMarkdown(
@@ -2377,11 +2423,38 @@ async function handleGetTranscript(
...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
...(record.platform ? { platform: record.platform } : {}),
};
- const md = transcriptToMarkdown(record, {
- timestamps: args.timestamps !== false,
- includeTags: true,
- linkForCue: (seconds) => momentLinkFor(source, link, seconds),
- });
+ // The record's tracks (lib/captionTracks.ts): the primary unless `track`
+ // names an alternate it holds.
+ const tracks = recordTracks(record);
+ const asked = typeof args.track === "string" ? args.track.trim() : "";
+ const cues = cuesOfTrack(record, asked);
+ if (asked && cues === undefined) {
+ return errorText(
+ tracks.length > 1
+ ? `video ${videoId} has no track "${asked}"; its tracks are: ${tracks.map((t) => `${t} (${trackLabel(t)})`).join(", ")}`
+ : `video ${videoId} has no track "${asked}": it has only its primary transcript`,
+ );
+ }
+ const shown = asked || record.track;
+ const extraMeta =
+ tracks.length > 1 && shown
+ ? [
+ `track: ${shown} (${trackLabel(shown)})${shown === record.track ? " — the primary" : ""}`,
+ `other tracks: ${tracks
+ .filter((t) => t !== shown)
+ .map((t) => `${t} (${trackLabel(t)}${t === record.track ? ", the primary" : ""})`)
+ .join(", ")} — pass track to read one`,
+ ]
+ : undefined;
+ const md = transcriptToMarkdown(
+ { ...record, cues },
+ {
+ timestamps: args.timestamps !== false,
+ includeTags: true,
+ linkForCue: (seconds) => momentLinkFor(source, link, seconds),
+ ...(extraMeta ? { extraMeta } : {}),
+ },
+ );
return text(md);
}
@@ -2405,12 +2478,23 @@ async function handleGetMetadata(
typeof args.channel === "string" ? args.channel : undefined,
);
if (!found) return errorText(`video not found: ${videoId}`);
- const { cues, ...meta } = found.record;
+ const { cues, altTracks, ...meta } = found.record;
const slug = found.record.slug;
+ // An alternate's cues are not metadata; its id and label are.
+ const otherTracks = (altTracks ?? []).map((t) => ({
+ track: t.track,
+ label: trackLabel(t.track),
+ cueCount: t.cues.length,
+ }));
const lines: string[] = [
JSON.stringify(
- { ...meta, channelName: found.ch.name, cueCount: cues?.length ?? 0 },
+ {
+ ...meta,
+ channelName: found.ch.name,
+ cueCount: cues?.length ?? 0,
+ ...(otherTracks.length > 0 ? { otherTracks } : {}),
+ },
null,
2,
),
@@ -2912,7 +2996,12 @@ function renderScopedSnippet(
s: ScopedSnippet,
): string {
if (s.seconds > 0) {
- const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `;
+ const tag =
+ s.scope === "transcripts"
+ ? s.track
+ ? `${inTrackLabel(s.track)} `
+ : ""
+ : `${s.track ?? s.scope} `;
return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`;
}
return ` - [${s.scope}] ${s.text}`;
diff --git a/umtool/report-to-video/cues.mjs b/umtool/report-to-video/cues.mjs
@@ -56,7 +56,7 @@
// a channel the stale copy lacked is found. Manifest and shard URLs carry no
// version, so a cue window does not move because of it.
-import { readFile, writeFile, mkdir, lstat, readlink } from "node:fs/promises";
+import { readFile, writeFile, mkdir, lstat, readlink, readdir, stat } from "node:fs/promises";
import path from "node:path";
import os from "node:os";
import { createHash } from "node:crypto";
@@ -114,6 +114,40 @@ export function siteOriginFromManifest(manifest) {
return null;
}
+// A TWIN OF `CAPTION_TRACK_RULE_VERSION` (`common/lib/videoStatus.ts`) — the
+// caption-track rule a `transcript.cues.json` records as `captionTrackRule`.
+// Copied for the same reason as the text guard below: umtool's bins run under
+// plain node. Change one, change the other.
+export const CAPTION_TRACK_RULE_VERSION = 1;
+const ENGLISH_VTT_RE = /^transcript\.en(?:-[^.]+)?\.vtt$/;
+
+// A local caption record normalized under an older caption-track rule, where
+// the rule now reads other words — the same test as `isCuesJsonFresh`
+// (`common/controller/normalizeTranscript.ts`, followsCaptionTrackRule): no
+// cues at all (a cue-block VTT used to parse to nothing; a lone track with word
+// timing is genuinely empty and is let through), or a transcript.en.vtt and
+// transcript.en-orig.vtt that differ (the older rule read en; en-orig is read
+// now). Cutting from it would widen clips on text the corpus no longer
+// publishes, so it is refused with the fix, not used.
+async function staleCaptionRecord(videoDir, record) {
+ if (record?.source !== "vtt" || record.captionTrackRule === CAPTION_TRACK_RULE_VERSION) {
+ return false;
+ }
+ const entries = await readdir(videoDir).catch(() => []);
+ const english = entries.filter((e) => ENGLISH_VTT_RE.test(e));
+ if (!Array.isArray(record.cues) || record.cues.length === 0) {
+ if (english.length !== 1) return true;
+ const head = (await readFile(path.join(videoDir, english[0]), "utf8").catch(() => "")).slice(0, 2048);
+ return !/<\d{2}:\d{2}:\d{2}\.\d{3}>/.test(head);
+ }
+ if (!entries.includes("transcript.en.vtt") || !entries.includes("transcript.en-orig.vtt")) return false;
+ const [en, orig] = await Promise.all([
+ stat(path.join(videoDir, "transcript.en.vtt")),
+ stat(path.join(videoDir, "transcript.en-orig.vtt")),
+ ]);
+ return en.size !== orig.size;
+}
+
export class CueLookupError extends Error {
constructor(message, { channelSlug, videoId, tried }) {
super(message);
@@ -423,6 +457,14 @@ export function createCueSource({
await assertChannelReachable(channelSlug);
const p = path.join(channelsDir, channelSlug, "data", videoId, "transcript.cues.json");
const parsed = JSON.parse(await readFile(p, "utf8"));
+ if (await staleCaptionRecord(path.dirname(p), parsed)) {
+ throw new CueLookupError(
+ `${channelSlug}/${videoId}: transcript.cues.json was normalized under an older caption-track rule ` +
+ `(it holds the served en track's words where en-orig is read now, or no words at all). ` +
+ `Run Normalize for channel ${channelSlug} in the editor, or pass --cue-source http.`,
+ { channelSlug, videoId, tried: [p] },
+ );
+ }
return { ...parsed, from: "local" };
}
diff --git a/umtool/report-to-video/cues.test.mjs b/umtool/report-to-video/cues.test.mjs
@@ -12,6 +12,7 @@ import { tmpdir } from "node:os";
import path from "node:path";
import {
+ CAPTION_TRACK_RULE_VERSION,
createCueSource,
pageFileName,
pageUrlFrom,
@@ -148,6 +149,58 @@ test("a local corpus is preferred over the network", async () => {
}
});
+// --- a caption record normalized under an older caption-track rule ---------
+
+test("a local caption record from before the caption-track rule is refused where the rule could read other words", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "cues-rule-"));
+ try {
+ const vdir = path.join(dir, "chan", "data", "vid1");
+ await mkdir(vdir, { recursive: true });
+ await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n\nserved words\n");
+ await writeFile(path.join(vdir, "transcript.en-orig.vtt"), "WEBVTT\n\nwhat was said\n");
+ await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" }));
+ const seen = [];
+ const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen) });
+ await assert.rejects(src.load("chan", "vid1"), (err) => {
+ assert.equal(err.name, "CueLookupError");
+ assert.match(err.message, /older caption-track rule.*Run Normalize for channel chan/s);
+ return true;
+ });
+ assert.deepEqual(seen, [], "never answered from the archive instead");
+
+ // Normalized under the current rule: read as usual.
+ await writeFile(
+ path.join(vdir, "transcript.cues.json"),
+ JSON.stringify({ ...RECORD, source: "vtt", captionTrackRule: CAPTION_TRACK_RULE_VERSION }),
+ );
+ assert.equal((await src.load("chan", "vid1")).from, "local");
+
+ // An unversioned record beside a byte-identical pair reads the same either way.
+ await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n\nwhat was said\n");
+ await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" }));
+ assert.equal((await src.load("chan", "vid1")).from, "local");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a lone-track caption record with cues needs no rule to be read", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "cues-rule-"));
+ try {
+ const vdir = path.join(dir, "chan", "data", "vid1");
+ await mkdir(vdir, { recursive: true });
+ await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n");
+ await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" }));
+ const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES) });
+ assert.equal((await src.load("chan", "vid1")).from, "local");
+ // …but one with no cues is refused: a cue-block track used to parse to none.
+ await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt", cues: [] }));
+ await assert.rejects(src.load("chan", "vid1"), /older caption-track rule/);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
// --- a channel the corpus holds but whose text it cannot read ---------------
//
// The bug these cover: on the RETIRED layout `data/` is a symlink to another