Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 8290da35c3131277080bff59049135f1da427b7e
parent 1245a148d42d1657342e42166a3b425474f45c36
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 09:19:14 -0400

captions: an unversioned cues.json is stale only where the rule reads other words

Measured on the corpus: 52,071 caption records have both transcript.en.vtt
and transcript.en-orig.vtt, and 50,461 of those pairs are byte-identical
(every equal-size pair was). Marking all of them stale would have made a
Normalize rewrite 52k files and held their digests in the meantime for text
that does not change.

An unversioned caption cues.json is now stale only when it holds no cues (a
cue-block parse, or the next track, may find text — unless its one track has
word timing), or when en and en-orig both exist and differ in size. One to
three small reads (head, tail, two stats), never a parse. umtool's local cue
lookup applies the same test.

Also: the index's helper comments in order.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/buildIndex.ts | 8++++----
Mcommon/controller/normalizeCaptionTrack.test.ts | 25++++++++++++++++++++++---
Mcommon/controller/normalizeTranscript.ts | 40++++++++++++++++++++++++++--------------
Meditor/CHANGELOG.md | 2+-
Mumtool/report-to-video/cues.mjs | 29+++++++++++++++++++++--------
Mumtool/report-to-video/cues.test.mjs | 9+++++++--
6 files changed, 81 insertions(+), 32 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -219,16 +219,16 @@ const PLATFORM_LABELS_KEY = "platformLabels"; // how many now read different text, how many had none and now do. const CAPTION_TRACK_KEY = "captionTrackRule"; -// Whether the cue list stored under a key is empty or absent, WITHOUT decoding -// it: a non-empty list of cues is far longer than the few bytes an empty -// msgpack array takes, and decoding every caption record of the corpus to ask -// this would cost the full read the pass exists to avoid. // A cue list's text, for "did the words change" — timing alone is not a // different transcript. function cueText(list: readonly Cue[]): string { return list.map((c) => c.text).join("\n"); } +// Whether the cue list stored under a key is empty or absent, WITHOUT decoding +// it: a non-empty list of cues is far longer than the few bytes an empty +// msgpack array takes, and decoding every caption record of the corpus to ask +// this would cost the full read the pass exists to avoid. function storedCuesEmpty(db: { getBinaryFast(key: IndexKey): Buffer | undefined }, key: IndexKey): boolean { const raw = db.getBinaryFast(key); return raw === undefined || raw.length <= 4; diff --git a/common/controller/normalizeCaptionTrack.test.ts b/common/controller/normalizeCaptionTrack.test.ts @@ -59,12 +59,31 @@ test("normalize reads en-orig, and records the track and the rule in the file's const n = await readNormalizedTranscript(path.join(dir, "transcript.cues.json")); assert.equal(n?.vttFile, "transcript.en-orig.vtt"); assert.equal(n?.captionTrackRule, CAPTION_TRACK_RULE_VERSION); - assert.equal(n?.cues[0].text, "are talking about the harbor"); + assert.equal(n?.cues?.[0].text, "are talking about the harbor"); assert.equal((await isCuesJsonFresh(dir)).fresh, true); assert.equal((await normalizeTranscript({ videoDir: dir, channelSlug: "c" })).status, "fresh"); }); -test("a cues.json from before the rule is stale beside two English tracks, and normalize rewrites it", async () => { +test("a cues.json from before the rule is fresh beside a byte-identical en/en-orig pair: it reads the same", async () => { + const dir = videoDir({ "transcript.en.vtt": ROLLING, "transcript.en-orig.vtt": ROLLING }); + oldCuesJson(dir, [{ start: 0, end: 1, text: "same words" }]); + assert.equal((await isCuesJsonFresh(dir)).fresh, true); + assert.equal((await normalizeTranscript({ videoDir: dir, channelSlug: "c" })).status, "fresh"); +}); + +test("a cues.json from before the rule is fresh beside two tracks the older rule ranked the same way", async () => { + const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-US.vtt": ROLLING }); + oldCuesJson(dir, [{ start: 0, end: 1, text: "kept" }]); + assert.equal((await isCuesJsonFresh(dir)).fresh, true); +}); + +test("a cues.json from before the rule with no cues is stale beside two tracks: the fallback may find text", async () => { + const dir = videoDir({ "transcript.en.vtt": ROLLING, "transcript.en-orig.vtt": ROLLING }); + oldCuesJson(dir, []); + assert.equal((await isCuesJsonFresh(dir)).fresh, false); +}); + +test("a cues.json from before the rule is stale beside an en and en-orig that differ, and normalize rewrites it", async () => { const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-orig.vtt": ROLLING }); oldCuesJson(dir, [{ start: 0, end: 1, text: "served words" }]); assert.deepEqual(await isCuesJsonFresh(dir), { @@ -82,7 +101,7 @@ test("a cues.json from before the rule is stale over a lone cue-block track (it assert.equal((await isCuesJsonFresh(dir)).fresh, false); await normalizeTranscript({ videoDir: dir, channelSlug: "c" }); const n = await readNormalizedTranscript(path.join(dir, "transcript.cues.json")); - assert.equal(n?.cues.length, 7); + assert.equal(n?.cues?.length, 7); assert.equal((await isCuesJsonFresh(dir)).fresh, true); }); diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts @@ -22,6 +22,8 @@ import { CAPTION_TRACK_RULE_VERSION, CUES_JSON_FILENAME, META_FILENAME, + ORIG_VTT_FILENAME, + VTT_FILENAME, WHISPER_FILENAME, captionInputs, englishVttsByPreference, @@ -227,9 +229,9 @@ export type CuesFreshReason = // operator's pin), and a cues.json that is new enough must also have been made // under the current caption-track rule. Made under an older one, it is `stale` // where the rule could read it differently: its cues may come from a track the -// rule no longer picks (a served `en` beside an `en-orig`), or be the zero cues -// a cue-block VTT used to parse to (followsCaptionTrackRule). Telling costs one -// or two small reads, of a file's head or tail, never a parse. +// rule no longer picks (a served `en` that differs from the `en-orig` beside +// it), or be the zero cues a cue-block VTT used to parse to +// (followsCaptionTrackRule). Telling costs a few small reads, never a parse. export async function isCuesJsonFresh( videoDir: string, ): Promise<{ fresh: boolean; reason: CuesFreshReason; cuesPath: string }> { @@ -294,17 +296,22 @@ async function readTail(file: string, bytes = 64): Promise<string> { } // Whether a caption cues.json, already new enough by mtime, was made under the -// current caption-track rule — or could not read differently under it: +// current caption-track rule — normalize records it as `captionTrackRule` in +// the file's first bytes — or, made before the rule had a version, holds what +// the rule would read anyway. The older rule ranked transcript.en.vtt first +// and parsed a cue-block VTT to no cues, so an unversioned file is stale only: // -// one English VTT with word timing — nothing to choose, and the parse of -// that shape did not change; -// one English VTT in cue blocks, and the file holds cues — the older parse -// read that shape as NO cues, so a file with some was not made by it. The -// cues are the last key normalize writes, so "none" is the file's tail. +// when it holds NO cues (the cue-block parse, or the fallback to the next +// track, may find text now) — the cues are the last key normalize writes, +// so that is the file's tail; unless its one English VTT has word timing, +// whose parse did not change and which had nothing to choose from; +// when transcript.en.vtt and transcript.en-orig.vtt are both present and +// differ — it was read from en, and en-orig is read now. A byte-identical +// pair (equal size: on this corpus every one of 50,461 equal-size pairs was +// byte-identical) reads the same either way. // -// Anything else must carry `captionTrackRule`, which normalize writes into the -// file's first bytes. A mis-read head only costs a rewrite, after which the -// record is there. +// One to three small reads (a head, a tail, two stats), never a parse. A +// mis-read only costs a rewrite, after which the record is there. async function followsCaptionTrackRule( videoDir: string, cuesPath: string, @@ -318,8 +325,13 @@ async function followsCaptionTrackRule( } const m = (await readHead(cuesPath)).match(/"captionTrackRule"\s*:\s*(\d+)/); if (m !== null && Number(m[1]) === CAPTION_TRACK_RULE_VERSION) return true; - if (vtts.length > 1) return false; - return !/"cues"\s*:\s*\[\s*\]\s*\}\s*$/.test(await readTail(cuesPath)); + if (/"cues"\s*:\s*\[\s*\]\s*\}\s*$/.test(await readTail(cuesPath))) return false; + if (!entries.includes(VTT_FILENAME) || !entries.includes(ORIG_VTT_FILENAME)) return true; + const [en, orig] = await Promise.all([ + stat(path.join(videoDir, VTT_FILENAME)), + stat(path.join(videoDir, ORIG_VTT_FILENAME)), + ]); + return en.size === orig.size; } catch { return false; } diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -3,7 +3,7 @@ ## [Unreleased] - **A video's captions are read from its original-audio track first, and a track with no text never hides one that has it.** Where YouTube serves both, `transcript.en-orig.vtt` (the captions of the original audio) is read before `transcript.en.vtt`, whose text can be a rewrite of what was said; then regional tracks (`en-US`, `en-GB`, …), then auto-translated `en-en-*` ones. The transcript is the first track in that order that has cues. One rule (`englishVttsByPreference` / `readEnglishVttCues` in `common/lib/videoStatus.ts`) serves the index, normalize and report compose, so search, the export, the MCP and report videos read the same words. **Set as transcript** on a video's page copies the chosen track to `transcript.en.vtt` and pins it there with `transcript-pin.json`, which ranks it first; deleting `transcript-pin.json` returns the video to the automatic pick. The served `en` track is listed as an alternate subtitle track where `en-orig` is the transcript. Videos with a human-made `en` track are not counted as auto-captions-only, so the replace-auto-captions lane still leaves them alone. - **Captions in cue blocks are read.** A VTT with no inline word timing (uploaded captions, and the `en` track YouTube serves for some livestream recordings: two lines a cue, `&nbsp;` at each line end) is read cue by cue; it used to parse to no cues, which left those videos with no text in the index. -- **The next index build re-reads the caption records the new rule reaches, once.** A record with more than one English track, or with no cues stored, is re-read from disk; nothing else is. The build log says `Caption track v1: N record(s) re-read.` and, per channel, how many now read different text and how many had none and now do. The version is recorded only when no channel is held. A `transcript.cues.json` written before the rule is stale where the rule could read something else (more than one English track, or a lone track with no word timing), so a **Normalize** run rewrites exactly those; a cues file now records the track it was read from (`vttFile`) and the rule (`captionTrackRule`). +- **The next index build re-reads the caption records the new rule reaches, once.** A record with more than one English track, or with no cues stored, is re-read from disk; nothing else is. The build log says `Caption track v1: N record(s) re-read.` and, per channel, how many now read different text and how many had none and now do. The version is recorded only when no channel is held. A `transcript.cues.json` written before the rule is stale only where the rule reads something else — its `transcript.en.vtt` and `transcript.en-orig.vtt` differ, or it holds no cues and a cue-block parse or the next track may have them — so a **Normalize** run rewrites exactly those, and the digest and attribution lanes hold them until it does; a cues file now records the track it was read from (`vttFile`) and the rule (`captionTrackRule`). umtool's report-to-video refuses such a stale local record by name rather than cut from it. - **A cited moment at the very end of a recording prepares.** Prepare evidence media cuts a clip whose padding runs past the recording's end at the end (the recording's duration from its metadata), where it found no media for the padded span; a span that starts past the end is still refused. report-to-video keeps its strict rule. - **Exporting a changed report records a new revision of it.** `reports export` (and **Export reports** on a site's Reports tab, and the end of a prepare) commits a revision to the report's own git history, `sites/<site>/reports/<id>/history-git/`, whenever its `report.json` changed since the last one: the `report.json`, its Markdown export and the checksums of every export file, with a message of `Revision N` and a summary of the change. A re-export of an unchanged report records nothing. The commits carry the site's name and a `noreply@<site>.invalid` address with dates in UTC, never your git name, email or time zone. The Reports tab shows each report's revision, its commit and the last change under **Exports**, and the site's next build publishes the history. Add `history-git/` to the corpus repository's `.gitignore`. - **archive.org files come over BitTorrent when possible, else straight from archive.org — never through yt-dlp.** The chosen file of an archive.org import is fetched from the item's own torrent (`<identifier>_archive.torrent`, which lists archive.org as a web seed, so other peers take load off archive.org) with aria2c, only that file of the item, and seeded afterwards for 10 minutes or to a ratio of 1, whichever comes first; the log shows "torrent: <file> (n of m pieces, peers p, web seed yes)" and "seeding 10 min…". With no aria2c, a torrent that does not carry the file, or no progress for 5 minutes, it is downloaded directly from `archive.org/download/…` instead (resumable, backing off on 429/503), and the log says "fell back to direct download: <reason>". Every file is checked against archive.org's sha1/md5: a mismatch is downloaded once more directly, a second one fails the record. The record is written from the item's metadata: `metadata.info.json` with the file's page, the canonical id, the duration ffprobe measures and archive.org's playable copies of the file, the `archiveorg.json` provenance (a mirror's original title, date and uploader), and `audio.<fmt>` — an audio file already in the channel's format is used as is, anything else goes through the app's audio extraction, a video kept in the saved-video store when the channel keeps sources. An .avi/.mpeg/.flac/.wav original is fetched as archive.org's mp4 or mp3 of it. aria2c runs in its own process group: cancelling the job stops it and everything it started, and it stops itself if the editor exits. New settings block `archiveOrg` (`torrent`, `seedMinutes`, `seedRatio`, `stallMinutes`, `maxPeers`, `maxDownloadKiBps`, `maxUploadKiBps`), `ARIA2C_BIN`, an aria2c row in `archilyzer doctor`, and `aria2` in the runtime Docker images. diff --git a/umtool/report-to-video/cues.mjs b/umtool/report-to-video/cues.mjs @@ -56,7 +56,7 @@ // a channel the stale copy lacked is found. Manifest and shard URLs carry no // version, so a cue window does not move because of it. -import { readFile, writeFile, mkdir, lstat, readlink, readdir } from "node:fs/promises"; +import { readFile, writeFile, mkdir, lstat, readlink, readdir, stat } from "node:fs/promises"; import path from "node:path"; import os from "node:os"; import { createHash } from "node:crypto"; @@ -122,17 +122,30 @@ export const CAPTION_TRACK_RULE_VERSION = 1; const ENGLISH_VTT_RE = /^transcript\.en(?:-[^.]+)?\.vtt$/; // A local caption record normalized under an older caption-track rule, where -// the rule could now read other words: more than one English VTT (the served -// `en` used to win over `en-orig`), or no cues at all (a cue-block VTT used to -// parse to nothing). Cutting from it would widen clips on text the corpus no -// longer publishes, so it is refused with the fix, not used. +// the rule now reads other words — the same test as `isCuesJsonFresh` +// (`common/controller/normalizeTranscript.ts`, followsCaptionTrackRule): no +// cues at all (a cue-block VTT used to parse to nothing; a lone track with word +// timing is genuinely empty and is let through), or a transcript.en.vtt and +// transcript.en-orig.vtt that differ (the older rule read en; en-orig is read +// now). Cutting from it would widen clips on text the corpus no longer +// publishes, so it is refused with the fix, not used. async function staleCaptionRecord(videoDir, record) { if (record?.source !== "vtt" || record.captionTrackRule === CAPTION_TRACK_RULE_VERSION) { return false; } - if (!Array.isArray(record.cues) || record.cues.length === 0) return true; const entries = await readdir(videoDir).catch(() => []); - return entries.filter((e) => ENGLISH_VTT_RE.test(e)).length > 1; + const english = entries.filter((e) => ENGLISH_VTT_RE.test(e)); + if (!Array.isArray(record.cues) || record.cues.length === 0) { + if (english.length !== 1) return true; + const head = (await readFile(path.join(videoDir, english[0]), "utf8").catch(() => "")).slice(0, 2048); + return !/<\d{2}:\d{2}:\d{2}\.\d{3}>/.test(head); + } + if (!entries.includes("transcript.en.vtt") || !entries.includes("transcript.en-orig.vtt")) return false; + const [en, orig] = await Promise.all([ + stat(path.join(videoDir, "transcript.en.vtt")), + stat(path.join(videoDir, "transcript.en-orig.vtt")), + ]); + return en.size !== orig.size; } export class CueLookupError extends Error { @@ -447,7 +460,7 @@ export function createCueSource({ if (await staleCaptionRecord(path.dirname(p), parsed)) { throw new CueLookupError( `${channelSlug}/${videoId}: transcript.cues.json was normalized under an older caption-track rule ` + - `(it may hold the served en track's words, or none, where en-orig is read now). ` + + `(it holds the served en track's words where en-orig is read now, or no words at all). ` + `Run Normalize for channel ${channelSlug} in the editor, or pass --cue-source http.`, { channelSlug, videoId, tried: [p] }, ); diff --git a/umtool/report-to-video/cues.test.mjs b/umtool/report-to-video/cues.test.mjs @@ -156,8 +156,8 @@ test("a local caption record from before the caption-track rule is refused where try { const vdir = path.join(dir, "chan", "data", "vid1"); await mkdir(vdir, { recursive: true }); - await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n"); - await writeFile(path.join(vdir, "transcript.en-orig.vtt"), "WEBVTT\n"); + await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n\nserved words\n"); + await writeFile(path.join(vdir, "transcript.en-orig.vtt"), "WEBVTT\n\nwhat was said\n"); await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" })); const seen = []; const src = createCueSource({ channelsDir: dir, siteOrigin: ORIGIN, cacheDir: null, fetchImpl: stubFetch(ROUTES, seen) }); @@ -174,6 +174,11 @@ test("a local caption record from before the caption-track rule is refused where JSON.stringify({ ...RECORD, source: "vtt", captionTrackRule: CAPTION_TRACK_RULE_VERSION }), ); assert.equal((await src.load("chan", "vid1")).from, "local"); + + // An unversioned record beside a byte-identical pair reads the same either way. + await writeFile(path.join(vdir, "transcript.en.vtt"), "WEBVTT\n\nwhat was said\n"); + await writeFile(path.join(vdir, "transcript.cues.json"), JSON.stringify({ ...RECORD, source: "vtt" })); + assert.equal((await src.load("chan", "vid1")).from, "local"); } finally { await rm(dir, { recursive: true, force: true }); }