Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 8a34d340a63f625381acdcc8c58a118bbaa81667
parent c5b8afaf99c09b3ec2068dd032abeec7dbbc51fd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 27 Aug 2026 10:31:40 -0400

backfill: the subtitle-channel re-acquire is pinned end to end

e2e case (11) in backfill.spec.ts seeds a handling: "youtube" channel with
captions and no audio — the shape every video on a subtitle channel has, and the
shape that makes diarization missing-input — arms allowRedownload, and runs the
speaker lane. It asserts all four properties the ~16,000 wasted fetches lacked:
diarization.json lands (so audio was downloaded), the audio does not survive the
item that fetched it, transcript.en.vtt is byte-identical to what was seeded (no
subtitle re-fetch), the last download attempt records handling "transcribe", and
transcript.cues.json exists with an mtime at or after metadata.info.json's.

The fake yt-dlp needed no changes: its prefetch branch rewrites metadata for any
invocation and its -x branch writes audio.mp3, which is the same path
auto-subs-replace.spec.ts already drives for the same override.

And the four places whose prose said "re-acquire" without saying "audio":
settings.ts's allowRedownload doc comment and the operator-facing hints on the
lane settings checkbox and the speakers stage card now say the fetch is audio,
that a subtitle-only channel gets a per-video transcribe override, and that the
channel's stored config is not changed. operations.ts's "backfillReacquire
answers it by fetching AUDIO" is now true as written and is left alone.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

Diffstat:
Mcommon/lib/settings.ts | 7+++++++
Meditor/app/channels/[slug]/components/stages/SpeakersStage.tsx | 2+-
Meditor/app/operations/components/settings/LaneSettingsForm.tsx | 5++++-
Meditor/e2e/backfill.spec.ts | 132++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
4 files changed, 143 insertions(+), 3 deletions(-)

diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -328,6 +328,13 @@ export type BackfillSettings = { // reachable work, against 45 GB free at 97% full. When on, each re-fetched // file is removed in a `finally` as soon as the backfill has used it, unless // the video is marked do-not-clean. + // + // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"` + // channel — which normally only fetches subtitles — the re-acquire applies a + // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read; + // the channel's stored config is not changed. Without that override the fetch + // re-downloads the captions the video already has and lands nothing, which is + // what happened to ~16,000 videos on eight channels in 2026-08. allowRedownload: boolean; }; diff --git a/editor/app/channels/[slug]/components/stages/SpeakersStage.tsx b/editor/app/channels/[slug]/components/stages/SpeakersStage.tsx @@ -163,7 +163,7 @@ export function SpeakersStage({ {missingInput === 1 ? "video needs" : "videos need"} their media re-acquired first {allowRedownload - ? " — re-download is on, so this run will fetch and then delete it, bounded by the free-disk floor." + ? " — re-download is on, so this run will fetch audio (even on a subtitle-only channel) and then delete it, bounded by the free-disk floor." : " — re-download is off, so this run skips them."} </> ), diff --git a/editor/app/operations/components/settings/LaneSettingsForm.tsx b/editor/app/operations/components/settings/LaneSettingsForm.tsx @@ -109,7 +109,10 @@ export function LaneSettingsForm({ the reachable work, against 45 GB free. With this on, each file is fetched, used, and <strong>deleted again immediately</strong>{" "} (unless the video is marked &quot;do not clean&quot;), and nothing - starts at all when free space is under the disk floor. + starts at all when free space is under the disk floor. What it + fetches is <strong>audio</strong>, including on channels that + normally only download subtitles — those use a per-video + transcribe override, and their stored config is not changed. </span> </span> </label> diff --git a/editor/e2e/backfill.spec.ts b/editor/e2e/backfill.spec.ts @@ -1,4 +1,4 @@ -import { mkdir, readdir, rm, writeFile } from "node:fs/promises"; +import { mkdir, readdir, readFile, rm, stat, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import type { APIRequestContext } from "@playwright/test"; import { baseUrl } from "./baseUrl"; @@ -814,6 +814,136 @@ test("the dashboard pauses and resumes the backfill lane", async ({ page }) => { await expect.poll(laneEnabled, { timeout: 30_000 }).toBe(true); }); +// (11) THE RE-DOWNLOAD ON A SUBTITLE CHANNEL: audio, not captions. +// +// A `handling: "youtube"` channel downloads with --skip-download --write-subs +// --write-auto-subs. Run a re-acquire with that config and yt-dlp re-fetches the +// captions the video already has, rewrites metadata.info.json, and lands nothing +// a diarizer can read — which is what happened to ~16,000 videos on eight +// channels between 2026-08-22 and 08-26: zero diarizations, and every touched +// video left reading `deferred` to the digest lane because the metadata rewrite +// had made transcript.cues.json stale. +// +// Both halves are pinned here: the transcribe override that makes there be audio +// at all, and the re-normalize that keeps the cues fresh afterwards. +const YT_SLUG = "test-yt-subs"; +const YT_ROOT = `test-transcripts/channels/${YT_SLUG}`; + +function ytRel(videoId: string, file: string): string { + return `${YT_ROOT}/data/${videoId}/${file}`; +} + +// A youtube-handling channel with captions and NO audio — the shape every video +// on a subtitle channel has, and the shape that makes diarization missing-input. +// audioFormat is pinned so the forced transcribe-handling download lands on +// audio.mp3, exactly as auto-subs-replace.spec.ts seeds it. +const SEEDED_VTT = + "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nSeeded caption line one.\n\n" + + "00:00:05.000 --> 00:00:10.000\nSeeded caption line two.\n"; + +async function seedSubtitleChannel(videoId: string): Promise<void> { + const dir = resolvePath(`${YT_ROOT}/data/${videoId}`); + await mkdir(dir, { recursive: true }); + await writeFile( + resolvePath(`${YT_ROOT}/config.json`), + JSON.stringify({ + handling: "youtube", + name: "Subtitle test channel", + url: "https://www.youtube.com/@subs/videos", + audioFormat: "mp3", + }) + "\n", + ); + // findVideoSourceUrl resolves the id's URL from the stored playlist. + await writeFile( + resolvePath(`${YT_ROOT}/playlist`), + `https://www.youtube.com/watch?v=${videoId}\n`, + ); + await writeFile(`${dir}/transcript.en.vtt`, SEEDED_VTT); + await writeFile( + `${dir}/metadata.info.json`, + JSON.stringify({ + id: videoId, + title: `Synthetic ${videoId}`, + upload_date: "20240101", + duration: 60, + extractor_key: "Youtube", + webpage_url: `https://www.youtube.com/watch?v=${videoId}`, + subtitles: {}, + automatic_captions: { en: [{ ext: "vtt", url: "fake://subs" }] }, + }) + "\n", + ); + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); +} + +test("re-acquiring on a subtitle channel downloads audio and re-normalizes the cues", async ({ + page, +}) => { + test.setTimeout(SLOW); + const VID = "ytsubs000001"; + await resetData(); + await writeSettings( + backfillSettings({ backfill: { allowRedownload: true } }), + ); + await seedSubtitleChannel(VID); + expect(await pathExists(ytRel(VID, "audio.mp3"))).toBe(false); + + await generateReport(page, YT_SLUG); + await page.goto(channelStage(YT_SLUG, "speakers")); + await page + .getByRole("button", { name: "Run speaker work", exact: true }) + .click(); + + // The lane got what it needed — which it can only have done by downloading + // AUDIO. With the channel's own config this assertion is what failed silently + // 16,000 times as "nothing usable landed". + await expect + .poll(async () => pathExists(ytRel(VID, "diarization.json")), { + timeout: 60_000, + }) + .toBe(true); + + // …and the audio still does not survive the item that fetched it (case (4)'s + // property, on a channel that never had audio to begin with). + await expect + .poll( + async () => { + const entries = await readdir( + resolvePath(`${YT_ROOT}/data/${VID}`), + ).catch(() => [] as string[]); + return entries.filter( + (e) => e.startsWith("audio.") && !e.endsWith(".info.json"), + ); + }, + { timeout: 30_000 }, + ) + .toEqual([]); + + // NO SUBTITLE RE-FETCH. The captions on disk are byte-identical to what was + // seeded; the fake yt-dlp's youtube-handling mode writes its own transcript, + // so any drift here means the override did not apply. + expect(await readFile(resolvePath(ytRel(VID, "transcript.en.vtt")), "utf8")) + .toBe(SEEDED_VTT); + + // The override, as the download itself recorded it. + const outcome = await readJson<{ + attempts: { kind: string; handling: string }[]; + }>(ytRel(VID, "download-outcome.json")); + const attempts = outcome.attempts ?? []; + expect(attempts.length).toBeGreaterThan(0); + expect(attempts[attempts.length - 1]?.handling).toBe("transcribe"); + + // THE SECOND HALF: the fetch rewrote metadata.info.json, and the cues were + // re-normalized after it. Without this the next lane in the same sweep skips + // the video as no-transcript and the digest lane defers it — 16,081 videos + // corpus-wide, until an operator runs Normalize. + expect(await pathExists(ytRel(VID, "transcript.cues.json"))).toBe(true); + const [cuesStat, metaStat] = await Promise.all([ + stat(resolvePath(ytRel(VID, "transcript.cues.json"))), + stat(resolvePath(ytRel(VID, "metadata.info.json"))), + ]); + expect(cuesStat.mtimeMs).toBeGreaterThanOrEqual(metaStat.mtimeMs); +}); + // (N) THE REGRESSION THIS STEP'S DESIGN EXISTS TO PREVENT. // // The channel snapshot now carries a work-list entry for EVERY catalog