Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 7f6e7e47cd6c213697e36aa2730192ee5621ef3b
parent 04b16332096449e83ef60ca3718a7af44f061fa4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  1 Jun 2026 16:38:49 -0400

handle non-standard vtt names

Diffstat:
Mcommon/controller/normalizeTranscript.ts | 8++++++--
Mcommon/controller/whisperBatch.ts | 20+++++++++-----------
Mcommon/lib/videoStatus.ts | 44+++++++++++++++++++++++++++++++++++++++++---
Meditor/CHANGELOG.md | 1+
Aeditor/e2e/regional-vtt-fallback.spec.ts | 64++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 3+++
6 files changed, 124 insertions(+), 16 deletions(-)

diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts @@ -4,7 +4,7 @@ // shape that buildIndex would emit, plus a `source` marker. import path from "node:path"; -import { readFile, rename, stat, writeFile } from "node:fs/promises"; +import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises"; import { parseVtt, type Cue } from "../lib/vtt"; import { parseWhisper } from "../lib/whisper"; import { summarize, type RawMetadata } from "../lib/transcripts-server"; @@ -16,6 +16,7 @@ import { WHISPER_FILENAME, pickIndexTranscript, readVideoFiles, + resolvePrimaryVtt, type IndexTranscript, } from "../lib/videoStatus"; @@ -134,7 +135,10 @@ export async function isCuesJsonFresh( ): Promise<{ fresh: boolean; cuesPath: string }> { const cuesPath = path.join(videoDir, CUES_JSON_FILENAME); const metaPath = path.join(videoDir, META_FILENAME); - const vttPath = path.join(videoDir, VTT_FILENAME); + // Resolve the actual primary VTT (may be a regional/auto English track like + // transcript.en-US.vtt) rather than assuming the literal transcript.en.vtt. + const entries = await readdir(videoDir).catch(() => [] as string[]); + const vttPath = path.join(videoDir, resolvePrimaryVtt(entries) ?? VTT_FILENAME); const whisperPath = path.join(videoDir, WHISPER_FILENAME); const [cuesMs, metaMs, vttMs, whisperMs] = await Promise.all([ mtimeMs(cuesPath), diff --git a/common/controller/whisperBatch.ts b/common/controller/whisperBatch.ts @@ -3,7 +3,7 @@ import fs from "fs-extra"; import pLimit from "p-limit"; import type { Paths } from "../lib/paths"; import type { AudioFormat } from "../lib/channelConfig"; -import { VTT_FILENAME, WHISPER_FILENAME } from "../lib/videoStatus"; +import { isVideoTranscribed, readVideoFiles } from "../lib/videoStatus"; import { transcribeOneVideo } from "./transcribeOne"; import { resolveShardItems } from "./shard"; import { countNotYetTranscribed } from "./channels"; @@ -106,8 +106,9 @@ export async function runWhisperBatch({ for (const id of candidateIds) { if (failedSet.has(id)) continue; const vp = path.join(dataDir, id); - if (await pathExists(path.join(vp, WHISPER_FILENAME))) continue; - if (await pathExists(path.join(vp, VTT_FILENAME))) continue; + // Already transcribed if whisper ran (transcript.json) OR yt-dlp wrote an + // English VTT (transcript.en.vtt or a regional/auto fallback like en-US). + if (isVideoTranscribed(await readVideoFiles(vp))) continue; fullItems.push(id); } const shardResult = await resolveShardItems({ @@ -175,14 +176,11 @@ export async function runWhisperBatch({ } // A transcript already exists if either whisper has run // (transcript.json) OR yt-dlp wrote auto/manual English subs - // (transcript.en.vtt). Without the VTT branch, videos on a - // youtube-handling channel with only VTT auto-subs end up - // attempted -> fail with "no audio file found" -> get added to - // failed-transcriptions on every run. - if ( - (await pathExists(path.join(videoPath, WHISPER_FILENAME))) || - (await pathExists(path.join(videoPath, VTT_FILENAME))) - ) { + // (transcript.en.vtt, or a regional/auto fallback like en-US). Without + // the VTT branch, videos on a youtube-handling channel with only VTT + // auto-subs end up attempted -> fail with "no audio file found" -> get + // added to failed-transcriptions on every run. + if (isVideoTranscribed(await readVideoFiles(videoPath))) { log(`Transcription for ${videoDir} already exists`); skipped++; return; diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -4,6 +4,10 @@ import { readdir, readFile, stat } from "node:fs/promises"; export type VideoFiles = { hasMeta: boolean; hasYtVtt: boolean; + // The resolved primary English VTT filename (transcript.en.vtt when present, + // otherwise the best regional/auto English track — see resolvePrimaryVtt). + // Null when no English VTT exists. hasYtVtt === (ytVttFile !== null). + ytVttFile: string | null; hasWhisper: boolean; hasCuesJson: boolean; isUntranscribable: boolean; @@ -16,7 +20,9 @@ export type VideoFiles = { export type IndexTranscript = | { kind: "whisper"; filename: "transcript.json" } - | { kind: "vtt"; filename: "transcript.en.vtt" }; + // filename is usually "transcript.en.vtt" but may be a regional/auto English + // track (e.g. "transcript.en-US.vtt") when YouTube served no plain `en` track. + | { kind: "vtt"; filename: string }; export const VTT_FILENAME = "transcript.en.vtt"; export const WHISPER_FILENAME = "transcript.json"; @@ -72,6 +78,32 @@ function isSubExt(value: string): value is SubExt { return (SUB_EXT_VALUES as readonly string[]).includes(value); } +// Resolve the primary English transcript VTT in a video dir. Normally this is +// the canonical transcript.en.vtt, but YouTube sometimes serves a video's +// English captions only under regional/auto codes (transcript.en-US.vtt, +// transcript.en-en-US.vtt, transcript.en-orig.vtt) with no plain `en` track. A +// file counts as English iff the FIRST segment of its language code is `en`, so +// translations like transcript.ab-en-US.vtt / transcript.es-en-US.vtt are +// excluded. Preference: en (canonical) > en-orig (original audio) > +// regional/manual en-US,en-GB,… > auto-translated en-en-* variants. +const EN_VTT_RE = /^transcript\.(en(?:-[^.]+)?)\.vtt$/; +function englishVttRank(track: string): number { + if (track === "en") return 0; + if (track === "en-orig") return 1; + if (/^en-en(?:-|$)/.test(track)) return 3; // auto-translated en→en variants + return 2; // regional/manual en-US, en-GB, … +} +export function resolvePrimaryVtt(entries: string[]): string | null { + let best: { name: string; rank: number } | null = null; + for (const e of entries) { + const m = e.match(EN_VTT_RE); + if (!m) continue; + const rank = englishVttRank(m[1]); + if (!best || rank < best.rank) best = { name: e, rank }; + } + return best?.name ?? null; +} + // Whisper "empty transcription" outputs include systeminfo/model/params/result // keys around an empty `transcription: []`, so they can be up to ~620B in // practice. Real transcripts observed start at ~2.9KB. A 4KB cutoff lets us @@ -84,7 +116,8 @@ export async function readVideoFiles( ): Promise<VideoFiles> { const entries = await readdir(videoDir).catch(() => [] as string[]); const hasMeta = entries.includes(META_FILENAME); - const hasYtVtt = entries.includes(VTT_FILENAME); + const ytVttFile = resolvePrimaryVtt(entries); + const hasYtVtt = ytVttFile !== null; const hasWhisper = entries.includes(WHISPER_FILENAME); const hasCuesJson = entries.includes(CUES_JSON_FILENAME); const audioFiles = entries.filter(isRealAudioFile); @@ -108,6 +141,7 @@ export async function readVideoFiles( return { hasMeta, hasYtVtt, + ytVttFile, hasWhisper, hasCuesJson, isUntranscribable, @@ -118,9 +152,13 @@ export async function readVideoFiles( export async function readSubTracks(videoDir: string): Promise<SubTrack[]> { const entries = await readdir(videoDir).catch(() => [] as string[]); + const primaryVtt = resolvePrimaryVtt(entries); const tracks: SubTrack[] = []; for (const entry of entries) { if (entry === VTT_FILENAME || entry === WHISPER_FILENAME) continue; + // The resolved primary English VTT (e.g. transcript.en-US.vtt when there's + // no transcript.en.vtt) is the main transcript, not an alternate sub-track. + if (entry === primaryVtt) continue; if (entry === CUES_JSON_FILENAME) continue; if (entry === LIVE_CHAT_CUES_FILENAME) continue; const m = entry.match(SUB_FILE_RE); @@ -134,7 +172,7 @@ export async function readSubTracks(videoDir: string): Promise<SubTrack[]> { export function pickIndexTranscript(files: VideoFiles): IndexTranscript | null { if (files.hasWhisper) return { kind: "whisper", filename: WHISPER_FILENAME }; - if (files.hasYtVtt) return { kind: "vtt", filename: VTT_FILENAME }; + if (files.ytVttFile) return { kind: "vtt", filename: files.ytVttFile }; return null; } diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Videos whose English captions only exist under a regional/auto code are now indexed.** YouTube occasionally serves a video's English subtitles only as `en-US`, `en-en-US`, or `en-orig` with no plain `en` track, so yt-dlp wrote e.g. `transcript.en-US.vtt` but never `transcript.en.vtt`. The index only recognized the literal `transcript.en.vtt`, so such a video looked untranscribed and never appeared in search. Build index now falls back to the best available English VTT (preferring `en`, then `en-orig`, then regional `en-US`/`en-GB`, then auto-translated `en-en-*`) while ignoring true translation tracks like `es-en-US`; whisper also treats these as already-transcribed. Re-run **Build index** to pick up affected videos already on disk. - **Managed downloads stream yt-dlp's full output again, and progress bars now read a structured progress template.** The per-video archive marker is captured with yt-dlp's `--print`, which silently implies `--quiet` — so managed downloads ran nearly silent: `download.log` held little more than the archive line, and the per-video progress bars on `/jobs/active` never advanced (the `[download]` lines they parsed were suppressed). Managed downloads now re-enable full logging (`--no-quiet`) and emit a machine-readable `--progress-template` line (throttled to ~1/sec) that the progress parser reads directly instead of scraping the human progress text — so the bars advance reliably during real downloads (with speed/ETA detail) and `download.log` captures the whole run. The legacy `[download]` line parser is kept as a fallback for non-managed single-video/subs-only downloads. - **One global site selector scopes the editor to a single site.** The sidebar gained a single **site** dropdown (with an **All sites** option) that scopes the **Dashboard**, **Channels**, **Charts**, and **Deploy** views to the chosen site: Dashboard stats and the channel table, and the Channels list, now show only that site's member channels; Charts edits that site's dashboard; and Build static export / Deploy target it. Views that act on the shared pool — **Jobs**, **Active**, **Build**, and **Actionable** — always show every site regardless of the selector. The nav is regrouped to match: a **Site** group first, then **Pool**, then **Manage**. The selection persists in `localStorage` and is mirrored into the `?site=` URL param. Under "All sites", Charts and Deploy ask you to pick a specific site; creating a channel while a specific site is active also adds it to that site's membership. This replaces the per-page site tabs on Charts and the per-button site dropdowns on Deploy. - **Date-range filter on the public site's search.** The exported site's search filters gained a **Date** row (From / To pickers) that restricts results to videos uploaded within an inclusive range. The range saves in profiles and shareable search links and carries over to "Chart this search", reusing the same date mechanism the charts dashboard already uses. No editor-chrome change; it ships in every built site. diff --git a/editor/e2e/regional-vtt-fallback.spec.ts b/editor/e2e/regional-vtt-fallback.spec.ts @@ -0,0 +1,64 @@ +import { rm, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { readJson, resetData, resolvePath, writeSite } from "./helpers"; + +const CHANNEL = "test-youtube"; +const VIDEO_DIR = "20240101_test1234567"; +const VIDEO_ID = "test1234567"; +const DATA = `test-transcripts/channels/${CHANNEL}/data/${VIDEO_DIR}`; +const TRANSCRIPTS_PAGE = `test-transcripts/.export-index/shared/transcripts/${CHANNEL}/page-0000.json`; +// parseVtt only emits cues from lines carrying YouTube's inline word-timing +// tags, so the body needs <hh:mm:ss.mmm> markers to produce a real cue. +const VTT = + "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\n" + + "<00:00:00.000><c> hello</c><00:00:02.000><c> regional</c>\n"; + +type Detail = { id: string; cues?: Array<unknown> }; + +async function buildIndex(page: import("@playwright/test").Page) { + await writeSite("testsite", { + channels: [{ slug: CHANNEL, groupId: "default" }], + }); + await page.goto("/build"); + await page.getByRole("button", { name: "Build index" }).click(); + await expect(page.getByLabel("Build index output")).toContainText("Done", { + timeout: 30_000, + }); +} + +// YouTube sometimes serves a video's English captions only under a regional or +// auto code (transcript.en-US.vtt) with no plain transcript.en.vtt. buildIndex +// should still parse it as the primary transcript via the English-VTT fallback, +// so the video's shared transcript page carries cues. +test("build index parses a video with only transcript.en-US.vtt", async ({ + page, +}) => { + await resetData("one-youtube-channel-with-data"); + // Replace the canonical transcript.en.vtt with a regional-only English track. + await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); + await writeFile(resolvePath(`${DATA}/transcript.en-US.vtt`), VTT); + + await buildIndex(page); + + const detailsPage = await readJson<Detail[]>(TRANSCRIPTS_PAGE); + const detail = detailsPage.find((d) => d.id === VIDEO_ID); + expect(detail).toBeDefined(); + // cues are written only when a primary transcript was recognized. + expect(detail?.cues?.length ?? 0).toBeGreaterThan(0); +}); + +// Regression guard: a translation track (transcript.es-en-US.vtt) is NOT English +// and must not be mistaken for the primary transcript — the video lands in the +// index (it has metadata) but carries no cues. +test("build index ignores a non-English translation track", async ({ page }) => { + await resetData("one-youtube-channel-with-data"); + await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); + await writeFile(resolvePath(`${DATA}/transcript.es-en-US.vtt`), VTT); + + await buildIndex(page); + + const detailsPage = await readJson<Detail[]>(TRANSCRIPTS_PAGE); + const detail = detailsPage.find((d) => d.id === VIDEO_ID); + expect(detail).toBeDefined(); + expect(detail?.cues?.length ?? 0).toBe(0); +}); diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,5 +1,8 @@ # Changelog +## [Unreleased] +- **Videos whose English captions only existed under a regional/auto code now appear.** A handful of videos had English subtitles only under codes like `en-US` or `en-en-US` (no plain `en`), which the index didn't recognize — so they were missing from the site even though they had a transcript. These now show up in browse and search like any other transcribed video. + ## [0.3.1] - 2026-05-31 - **Date-range filter in search.** The filters section has a new **Date** row with From / To pickers that limit results to videos uploaded within the range (inclusive bounds; leave either end blank for open-ended). It works the same way the charts date range does and applies to the browse list and every search layer. The range saves in profiles and the working snapshot, rides along in shared search links, and carries over to **"Chart this search"** so the chart opens scoped to the same window.