Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 4ff71bb2324b38f456e66a2c390a929d4bdfd392
parent 7f6e7e47cd6c213697e36aa2730192ee5621ef3b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  2 Jun 2026 13:42:56 -0400

transcript switching

Diffstat:
Mcommon/controller/channelSnapshot.ts | 21+++++++++++++++++++++
Mcommon/lib/videoStatus.ts | 18++++++++++++++++++
Meditor/CHANGELOG.md | 4+++-
Meditor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx | 9+++++++++
Meditor/app/channels/[slug]/lib/stageStatus.ts | 1+
Meditor/app/channels/[slug]/page.tsx | 3+++
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 142+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 2++
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 50++++++++++++++++++++++++++++++++++++++++++++++++--
Aeditor/e2e/transcript-source.spec.ts | 85+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 2+-
11 files changed, 331 insertions(+), 6 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -7,6 +7,7 @@ import { isVideoDownloaded, isVideoTranscribed, readVideoFiles, + VTT_FILENAME, type VideoFiles, } from "../lib/videoStatus"; import { @@ -59,6 +60,13 @@ export type ChannelSnapshot = { missingFromArchive: string[]; duplicateDirs: string[]; partialDownloads: string[]; + // Videos whose transcript rides on a non-canonical VTT name (e.g. only + // transcript.en-US.vtt, or a foreign-only transcript.<lang>.vtt) instead of + // the standard transcript.en.vtt — surfaced so the user can normalize/switch + // the primary transcript. Excludes videos that already have a whisper + // transcript or the canonical transcript.en.vtt. Optional: older snapshots + // lack it; readers must default to []. + nonStandardVtt: string[]; }; undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; @@ -254,6 +262,7 @@ export async function generateChannelSnapshot( const untranscribable: string[] = []; const noMetadata: string[] = []; const partialDownloads: string[] = []; + const nonStandardVtt: string[] = []; let transcribed = 0; let downloaded = 0; for (const { id, files } of perVideo) { @@ -289,6 +298,17 @@ export async function generateChannelSnapshot( ) { partialDownloads.push(id); } + // A transcript that exists only under a non-standard VTT name — either a + // regional/auto English track (transcript.en-US.vtt) now picked up by the + // fallback, or a foreign-only transcript.<lang>.vtt that isn't recognized as + // English. Whisper transcripts and the canonical transcript.en.vtt are fine. + if ( + !files.hasWhisper && + files.hasNonCanonicalVtt && + files.ytVttFile !== VTT_FILENAME + ) { + nonStandardVtt.push(id); + } if (files.isUntranscribable) { untranscribable.push(id); continue; @@ -357,6 +377,7 @@ export async function generateChannelSnapshot( missingFromArchive: missingFromArchive.sort(), duplicateDirs: [], partialDownloads: partialDownloads.sort(), + nonStandardVtt: nonStandardVtt.sort(), }, undownloadedIds, excludedFromDownload, diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -8,6 +8,10 @@ export type VideoFiles = { // otherwise the best regional/auto English track — see resolvePrimaryVtt). // Null when no English VTT exists. hasYtVtt === (ytVttFile !== null). ytVttFile: string | null; + // True when a transcript.<lang>.vtt exists under a non-canonical name (i.e. + // anything other than transcript.en.vtt). Lets diagnostics surface videos + // whose only/primary transcript rides on a non-standard VTT name. + hasNonCanonicalVtt: boolean; hasWhisper: boolean; hasCuesJson: boolean; isUntranscribable: boolean; @@ -104,6 +108,16 @@ export function resolvePrimaryVtt(entries: string[]): string | null { return best?.name ?? null; } +// Any transcript.<lang>.vtt file (any language code). These are the candidate +// transcript tracks a user can promote to the canonical transcript.en.vtt. +const TRANSCRIPT_VTT_RE = /^transcript\.[^.]+\.vtt$/; +export function isTranscriptVtt(name: string): boolean { + return TRANSCRIPT_VTT_RE.test(name); +} +export function listTranscriptVtts(entries: string[]): string[] { + return entries.filter(isTranscriptVtt).sort(); +} + // Whisper "empty transcription" outputs include systeminfo/model/params/result // keys around an empty `transcription: []`, so they can be up to ~620B in // practice. Real transcripts observed start at ~2.9KB. A 4KB cutoff lets us @@ -118,6 +132,9 @@ export async function readVideoFiles( const hasMeta = entries.includes(META_FILENAME); const ytVttFile = resolvePrimaryVtt(entries); const hasYtVtt = ytVttFile !== null; + const hasNonCanonicalVtt = entries.some( + (e) => isTranscriptVtt(e) && e !== VTT_FILENAME, + ); const hasWhisper = entries.includes(WHISPER_FILENAME); const hasCuesJson = entries.includes(CUES_JSON_FILENAME); const audioFiles = entries.filter(isRealAudioFile); @@ -142,6 +159,7 @@ export async function readVideoFiles( hasMeta, hasYtVtt, ytVttFile, + hasNonCanonicalVtt, hasWhisper, hasCuesJson, isUntranscribable, diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,7 +1,9 @@ # Changelog ## [Unreleased] -- **Videos whose English captions only exist under a regional/auto code are now indexed.** YouTube occasionally serves a video's English subtitles only as `en-US`, `en-en-US`, or `en-orig` with no plain `en` track, so yt-dlp wrote e.g. `transcript.en-US.vtt` but never `transcript.en.vtt`. The index only recognized the literal `transcript.en.vtt`, so such a video looked untranscribed and never appeared in search. Build index now falls back to the best available English VTT (preferring `en`, then `en-orig`, then regional `en-US`/`en-GB`, then auto-translated `en-en-*`) while ignoring true translation tracks like `es-en-US`; whisper also treats these as already-transcribed. Re-run **Build index** to pick up affected videos already on disk. +- **Videos whose English captions only exist under a regional/auto code are now indexed.** YouTube occasionally serves a video's English subtitles only as `en-US`, `en-en-US`, or `en-orig` with no plain `en` track, so yt-dlp wrote e.g. `transcript.en-US.vtt` but never `transcript.en.vtt`. The index only recognized the literal `transcript.en.vtt`, so such a video looked untranscribed and never appeared in search. Build index now falls back to the best available English VTT (preferring `en`, then `en-orig`, then regional `en-US`/`en-GB`, then auto-translated `en-en-*`) while ignoring true translation tracks like `es-en-US`; whisper also treats these as already-transcribed. Re-run **Build index** to pick up affected videos already on disk. The video detail page now reflects the same fallback (it previously hardcoded `transcript.en.vtt`, so a regional-only video showed as untranscribed there). +- **Pick which subtitle track is a video's transcript.** The video page has a new **Transcript source** section listing every `transcript.<lang>.vtt` track, marking the current primary, with a **Set as transcript** button that promotes any track to the canonical `transcript.en.vtt` (the chosen track is copied, so the original stays and the choice is reversible — delete `transcript.en.vtt` to fall back to the automatic English pick, or pick another track to switch). Useful when the auto-picked track isn't the one you want, or when a video's only captions are a non-English track. +- **Diagnostics: "Non-standard transcript VTT name" bucket.** A channel's Diagnostics now lists videos whose transcript rides on a non-canonical VTT name (e.g. only `transcript.en-US.vtt`, or a foreign-language track) rather than the standard `transcript.en.vtt` — each links to the video so you can normalize/switch the primary transcript. Refresh the channel snapshot to populate it. - **Managed downloads stream yt-dlp's full output again, and progress bars now read a structured progress template.** The per-video archive marker is captured with yt-dlp's `--print`, which silently implies `--quiet` — so managed downloads ran nearly silent: `download.log` held little more than the archive line, and the per-video progress bars on `/jobs/active` never advanced (the `[download]` lines they parsed were suppressed). Managed downloads now re-enable full logging (`--no-quiet`) and emit a machine-readable `--progress-template` line (throttled to ~1/sec) that the progress parser reads directly instead of scraping the human progress text — so the bars advance reliably during real downloads (with speed/ETA detail) and `download.log` captures the whole run. The legacy `[download]` line parser is kept as a fallback for non-managed single-video/subs-only downloads. - **One global site selector scopes the editor to a single site.** The sidebar gained a single **site** dropdown (with an **All sites** option) that scopes the **Dashboard**, **Channels**, **Charts**, and **Deploy** views to the chosen site: Dashboard stats and the channel table, and the Channels list, now show only that site's member channels; Charts edits that site's dashboard; and Build static export / Deploy target it. Views that act on the shared pool — **Jobs**, **Active**, **Build**, and **Actionable** — always show every site regardless of the selector. The nav is regrouped to match: a **Site** group first, then **Pool**, then **Manage**. The selection persists in `localStorage` and is mirrored into the `?site=` URL param. Under "All sites", Charts and Deploy ask you to pick a specific site; creating a channel while a specific site is active also adds it to that site's membership. This replaces the per-page site tabs on Charts and the per-button site dropdowns on Deploy. - **Date-range filter on the public site's search.** The exported site's search filters gained a **Date** row (From / To pickers) that restricts results to videos uploaded within an inclusive range. The range saves in profiles and shareable search links and carries over to "Chart this search", reusing the same date mechanism the charts dashboard already uses. No editor-chrome change; it ships in every built site. diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx @@ -43,6 +43,7 @@ type Props = { noMetadataIds: string[]; missingFromArchiveIds: string[]; duplicateDirIds: string[]; + nonStandardVttIds: string[]; totals: { videos: number; transcribed: number; downloaded: number }; availability: AvailabilitySnapshot; availabilityShard: ShardConfigSummary | null; @@ -56,6 +57,7 @@ export function DiagnosticsStage({ noMetadataIds, missingFromArchiveIds, duplicateDirIds, + nonStandardVttIds, totals, availability, availabilityShard, @@ -64,6 +66,13 @@ export function DiagnosticsStage({ }: Props) { const buckets: DiagnosticBucket[] = [ { + ids: nonStandardVttIds, + label: "Non-standard transcript VTT name", + description: + "Transcript exists only under a non-canonical VTT name (e.g. transcript.en-US.vtt) or a foreign-language track. Open the video to set the primary transcript.", + ariaLabel: "non-standard vtt", + }, + { ids: noMetadataIds, label: "Missing metadata.info.json", description: diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -24,6 +24,7 @@ export function normalizeBuckets( missingFromArchive: raw?.missingFromArchive ?? [], duplicateDirs: raw?.duplicateDirs ?? [], partialDownloads: raw?.partialDownloads ?? [], + nonStandardVtt: raw?.nonStandardVtt ?? [], }; } diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -21,6 +21,7 @@ import { } from "yt-dlp-transcript-common/controller/shard"; import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcome-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { resolvePrimaryVtt } from "yt-dlp-transcript-common/lib/videoStatus"; import { platformQueueKey, queueKeyForUrl, @@ -263,6 +264,7 @@ export default async function ChannelDetailPage({ noMetadataIds={buckets.noMetadata} missingFromArchiveIds={buckets.missingFromArchive} duplicateDirIds={buckets.duplicateDirs} + nonStandardVttIds={buckets.nonStandardVtt} totals={snapshot.totals} availability={availability} availabilityShard={availabilityShard} @@ -342,6 +344,7 @@ export default async function ChannelDetailPage({ slug={slug} videoId={selectedVideoId} files={dirData.files} + primaryVtt={resolvePrimaryVtt(dirData.files.map((f) => f.name))} handling={config.handling} defaultQueueKey={platformDefaultQueueKey} existingQueues={existingQueues} diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -19,6 +19,7 @@ import { deleteVideoFileAction, downloadVideoPipelineAction, markVideoUntranscribableAction, + setPrimaryTranscriptAction, toggleDoNotCleanAction, transcodeAudioAction, transcribeOneAction, @@ -32,10 +33,20 @@ export type VideoFile = { mtime: number; }; +const WHISPER_FILENAME = "transcript.json"; +const CANONICAL_VTT = "transcript.en.vtt"; + +function isTranscriptVttName(name: string): boolean { + return /^transcript\.[^.]+\.vtt$/.test(name); +} + type Props = { slug: string; videoId: string; files: VideoFile[]; + // Resolved primary English VTT filename (transcript.en.vtt, or a regional/auto + // fallback like transcript.en-US.vtt), or null when no English VTT exists. + primaryVtt: string | null; handling: ChannelHandling; defaultQueueKey: string; existingQueues: string[]; @@ -117,6 +128,7 @@ export function VideoPanel({ slug, videoId, files, + primaryVtt, handling, defaultQueueKey, existingQueues, @@ -129,9 +141,15 @@ export function VideoPanel({ }: Props) { const audioFiles = files.filter((f) => isAudioFile(f.name)); const transcodeSources = files.filter((f) => isTranscodeSource(f.name)); - const hasTranscriptJson = files.some((f) => f.name === "transcript.json"); - const hasYtVtt = files.some((f) => f.name === "transcript.en.vtt"); + const hasTranscriptJson = files.some((f) => f.name === WHISPER_FILENAME); + const hasYtVtt = primaryVtt !== null; const hasTranscript = hasTranscriptJson || hasYtVtt; + const vttTracks = files + .filter((f) => isTranscriptVttName(f.name)) + .map((f) => f.name) + .sort(); + // The transcript the index/viewer will actually use: whisper wins over VTT. + const activeTranscript = hasTranscriptJson ? WHISPER_FILENAME : primaryVtt; const noAudio = audioFiles.length === 0; const downloadFailed = downloadOutcome?.status === "failed"; @@ -239,6 +257,32 @@ export function VideoPanel({ </div> </PipelineStageCard> + {vttTracks.length > 0 && ( + <PipelineStageCard + id="transcript-source" + title="Transcript source" + summary={ + activeTranscript + ? `Primary transcript: ${activeTranscript}` + : "No primary transcript selected." + } + defaultOpen={vttTracks.length > 1 || primaryVtt !== CANONICAL_VTT} + tone={ + primaryVtt && primaryVtt !== CANONICAL_VTT && !hasTranscriptJson + ? "attention" + : "neutral" + } + > + <TranscriptSourceSection + slug={slug} + videoId={videoId} + vttTracks={vttTracks} + primaryVtt={primaryVtt} + hasWhisper={hasTranscriptJson} + /> + </PipelineStageCard> + )} + {!hasTranscript && ( <PipelineStageCard id="mark-untranscribable" @@ -657,6 +701,100 @@ function FilesList({ ); } +function TranscriptSourceSection({ + slug, + videoId, + vttTracks, + primaryVtt, + hasWhisper, +}: { + slug: string; + videoId: string; + vttTracks: string[]; + primaryVtt: string | null; + hasWhisper: boolean; +}) { + const [pending, startTransition] = useTransition(); + const [busyFile, setBusyFile] = useState<string | null>(null); + const [error, setError] = useState<string | null>(null); + + const setPrimary = (filename: string) => { + setError(null); + setBusyFile(filename); + startTransition(async () => { + const res = await setPrimaryTranscriptAction(slug, videoId, filename); + if (!res.ok) setError(res.error); + setBusyFile(null); + }); + }; + + return ( + <div className="flex flex-col gap-3"> + <p className="text-sm text-zinc-500"> + Pick which subtitle track is this video&apos;s primary transcript. The + chosen track is copied to <code>{CANONICAL_VTT}</code> — the canonical + name the index and viewer read. Reversible: delete{" "} + <code>{CANONICAL_VTT}</code> in Files to fall back to the automatic + English pick, or choose another track to switch. + </p> + {hasWhisper && ( + <p + className="text-sm text-amber-700 dark:text-amber-400" + aria-label="whisper precedence note" + > + A whisper transcript (<code>{WHISPER_FILENAME}</code>) is present and + takes precedence over any VTT when indexing. + </p> + )} + <ul aria-label="transcript tracks" className="flex flex-col gap-2"> + {vttTracks.map((name) => { + const isPrimary = name === primaryVtt; + return ( + <li + key={name} + aria-label={`transcript track ${name}`} + className="flex items-center justify-between gap-3 rounded border border-zinc-200 dark:border-zinc-800 px-3 py-2" + > + <span className="flex items-center gap-2 font-mono text-sm"> + {name} + {isPrimary && ( + <span + aria-label={`primary transcript ${name}`} + className="rounded bg-emerald-100 dark:bg-emerald-950 text-emerald-700 dark:text-emerald-300 px-1.5 py-0.5 text-[10px] font-sans font-medium uppercase tracking-wide" + > + Primary + </span> + )} + </span> + {!isPrimary && ( + <button + type="button" + disabled={pending} + aria-label={`set ${name} as primary transcript`} + onClick={() => setPrimary(name)} + className="shrink-0 px-2 py-1 rounded border border-zinc-300 dark:border-zinc-700 text-xs hover:bg-zinc-100 dark:hover:bg-zinc-800 disabled:opacity-50" + > + {pending && busyFile === name + ? "Setting…" + : "Set as transcript"} + </button> + )} + </li> + ); + })} + </ul> + {error && ( + <span + className="text-sm text-red-600 dark:text-red-400" + aria-label="set primary transcript error" + > + {error} + </span> + )} + </div> + ); +} + function MarkUntranscribableSection({ slug, videoId, diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -8,6 +8,7 @@ import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels" import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcome-server"; import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { resolvePrimaryVtt } from "yt-dlp-transcript-common/lib/videoStatus"; import { platformQueueKey, queueKeyForUrl, @@ -167,6 +168,7 @@ export default async function VideoDetailPage({ slug={slug} videoId={id} files={dirData.files} + primaryVtt={resolvePrimaryVtt(dirData.files.map((f) => f.name))} handling={config.handling} defaultQueueKey={defaultQueueKey} existingQueues={existingQueues} diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -1,7 +1,7 @@ "use server"; import path from "node:path"; -import { readdir, rename, rm, stat, writeFile } from "node:fs/promises"; +import { readdir, readFile, rename, rm, stat, writeFile } from "node:fs/promises"; import { revalidatePath } from "next/cache"; import { redirect } from "next/navigation"; import type { @@ -10,7 +10,11 @@ import type { } from "yt-dlp-transcript-common/lib/channelConfig"; import { AUDIO_FORMAT_VALUES } from "yt-dlp-transcript-common/lib/channelConfig"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { isRealAudioFile } from "yt-dlp-transcript-common/lib/videoStatus"; +import { + isRealAudioFile, + isTranscriptVtt, + VTT_FILENAME, +} from "yt-dlp-transcript-common/lib/videoStatus"; import { downloadQueueKey, resolveQueueKey, @@ -273,6 +277,48 @@ export async function deleteVideoFileAction( return { ok: true }; } +// Promote a transcript.<lang>.vtt track to the canonical transcript.en.vtt so +// the index, snapshot, and viewer all treat it as the primary transcript. The +// chosen file is copied (not moved) so the original language-coded track is kept +// and the choice stays reversible/repeatable — delete transcript.en.vtt to fall +// back to the automatic regional-English pick, or pick a different track to +// switch again. +export async function setPrimaryTranscriptAction( + slug: string, + videoId: string, + filename: string, +): Promise<{ ok: true } | { ok: false; error: string }> { + if (!isTranscriptVtt(filename)) { + return { + ok: false, + error: `Not a transcript VTT file: ${filename}`, + }; + } + const videoDir = videoDirOf(slug, videoId); + const source = safeJoinUnderDir(videoDir, filename); + if (!source) { + return { ok: false, error: `Refusing to read suspicious filename "${filename}"` }; + } + let raw: string; + try { + raw = await readFile(source, "utf8"); + } catch { + return { ok: false, error: `File not found: ${filename}` }; + } + if (filename !== VTT_FILENAME) { + const dest = path.join(videoDir, VTT_FILENAME); + const tmp = path.join(videoDir, `${VTT_FILENAME}.tmp-${process.pid}`); + await writeFile(tmp, raw); + await rename(tmp, dest); + } + // The video page reads the dir directly, so it reflects the new primary right + // away. The channel list + diagnostics bucket read the cached snapshot and + // refresh on the next snapshot regeneration (same as the other video actions). + revalidatePath(`/channels/${slug}/videos/${videoId}`); + revalidatePath(`/channels/${slug}`); + return { ok: true }; +} + export type DeleteDirActionResult = { error: string } | undefined; export async function deleteVideoDirAction( diff --git a/editor/e2e/transcript-source.spec.ts b/editor/e2e/transcript-source.spec.ts @@ -0,0 +1,85 @@ +import { rm, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { pathExists, resetData, resolvePath } from "./helpers"; + +const CHANNEL = "test-youtube"; +const VIDEO_DIR = "20240101_test1234567"; +const DATA = `test-transcripts/channels/${CHANNEL}/data/${VIDEO_DIR}`; +const VIDEO_URL = `/channels/${CHANNEL}/videos/${VIDEO_DIR}`; +const VTT = "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nhello\n"; + +// The video page must recognize a regional/auto English VTT as the transcript, +// not just the canonical transcript.en.vtt. +test("video page treats transcript.en-US.vtt as the transcript", async ({ + page, +}) => { + await resetData("one-youtube-channel-with-data"); + await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); + await writeFile(resolvePath(`${DATA}/transcript.en-US.vtt`), VTT); + + await page.goto(VIDEO_URL); + + // hasTranscript === true: the transcribe card says so and the + // "Mark untranscribable" card (shown only when untranscribed) is absent. + await expect(page.getByText("Transcript present.")).toBeVisible(); + await expect( + page.getByRole("heading", { name: "Mark untranscribable" }), + ).toHaveCount(0); + // The regional track is marked the primary transcript. + await expect( + page.getByLabel("primary transcript transcript.en-US.vtt"), + ).toBeVisible(); +}); + +// The user can switch which VTT track is the primary transcript; the choice is +// copied to the canonical transcript.en.vtt. +test("switching the transcript source promotes a track to transcript.en.vtt", async ({ + page, +}) => { + await resetData("one-youtube-channel-with-data"); + await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); + await writeFile(resolvePath(`${DATA}/transcript.en-US.vtt`), VTT); + await writeFile(resolvePath(`${DATA}/transcript.en-en-US.vtt`), VTT); + + await page.goto(VIDEO_URL); + + // en-US (manual/regional) outranks en-en-US (auto-translated) as the default. + await expect( + page.getByLabel("primary transcript transcript.en-US.vtt"), + ).toBeVisible(); + + // Switch the primary to the auto track. + await page + .getByRole("button", { + name: "set transcript.en-en-US.vtt as primary transcript", + }) + .click(); + + // The chosen track is copied to the canonical name… + await expect + .poll(() => pathExists(`${DATA}/transcript.en.vtt`), { timeout: 15_000 }) + .toBe(true); + // …and transcript.en.vtt becomes the primary (rank 0) after revalidation. + await expect( + page.getByLabel("primary transcript transcript.en.vtt"), + ).toBeVisible(); +}); + +// The channel diagnostics surface a "non-standard transcript VTT name" bucket +// for videos whose transcript rides on a non-canonical VTT name. +test("diagnostics list a video with only transcript.en-US.vtt", async ({ + page, +}) => { + await resetData("one-youtube-channel-with-data"); + await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); + await writeFile(resolvePath(`${DATA}/transcript.en-US.vtt`), VTT); + + // First channel-page load generates a fresh snapshot including the bucket. + await page.goto(`/channels/${CHANNEL}`); + + await expect( + page.getByRole("heading", { + name: /Non-standard transcript VTT name \(1\)/, + }), + ).toBeVisible(); +}); diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,6 +1,6 @@ # Changelog -## [Unreleased] +## [0.3.3] - 2026-06-01 - **Videos whose English captions only existed under a regional/auto code now appear.** A handful of videos had English subtitles only under codes like `en-US` or `en-en-US` (no plain `en`), which the index didn't recognize — so they were missing from the site even though they had a transcript. These now show up in browse and search like any other transcribed video. ## [0.3.1] - 2026-05-31