Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 6198bcebf6756650febb7eb5f4d848158bee0e00
parent 12353c3cc6b1c40d6bace8529aa5ec0e88d144a6
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 11:30:42 -0400

Merge transcripts/multi-track (every English track searchable; the reader and MCP choose a track; alternates only where words differ)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
MPUBLISH.md | 10++++++++++
MREADME.md | 5++++-
Mcommon/components/PlayerProvider.tsx | 48+++++++++++++++++++++++++++++++++++++++++-------
Mcommon/components/SearchResults.tsx | 5++++-
Mcommon/components/SearchSessionContext.tsx | 7++++++-
Mcommon/components/TranscriptModal.tsx | 50+++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/components/searchIndex.worker.ts | 18+++++++++++++++++-
Mcommon/components/searchIndexWorkerProtocol.ts | 4+++-
Mcommon/components/urlState.ts | 10++++++++++
Mcommon/controller/buildIndex.ts | 91++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Acommon/controller/buildIndexAltTracks.test.ts | 211+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/captionTrack.test.ts | 4++--
Acommon/lib/captionTracks-server.ts | 129+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/captionTracks.test.ts | 189+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/captionTracks.ts | 225+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/corpus.ts | 5++++-
Mcommon/lib/search/evalTree.test.ts | 19+++++++++++++++++++
Mcommon/lib/search/evalTree.ts | 13+++++++++++--
Mcommon/lib/search/leafPipeline.test.ts | 25+++++++++++++++++++++++++
Mcommon/lib/search/leafPipeline.ts | 9++++++---
Mcommon/lib/search/window.ts | 4+++-
Mcommon/lib/transcripts.ts | 6+++++-
Mcommon/lib/videoStatus.ts | 17++++++++---------
Meditor/CHANGELOG.md | 5++++-
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 17+++++++++++++++++
Aeditor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx | 100+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 16++++++++++++++++
Meditor/e2e/transcript-source.spec.ts | 34++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 1+
Mexport/app/components/OfflineManager.tsx | 2++
Mexport/e2e/fixtures/data.ts | 10++++++++++
Aexport/e2e/transcript-tracks.spec.ts | 72++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/src/instructions.ts | 10++++++++++
Mmcp/src/search.test.ts | 79++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mmcp/src/search.ts | 14++++++++++++--
Mmcp/src/server.ts | 111+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
36 files changed, 1525 insertions(+), 50 deletions(-)

diff --git a/PUBLISH.md b/PUBLISH.md @@ -38,6 +38,16 @@ the stats datasets and the chart templates — `archilyzer index`, `build stats` `export/public`, plus its download archives — `archilyzer compose site <id>`) and `next build`. `--nodata` skips the data phase and reuses the last one's staging. +A transcript record in the shared pages carries its other English caption tracks +(`altTracks`, with the transcript's own `track`) only where one's words differ from +the transcript's — the served `en` beside `en-orig`, a regional or auto-translated +track, the captions a local transcription replaced (`common/lib/captionTracks.ts`). +Identical tracks add nothing, so most records' bytes are what they were. Search reads +those tracks with the transcript and names the track of a hit only one holds; the +reader switches to them. English VTTs are not published as subtitle tracks. The first +index build after this reads them once (`Alternate tracks v1: N record(s) re-read.`), +for exactly the records that can hold one. + ## The three ways to drive it The editor's **/sites** page, `pnpm ops` (HTTP to a running editor, with its diff --git a/README.md b/README.md @@ -385,7 +385,10 @@ differs is emphasis: where you would rather have the better text. Where YouTube serves both, the original-audio captions (`en-orig`) are read before the served `en` track, which can reword what was said; a track with no text falls through to the next, and a video's - page can pin another (`common/lib/videoStatus.ts`, "the caption-track rule"). + page can pin another (`common/lib/videoStatus.ts`, "the caption-track rule"). The + other English tracks are kept where their words differ — uploaded captions are not + always what was said: search reads them too and says which track a hit is in, and + the transcript reader switches to them (`common/lib/captionTracks.ts`). - **The output is yours to brand.** Site title, header, description, tagline, social links and channel grouping are all per-site configuration. - **One corpus can publish several sites.** Channels are grouped into sites, so a diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx @@ -25,6 +25,7 @@ import { cuesToSrt, cuesToText } from "../lib/vtt"; import { transcriptToMarkdown } from "../lib/transcriptToMarkdown"; import type { Platform } from "../lib/transcripts"; import { vodExpiry } from "../lib/vodExpiry"; +import { cuesOfTrack, recordTracks, type TrackFields } from "../lib/captionTracks"; import { VodExpiredBadge } from "./badges"; import type { RumblePlayerHandle } from "./RumblePlayer"; import type { OdyseePlayerHandle } from "./OdyseePlayer"; @@ -92,7 +93,15 @@ export type TranscriptData = { webpageUrl: string; hlsUrl?: string; mediaUrl?: string; + // The cues of the track on show: the primary, or the alternate the viewer + // picked (`track`). cues?: Cue[]; + // The record's tracks, primary first (lib/captionTracks.ts) — more than one + // only when an alternate's words differ from the primary's. Empty when the + // record names none. + tracks: string[]; + // The track on show; null when the record names no tracks. + track: string | null; }; type Status = "idle" | "loading" | "ready"; @@ -117,7 +126,7 @@ type Detail = { hlsUrl?: string; mediaUrl?: string; cues?: Cue[]; -}; +} & TrackFields; type ClipState = { slug: string | null; @@ -164,8 +173,11 @@ type PlayerState = { openTranscript: ( slug: string, start?: number, - opts?: { mode?: ModalMode }, + opts?: { mode?: ModalMode; track?: string }, ) => void; + // Show another of the record's tracks (null: the primary). A viewer's + // choice — it changes the URL (`vt`), never the record. + setViewTrack: (track: string | null) => void; setDisplayMode: (mode: DisplayMode) => void; setModalMode: (mode: ModalMode) => void; closePlayer: () => void; @@ -395,7 +407,7 @@ export function PlayerProvider({ children: React.ReactNode; features?: PlayerFeatures; }) { - const { v: urlSlug, t: urlTime, vm: urlVm } = useUrlParams(); + const { v: urlSlug, t: urlTime, vm: urlVm, vt: urlVt } = useUrlParams(); const [core, dispatch] = useReducer(coreReducer, INITIAL_CORE); const { detail, displayState, clip, chat, digest } = core; const [playing, setPlaying] = useState(false); @@ -450,9 +462,15 @@ export function PlayerProvider({ webpageUrl: detail.webpageUrl, hlsUrl: detail.hlsUrl, mediaUrl: detail.mediaUrl, - cues: detail.cues, + // A `vt` the record does not hold reads as the primary. + cues: cuesOfTrack(detail, urlVt) ?? detail.cues, + tracks: recordTracks(detail), + track: + urlVt && cuesOfTrack(detail, urlVt) !== undefined + ? urlVt + : (detail.track ?? null), }; - }, [detail, detailMatches]); + }, [detail, detailMatches, urlVt]); const status: Status = !activeSlug ? "idle" : detailMatches @@ -463,7 +481,7 @@ export function PlayerProvider({ const clipEnd = clip.slug && clip.slug === activeSlug ? clip.end : null; const openTranscript = useCallback( - (slug: string, start?: number, opts?: { mode?: ModalMode }) => { + (slug: string, start?: number, opts?: { mode?: ModalMode; track?: string }) => { dispatch({ type: "DISPLAY_SET", slug, mode: "modal" }); const t = typeof start === "number" ? Math.round(start) : null; if (slug === activeSlug && t !== null && playerRef.current) { @@ -475,11 +493,18 @@ export function PlayerProvider({ v: slug, t, vm: opts?.mode ?? "transcript", + // A hit from an alternate track opens on that track; anything else + // opens on the primary. + vt: opts?.track ?? null, }); }, [activeSlug], ); + const setViewTrack = useCallback((track: string | null) => { + writeUrlParams({ vt: track }); + }, []); + const setModalMode = useCallback((mode: ModalMode) => { writeUrlParams({ vm: mode }); }, []); @@ -494,7 +519,7 @@ export function PlayerProvider({ const closePlayer = useCallback(() => { dispatch({ type: "DISPLAY_CLEAR" }); - writeUrlParams({ v: null, t: null, vm: "transcript" }); + writeUrlParams({ v: null, t: null, vm: "transcript", vt: null }); }, []); const markClipStart = useCallback(() => { @@ -542,6 +567,10 @@ export function PlayerProvider({ // Carry the current panel so a shared link reopens what the sharer was // looking at. "transcript" is the default and stays absent from the URL. if (modalMode !== "transcript") params.set("vm", modalMode); + // And the track, when it is not the primary. + if (data.track && data.tracks.length > 1 && data.track !== data.tracks[0]) { + params.set("vt", data.track); + } const url = `${window.location.origin}${window.location.pathname}?${params.toString()}`; try { await navigator.clipboard.writeText(url); @@ -688,6 +717,9 @@ export function PlayerProvider({ hlsUrl: full.hlsUrl, mediaUrl: full.mediaUrl, cues: full.cues, + ...(full.altTracks?.length + ? { track: full.track, altTracks: full.altTracks } + : {}), }, }); }) @@ -922,6 +954,7 @@ export function PlayerProvider({ clipStart, clipEnd, openTranscript, + setViewTrack, setDisplayMode, setModalMode, closePlayer, @@ -948,6 +981,7 @@ export function PlayerProvider({ clipStart, clipEnd, openTranscript, + setViewTrack, setDisplayMode, setModalMode, closePlayer, diff --git a/common/components/SearchResults.tsx b/common/components/SearchResults.tsx @@ -45,6 +45,7 @@ import { } from "./SearchSessionContext"; import { useSearchData } from "./SearchDataContext"; import type { PublishedTag } from "../lib/curatedTags"; +import { inTrackLabel } from "../lib/captionTracks"; // Rough first-paint guess for one full result card (header + a couple of // leaf sections + a handful of hits). After mount, ResizeObserver measures @@ -969,7 +970,9 @@ function PostBadge({ platform }: { platform: string }) { } function TrackBadge({ track }: { track: string }) { - const label = track === "live_chat" ? "live chat" : track; + // An alternate English track of a transcript (lib/captionTracks.ts) says + // which one: "in uploaded captions". + const label = track === "live_chat" ? "live chat" : inTrackLabel(track); return ( <span className="shrink-0 text-[10px] uppercase tracking-wide font-medium px-1.5 py-0.5 rounded bg-muted text-muted-foreground self-center"> {label} diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx @@ -1925,7 +1925,12 @@ function useSearchSessionState() { : hit?.scope === "chat" ? "chat" : "transcript"; - openTranscript(slug, hit?.start, { mode: modalMode }); + // A transcript hit from an alternate track (lib/captionTracks.ts) opens + // the reader on that track, where its words are. + openTranscript(slug, hit?.start, { + mode: modalMode, + ...(modalMode === "transcript" && hit?.track ? { track: hit.track } : {}), + }); }, [openTranscript], ); diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx @@ -36,6 +36,7 @@ import { AgeRestrictedBadge, LivestreamBadge } from "./badges"; import { VirtualRow } from "./VirtualRow"; import { formatTimestamp } from "../lib/vtt"; import { formatDate } from "../lib/format"; +import { trackLabels } from "../lib/captionTracks"; type DisplayCue = { start: number; @@ -53,6 +54,7 @@ export default function TranscriptModal() { status, modalMode, setModalMode, + setViewTrack, chatCues, chatStatus, modalNotice, @@ -320,7 +322,22 @@ export default function TranscriptModal() { )} </div> - <div className="px-3 sm:px-0">{modeStrip}</div> + <div className="px-3 sm:px-0 flex flex-wrap items-center gap-x-3 gap-y-1"> + {modeStrip} + {/* The record's other English tracks (lib/captionTracks.ts), only + where one's words differ from the primary's. A viewer's choice: + it swaps the cues on show and changes nothing else. */} + {!isChat && !isDigest && data && data.tracks.length > 1 && ( + <TrackSwitcher + tracks={data.tracks} + track={data.track ?? data.tracks[0]} + onChange={(t) => { + scrollKindRef.current = "smooth"; + setViewTrack(t === data.tracks[0] ? null : t); + }} + /> + )} + </div> {modalNotice && ( <p @@ -793,3 +810,34 @@ function findActiveIndex( } return found; } + +// "Track: original audio captions ▾" — small and quiet, beside the mode strip. +// The first track is the primary (the default); its label says so in the menu. +function TrackSwitcher({ + tracks, + track, + onChange, +}: { + tracks: string[]; + track: string; + onChange: (track: string) => void; +}) { + return ( + <label className="inline-flex items-center gap-1.5 text-xs text-zinc-400 pointer-events-auto"> + <span>Track:</span> + <select + data-testid="track-switcher" + value={track} + onChange={(e) => onChange(e.target.value)} + className="bg-transparent text-zinc-200 rounded border border-zinc-700 px-1 py-0.5 text-xs focus:outline-none focus:ring-1 focus:ring-zinc-500" + > + {trackLabels(tracks).map((label, i) => ( + <option key={tracks[i]} value={tracks[i]} className="bg-zinc-900"> + {label} + {i === 0 ? " (default)" : ""} + </option> + ))} + </select> + </label> + ); +} diff --git a/common/components/searchIndex.worker.ts b/common/components/searchIndex.worker.ts @@ -51,7 +51,7 @@ function mkIndex(): Index { } type RawCue = { start?: number; text?: unknown }; -type RawEntry = { slug?: unknown; cues?: unknown }; +type RawEntry = { slug?: unknown; cues?: unknown; altTracks?: unknown }; async function build(reqId: number, slug: string): Promise<void> { // Manifest → page count. @@ -75,13 +75,29 @@ async function build(reqId: number, slug: string): Promise<void> { for (const entry of entries) { const eslug = typeof entry.slug === "string" ? entry.slug : slug; const cues = Array.isArray(entry.cues) ? entry.cues : []; + const said = new Set<string>(); for (const c of cues as RawCue[]) { const text = typeof c.text === "string" ? c.text : ""; if (!text) continue; + said.add(text); idx.add(id, text); meta[id] = { slug: eslug, start: c.start, text }; id++; } + // The record's alternate English tracks (lib/captionTracks.ts): a line + // the transcript itself does not say is indexed too, under its track. + const alts = Array.isArray(entry.altTracks) ? entry.altTracks : []; + for (const alt of alts as { track?: unknown; cues?: unknown }[]) { + const track = typeof alt.track === "string" ? alt.track : ""; + if (!track || !Array.isArray(alt.cues)) continue; + for (const c of alt.cues as RawCue[]) { + const text = typeof c.text === "string" ? c.text : ""; + if (!text || said.has(text)) continue; + idx.add(id, text); + meta[id] = { slug: eslug, start: c.start, text, track }; + id++; + } + } } post({ type: "progress", reqId, slug, done: p + 1, total: pageCount }); } diff --git a/common/components/searchIndexWorkerProtocol.ts b/common/components/searchIndexWorkerProtocol.ts @@ -2,7 +2,9 @@ // the FlexSearch web worker (searchIndex.worker). Types only — safe to import // from both sides. -export type IndexHit = { slug: string; start?: number; text: string }; +// `track`: a cue from one of the record's alternate English tracks +// (lib/captionTracks.ts) — absent for the transcript's own. +export type IndexHit = { slug: string; start?: number; text: string; track?: string }; // Main thread → worker. export type WorkerRequest = diff --git a/common/components/urlState.ts b/common/components/urlState.ts @@ -26,6 +26,10 @@ export type UrlParams = { v: string | null; t: number | null; vm: ModalMode; + // The transcript TRACK the reader shows (lib/captionTracks.ts) when it is + // not the record's primary — a viewer's choice, never a data change. Absent + // (null) for the primary, which is the default and stays off the URL. + vt: string | null; ch: string[]; nov: boolean; nol: boolean; @@ -94,6 +98,7 @@ function parse(search: string): UrlParams { v: p.get("v"), t: t !== null && Number.isFinite(t) ? t : null, vm, + vt: p.get("vt") || null, ch: p.getAll("ch"), nov: p.get("nov") === "1", nol: p.get("nol") === "1", @@ -126,6 +131,7 @@ type Patch = Partial<{ v: string | null; t: number | null; vm: ModalMode; + vt: string | null; ch: string[]; nov: boolean; nol: boolean; @@ -165,6 +171,10 @@ export function writeUrlParams(patch: Patch) { // setting vm=transcript — the default must stay absent from the URL. else params.delete("vm"); } + if (patch.vt !== undefined) { + if (patch.vt) params.set("vt", patch.vt); + else params.delete("vt"); + } if (patch.ch !== undefined) { params.delete("ch"); for (const v of patch.ch) params.append("ch", v); diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -65,6 +65,8 @@ import type { DisplaySummary, } from "../lib/transcripts"; import type { StoredSubs, SubsDetail } from "../lib/subs"; +import type { TrackFields } from "../lib/captionTracks"; +import { readTrackFields } from "../lib/captionTracks-server"; import type { Manifest, ChannelEntry, @@ -234,6 +236,28 @@ function storedCuesEmpty(db: { getBinaryFast(key: IndexKey): Buffer | undefined return raw === undefined || raw.length <= 4; } +// ALTERNATE TRACKS — the same one-shot shape again (lib/captionTracks.ts). A +// record's other English tracks, kept where their words differ from the +// primary's, live in the `alts` sub-DB and ride on its transcript page +// (`track` + `altTracks`). The first build that sees a new ALT_TRACKS_VERSION +// re-reads every record that CAN hold one — a caption record with two or more +// English VTTs, a transcribed record with any — and nothing else; then records +// the version, only when no channel is held. Bump it when what an alternate is +// changes. +const ALT_TRACKS_VERSION = 1; +const ALT_TRACKS_KEY = "altTracks"; + +// Whether a scanned record can hold an alternate track at all — no file read. +function canHoldAltTracks(s: { + transcriptKind: IndexTranscript["kind"] | null; + englishVttCount: number; +}): boolean { + return ( + (s.transcriptKind === "vtt" && s.englishVttCount >= 2) || + (s.transcriptKind === "whisper" && s.englishVttCount >= 1) + ); +} + export type CaptionTrackChannelReport = { reread: number; // Re-read records whose caption text is not what the index held. @@ -469,9 +493,13 @@ async function scanSource( let transcriptMs: number | null = null; if (picked) { // Captions: the newest of every caption input (each English VTT and - // the operator's pin), since the cues may come from any of them. + // the operator's pin), since the cues may come from any of them. A + // transcribed record's captions are its alternate tracks + // (lib/captionTracks.ts), so they count for it too. const inputs = - picked.kind === "vtt" ? captionInputs(files.entries) : [picked.filename]; + picked.kind === "vtt" + ? captionInputs(files.entries) + : [picked.filename, ...captionInputs(files.entries)]; for (const name of inputs) { try { const ms = (await stat(path.join(fullVideoDir, name))).mtimeMs; @@ -651,6 +679,12 @@ export async function buildIndex({ name: "subs", encoding: "msgpack", }); + // A record's alternate English tracks (ALT_TRACKS_KEY), only where one + // differs from the primary — sparse, like `subs`. + const alts = root.openDB<TrackFields, IndexKey>({ + name: "alts", + encoding: "msgpack", + }); const mtimes = root.openDB<MtimeRecord, PathKey>({ name: "mtimes", encoding: "msgpack", @@ -761,6 +795,7 @@ export async function buildIndex({ await sums.clearAsync(); await cues.clearAsync(); await subs.clearAsync(); + await alts.clearAsync(); await mtimes.clearAsync(); await byChannel.clearAsync(); await pageHashes.clearAsync(); @@ -878,6 +913,25 @@ export async function buildIndex({ } const captionReport = new Map<string, CaptionTrackChannelReport>(); + // Records that can hold an alternate track (ALT_TRACKS_KEY). A schema bump + // re-reads everything anyway. + const altTracksDue = meta.get(ALT_TRACKS_KEY) !== ALT_TRACKS_VERSION; + if (altTracksDue && !schemaBumped) { + const queued = new Set( + [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])), + ); + let reread = 0; + for (const s of live) { + if (!canHoldAltTracks(s)) continue; + const pk: PathKey = [s.channelSlug, s.videoDir]; + if (!mtimes.get(pk)) continue; + reread++; + if (!queued.has(pathKeyId(pk))) changed.push(s); + } + log(`Alternate tracks v${ALT_TRACKS_VERSION}: ${reread} record(s) re-read.`); + } + let altTrackRecords = 0; + const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; @@ -1012,6 +1066,7 @@ export async function buildIndex({ sums.remove(prev.indexKey); cues.remove(prev.indexKey); subs.remove(prev.indexKey); + alts.remove(prev.indexKey); digests.remove(prev.indexKey); byChannel.remove(indexToChannelKey(prev.indexKey)); } @@ -1019,6 +1074,18 @@ export async function buildIndex({ if (cueList) cues.put(indexKey, cueList); else cues.remove(indexKey); + // The other English tracks, where their words differ from the + // primary's (lib/captionTracks.ts). Only a record that can hold one + // reads anything. + const trackFields: TrackFields = + canHoldAltTracks(s) && s.transcriptKind + ? await readTrackFields(videoFullDir, s.transcriptKind, cueList) + : {}; + if (trackFields.altTracks) { + alts.put(indexKey, trackFields); + altTrackRecords++; + } else alts.remove(indexKey); + if (heldBefore !== undefined) { const r = captionReport.get(s.channelSlug) ?? { reread: 0, changed: 0, zeroToText: 0 }; r.reread++; @@ -1211,10 +1278,15 @@ export async function buildIndex({ ); } + if (altTrackRecords > 0) { + log(`Alternate tracks: ${altTrackRecords} record(s) read this build hold a track whose words differ from the primary's.`); + } + for (const { pathKey, indexKey } of removed) { sums.remove(indexKey); cues.remove(indexKey); subs.remove(indexKey); + alts.remove(indexKey); digests.remove(indexKey); byChannel.remove(indexToChannelKey(indexKey)); mtimes.remove(pathKey); @@ -1254,6 +1326,7 @@ export async function buildIndex({ await sums.flushed; await cues.flushed; await subs.flushed; + await alts.flushed; await digests.flushed; await byChannel.flushed; await mtimes.flushed; @@ -1436,7 +1509,16 @@ export async function buildIndex({ const summary = sums.get(indexKey); if (!summary) continue; const cueList = cues.get(indexKey); - const detail: TranscriptDetail = { ...summary, cues: cueList }; + // `track` + `altTracks` only on a record that has an alternate, so every + // other record's page bytes are what they were. + const trackFields = alts.get(indexKey); + const detail: TranscriptDetail = { + ...summary, + cues: cueList, + ...(trackFields?.altTracks?.length + ? { track: trackFields.track, altTracks: trackFields.altTracks } + : {}), + }; const encoded = JSON.stringify(detail); await writer.push(encoded, summary.id); } @@ -2434,6 +2516,9 @@ export async function buildIndex({ if (captionTrackDue && held.size === 0) { await meta.put(CAPTION_TRACK_KEY, CAPTION_TRACK_RULE_VERSION); } + if (altTracksDue && held.size === 0) { + await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION); + } await meta.flushed; await root.close(); diff --git a/common/controller/buildIndexAltTracks.test.ts b/common/controller/buildIndexAltTracks.test.ts @@ -0,0 +1,211 @@ +// Integration: ALTERNATE TRACKS (lib/captionTracks.ts) through the REAL +// buildIndex over a temp corpus. A record whose served `en` says something its +// en-orig does not carries that track on its transcript page — so a word only +// in `en` is findable, with the track named — and a record whose tracks are +// identical carries nothing extra; no English VTT is shipped again as a +// subtitle track; and the one-shot pass (ALT_TRACKS_VERSION) re-reads exactly +// the records that can hold an alternate. +// +// Run with: node_modules/.bin/tsx --test common/controller/buildIndexAltTracks.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-alts-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("./buildIndex"); +const { open } = await import("lmdb"); +const { hitsAcrossTracks } = await import("../lib/captionTracks"); + +const paths = getPaths(); +const CHANNEL = "example-channel"; +const SITE = "testsite"; +const DIFFER = "DifferTrack1"; // en-orig + an en that says other words +const SAME = "SameTracks01"; // en-orig + an identical en +const WHISPER = "Transcribed1"; // transcript.json + captions that differ +const SPANISH = "SpanishSub01"; // en-orig + a Spanish subtitle track + +const fixture = (name: string) => + readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8"); +const ROLLING = fixture("vtt-rolling.vtt"); +const vtt = (...lines: [string, string, string][]) => + "WEBVTT\nKind: captions\nLanguage: en\n\n" + + lines.map(([a, b, t]) => `${a} --> ${b}\n${t}\n`).join("\n"); +// What the served `en` says: the same opening, then a word en-orig never has, +// far from anything en-orig matches. +const SERVED_EN = vtt( + ["00:00:03.080", "00:00:05.670", "are talking about the harbor"], + ["00:01:40.000", "00:01:43.000", "the zeppelin landed in nineteen thirty"], +); +const WHISPER_JSON = JSON.stringify({ + transcription: [ + { offsets: { from: 0, to: 2000 }, text: " a local transcription says hello" }, + { offsets: { from: 2000, to: 4000 }, text: " and nothing else" }, + ], +}); + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); + +function seed(): void { + rmSync(paths.transcriptsDir, { recursive: true, force: true }); + rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true }); + writeFileSync(paths.settingsFile, "{}"); + writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { + handling: "youtube", + name: CHANNEL, + }); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [{ slug: CHANNEL, groupId: "default" }], + }); + for (const [i, id] of [DIFFER, SAME, WHISPER, SPANISH].entries()) { + writeJson(path.join(dirOf(id), "metadata.info.json"), { + id, + title: `Video ${id}`, + upload_date: `2026060${i + 1}`, + duration: 200, + webpage_url: `https://www.youtube.com/watch?v=${id}`, + extractor_key: "Youtube", + }); + } + writeFileSync(path.join(dirOf(DIFFER), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(DIFFER), "transcript.en.vtt"), SERVED_EN); + writeFileSync(path.join(dirOf(SAME), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(SAME), "transcript.en.vtt"), ROLLING); + writeFileSync(path.join(dirOf(WHISPER), "transcript.json"), WHISPER_JSON); + writeFileSync(path.join(dirOf(WHISPER), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(SPANISH), "transcript.en-orig.vtt"), ROLLING); + writeFileSync( + path.join(dirOf(SPANISH), "transcript.es.vtt"), + vtt(["00:00:01.000", "00:00:02.000", "hola a todos"]), + ); +} + +type Cue = { start: number; end: number; text: string }; +type Rec = { + id: string; + cues?: Cue[]; + track?: string; + altTracks?: { track: string; cues: Cue[] }[]; +}; + +// Every record of the channel's shared transcript pages, by id. +function pageRecords(): Map<string, Rec> { + const dir = path.join(paths.exportSharedTranscriptsDir, CHANNEL); + const out = new Map<string, Rec>(); + for (const name of readdirSync(dir)) { + if (!/^page-\d+\.json$/.test(name)) continue; + for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as Rec[]) { + out.set(r.id, r); + } + } + return out; +} +function subsTracks(): Record<string, string[]> { + const dir = path.join(paths.exportSharedSubsDir, CHANNEL); + const out: Record<string, string[]> = {}; + if (!existsSync(dir)) return out; + for (const name of readdirSync(dir)) { + if (!/^page-\d+\.json$/.test(name)) continue; + for (const r of JSON.parse(readFileSync(path.join(dir, name), "utf8")) as { + id: string; + tracks: Record<string, unknown>; + }[]) { + out[r.id] = Object.keys(r.tracks).sort(); + } + } + return out; +} + +async function runIndex(): Promise<string[]> { + const log: string[] = []; + await buildIndex({ paths, onLog: (s) => log.push(s) }); + return log; +} + +test("a differing served en rides on the page as an alternate; identical tracks add nothing", async () => { + seed(); + await runIndex(); + const recs = pageRecords(); + + const differ = recs.get(DIFFER)!; + assert.equal(differ.track, "en-orig"); + assert.deepEqual(differ.altTracks?.map((t) => t.track), ["en"]); + // The word only `en` has is found — in `en`, named — and the opening both + // say is found once, in the primary. + const find = (q: string) => + hitsAcrossTracks(differ, (cues) => cues.filter((c) => c.text.includes(q))); + assert.deepEqual( + find("zeppelin").map((h) => [h.track, Math.round(h.start)]), + [["en", 100]], + ); + assert.deepEqual(find("harbor").map((h) => h.track), [undefined]); + + // Identical tracks: no fields at all, so the record is what it always was. + const same = recs.get(SAME)!; + assert.equal("track" in same, false); + assert.equal("altTracks" in same, false); + + // A transcription is the primary; the captions it replaced are an alternate. + const whisper = recs.get(WHISPER)!; + assert.equal(whisper.track, "transcription"); + assert.equal(whisper.cues?.[0].text, "a local transcription says hello"); + assert.deepEqual(whisper.altTracks?.map((t) => t.track), ["en-orig"]); + + // No English VTT is a subtitle track any more; a Spanish one still is. + assert.deepEqual(subsTracks(), { [SPANISH]: ["es"] }); +}); + +test("the one-shot pass re-reads exactly the records that can hold an alternate, once", async () => { + // The index as a build before alternate tracks left it: no `alts` records, + // no version. + const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); + try { + await root.openDB({ name: "alts", encoding: "msgpack" }).clearAsync(); + await root.openDB({ name: "meta", encoding: "msgpack" }).remove("altTracks"); + } finally { + await root.close(); + } + const log = await runIndex(); + // DIFFER, SAME (two English VTTs) and WHISPER (a transcription beside + // captions); not SPANISH. + assert.ok(log.includes("Alternate tracks v1: 3 record(s) re-read."), log.join("\n")); + assert.deepEqual(pageRecords().get(DIFFER)?.altTracks?.map((t) => t.track), ["en"]); + + const again = await runIndex(); + assert.equal(again.some((l) => /Alternate tracks v1/.test(l)), false, again.join("\n")); +}); diff --git a/common/lib/captionTrack.test.ts b/common/lib/captionTrack.test.ts @@ -97,14 +97,14 @@ test("readEnglishVttCues: every track empty is the first track with no cues; non assert.equal(await readEnglishVttCues(videoDir({ "transcript.es.vtt": ROLLING })), null); }); -test("readSubTracks lists the served en as an alternate beside an en-orig primary", async () => { +test("readSubTracks lists no English VTT: those are caption tracks (lib/captionTracks.ts)", async () => { const dir = videoDir({ "transcript.en.vtt": CUE_BLOCKS, "transcript.en-orig.vtt": ROLLING, "transcript.es.vtt": ROLLING, }); const tracks = (await readSubTracks(dir)).map((t) => t.track).sort(); - assert.deepEqual(tracks, ["en", "es"]); + assert.deepEqual(tracks, ["es"]); }); test("umtool's copy of the caption-track rule version matches", () => { diff --git a/common/lib/captionTracks-server.ts b/common/lib/captionTracks-server.ts @@ -0,0 +1,129 @@ +// The alternate tracks of one video dir, read from disk (lib/captionTracks.ts +// says what an alternate is). SERVER-ONLY (node:fs). + +import path from "node:path"; +import { readdir, readFile } from "node:fs/promises"; +import { parseVtt, type Cue } from "./vtt"; +import { parseTranscriptJson } from "./whisper"; +import { + TRANSCRIPT_PIN_FILENAME, + VTT_FILENAME, + WHISPER_FILENAME, + englishVttsByPreference, + readEnglishVttCues, +} from "./videoStatus"; +import { + PINNED_TRACK, + TRANSCRIPTION_TRACK, + distinctAltTracks, + trackOfVttFile, + type AltTrack, + type TrackFields, +} from "./captionTracks"; + +// The track id of an English caption file in a listing: its language code, or +// `pinned` for transcript.en.vtt while the operator's pin stands. +export function trackIdOfVtt(filename: string, entries: readonly string[]): string { + if (filename === VTT_FILENAME && entries.includes(TRANSCRIPT_PIN_FILENAME)) { + return PINNED_TRACK; + } + return trackOfVttFile(filename) ?? filename; +} + +// Whether a listing can hold an alternate at all — no file is read. A caption +// record needs two English VTTs; a transcribed one, one. +export function mayHaveAltTracks( + primaryKind: "vtt" | "whisper", + entries: readonly string[], +): boolean { + const n = englishVttsByPreference(entries).length; + return primaryKind === "whisper" ? n >= 1 : n >= 2; +} + +// Every English VTT of a listing, parsed, in preference order. One that cannot +// be read is left out. +export async function readEnglishVttTracks( + videoDir: string, + entries: readonly string[], +): Promise<{ filename: string; cues: Cue[] }[]> { + const out: { filename: string; cues: Cue[] }[] = []; + for (const filename of englishVttsByPreference(entries)) { + try { + out.push({ + filename, + cues: parseVtt(await readFile(path.join(videoDir, filename), "utf8")), + }); + } catch { + // unreadable: not a track + } + } + return out; +} + +// A record's track fields: the primary's id and the English tracks whose words +// differ from it. `primaryCues` is what the record's transcript holds (the +// dedupe compares against it); for a caption record, the primary is the first +// track in preference order that has a cue — the caption-track rule's content +// fallback, so the id names the track the words really came from. Empty +// (no fields) when nothing differs. +export async function readTrackFields( + videoDir: string, + primaryKind: "vtt" | "whisper", + primaryCues: readonly Cue[] | undefined, + entries?: readonly string[], +): Promise<TrackFields> { + const listing = entries ?? (await readdir(videoDir).catch(() => [] as string[])); + if (!mayHaveAltTracks(primaryKind, listing)) return {}; + const vtts = await readEnglishVttTracks(videoDir, listing); + let primaryTrack: string; + let primary: readonly Cue[] | undefined = primaryCues; + let candidates: { filename: string; cues: Cue[] }[]; + if (primaryKind === "whisper") { + primaryTrack = TRANSCRIPTION_TRACK; + candidates = vtts; + } else { + if (vtts.length === 0) return {}; + const idx = Math.max(0, vtts.findIndex((t) => t.cues.length > 0)); + primaryTrack = trackIdOfVtt(vtts[idx].filename, listing); + primary ??= vtts[idx].cues; + candidates = vtts.filter((_, i) => i !== idx); + } + const alts: AltTrack[] = distinctAltTracks( + primary, + candidates.map((t) => ({ track: trackIdOfVtt(t.filename, listing), cues: t.cues })), + ); + if (alts.length === 0) return {}; + return { track: primaryTrack, altTracks: alts }; +} + +// Every track of a video dir, read from disk, primary first: what the editor's +// transcript reader shows and switches between. The primary is the record's +// transcript by the same rules the index reads it with — a local +// transcription (transcript.json) over captions, captions by the caption-track +// rule — and the rest are the alternates readTrackFields keeps. Null when the +// dir holds no transcript. +export async function readVideoTracks( + videoDir: string, +): Promise<{ tracks: AltTrack[] } | null> { + const entries = await readdir(videoDir).catch(() => [] as string[]); + let primary: AltTrack | null = null; + let kind: "vtt" | "whisper" = "vtt"; + if (entries.includes(WHISPER_FILENAME)) { + try { + primary = { + track: TRANSCRIPTION_TRACK, + cues: parseTranscriptJson(await readFile(path.join(videoDir, WHISPER_FILENAME), "utf8")), + }; + kind = "whisper"; + } catch { + primary = null; + } + } + if (!primary) { + const read = await readEnglishVttCues(videoDir, entries); + if (!read) return null; + primary = { track: trackIdOfVtt(read.filename, entries), cues: read.cues }; + } + const fields = await readTrackFields(videoDir, kind, primary.cues, entries); + return { tracks: [primary, ...(fields.altTracks ?? [])] }; +} diff --git a/common/lib/captionTracks.test.ts b/common/lib/captionTracks.test.ts @@ -0,0 +1,189 @@ +// The tracks of a transcript (lib/captionTracks.ts + captionTracks-server.ts): +// labels from ids, which alternates are kept, and how a search finds a word +// across them. +// +// Run with: node_modules/.bin/tsx --test common/lib/captionTracks.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + cuesOfTrack, + distinctAltTracks, + hitsAcrossTracks, + inTrackLabel, + recordTracks, + trackKind, + trackLabel, + trackLabels, + uncoveredAltHits, +} from "./captionTracks"; +import { readTrackFields, readVideoTracks } from "./captionTracks-server"; +import { TRANSCRIPT_PIN_FILENAME } from "./videoStatus"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "caption-tracks-")); +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const cue = (start: number, text: string) => ({ start, end: start + 2, text }); +const vtt = (...cues: [number, string][]) => + "WEBVTT\n\n" + + cues + .map(([s, t]) => { + const ts = (x: number) => `00:${String(Math.floor(x / 60)).padStart(2, "0")}:${String(x % 60).padStart(2, "0")}.000`; + return `${ts(s)} --> ${ts(s + 2)}\n${t}\n`; + }) + .join("\n"); + +let n = 0; +function videoDir(files: Record<string, string>): string { + const dir = path.join(ROOT, `v${n++}`); + mkdirSync(dir, { recursive: true }); + for (const [name, body] of Object.entries(files)) writeFileSync(path.join(dir, name), body); + return dir; +} + +test("labels are plain words derived from the track id", () => { + assert.equal(trackLabel("en-orig"), "original audio captions"); + assert.equal(trackLabel("en"), "uploaded captions"); + assert.equal(trackLabel("en-en-US"), "auto-translated captions"); + assert.equal(trackLabel("en-GB"), "UK English captions"); + assert.equal(trackLabel("en-NG"), "regional captions (en-NG)"); + assert.equal(trackLabel("en-US-orig"), "original audio captions (US)"); + // A track YouTube named itself: another uploaded one. + assert.equal(trackLabel("en-uYU-mmqFLq8"), "other uploaded captions"); + assert.equal(trackKind("en-JkeT_87f4cc"), "uploaded"); + assert.deepEqual(trackLabels(["en-orig", "en-aa-1", "en-bb-2"]), [ + "original audio captions", + "other uploaded captions (en-aa-1)", + "other uploaded captions (en-bb-2)", + ]); + assert.equal(trackLabel("transcription"), "transcription"); + assert.equal(trackLabel("pinned"), "chosen captions"); + assert.equal(inTrackLabel("en"), "in uploaded captions"); + assert.equal(trackKind("en-US"), "regional"); + assert.equal(trackKind("live_chat"), "other"); +}); + +test("an alternate is kept only where its words differ from the primary and every kept one", () => { + const primary = [cue(0, "hello there")]; + const kept = distinctAltTracks(primary, [ + { track: "en", cues: [cue(0, "hello there")] }, // identical words + { track: "en-GB", cues: [cue(5, "hello there")] }, // timing alone differs + { track: "en-US", cues: [cue(0, "hello their")] }, // other words: kept + { track: "en-en-US", cues: [cue(0, "hello their")] }, // same as en-US + { track: "en-CA", cues: [] }, // empty + ]); + assert.deepEqual(kept.map((t) => t.track), ["en-US"]); +}); + +test("recordTracks and cuesOfTrack: primary first; an unknown track is undefined", () => { + const rec = { + cues: [cue(0, "a")], + track: "en-orig", + altTracks: [{ track: "en", cues: [cue(0, "b")] }], + }; + assert.deepEqual(recordTracks(rec), ["en-orig", "en"]); + assert.equal(cuesOfTrack(rec, null)?.[0].text, "a"); + assert.equal(cuesOfTrack(rec, "en-orig")?.[0].text, "a"); + assert.equal(cuesOfTrack(rec, "en")?.[0].text, "b"); + assert.equal(cuesOfTrack(rec, "en-GB"), undefined); + assert.deepEqual(recordTracks({}), []); +}); + +test("a search finds a word every track says once, in the primary, and an alternate's own word there", () => { + const rec = { + cues: [cue(10, "the bridge opened"), cue(300, "and then we left")], + track: "en-orig", + altTracks: [ + { + track: "en", + cues: [cue(11, "the bridge opened in 1932"), cue(200, "a zeppelin flew over the bridge")], + }, + ], + }; + const find = (q: string) => + hitsAcrossTracks(rec, (cues) => cues.filter((c) => c.text.includes(q))); + assert.deepEqual( + find("bridge").map((h) => [h.start, h.track]), + [ + [10, undefined], + [200, "en"], + ], + ); + assert.deepEqual(find("1932"), [{ ...cue(11, "the bridge opened in 1932"), track: "en" }]); + assert.deepEqual(find("nothing"), []); + // A record with no alternates is its primary alone. + assert.deepEqual( + hitsAcrossTracks({ cues: rec.cues }, (cues) => cues.filter((c) => c.text.includes("left"))), + [cue(300, "and then we left")], + ); +}); + +test("uncoveredAltHits drops an alternate hit within the window of a primary one", () => { + assert.deepEqual( + uncoveredAltHits([{ start: 100 }, { start: 500 }], [{ start: 90 }, { start: 130 }, { start: 515 }, { start: 900 }]), + [{ start: 130 }, { start: 900 }], + ); + assert.deepEqual(uncoveredAltHits([], [{ start: 1 }]), [{ start: 1 }]); +}); + +test("readTrackFields: a differing en beside en-orig is an alternate; identical tracks are none", async () => { + const differ = videoDir({ + "transcript.en-orig.vtt": vtt([1, "said words"]), + "transcript.en.vtt": vtt([1, "uploaded words"]), + }); + assert.deepEqual(await readTrackFields(differ, "vtt", undefined), { + track: "en-orig", + altTracks: [{ track: "en", cues: [{ start: 1, end: 3, text: "uploaded words" }] }], + }); + const same = videoDir({ + "transcript.en-orig.vtt": vtt([1, "said words"]), + "transcript.en.vtt": vtt([1, "said words"]), + }); + assert.deepEqual(await readTrackFields(same, "vtt", undefined), {}); + // One English VTT: nothing to read. + const lone = videoDir({ "transcript.en.vtt": vtt([1, "x"]) }); + assert.deepEqual(await readTrackFields(lone, "vtt", undefined), {}); +}); + +test("readTrackFields: an empty en-orig falls through to en as the primary, as the rule reads it", async () => { + const dir = videoDir({ + "transcript.en-orig.vtt": "WEBVTT\n\n", + "transcript.en.vtt": vtt([1, "served words"]), + "transcript.en-GB.vtt": vtt([1, "british words"]), + }); + assert.deepEqual(await readTrackFields(dir, "vtt", undefined), { + track: "en", + altTracks: [{ track: "en-GB", cues: [{ start: 1, end: 3, text: "british words" }] }], + }); +}); + +test("readTrackFields: the operator's pin names the primary `pinned`; its source is not repeated", async () => { + const dir = videoDir({ + "transcript.en-orig.vtt": vtt([1, "said words"]), + "transcript.en.vtt": vtt([1, "said words"]), // the copy of en-orig the pin made + "transcript.en-US.vtt": vtt([1, "other words"]), + [TRANSCRIPT_PIN_FILENAME]: JSON.stringify({ from: "transcript.en-orig.vtt", pinnedAt: "" }), + }); + assert.deepEqual(await readTrackFields(dir, "vtt", undefined), { + track: "pinned", + altTracks: [{ track: "en-US", cues: [{ start: 1, end: 3, text: "other words" }] }], + }); +}); + +test("readVideoTracks: a transcription is the primary and its captions the alternate", async () => { + const dir = videoDir({ + "transcript.json": JSON.stringify({ + transcription: [{ offsets: { from: 0, to: 1000 }, text: " machine words" }], + }), + "transcript.en-orig.vtt": vtt([1, "caption words"]), + }); + const read = await readVideoTracks(dir); + assert.deepEqual(read?.tracks.map((t) => [t.track, t.cues[0].text]), [ + ["transcription", "machine words"], + ["en-orig", "caption words"], + ]); + assert.equal(await readVideoTracks(videoDir({ "metadata.info.json": "{}" })), null); +}); diff --git a/common/lib/captionTracks.ts b/common/lib/captionTracks.ts @@ -0,0 +1,225 @@ +// THE TRACKS OF A TRANSCRIPT — one notion of "track" for every reader. +// +// A record's transcript is read from ONE track, its primary, chosen by the +// caption-track rule (lib/videoStatus.ts: en-orig first, the operator's pin +// above all, a local transcription above captions). A record may hold other +// English tracks beside it: the served `en`, a regional en-GB, an en→en +// auto-translation, or the captions a local transcription replaced. Human +// captions are not always a transcript of what was said, so those stay +// readable and searchable as ALTERNATES — a viewer's choice, never a change to +// the primary (the pin, transcript-pin.json, is how the primary changes). +// +// An alternate is kept only where its words differ from the primary's and from +// every alternate kept before it: most served `en` tracks are byte-identical to +// en-orig, and shipping or indexing those would double the corpus for nothing. +// +// Track ids are the caption's language code as its file names it +// (transcript.<id>.vtt), plus two that no file names: `transcription` (a local +// whisper/parakeet transcript) and `pinned` (transcript.en.vtt while the +// operator's pin stands — a copy of whichever track was picked). Labels are +// derived from the id, here and nowhere else, so the editor, the export site, +// a search hit and the MCP all say the same words. +// +// Pure: no node built-ins, safe in a client bundle. + +import type { Cue } from "./vtt"; + +export const TRANSCRIPTION_TRACK = "transcription"; +export const PINNED_TRACK = "pinned"; + +export type AltTrack = { track: string; cues: Cue[] }; + +// What a transcript record carries about its tracks. Both fields are OMITTED +// when the record has no alternate that differs (the common case), so those +// records' pages stay byte-identical to the ones already on disk. +export type TrackFields = { + // The primary's track id. + track?: string; + // The English tracks whose words differ from the primary's, in preference + // order. + altTracks?: AltTrack[]; +}; + +const REGIONS: Record<string, string> = { + US: "US", + GB: "UK", + UK: "UK", + CA: "Canadian", + AU: "Australian", + IE: "Irish", + IN: "Indian", + NZ: "New Zealand", + ZA: "South African", +}; + +// The kind of a track, from its id. +export type TrackKind = + | "original" + | "uploaded" + | "regional" + | "translated" + | "transcription" + | "pinned" + | "other"; + +// A region subtag as YouTube writes one: two letters (en-US) or three digits +// (en-419). Anything else after `en-` is a track YouTube named itself — an +// uploader's extra caption track (en-uYU-mmqFLq8). +const REGION_RE = /^(?:[A-Za-z]{2}|\d{3})$/; + +export function trackKind(track: string): TrackKind { + if (track === TRANSCRIPTION_TRACK) return "transcription"; + if (track === PINNED_TRACK) return "pinned"; + if (track === "en-orig" || /^en-[^-]+-orig$/.test(track)) return "original"; + if (track === "en") return "uploaded"; + if (/^en-en(?:-|$)/.test(track)) return "translated"; + const m = track.match(/^en-(.+)$/); + if (m) return REGION_RE.test(m[1]) ? "regional" : "uploaded"; + return "other"; +} + +// "UK" for en-GB, the code itself for a region this table does not name. +function regionName(code: string): string { + return REGIONS[code.toUpperCase()] ?? code; +} + +// A plain label for a track — what a person reads in the switcher and on a +// search hit ("in uploaded captions"). +export function trackLabel(track: string): string { + switch (trackKind(track)) { + case "original": { + // en-US-orig: the original audio's captions, under a region. + const m = track.match(/^en-([^-]+)-orig$/); + return m ? `original audio captions (${regionName(m[1])})` : "original audio captions"; + } + case "uploaded": + // A track YouTube named (en-uYU-mmqFLq8) is another uploaded one. + return track === "en" ? "uploaded captions" : "other uploaded captions"; + case "translated": + return "auto-translated captions"; + case "transcription": + return "transcription"; + case "pinned": + return "chosen captions"; + case "regional": { + const name = REGIONS[track.slice(3).toUpperCase()]; + return name ? `${name} English captions` : `regional captions (${track})`; + } + default: + return `captions (${track})`; + } +} + +// The labels of a switcher's tracks, in order: trackLabel, with the id added +// where two tracks would otherwise read the same (two tracks YouTube named). +export function trackLabels(tracks: readonly string[]): string[] { + const labels = tracks.map(trackLabel); + return labels.map((l, i) => + labels.indexOf(l) !== labels.lastIndexOf(l) ? `${l} (${tracks[i]})` : l, + ); +} + +// The track id of a caption file name (transcript.<id>.vtt), or null. +export function trackOfVttFile(filename: string): string | null { + const m = filename.match(/^transcript\.([^.]+)\.vtt$/); + return m ? m[1] : null; +} + +// A cue list's words, for "do these two tracks say the same thing" — timing +// alone is not a different transcript. +export function cueWords(cues: readonly Cue[]): string { + return cues.map((c) => c.text).join("\n"); +} + +// The alternates worth keeping: tracks with at least one cue whose words differ +// from the primary's and from every track kept before them. `candidates` in +// preference order; the primary is not among them. +export function distinctAltTracks( + primary: readonly Cue[] | undefined, + candidates: readonly AltTrack[], +): AltTrack[] { + const seen = new Set<string>([cueWords(primary ?? [])]); + const out: AltTrack[] = []; + for (const c of candidates) { + if (c.cues.length === 0) continue; + const words = cueWords(c.cues); + if (seen.has(words)) continue; + seen.add(words); + out.push(c); + } + return out; +} + +// The tracks a record offers, primary first: what a switcher lists. A record +// with no alternates offers just its primary (or nothing, with no track id). +export function recordTracks(rec: TrackFields): string[] { + if (!rec.altTracks || rec.altTracks.length === 0) return rec.track ? [rec.track] : []; + return [rec.track ?? "", ...rec.altTracks.map((t) => t.track)].filter((t) => t !== ""); +} + +// The cues of one track of a record: the primary when `track` is absent or +// names it, an alternate when it names one, undefined when the record has no +// such track. +export function cuesOfTrack<R extends TrackFields & { cues?: Cue[] | undefined }>( + rec: R, + track: string | null | undefined, +): Cue[] | undefined { + if (!track || track === rec.track) return rec.cues; + return rec.altTracks?.find((t) => t.track === track)?.cues; +} + +// How far (seconds) a primary hit covers an alternate's. An alternate hit with +// a primary hit this close is the same moment found twice — the primary's is +// the one shown. Only a match the primary has nowhere near is the alternate's +// to report. +export const ALT_HIT_COVERED_SEC = 20; + +// Drop the alternate hits a primary hit already covers (ALT_HIT_COVERED_SEC). +// Both lists carry `start` in seconds; the primary's needs no order. +export function uncoveredAltHits<H extends { start: number }>( + primary: readonly { start: number }[], + alt: readonly H[], +): H[] { + if (primary.length === 0) return [...alt]; + const starts = primary.map((h) => h.start).sort((a, b) => a - b); + return alt.filter((h) => { + // Binary search for the nearest primary start. + let lo = 0; + let hi = starts.length - 1; + while (lo < hi) { + const mid = (lo + hi) >> 1; + if (starts[mid] < h.start) lo = mid + 1; + else hi = mid; + } + const near = [starts[lo], starts[lo - 1]].filter((s) => s !== undefined); + return !near.some((s) => Math.abs(s - h.start) <= ALT_HIT_COVERED_SEC); + }); +} + +// THE ONE RULE for matching a record's words across its tracks, used by every +// search (the viewer's leaf pipeline, the MCP, the query-tree evaluator): +// `find` is run over the primary's cues, then over each alternate's, and an +// alternate's hit is kept only where no hit already kept is within +// ALT_HIT_COVERED_SEC — so a word every track says is found once, in the +// primary, and a word only an alternate says is found there, wearing that +// alternate's track id. Ordered by start when an alternate adds anything. +export function hitsAcrossTracks<H extends { start: number }>( + rec: TrackFields & { cues?: readonly Cue[] | undefined }, + find: (cues: readonly Cue[]) => H[], +): (H & { track?: string })[] { + const out: (H & { track?: string })[] = rec.cues ? find(rec.cues) : []; + let added = false; + for (const alt of rec.altTracks ?? []) { + for (const h of uncoveredAltHits(out, find(alt.cues))) { + out.push({ ...h, track: alt.track }); + added = true; + } + } + if (added) out.sort((a, b) => a.start - b.start); + return out; +} + +// "in uploaded captions" — what a hit from an alternate says about itself. +export function inTrackLabel(track: string): string { + return `in ${trackLabel(track)}`; +} diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts @@ -80,7 +80,10 @@ const SHARD_SCHEME = { "<channel.manifests.transcripts> -> { pageCount, slugToPage: { <videoId>: <pageNumber> } }", transcriptPage: "/transcripts/<slug>/page-<NNNN>.json -> array of { id, title, uploadDate, " + - "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }", + "duration, channel, description, tags, webpageUrl, platform, cues: [{ start, end, text }] }; " + + "a record with other English caption tracks whose words differ from its transcript " + + "also carries `track` (the transcript's track id, e.g. \"en-orig\") and " + + "`altTracks: [{ track, cues }]` (e.g. \"en\", the uploaded captions) — absent otherwise", subsManifest: "<channel.manifests.subs> -> lighter list-view records under the same slugToPage scheme", summariesIndex: diff --git a/common/lib/search/evalTree.test.ts b/common/lib/search/evalTree.test.ts @@ -374,3 +374,22 @@ test("a non-empty curated-tag selection makes a filter selective", () => { // a tag-only query would plan a full scan. assert.equal(filterIsSelective(openFilters({ curatedTags: ["eva-collab"] })), true); }); + +test("a transcripts leaf reads the record's alternate tracks; a hit only one holds wears its track", () => { + const r = evalLeaf( + newLeaf({ id: "l", query: "zeppelin" }), + { scope: "transcripts", test: (t) => t.includes("zeppelin") || t.includes("bridge") }, + ctx({ + cues: [cue(3, "the bridge")], + altTracks: [{ track: "en", cues: [cue(4, "the bridge"), cue(100, "a zeppelin")] }], + }), + ); + assert.equal(r.count, 2, "the bridge once (primary), the zeppelin once (en)"); + assert.deepEqual( + r.hits.map((h) => [h.seconds, h.track]), + [ + [3, undefined], + [100, "en"], + ], + ); +}); diff --git a/common/lib/search/evalTree.ts b/common/lib/search/evalTree.ts @@ -32,6 +32,7 @@ import { import { matchAliases, type SearchAlias } from "../searchAliases"; import { VIDEO_STATES, type VideoState } from "../availability"; import type { Cue } from "../vtt"; +import { hitsAcrossTracks, type AltTrack } from "../captionTracks"; import { clock, truncate, type Matcher } from "./window"; export type { Matcher }; @@ -131,6 +132,9 @@ export type RecordCtx = { description: string; tags: string; cues: Cue[]; + // The record's alternate English tracks (lib/captionTracks.ts), when it has + // any: the transcripts scope reads them too. + altTracks?: AltTrack[]; chatCues: Cue[]; // The post body, when this record IS a post rather than a video. Empty for a // video record, so a posts-scope leaf never matches one. @@ -159,11 +163,16 @@ export function evalLeaf( let count = 0; switch (m.scope) { case "transcripts": - for (const cue of ctx.cues) { - if (!m.test(cue.text)) continue; + // Every English track of the record (lib/captionTracks.ts): a match only + // an alternate holds wears that alternate's track id. + for (const cue of hitsAcrossTracks( + { cues: ctx.cues, altTracks: ctx.altTracks }, + (cues) => cues.filter((c) => m.test(c.text)), + )) { count++; push({ scope: "transcripts", + ...(cue.track ? { track: cue.track } : {}), clock: clock(cue.start), seconds: cue.start, text: truncate(cue.text, max), diff --git a/common/lib/search/leafPipeline.test.ts b/common/lib/search/leafPipeline.test.ts @@ -284,3 +284,28 @@ test("a regex leaf compiles once; a malformed pattern matches nothing", async () const bad = await run(newLeaf({ query: "n([edle", useRegex: true }), ["c/v1"], f); assert.deepEqual(bad.slugs, []); }); + +test("a transcripts leaf reads a record's alternate tracks too, and names the track of a hit only one holds", async () => { + const rec = { + ...record("v1", [{ start: 3, text: "the harbor bridge" }]), + track: "en-orig", + altTracks: [ + { + track: "en", + cues: [ + { start: 4, end: 6, text: "the harbor bridge" }, + { start: 100, end: 102, text: "a zeppelin overhead" }, + ], + }, + ], + } as TranscriptDetail; + const a = archive({ pages: [[rec, record("v2", [{ start: 0, text: "nothing" }])]] }); + const only = await run(newLeaf({ query: "zeppelin" }), ["c/v1", "c/v2"], fetchersFor(a)); + assert.deepEqual(only.slugs, ["c/v1"]); + const hits = only.hits.get("c/v1") as { start: number; track?: string; scope: string }[]; + assert.deepEqual(hits.map((h) => [h.start, h.track, h.scope]), [[100, "en", "transcripts"]]); + // A word both tracks say near the same moment: one hit, the primary's. + const both = await run(newLeaf({ query: "harbor" }), ["c/v1"], fetchersFor(a)); + const bh = both.hits.get("c/v1") as { track?: string }[]; + assert.deepEqual(bh.map((h) => h.track), [undefined]); +}); diff --git a/common/lib/search/leafPipeline.ts b/common/lib/search/leafPipeline.ts @@ -25,6 +25,7 @@ import type { SubsDetail } from "../subs"; import type { Post } from "../posts"; import type { LayerScope, LeafNode } from "../searchQuery"; import { findHitsInCues, findHitsInText, type Hit, type SubsHit } from "./window"; +import { hitsAcrossTracks } from "../captionTracks"; export type { Hit, SubsHit }; @@ -163,9 +164,10 @@ export function createSearchPipeline( useRegex, regex, ) - : full.cues - ? findHitsInCues(full.cues, query, useRegex, regex, remaining) - : []; + : // Every English track of the record (lib/captionTracks.ts). + hitsAcrossTracks(full, (cues) => + findHitsInCues(cues, query, useRegex, regex, remaining), + ); if (hits.length > 0) { localHits[slug] = hits; totalSoFar += hits.length; @@ -665,6 +667,7 @@ export function runLeafPipeline(opts: LeafPipelineOptions): LeafController { list.map((h) => ({ leafId: leaf.id, scope: leaf.scope, + ...(h.track ? { track: h.track } : {}), start: h.start, text: h.text, })), diff --git a/common/lib/search/window.ts b/common/lib/search/window.ts @@ -110,7 +110,9 @@ export function windowedTranscript( // — on the cue it starts in — and the emitted text widens to the window only // in that case. Everything else emits the matched cue verbatim. -export type Hit = { start: number; text: string }; +// `track`: a transcript hit from one of the record's ALTERNATE English tracks +// (lib/captionTracks.ts, hitsAcrossTracks). Absent for a hit in the primary. +export type Hit = { start: number; text: string; track?: string }; export type SubsHit = { track: string; start: number; text: string }; diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts @@ -1,5 +1,6 @@ import path from "node:path"; import type { Cue } from "./vtt"; +import type { TrackFields } from "./captionTracks"; import { readFile } from "fs-extra"; import { getPaths } from "./paths"; @@ -69,9 +70,12 @@ export type DisplaySummary = { curatedTags?: string[]; }; +// `track` / `altTracks` (lib/captionTracks.ts): the primary's track id and the +// other English tracks whose words differ from it — present only on a record +// that has such a track. export type TranscriptDetail = TranscriptSummary & { cues: Cue[] | undefined; -}; +} & TrackFields; export type TranscriptPage = TranscriptDetail[]; diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -82,9 +82,9 @@ export type SubTrack = { ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3"; }; -// Treat any transcript.<x>.<y> file as a sub track when it isn't one of the -// primary transcript outputs (transcript.en.vtt, transcript.json) or a -// derived/auxiliary file (transcript.cues.json). Live chat lands as +// Treat any transcript.<x>.<y> file as a sub track when it isn't a transcript +// (transcript.json, or ANY English VTT — the primary and its alternate tracks, +// lib/captionTracks.ts) or a derived/auxiliary file (transcript.cues.json). Live chat lands as // transcript.live_chat.json; non-en languages as transcript.<lang>.vtt. // Exported for lib/sidecar-server.ts, which refuses to declare a sidecar this // matches (a sidecar so named would be read as a subtitle track). @@ -274,15 +274,14 @@ export async function readVideoFiles( export async function readSubTracks(videoDir: string): Promise<SubTrack[]> { const entries = await readdir(videoDir).catch(() => [] as string[]); - const primaryVtt = resolvePrimaryVtt(entries); const tracks: SubTrack[] = []; for (const entry of entries) { if (entry === WHISPER_FILENAME) continue; - // The resolved primary English VTT (transcript.en-orig.vtt, or e.g. - // transcript.en-US.vtt when there is nothing better) is the main - // transcript, not an alternate sub-track. Any other English track — the - // served transcript.en.vtt beside an en-orig — is an alternate. - if (entry === primaryVtt) continue; + // Every English VTT is a CAPTION track, not a sub track: the primary is + // the transcript, and the others are its alternate tracks, kept only + // where their words differ (lib/captionTracks.ts) — not shipped again, + // identical or not, as subtitles. + if (isEnglishVtt(entry)) continue; if (entry === CUES_JSON_FILENAME) continue; if (entry === LIVE_CHAT_CUES_FILENAME) continue; const m = entry.match(SUB_FILE_RE); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,9 +1,12 @@ # Changelog ## [Unreleased] +- **A video's other English tracks are readable and searchable where their words differ.** Uploaded captions are not always a transcript of what was said, so the tracks beside the transcript stay: the served `en` beside `en-orig`, a regional or auto-translated track, and the captions a local transcription replaced. One is kept where its words differ from the transcript's and from every track kept before it; identical tracks, most of them, add nothing. The index keeps them in an `alts` sub-DB and writes `track` and `altTracks` onto the transcript record only then, so every other record's page is what it was. A search hit in a word only an alternate holds names the track; one every track says is found once, in the transcript. The video page's **Transcript** card reads the transcript and switches tracks ("Track: original audio captions ▾"); switching changes nothing on disk, and **Set as transcript** stays the way the transcript itself changes. English VTTs are no longer shipped as subtitle tracks. One notion of a track — ids, plain labels, which are kept, how a hit across them is found — lives in `common/lib/captionTracks.ts`. +- **The next index build reads the alternate tracks once.** Every record that can hold one — two or more English VTTs, or a transcription beside captions — is re-read from disk, and nothing else; the log says `Alternate tracks v1: N record(s) re-read.` and how many hold a track whose words differ. The version is recorded only when no channel is held. A transcribed video's captions now count toward its change time, so a later caption fetch reaches the index. +- **The MCP reads every English track.** `search_transcripts` and the query-tree tools match a record's alternate tracks and tag a snippet from one (`[in uploaded captions 1:30]`); `get_transcript` names a video's tracks in its header and reads another with `track`; `get_transcripts` windows a match only an alternate holds, under its name; `get_video_metadata` lists the other tracks without their cues. The sweep plan says what such a hit is before it is quoted. - **A Wayback Machine capture is a copy, and says of what.** A capture URL (`web.archive.org/web/<timestamp>[id_|im_|…]/<original>`) names its record by what it is a capture of: an archived YouTube page by its YouTube id (no longer `watch`), a JW Player file by its media id (no longer `<id>-<rendition>.mp4`). Every download of a capture writes `wayback.json` (the original URL, the capture's timestamp, the capture page and its raw bytes); the video page says "Archived copy (Wayback Machine, <date>) of <original>"; a citation links the original, marked as possibly gone, and the Wayback copy, and its moment link is the capture, which plays (a capture URL never takes a time param). An existing record whose page is a capture is renamed to its id by the next snapshot. - **`archilyzer wayback refresh <slug> [--titles <file>] [--dry-run]`** brings a channel's Wayback copies up to that offline: `wayback.json`, the dir renamed through the snapshot's own reconcile pass with its roster entry moved, and with `--titles` (`id → {title, upload_date}`) the title and date of a raw file that has none, recorded in the metadata history as `wayback-provenance`. A record a live job holds is skipped and named; a second run changes nothing. -- **A video's captions are read from its original-audio track first, and a track with no text never hides one that has it.** Where YouTube serves both, `transcript.en-orig.vtt` (the captions of the original audio) is read before `transcript.en.vtt`, whose text can be a rewrite of what was said; then regional tracks (`en-US`, `en-GB`, …), then auto-translated `en-en-*` ones. The transcript is the first track in that order that has cues. One rule (`englishVttsByPreference` / `readEnglishVttCues` in `common/lib/videoStatus.ts`) serves the index, normalize and report compose, so search, the export, the MCP and report videos read the same words. **Set as transcript** on a video's page copies the chosen track to `transcript.en.vtt` and pins it there with `transcript-pin.json`, which ranks it first; deleting `transcript-pin.json` returns the video to the automatic pick. The served `en` track is listed as an alternate subtitle track where `en-orig` is the transcript. Videos with a human-made `en` track are not counted as auto-captions-only, so the replace-auto-captions lane still leaves them alone. +- **A video's captions are read from its original-audio track first, and a track with no text never hides one that has it.** Where YouTube serves both, `transcript.en-orig.vtt` (the captions of the original audio) is read before `transcript.en.vtt`, whose text can be a rewrite of what was said; then regional tracks (`en-US`, `en-GB`, …), then auto-translated `en-en-*` ones. The transcript is the first track in that order that has cues. One rule (`englishVttsByPreference` / `readEnglishVttCues` in `common/lib/videoStatus.ts`) serves the index, normalize and report compose, so search, the export, the MCP and report videos read the same words. **Set as transcript** on a video's page copies the chosen track to `transcript.en.vtt` and pins it there with `transcript-pin.json`, which ranks it first; deleting `transcript-pin.json` returns the video to the automatic pick. The served `en` track stays readable and searchable as an alternate track where its words differ (below). Videos with a human-made `en` track are not counted as auto-captions-only, so the replace-auto-captions lane still leaves them alone. - **Captions in cue blocks are read.** A VTT with no inline word timing (uploaded captions, and the `en` track YouTube serves for some livestream recordings: two lines a cue, `&nbsp;` at each line end) is read cue by cue; it used to parse to no cues, which left those videos with no text in the index. - **The next index build re-reads the caption records the new rule reaches, once.** A record with more than one English track, or with no cues stored, is re-read from disk; nothing else is. The build log says `Caption track v1: N record(s) re-read.` and, per channel, how many now read different text and how many had none and now do. The version is recorded only when no channel is held. A `transcript.cues.json` written before the rule is stale only where the rule reads something else — its `transcript.en.vtt` and `transcript.en-orig.vtt` differ, or it holds no cues and a cue-block parse or the next track may have them — so a **Normalize** run rewrites exactly those, and the digest and attribution lanes hold them until it does; a cues file now records the track it was read from (`vttFile`) and the rule (`captionTrackRule`). umtool's report-to-video refuses such a stale local record by name rather than cut from it. - **A cited moment at the very end of a recording prepares.** Prepare evidence media cuts a clip whose padding runs past the recording's end at the end (the recording's duration from its metadata), where it found no media for the padded span; a span that starts past the end is still refused. report-to-video keeps its strict rule. diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -19,6 +19,7 @@ import type { AvailabilityHistoryEntry } from "yt-dlp-transcript-common/lib/avai import type { SubtitleProvenance } from "yt-dlp-transcript-common/lib/subtitleProvenance"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { PipelineStageCard } from "../../../components/PipelineStageCard"; +import { TranscriptTracksReader } from "./cards/TranscriptTracksReader"; import { VideoNavStrip } from "./VideoNavStrip"; import { PipelineStatusStrip } from "./PipelineStatusStrip"; import { AvailabilityHistoryList } from "./cards/AvailabilityHistoryList"; @@ -322,6 +323,22 @@ export function VideoPanel({ <SubtitleDeferralLine slug={slug} videoId={videoId} /> + {hasTranscript && ( + <PipelineStageCard + id="transcript-read" + title="Transcript" + summary={ + vttTracks.length + (hasTranscriptJson ? 1 : 0) > 1 + ? "Read it; switch to another track where one says something else." + : "Read it." + } + defaultOpen={false} + tone="neutral" + > + <TranscriptTracksReader slug={slug} videoId={videoId} /> + </PipelineStageCard> + )} + {vttTracks.length > 0 && ( <PipelineStageCard id="transcript-source" diff --git a/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx b/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx @@ -0,0 +1,100 @@ +"use client"; + +// The video's transcript, read on demand, with the small track switcher when it +// has more than one English track whose words differ (lib/captionTracks.ts). +// Switching is a reader's choice and changes nothing on disk — "Set as +// transcript" in the Transcript source card is how the primary changes. + +import { useState, useTransition } from "react"; +import type { AltTrack } from "yt-dlp-transcript-common/lib/captionTracks"; +import { trackLabels } from "yt-dlp-transcript-common/lib/captionTracks"; +import { formatTimestamp } from "yt-dlp-transcript-common/lib/vtt"; +import { readTranscriptTracksAction } from "../../videoActions"; + +export function TranscriptTracksReader({ + slug, + videoId, +}: { + slug: string; + videoId: string; +}) { + const [pending, startTransition] = useTransition(); + const [tracks, setTracks] = useState<AltTrack[] | null>(null); + const [shown, setShown] = useState<string | null>(null); + const [error, setError] = useState<string | null>(null); + + const load = () => { + setError(null); + startTransition(async () => { + const res = await readTranscriptTracksAction(slug, videoId); + if (!res.ok) { + setError(res.error); + return; + } + setTracks(res.tracks); + setShown(res.tracks[0]?.track ?? null); + }); + }; + + if (!tracks) { + return ( + <div className="flex flex-col gap-2"> + <button + type="button" + onClick={load} + disabled={pending} + aria-label="read transcript" + className="self-start px-2 py-1 rounded border border-border text-xs hover:bg-muted disabled:opacity-50" + > + {pending ? "Reading…" : "Read transcript"} + </button> + {error && ( + <span className="text-sm text-destructive" aria-label="read transcript error"> + {error} + </span> + )} + </div> + ); + } + + const current = tracks.find((t) => t.track === shown) ?? tracks[0]; + return ( + <div className="flex flex-col gap-2"> + {tracks.length > 1 && ( + <label className="inline-flex items-center gap-1.5 self-start text-xs text-muted-foreground"> + <span>Track:</span> + <select + aria-label="transcript track" + value={current.track} + onChange={(e) => setShown(e.target.value)} + className="rounded border border-border bg-background px-1 py-0.5 text-xs text-foreground" + > + {trackLabels(tracks.map((t) => t.track)).map((label, i) => ( + <option key={tracks[i].track} value={tracks[i].track}> + {label} + {i === 0 ? " (default)" : ""} + </option> + ))} + </select> + </label> + )} + {current.cues.length === 0 ? ( + <p className="text-sm text-muted-foreground">No cues in this track.</p> + ) : ( + <ol + aria-label="transcript cues" + className="max-h-96 overflow-y-auto rounded border border-border divide-y divide-border text-sm" + > + {current.cues.map((c, i) => ( + <li key={i} className="flex gap-3 px-3 py-1"> + <span className="shrink-0 w-16 font-mono text-xs text-muted-foreground pt-0.5"> + {formatTimestamp(c.start)} + </span> + <span className="min-w-0">{c.text}</span> + </li> + ))} + </ol> + )} + </div> + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -47,6 +47,8 @@ import { onDrive } from "yt-dlp-transcript-common/lib/storageHealth"; import { isTierable } from "yt-dlp-transcript-common/lib/mediaTier"; import { setExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server"; import { pinTranscript } from "yt-dlp-transcript-common/lib/transcriptPin-server"; +import { readVideoTracks } from "yt-dlp-transcript-common/lib/captionTracks-server"; +import type { AltTrack } from "yt-dlp-transcript-common/lib/captionTracks"; import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode"; import { @@ -623,6 +625,20 @@ export async function setPrimaryTranscriptAction( return { ok: true }; } +// The video's transcript tracks, primary first (lib/captionTracks.ts): the +// primary and every other English track whose words differ from it. Read-only — +// what the page's transcript reader shows and switches between. Choosing a +// track there changes nothing on disk; "Set as transcript" above is how the +// primary changes. +export async function readTranscriptTracksAction( + slug: string, + videoId: string, +): Promise<{ ok: true; tracks: AltTrack[] } | { ok: false; error: string }> { + const read = await readVideoTracks(videoDirOf(slug, videoId)); + if (!read) return { ok: false, error: "This video has no transcript to read." }; + return { ok: true, tracks: read.tracks }; +} + // A refusal carries what was submitted (lib/formState.ts). export type DeleteDirActionResult = FormErrorState; diff --git a/editor/e2e/transcript-source.spec.ts b/editor/e2e/transcript-source.spec.ts @@ -72,6 +72,40 @@ test("switching the transcript source promotes a track to transcript.en.vtt", as ).toBeVisible(); }); +// The page's transcript reader shows the primary and switches to another +// English track only where its words differ — a reader's choice, nothing on +// disk changes. +test("the transcript reader switches to a track whose words differ", async ({ + page, +}) => { + await resetData("one-youtube-channel-with-data"); + await writeFile( + resolvePath(`${DATA}/transcript.en-orig.vtt`), + "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nwords as spoken\n", + ); + await writeFile( + resolvePath(`${DATA}/transcript.en.vtt`), + "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nwords as uploaded\n", + ); + + await page.goto(VIDEO_URL); + await page.getByLabel("Transcript stage summary", { exact: true }).click(); + await page.getByRole("button", { name: "read transcript" }).click(); + + const cuesList = page.getByLabel("transcript cues", { exact: true }); + await expect(cuesList).toContainText("words as spoken"); + const switcher = page.getByLabel("transcript track", { exact: true }); + await expect(switcher).toHaveValue("en-orig"); + await expect(switcher.locator("option")).toHaveText([ + "original audio captions (default)", + "uploaded captions", + ]); + await switcher.selectOption("en"); + await expect(cuesList).toContainText("words as uploaded"); + // Nothing was pinned. + expect(await pathExists(`${DATA}/transcript-pin.json`)).toBe(false); +}); + // The channel diagnostics surface a "non-standard transcript VTT name" bucket // for videos whose transcript rides on a non-canonical VTT name. test("diagnostics list a video with only transcript.en-US.vtt", async ({ diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Search reads every English track of a video, and the transcript switches tracks.** Where a video has another English caption track whose words differ from its transcript — the uploaded captions beside the original audio's, a regional or auto-translated track — a query matches it too: a hit only that track holds says so ("in uploaded captions") and opens the transcript on that track at that moment, and a word both say is found once, in the transcript. The transcript reader shows a small "Track:" switcher beside the mode buttons on such a video; the transcript stays the default, and the choice rides on the share link (`vt`). Downloads and Copy MD take the track on show. Needs an index build and a rebuild and deploy of each site. - **A citation of a Wayback Machine copy links its original and the copy.** A cited record downloaded from a Wayback capture shows "Original (may be gone)", the original at the cited second where its platform takes one, and "Wayback Machine copy, <capture date>", the capture page, which plays. Its moment link is the capture: a capture URL never takes a time param. - **Transcripts read the original-audio captions.** Where a video has both, its transcript is YouTube's `en-orig` track (the captions of what was said) rather than the served `en`, which can reword it; a track with no text falls through to the next. Videos whose only captions are in cue blocks (some livestream recordings) have their text. - **A report shows its revision, and every edit to it can be checked.** A report's date line ends with "revision N", linking to its history ("edited since revision N" when the report has changed since). The history page, `/reports/<id>/history/`, lists every revision, newest first: its number, date (UTC), commit hash and the sha256 of its `report.json`, what changed (claims added or removed, verdicts changed, claims edited, citations added or removed, quotes edited, title, series or subtitle changed), and each changed claim's title, text, verdict and findings with the words removed struck through and the words added marked. The same data is in `history.json` beside the page. Each report's history is its own git repository, published for cloning: `git clone <site>/reports/<id>/history/repo`. Each commit names the site as its author, with its date in UTC. The footer of the report's HTML, PDF and Markdown downloads begins with the revision number, and its sha256 can be looked up on the history page. Needs `reports export` and a rebuild and deploy of each site with reports. diff --git a/export/app/components/OfflineManager.tsx b/export/app/components/OfflineManager.tsx @@ -15,6 +15,7 @@ import { workerSupported, type IndexHit, } from "yt-dlp-transcript-common/components/searchIndexWorkerClient"; +import { inTrackLabel } from "yt-dlp-transcript-common/lib/captionTracks"; export type OfflineChannel = { slug: string; name: string }; @@ -336,6 +337,7 @@ function OfflineSearch({ {typeof hit.start === "number" ? ` · ${formatTime(hit.start)}` : ""} + {hit.track ? ` · ${inTrackLabel(hit.track)}` : ""} </span> </li> ))} diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts @@ -292,6 +292,16 @@ export function transcriptPage() { description: "Filmed on location with a zebra in the background.", tags: ["news"], cues: transcriptCues("transcript-only video"), + // An ALTERNATE English track (lib/captionTracks.ts): the uploaded + // captions, whose words differ from the primary's. "zeppelin" is said + // only here, at 200 s — what transcript-tracks.spec.ts searches for. + track: "en-orig", + altTracks: [ + { + track: "en", + cues: [{ start: 200, end: 204, text: "uploaded words about a zeppelin" }], + }, + ], }, { ...makeSummary(VIDEO_CHAT_SMALL, "Small live chat"), diff --git a/export/e2e/transcript-tracks.spec.ts b/export/e2e/transcript-tracks.spec.ts @@ -0,0 +1,72 @@ +import { expect, test, type Page } from "@playwright/test"; +import { CHANNEL_SLUG, VIDEO_TRANSCRIPT_ONLY } from "./fixtures/data"; +import { expectModalOpen, installRoutes } from "./helpers"; + +// A record's other English tracks (lib/captionTracks.ts). VIDEO_TRANSCRIPT_ONLY +// carries an en-orig primary and an uploaded `en` that says "zeppelin" at +// 200 s, which its primary never does; the other fixture videos have no +// alternate. + +test.use({ + permissions: ["clipboard-read", "clipboard-write"], +}); + +const SLUG = `${CHANNEL_SLUG}/${VIDEO_TRANSCRIPT_ONLY}`; +const leafInput = (page: Page) => + page.locator('input[data-testid^="leaf-query-"]').first(); + +test.describe("transcript tracks", () => { + test.beforeEach(async ({ page }) => { + await installRoutes(page); + }); + + test("the reader shows the primary and switches to the uploaded captions", async ({ page }) => { + await page.goto(`/?v=${SLUG}`); + await expectModalOpen(page); + const cues = page.getByTestId("cue-list"); + await expect(cues).toContainText("transcript-only video — alpha line"); + + const switcher = page.getByTestId("track-switcher"); + await expect(switcher).toHaveValue("en-orig"); + await expect(switcher.locator("option")).toHaveText([ + "original audio captions (default)", + "uploaded captions", + ]); + await switcher.selectOption("en"); + await expect(cues).toContainText("uploaded words about a zeppelin"); + await expect(cues).not.toContainText("alpha line"); + expect(new URL(page.url()).searchParams.get("vt")).toBe("en"); + + // The share link reopens this track; back on the primary, it is off the URL. + await page.getByRole("button", { name: "Copy share link at current time" }).click(); + const clip = new URL(await page.evaluate(() => navigator.clipboard.readText())); + expect(clip.searchParams.get("vt")).toBe("en"); + await switcher.selectOption("en-orig"); + await expect(cues).toContainText("alpha line"); + expect(new URL(page.url()).searchParams.get("vt")).toBeNull(); + }); + + test("a record with no alternate shows no switcher", async ({ page }) => { + await page.goto(`/?v=${CHANNEL_SLUG}/vid-chat-small`); + await expectModalOpen(page); + await expect(page.getByTestId("cue-list")).toContainText("small chat video"); + await expect(page.getByTestId("track-switcher")).toHaveCount(0); + }); + + test("search finds a word only the uploaded captions hold, says so, and opens that track", async ({ page }) => { + await page.goto("/"); + await leafInput(page).fill("zeppelin"); + await page.getByTestId("search-submit").click(); + + const card = page.locator(`[data-result-slug="${SLUG}"]`); + await expect(card).toBeVisible({ timeout: 15_000 }); + await expect(page.locator("[data-card-header]")).toHaveCount(1); + const hit = page.getByRole("button", { name: /in uploaded captions/i }).first(); + await expect(hit).toContainText("zeppelin"); + await hit.click(); + + await expectModalOpen(page); + await expect(page.getByTestId("track-switcher")).toHaveValue("en"); + await expect(page.getByTestId("cue-list")).toContainText("uploaded words about a zeppelin"); + }); +}); diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts @@ -232,6 +232,16 @@ export function buildSweepInstructions( ); steps.push( + `**A hit can come from another caption track.** Where a video has more ` + + `than one English track whose words differ, search reads them all; a ` + + `snippet tagged \`in uploaded captions\` (or another track) matched ` + + `words the primary transcript — the original audio's captions — does ` + + `not have there. Uploaded captions are not always what was said: read ` + + `the primary around that moment (\`get_transcript\`; \`track\` reads ` + + `the other one) before quoting, and say which track the words are from.`, + ); + + steps.push( `**State the plan.** Report N (the enumerated total) and ` + `\`ceil(N / ${req.batchSize})\` batches before you start. The report may ` + `only ever claim the coverage this number justifies: N videos ` + diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts @@ -237,7 +237,7 @@ class StubSource implements ShardSource { ]; } - private pages(ch: ChannelRef): TranscriptDetail[][] { + protected pages(ch: ChannelRef): TranscriptDetail[][] { // One record per page so paging exercises multiple shard pages. const recs = ch.slug === "chan-a" ? CHAN_A : CHAN_B; return recs.map((r) => [r]); @@ -1685,3 +1685,80 @@ test("server: the unadvertised channel/group singulars are still parsed", async assert.match(out, /scope: 1 channel/); await client.close(); }); + +// ─── Alternate tracks (lib/captionTracks.ts) ─── + +// chan-b plus b2: an en-orig primary and an uploaded `en` that says a word the +// primary never does. +const ALT_REC = vid( + "b2", + "Two tracks", + "chan-b", + cues([5, "the harbor bridge opened"]), + { + track: "en-orig", + altTracks: [{ track: "en", cues: cues([6, "the harbor bridge opened"], [90, "a zeppelin flew over"]) }], + }, +); +class AltTrackSource extends StubSource { + protected override pages(ch: ChannelRef): TranscriptDetail[][] { + const base = super.pages(ch); + return ch.slug === "chan-b" ? [...base, [ALT_REC]] : base; + } +} + +test("searchTranscripts: a word only an alternate track holds is found there, the track named", async () => { + const src = new AltTrackSource(); + const res = await searchTranscripts(src, { query: "zeppelin" }); + assert.equal(res.hits.length, 1); + assert.equal(res.hits[0].videoId, "b2"); + assert.deepEqual( + res.hits[0].snippets.map((s) => [s.seconds, s.track]), + [[90, "en"]], + ); + // Said by both tracks at the same moment: once, from the primary. + const both = await searchTranscripts(src, { query: "harbor bridge" }); + assert.deepEqual(both.hits[0].snippets.map((s) => s.track), [undefined]); +}); + +test("server: a hit from an alternate track says which one", async () => { + const client = await connectClient(new AltTrackSource()); + const out = firstText( + await client.callTool({ name: "search_transcripts", arguments: { query: "zeppelin" } }), + ); + assert.match(out, /\[in uploaded captions \[1:30\]\(https:\/\/example.test\/b2\?t=90s\)\] a zeppelin flew over/); + await client.close(); +}); + +test("server: get_transcript reads the primary by default and another track by `track`", async () => { + const client = await connectClient(new AltTrackSource()); + const primary = firstText( + await client.callTool({ name: "get_transcript", arguments: { video_id: "b2" } }), + ); + assert.doesNotMatch(primary, /zeppelin/); + assert.match(primary, /track: en-orig \(original audio captions\) — the primary/); + assert.match(primary, /other tracks: en \(uploaded captions\) — pass track to read one/); + + const en = firstText( + await client.callTool({ name: "get_transcript", arguments: { video_id: "b2", track: "en" } }), + ); + assert.match(en, /a zeppelin flew over/); + assert.match(en, /track: en \(uploaded captions\)/); + + const bad = await client.callTool({ + name: "get_transcript", + arguments: { video_id: "b2", track: "en-GB" }, + }); + assert.match(firstText(bad), /has no track "en-GB"; its tracks are: en-orig \(original audio captions\), en \(uploaded captions\)/); + + // get_transcripts with a query windows a match only the alternate holds. + const batch = firstText( + await client.callTool({ + name: "get_transcripts", + arguments: { video_ids: ["b2"], query: "zeppelin" }, + }), + ); + assert.match(batch, /1 matching line\(s\) only in uploaded captions \(track en\), windowed/); + assert.match(batch, /a zeppelin flew over/); + await client.close(); +}); diff --git a/mcp/src/search.ts b/mcp/src/search.ts @@ -10,6 +10,7 @@ import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; import { postConversation, type Post } from "yt-dlp-transcript-common/lib/posts"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; +import { hitsAcrossTracks } from "yt-dlp-transcript-common/lib/captionTracks"; import type { Platform } from "yt-dlp-transcript-common/lib/platform"; import { type GroupNode, @@ -65,6 +66,10 @@ export type Snippet = { // reader has to be able to tell "the word appears in the description" from // "the word was said at 0:00" — they license completely different citations. scope?: LayerScope; + // A transcript hit from one of the record's ALTERNATE English tracks + // (lib/captionTracks.ts) — words its primary does not have there. Absent for + // a hit in the primary. + track?: string; }; // A scope selector for a search/sweep: any mix of channel handles (slug / key / @@ -614,13 +619,17 @@ export async function searchTranscripts( let otherHit = false; if (wantCues) { - for (const cue of rec.cues ?? []) { - if (!match(cue.text)) continue; + // Every English track of the record (lib/captionTracks.ts): a + // match only an alternate holds names that alternate. + for (const cue of hitsAcrossTracks(rec, (cues) => + cues.filter((c) => match(c.text)), + )) { matches++; push({ clock: clock(cue.start), seconds: cue.start, text: truncate(cue.text, policy.snippetChars), + ...(cue.track ? { track: cue.track } : {}), }); } } @@ -1124,6 +1133,7 @@ export async function runSearchSpec( description: rec.description ?? "", tags: (rec.tags ?? []).join(", "), cues: rec.cues ?? [], + ...(rec.altTracks ? { altTracks: rec.altTracks } : {}), chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [], snippetsPerVideo, includeSnippets, diff --git a/mcp/src/server.ts b/mcp/src/server.ts @@ -15,6 +15,14 @@ import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { isTagId, type PublishedTag } from "yt-dlp-transcript-common/lib/curatedTags"; import { groupPublishedTags } from "yt-dlp-transcript-common/lib/publishedTags"; import { + cuesOfTrack, + hitsAcrossTracks, + inTrackLabel, + recordTracks, + trackLabel, +} from "yt-dlp-transcript-common/lib/captionTracks"; +import { windowedTranscript } from "yt-dlp-transcript-common/lib/search/window"; +import { VIDEO_STATES, isVideoState, type VideoState, @@ -44,6 +52,7 @@ import { findVideo, buildMatcher, getWindowedTranscript, + MCP_POLICY, runSearchSpec, type SearchFilters, type SearchResult, @@ -582,6 +591,14 @@ export const TOOLS: Tool[] = [ type: "boolean", description: "Prefix each caption line with a timestamp (default true).", }, + track: { + type: "string", + description: + "Optional: read one of the video's other English tracks instead of its " + + "primary transcript (e.g. \"en\" for the uploaded captions beside the " + + "original-audio \"en-orig\"). The header lists the tracks a video has; " + + "only tracks whose words differ from the primary are kept. Omit for the primary.", + }, }, required: ["video_id"], additionalProperties: false, @@ -1635,7 +1652,11 @@ async function handleSearch( const stamp = base ? baseStamp(s.clock, s.seconds) : stampMarkup(source, h, s.clock, s.seconds); - const tag = s.scope && s.scope !== "transcripts" ? `${s.scope} ` : ""; + const tag = s.scope && s.scope !== "transcripts" + ? `${s.scope} ` + : s.track + ? `${inTrackLabel(s.track)} ` + : ""; return ` - [${tag}${stamp}] ${s.text}`; }) .join("\n"); @@ -2207,12 +2228,37 @@ async function handleGetTranscripts( counts.length > 1 ? counts.map((c) => `"${c.query}": ${c.n}`).join(", ") : `${matchCount} matching line(s)`; + // The record's alternate tracks (lib/captionTracks.ts): a match only + // an alternate holds is windowed from that track, under its name. + const altBlocks: string[] = []; + let altMatches = 0; + const across = hitsAcrossTracks(record, (list) => + list.filter((c) => matcher.match(c.text)), + ); + for (const alt of record.altTracks ?? []) { + const own = new Set(across.filter((h) => h.track === alt.track).map((h) => h.text)); + if (own.size === 0) continue; + const w = windowedTranscript(alt.cues, (t) => own.has(t), { + before, + after, + timestamps, + stamp, + maxLines: maxLines ?? MCP_POLICY.windowLineCap, + }); + altMatches += w.matchCount; + altBlocks.push( + `_(${w.matchCount} matching line(s) only ${inTrackLabel(alt.track)} (track ${alt.track}), windowed)_\n${w.lines.join("\n")}`, + ); + } const body = - matchCount === 0 + matchCount === 0 && altMatches === 0 ? `_(no lines matched ${ counts.length > 1 ? "any query" : "the query" } in this transcript${counts.length > 1 ? ` — ${countNote}` : ""})_` - : `_(${countNote}, windowed)_\n${lines.join("\n")}`; + : [ + ...(matchCount > 0 ? [`_(${countNote}, windowed)_\n${lines.join("\n")}`] : []), + ...altBlocks, + ].join("\n\n"); blocks.push(`${head}\n\n${body}`); } else { const md = transcriptToMarkdown( @@ -2377,11 +2423,38 @@ async function handleGetTranscript( ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), ...(record.platform ? { platform: record.platform } : {}), }; - const md = transcriptToMarkdown(record, { - timestamps: args.timestamps !== false, - includeTags: true, - linkForCue: (seconds) => momentLinkFor(source, link, seconds), - }); + // The record's tracks (lib/captionTracks.ts): the primary unless `track` + // names an alternate it holds. + const tracks = recordTracks(record); + const asked = typeof args.track === "string" ? args.track.trim() : ""; + const cues = cuesOfTrack(record, asked); + if (asked && cues === undefined) { + return errorText( + tracks.length > 1 + ? `video ${videoId} has no track "${asked}"; its tracks are: ${tracks.map((t) => `${t} (${trackLabel(t)})`).join(", ")}` + : `video ${videoId} has no track "${asked}": it has only its primary transcript`, + ); + } + const shown = asked || record.track; + const extraMeta = + tracks.length > 1 && shown + ? [ + `track: ${shown} (${trackLabel(shown)})${shown === record.track ? " — the primary" : ""}`, + `other tracks: ${tracks + .filter((t) => t !== shown) + .map((t) => `${t} (${trackLabel(t)}${t === record.track ? ", the primary" : ""})`) + .join(", ")} — pass track to read one`, + ] + : undefined; + const md = transcriptToMarkdown( + { ...record, cues }, + { + timestamps: args.timestamps !== false, + includeTags: true, + linkForCue: (seconds) => momentLinkFor(source, link, seconds), + ...(extraMeta ? { extraMeta } : {}), + }, + ); return text(md); } @@ -2405,12 +2478,23 @@ async function handleGetMetadata( typeof args.channel === "string" ? args.channel : undefined, ); if (!found) return errorText(`video not found: ${videoId}`); - const { cues, ...meta } = found.record; + const { cues, altTracks, ...meta } = found.record; const slug = found.record.slug; + // An alternate's cues are not metadata; its id and label are. + const otherTracks = (altTracks ?? []).map((t) => ({ + track: t.track, + label: trackLabel(t.track), + cueCount: t.cues.length, + })); const lines: string[] = [ JSON.stringify( - { ...meta, channelName: found.ch.name, cueCount: cues?.length ?? 0 }, + { + ...meta, + channelName: found.ch.name, + cueCount: cues?.length ?? 0, + ...(otherTracks.length > 0 ? { otherTracks } : {}), + }, null, 2, ), @@ -2912,7 +2996,12 @@ function renderScopedSnippet( s: ScopedSnippet, ): string { if (s.seconds > 0) { - const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `; + const tag = + s.scope === "transcripts" + ? s.track + ? `${inTrackLabel(s.track)} ` + : "" + : `${s.track ?? s.scope} `; return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`; } return ` - [${s.scope}] ${s.text}`;