Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 3cb53fc773c769dcda00e05b3b2b4492cfd3a657
parent a9c364f637e3df63fa66dfab46f48edb49c214cb
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 11:21:04 -0400

tracks: labels for tracks YouTube named, and en-<region>-orig

The corpus holds caption tracks YouTube names itself (en-<id>, an uploader's
extra track) and en-US-orig beside the plain ones. A region subtag is two
letters or three digits; anything else after `en-` is "other uploaded
captions", en-<region>-orig is "original audio captions (US)", and a switcher
whose two tracks would read the same adds the id (trackLabels).

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/TranscriptModal.tsx | 8++++----
Mcommon/lib/captionTracks.test.ts | 12+++++++++++-
Mcommon/lib/captionTracks.ts | 37++++++++++++++++++++++++++++++-------
Meditor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx | 8++++----
4 files changed, 49 insertions(+), 16 deletions(-)

diff --git a/common/components/TranscriptModal.tsx b/common/components/TranscriptModal.tsx @@ -36,7 +36,7 @@ import { AgeRestrictedBadge, LivestreamBadge } from "./badges"; import { VirtualRow } from "./VirtualRow"; import { formatTimestamp } from "../lib/vtt"; import { formatDate } from "../lib/format"; -import { trackLabel } from "../lib/captionTracks"; +import { trackLabels } from "../lib/captionTracks"; type DisplayCue = { start: number; @@ -831,9 +831,9 @@ function TrackSwitcher({ onChange={(e) => onChange(e.target.value)} className="bg-transparent text-zinc-200 rounded border border-zinc-700 px-1 py-0.5 text-xs focus:outline-none focus:ring-1 focus:ring-zinc-500" > - {tracks.map((t, i) => ( - <option key={t} value={t} className="bg-zinc-900"> - {trackLabel(t)} + {trackLabels(tracks).map((label, i) => ( + <option key={tracks[i]} value={tracks[i]} className="bg-zinc-900"> + {label} {i === 0 ? " (default)" : ""} </option> ))} diff --git a/common/lib/captionTracks.test.ts b/common/lib/captionTracks.test.ts @@ -17,6 +17,7 @@ import { recordTracks, trackKind, trackLabel, + trackLabels, uncoveredAltHits, } from "./captionTracks"; import { readTrackFields, readVideoTracks } from "./captionTracks-server"; @@ -48,7 +49,16 @@ test("labels are plain words derived from the track id", () => { assert.equal(trackLabel("en"), "uploaded captions"); assert.equal(trackLabel("en-en-US"), "auto-translated captions"); assert.equal(trackLabel("en-GB"), "UK English captions"); - assert.equal(trackLabel("en-x-foo"), "regional captions (en-x-foo)"); + assert.equal(trackLabel("en-NG"), "regional captions (en-NG)"); + assert.equal(trackLabel("en-US-orig"), "original audio captions (US)"); + // A track YouTube named itself: another uploaded one. + assert.equal(trackLabel("en-uYU-mmqFLq8"), "other uploaded captions"); + assert.equal(trackKind("en-JkeT_87f4cc"), "uploaded"); + assert.deepEqual(trackLabels(["en-orig", "en-aa-1", "en-bb-2"]), [ + "original audio captions", + "other uploaded captions (en-aa-1)", + "other uploaded captions (en-bb-2)", + ]); assert.equal(trackLabel("transcription"), "transcription"); assert.equal(trackLabel("pinned"), "chosen captions"); assert.equal(inTrackLabel("en"), "in uploaded captions"); diff --git a/common/lib/captionTracks.ts b/common/lib/captionTracks.ts @@ -62,24 +62,39 @@ export type TrackKind = | "pinned" | "other"; +// A region subtag as YouTube writes one: two letters (en-US) or three digits +// (en-419). Anything else after `en-` is a track YouTube named itself — an +// uploader's extra caption track (en-uYU-mmqFLq8). +const REGION_RE = /^(?:[A-Za-z]{2}|\d{3})$/; + export function trackKind(track: string): TrackKind { if (track === TRANSCRIPTION_TRACK) return "transcription"; if (track === PINNED_TRACK) return "pinned"; - if (track === "en-orig") return "original"; + if (track === "en-orig" || /^en-[^-]+-orig$/.test(track)) return "original"; if (track === "en") return "uploaded"; if (/^en-en(?:-|$)/.test(track)) return "translated"; - if (/^en-/.test(track)) return "regional"; + const m = track.match(/^en-(.+)$/); + if (m) return REGION_RE.test(m[1]) ? "regional" : "uploaded"; return "other"; } +// "UK" for en-GB, the code itself for a region this table does not name. +function regionName(code: string): string { + return REGIONS[code.toUpperCase()] ?? code; +} + // A plain label for a track — what a person reads in the switcher and on a // search hit ("in uploaded captions"). export function trackLabel(track: string): string { switch (trackKind(track)) { - case "original": - return "original audio captions"; + case "original": { + // en-US-orig: the original audio's captions, under a region. + const m = track.match(/^en-([^-]+)-orig$/); + return m ? `original audio captions (${regionName(m[1])})` : "original audio captions"; + } case "uploaded": - return "uploaded captions"; + // A track YouTube named (en-uYU-mmqFLq8) is another uploaded one. + return track === "en" ? "uploaded captions" : "other uploaded captions"; case "translated": return "auto-translated captions"; case "transcription": @@ -87,8 +102,7 @@ export function trackLabel(track: string): string { case "pinned": return "chosen captions"; case "regional": { - const region = track.slice(3); - const name = REGIONS[region.toUpperCase()]; + const name = REGIONS[track.slice(3).toUpperCase()]; return name ? `${name} English captions` : `regional captions (${track})`; } default: @@ -96,6 +110,15 @@ export function trackLabel(track: string): string { } } +// The labels of a switcher's tracks, in order: trackLabel, with the id added +// where two tracks would otherwise read the same (two tracks YouTube named). +export function trackLabels(tracks: readonly string[]): string[] { + const labels = tracks.map(trackLabel); + return labels.map((l, i) => + labels.indexOf(l) !== labels.lastIndexOf(l) ? `${l} (${tracks[i]})` : l, + ); +} + // The track id of a caption file name (transcript.<id>.vtt), or null. export function trackOfVttFile(filename: string): string | null { const m = filename.match(/^transcript\.([^.]+)\.vtt$/); diff --git a/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx b/editor/app/channels/[slug]/videos/[id]/components/cards/TranscriptTracksReader.tsx @@ -7,7 +7,7 @@ import { useState, useTransition } from "react"; import type { AltTrack } from "yt-dlp-transcript-common/lib/captionTracks"; -import { trackLabel } from "yt-dlp-transcript-common/lib/captionTracks"; +import { trackLabels } from "yt-dlp-transcript-common/lib/captionTracks"; import { formatTimestamp } from "yt-dlp-transcript-common/lib/vtt"; import { readTranscriptTracksAction } from "../../videoActions"; @@ -69,9 +69,9 @@ export function TranscriptTracksReader({ onChange={(e) => setShown(e.target.value)} className="rounded border border-border bg-background px-1 py-0.5 text-xs text-foreground" > - {tracks.map((t, i) => ( - <option key={t.track} value={t.track}> - {trackLabel(t.track)} + {trackLabels(tracks.map((t) => t.track)).map((label, i) => ( + <option key={tracks[i].track} value={tracks[i].track}> + {label} {i === 0 ? " (default)" : ""} </option> ))}