commit 21c44a807157a99a4e56de7ccb37f6b92a2ae44f
parent 0f614b3a43376d07d562101aae4c83424749ccd3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 01:11:15 -0400
common: release 20 D2 — one Twitch id: records carry the canonical <n> their directory does, not the native v<n>
canonicalTwitchVideoId (lib/videoId.ts) is the one normalization: summarize
uses it for a Twitch record, readNormalizedTranscript applies it to a
transcript.cues.json written before (normalizeSummaryId), and the index
re-processes once every record it holds under a v-id ("Twitch ids v1", the
platform-labels shape), re-keying it. The Twitch player is still handed v<n>.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 292 insertions(+), 5 deletions(-)
diff --git a/common/components/TwitchPlayer.tsx b/common/components/TwitchPlayer.tsx
@@ -47,8 +47,11 @@ const TwitchPlayer = forwardRef<TwitchPlayerHandle, Props>(function TwitchPlayer
const parent =
typeof window !== "undefined" ? window.location.hostname : "localhost";
+ // The record carries the canonical `<n>` (release 20 D2); the player is
+ // given the `v<n>` it has always been given.
+ const video = /^\d+$/.test(videoId) ? `v${videoId}` : videoId;
const src = `https://player.twitch.tv/?video=${encodeURIComponent(
- videoId,
+ video,
)}&parent=${encodeURIComponent(parent)}&time=${toTwitchTime(start)}&autoplay=true`;
return (
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -47,6 +47,7 @@ import { parseTranscriptJson } from "../lib/whisper";
import { parseLiveChat } from "../lib/liveChat";
import {
platformLabelStale,
+ summaryIdStale,
summarize,
toDisplaySummary,
type RawMetadata,
@@ -230,6 +231,13 @@ const CAPTION_TRACK_KEY = "captionTrackRule";
// pattern.
const RECORDED_DATE_RULE_KEY_PREFIX = "recordedDateRule:";
+// TWITCH IDS (release 20 D2) — the platform-labels shape once more: records
+// indexed under yt-dlp's native `v<n>` (summaryIdStale) are re-processed once,
+// so they take the canonical `<n>` their directory carries and are re-keyed;
+// then the version is recorded, only when no channel is held.
+const TWITCH_IDS_VERSION = 1;
+const TWITCH_IDS_KEY = "twitchIds";
+
// A cue list's text, for "did the words change" — timing alone is not a
// different transcript.
function cueText(list: readonly Cue[]): string {
@@ -946,6 +954,24 @@ export async function buildIndex({
// records were derived under (RECORDED_DATE_RULE_KEY_PREFIX) — every record
// of those is re-processed like a changed one. A held channel is left for
// the build that can read it.
+ const twitchIdsDue = meta.get(TWITCH_IDS_KEY) !== TWITCH_IDS_VERSION;
+ if (twitchIdsDue && !schemaBumped) {
+ const queued = new Set(
+ [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])),
+ );
+ let rekeyed = 0;
+ for (const s of live) {
+ const pk: PathKey = [s.channelSlug, s.videoDir];
+ const prev = mtimes.get(pk);
+ if (!prev) continue;
+ const sum = sums.get(prev.indexKey);
+ if (!sum || !summaryIdStale(sum)) continue;
+ rekeyed++;
+ if (!queued.has(pathKeyId(pk))) changed.push(s);
+ }
+ log(`Twitch ids v${TWITCH_IDS_VERSION}: ${rekeyed} record(s) indexed under the native v-id, re-keyed.`);
+ }
+
const recordedDateRes = new Map<string, RegExp>();
const recordedDateRuleChanged: string[] = [];
for (const [slug, cfg] of channelConfigs) {
@@ -2581,6 +2607,9 @@ export async function buildIndex({
if (altTracksDue && held.size === 0) {
await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION);
}
+ if (twitchIdsDue && held.size === 0) {
+ await meta.put(TWITCH_IDS_KEY, TWITCH_IDS_VERSION);
+ }
for (const slug of recordedDateRuleChanged) {
const rule = channelConfigs.get(slug)?.recordedDate;
const key = `${RECORDED_DATE_RULE_KEY_PREFIX}${slug}`;
diff --git a/common/controller/buildIndexTwitchIds.test.ts b/common/controller/buildIndexTwitchIds.test.ts
@@ -0,0 +1,146 @@
+// Integration: Twitch ids (release 20 D2) through the REAL buildIndex over a
+// temp corpus — a record is indexed under the canonical `<n>` its directory
+// carries, and an index that holds one under the native `v<n>` re-keys it once.
+// Synthetic ids.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildIndexTwitchIds.test.ts
+
+import { after, test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-twitch-"));
+const PINNED: Record<string, string> = {
+ TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"),
+ SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"),
+ SITES_DIR: path.join(ROOT, "transcripts", "sites"),
+ SETTINGS_FILE: path.join(ROOT, "settings.json"),
+ EXPORT_PUBLIC_DIR: path.join(ROOT, "public"),
+ EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"),
+ EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"),
+ EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"),
+ EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"),
+ CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"),
+ SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"),
+ CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"),
+ ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"),
+ ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"),
+};
+Object.assign(process.env, PINNED);
+delete process.env.ARCHILYZER_INDEX_ALLOW_HELD;
+after(() => rmSync(ROOT, { recursive: true, force: true }));
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { open } = await import("lmdb");
+
+const paths = getPaths();
+const CHANNEL = "example-twitch";
+const SITE = "testsite";
+const N = "2700000001";
+
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+
+function seed(): void {
+ writeFileSync(paths.settingsFile, "{}");
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "transcribe",
+ name: CHANNEL,
+ platform: "twitch",
+ url: "https://www.twitch.tv/example",
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+ const dir = path.join(paths.channelsDir, CHANNEL, "data", N);
+ writeJson(path.join(dir, "metadata.info.json"), {
+ id: `v${N}`,
+ title: "Example stream",
+ upload_date: "20260105",
+ duration: 30,
+ webpage_url: `https://www.twitch.tv/videos/${N}`,
+ extractor_key: "TwitchVod",
+ });
+ writeFileSync(
+ path.join(dir, "transcript.en.vtt"),
+ readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", "vtt-rolling.vtt"), "utf8"),
+ );
+}
+
+type Summary = { id: string; slug: string };
+type Key = [string, string, string];
+
+function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T {
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ try {
+ return fn((name) => root.openDB({ name, encoding: "msgpack" }));
+ } finally {
+ root.close();
+ }
+}
+
+function records(): { key: Key; id: string; slug: string; cues: number }[] {
+ return withIndex((db) =>
+ [...db("sums").getRange()].map(({ key, value }) => ({
+ key: key as Key,
+ id: (value as Summary).id,
+ slug: (value as Summary).slug,
+ cues: ((db("cues").get(key) as unknown[] | undefined) ?? []).length,
+ })),
+ );
+}
+
+async function runIndex(): Promise<string[]> {
+ const log: string[] = [];
+ await buildIndex({ paths, onLog: (s) => log.push(s) });
+ return log;
+}
+
+test("a Twitch record is indexed under its directory's id", async () => {
+ seed();
+ await runIndex();
+ assert.deepEqual(records(), [
+ { key: ["20260105", CHANNEL, N], id: N, slug: `${CHANNEL}/${N}`, cues: 3 },
+ ]);
+});
+
+test("an index holding the native v-id re-keys it once", async () => {
+ // The index as a build before canonical ids left it.
+ withIndex((db) => {
+ const key: Key = ["20260105", CHANNEL, N];
+ const old: Key = ["20260105", CHANNEL, `v${N}`];
+ const sum = db("sums").get(key) as Summary;
+ db("sums").putSync(old, { ...sum, id: `v${N}`, slug: `${CHANNEL}/v${N}` });
+ db("cues").putSync(old, db("cues").get(key));
+ db("sums").removeSync(key);
+ db("cues").removeSync(key);
+ const mtimes = db("mtimes");
+ for (const { key: pk, value } of mtimes.getRange()) {
+ mtimes.putSync(pk, { ...(value as object), indexKey: old });
+ }
+ db("meta").removeSync("twitchIds");
+ });
+ assert.equal(records()[0].id, `v${N}`);
+
+ const log = await runIndex();
+ assert.ok(log.includes("Twitch ids v1: 1 record(s) indexed under the native v-id, re-keyed."), log.join("\n"));
+ assert.deepEqual(records(), [
+ { key: ["20260105", CHANNEL, N], id: N, slug: `${CHANNEL}/${N}`, cues: 3 },
+ ]);
+
+ const again = await runIndex();
+ assert.equal(again.some((l) => l.startsWith("Twitch ids")), false, again.join("\n"));
+});
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -12,7 +12,7 @@ import {
parseTranscriptJson,
} from "../lib/whisper";
import type { TranscriptOutputFormat } from "../lib/transcriptionApps";
-import { summarize, type RawMetadata } from "../lib/transcripts-server";
+import { normalizeSummaryId, summarize, type RawMetadata } from "../lib/transcripts-server";
import type { TranscriptDetail } from "../lib/transcripts";
import {
transcriptCoverage,
@@ -174,7 +174,9 @@ export async function readNormalizedTranscript(
const raw = await readFile(cuesPath, "utf8");
const parsed = JSON.parse(raw) as NormalizedTranscript;
if (typeof parsed.version !== "number") return null;
- return parsed;
+ // Frozen when it was written: a Twitch record from before canonical ids
+ // reads with its canonical id (release 20 D2).
+ return normalizeSummaryId(parsed);
} catch {
return null;
}
diff --git a/common/controller/twitchId.test.ts b/common/controller/twitchId.test.ts
@@ -0,0 +1,77 @@
+// One Twitch id (release 20 D2): the canonical `<n>` its directory carries,
+// never yt-dlp's native `v<n>`, wherever a record is made or read back.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/twitchId.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { canonicalTwitchVideoId, extractVideoId } from "../lib/videoId";
+import { normalizeSummaryId, summarize, summaryIdStale } from "../lib/transcripts-server";
+import { readNormalizedTranscript } from "./normalizeTranscript";
+
+const META = {
+ id: "v1234567890",
+ title: "Example stream",
+ upload_date: "20260101",
+ duration: 60,
+ webpage_url: "https://www.twitch.tv/videos/1234567890",
+ extractor_key: "TwitchVod",
+};
+
+test("canonicalTwitchVideoId drops only a v before digits", () => {
+ assert.equal(canonicalTwitchVideoId("v1234567890"), "1234567890");
+ assert.equal(canonicalTwitchVideoId("1234567890"), "1234567890");
+ assert.equal(canonicalTwitchVideoId("vod-clip-slug"), "vod-clip-slug");
+ assert.equal(canonicalTwitchVideoId("v"), "v");
+ // The same id the URL gives, which is what data/<id>/ is named.
+ assert.equal(extractVideoId(META.webpage_url), canonicalTwitchVideoId(META.id));
+});
+
+test("summarize gives a Twitch record its directory's id", () => {
+ const s = summarize("example-twitch", "1234567890", META);
+ assert.equal(s.platform, "twitch");
+ assert.equal(s.id, "1234567890");
+ assert.equal(s.slug, "example-twitch/1234567890");
+ assert.equal(summaryIdStale(s), false);
+});
+
+test("a summary from before canonical ids is normalized; other platforms are untouched", () => {
+ const old = { id: "v1234567890", slug: "example-twitch/v1234567890", channelSlug: "example-twitch", platform: "twitch" as const };
+ assert.equal(summaryIdStale(old), true);
+ assert.deepEqual(normalizeSummaryId(old), {
+ id: "1234567890",
+ slug: "example-twitch/1234567890",
+ channelSlug: "example-twitch",
+ platform: "twitch",
+ });
+ const yt = { id: "v12345678901", slug: "c/v12345678901", channelSlug: "c", platform: "youtube" as const };
+ assert.equal(normalizeSummaryId(yt), yt);
+ assert.equal(summaryIdStale(yt), false);
+});
+
+test("a normalized transcript.cues.json written with the native id reads back canonical", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "twitch-cues-"));
+ try {
+ const file = path.join(dir, "transcript.cues.json");
+ await writeFile(
+ file,
+ JSON.stringify({
+ version: 1,
+ source: "vtt",
+ ...summarize("example-twitch", "1234567890", META),
+ id: "v1234567890",
+ slug: "example-twitch/v1234567890",
+ cues: [{ start: 0, end: 1, text: "hello" }],
+ }),
+ );
+ const t = await readNormalizedTranscript(file);
+ assert.equal(t?.id, "1234567890");
+ assert.equal(t?.slug, "example-twitch/1234567890");
+ assert.equal(t?.cues?.length, 1);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -6,7 +6,7 @@ import { archiveOrgPlayableUrl } from "./archiveOrg";
import { bitchutePlayableUrl } from "./bitchute";
import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId";
import { parseWaybackUrl } from "./wayback";
-import { extractVideoId } from "./videoId";
+import { canonicalTwitchVideoId, extractVideoId } from "./videoId";
import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
import type { MediaType, VideoStat, VideoStatus } from "./stats";
import type { VideoState } from "./availability";
@@ -113,6 +113,25 @@ const PAGE_DECIDES: ReadonlyArray<Platform> = ["archiveorg", "bitchute"];
// the record's metadata (controller/buildIndex.ts), and so does every reader
// of a normalized transcript.cues.json, whose summary was frozen when it was
// written. Pure: the summary alone decides.
+// THE RECORD ID AT THE BOUNDARY (release 20 D2): a summary written before
+// Twitch ids were canonical (a normalized transcript.cues.json, an index
+// record) carries the native `v<n>`; this gives it the canonical `<n>` and the
+// slug that goes with it (lib/videoId.ts, canonicalTwitchVideoId). The same
+// object when nothing changes.
+export function normalizeSummaryId<T extends { id: string; slug: string; channelSlug: string; platform: Platform }>(
+ summary: T,
+): T {
+ if (summary.platform !== "twitch") return summary;
+ const id = canonicalTwitchVideoId(summary.id);
+ if (id === summary.id) return summary;
+ return { ...summary, id, slug: `${summary.channelSlug}/${id}` };
+}
+
+// Whether a stored summary's id is one normalizeSummaryId would change.
+export function summaryIdStale(summary: { id: string; platform: Platform }): boolean {
+ return summary.platform === "twitch" && canonicalTwitchVideoId(summary.id) !== summary.id;
+}
+
export function platformLabelStale(summary: {
platform?: Platform;
webpageUrl?: string;
@@ -157,7 +176,9 @@ export function summarize(
? (meta.webpage_url_basename ?? meta.id ?? videoDir)
: platform === "archiveorg"
? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir)
- : (meta.id ?? videoDir));
+ : platform === "twitch"
+ ? canonicalTwitchVideoId(meta.id ?? videoDir)
+ : (meta.id ?? videoDir));
const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1];
return {
slug: `${channelSlug}/${id}`,
diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts
@@ -12,6 +12,15 @@
import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId";
import { isJwPlayerHost, isWaybackHost, jwPlayerMediaId, parseWaybackUrl } from "./wayback";
+// A TWITCH VOD'S ONE ID (release 20 D2). yt-dlp's native id is `v<n>`; the
+// canonical id — the URL's `/videos/<n>`, and so the name of the video's
+// `data/<n>/` — is `<n>`. A record that carried the native id could not be
+// joined to its own directory, so every record id is normalized here, once:
+// a `v` followed by digits loses the `v`; anything else is returned as it is.
+export function canonicalTwitchVideoId(id: string): string {
+ return /^v\d+$/.test(id) ? id.slice(1) : id;
+}
+
export function extractVideoId(url: string): string | null {
return extractVideoIdAt(url, 0);
}