Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 21c44a807157a99a4e56de7ccb37f6b92a2ae44f
parent 0f614b3a43376d07d562101aae4c83424749ccd3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 10 Oct 2026 01:11:15 -0400

common: release 20 D2 — one Twitch id: records carry the canonical <n> their directory does, not the native v<n>

canonicalTwitchVideoId (lib/videoId.ts) is the one normalization: summarize
uses it for a Twitch record, readNormalizedTranscript applies it to a
transcript.cues.json written before (normalizeSummaryId), and the index
re-processes once every record it holds under a v-id ("Twitch ids v1", the
platform-labels shape), re-keying it. The Twitch player is still handed v<n>.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/TwitchPlayer.tsx | 5++++-
Mcommon/controller/buildIndex.ts | 29+++++++++++++++++++++++++++++
Acommon/controller/buildIndexTwitchIds.test.ts | 146+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/normalizeTranscript.ts | 6++++--
Acommon/controller/twitchId.test.ts | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/transcripts-server.ts | 25+++++++++++++++++++++++--
Mcommon/lib/videoId.ts | 9+++++++++
7 files changed, 292 insertions(+), 5 deletions(-)

diff --git a/common/components/TwitchPlayer.tsx b/common/components/TwitchPlayer.tsx @@ -47,8 +47,11 @@ const TwitchPlayer = forwardRef<TwitchPlayerHandle, Props>(function TwitchPlayer const parent = typeof window !== "undefined" ? window.location.hostname : "localhost"; + // The record carries the canonical `<n>` (release 20 D2); the player is + // given the `v<n>` it has always been given. + const video = /^\d+$/.test(videoId) ? `v${videoId}` : videoId; const src = `https://player.twitch.tv/?video=${encodeURIComponent( - videoId, + video, )}&parent=${encodeURIComponent(parent)}&time=${toTwitchTime(start)}&autoplay=true`; return ( diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -47,6 +47,7 @@ import { parseTranscriptJson } from "../lib/whisper"; import { parseLiveChat } from "../lib/liveChat"; import { platformLabelStale, + summaryIdStale, summarize, toDisplaySummary, type RawMetadata, @@ -230,6 +231,13 @@ const CAPTION_TRACK_KEY = "captionTrackRule"; // pattern. const RECORDED_DATE_RULE_KEY_PREFIX = "recordedDateRule:"; +// TWITCH IDS (release 20 D2) — the platform-labels shape once more: records +// indexed under yt-dlp's native `v<n>` (summaryIdStale) are re-processed once, +// so they take the canonical `<n>` their directory carries and are re-keyed; +// then the version is recorded, only when no channel is held. +const TWITCH_IDS_VERSION = 1; +const TWITCH_IDS_KEY = "twitchIds"; + // A cue list's text, for "did the words change" — timing alone is not a // different transcript. function cueText(list: readonly Cue[]): string { @@ -946,6 +954,24 @@ export async function buildIndex({ // records were derived under (RECORDED_DATE_RULE_KEY_PREFIX) — every record // of those is re-processed like a changed one. A held channel is left for // the build that can read it. + const twitchIdsDue = meta.get(TWITCH_IDS_KEY) !== TWITCH_IDS_VERSION; + if (twitchIdsDue && !schemaBumped) { + const queued = new Set( + [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])), + ); + let rekeyed = 0; + for (const s of live) { + const pk: PathKey = [s.channelSlug, s.videoDir]; + const prev = mtimes.get(pk); + if (!prev) continue; + const sum = sums.get(prev.indexKey); + if (!sum || !summaryIdStale(sum)) continue; + rekeyed++; + if (!queued.has(pathKeyId(pk))) changed.push(s); + } + log(`Twitch ids v${TWITCH_IDS_VERSION}: ${rekeyed} record(s) indexed under the native v-id, re-keyed.`); + } + const recordedDateRes = new Map<string, RegExp>(); const recordedDateRuleChanged: string[] = []; for (const [slug, cfg] of channelConfigs) { @@ -2581,6 +2607,9 @@ export async function buildIndex({ if (altTracksDue && held.size === 0) { await meta.put(ALT_TRACKS_KEY, ALT_TRACKS_VERSION); } + if (twitchIdsDue && held.size === 0) { + await meta.put(TWITCH_IDS_KEY, TWITCH_IDS_VERSION); + } for (const slug of recordedDateRuleChanged) { const rule = channelConfigs.get(slug)?.recordedDate; const key = `${RECORDED_DATE_RULE_KEY_PREFIX}${slug}`; diff --git a/common/controller/buildIndexTwitchIds.test.ts b/common/controller/buildIndexTwitchIds.test.ts @@ -0,0 +1,146 @@ +// Integration: Twitch ids (release 20 D2) through the REAL buildIndex over a +// temp corpus — a record is indexed under the canonical `<n>` its directory +// carries, and an index that holds one under the native `v<n>` re-keys it once. +// Synthetic ids. +// +// Run with: node_modules/.bin/tsx --test common/controller/buildIndexTwitchIds.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-twitch-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("./buildIndex"); +const { open } = await import("lmdb"); + +const paths = getPaths(); +const CHANNEL = "example-twitch"; +const SITE = "testsite"; +const N = "2700000001"; + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; + +function seed(): void { + writeFileSync(paths.settingsFile, "{}"); + writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { + handling: "transcribe", + name: CHANNEL, + platform: "twitch", + url: "https://www.twitch.tv/example", + }); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [{ slug: CHANNEL, groupId: "default" }], + }); + const dir = path.join(paths.channelsDir, CHANNEL, "data", N); + writeJson(path.join(dir, "metadata.info.json"), { + id: `v${N}`, + title: "Example stream", + upload_date: "20260105", + duration: 30, + webpage_url: `https://www.twitch.tv/videos/${N}`, + extractor_key: "TwitchVod", + }); + writeFileSync( + path.join(dir, "transcript.en.vtt"), + readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", "vtt-rolling.vtt"), "utf8"), + ); +} + +type Summary = { id: string; slug: string }; +type Key = [string, string, string]; + +function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T { + const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); + try { + return fn((name) => root.openDB({ name, encoding: "msgpack" })); + } finally { + root.close(); + } +} + +function records(): { key: Key; id: string; slug: string; cues: number }[] { + return withIndex((db) => + [...db("sums").getRange()].map(({ key, value }) => ({ + key: key as Key, + id: (value as Summary).id, + slug: (value as Summary).slug, + cues: ((db("cues").get(key) as unknown[] | undefined) ?? []).length, + })), + ); +} + +async function runIndex(): Promise<string[]> { + const log: string[] = []; + await buildIndex({ paths, onLog: (s) => log.push(s) }); + return log; +} + +test("a Twitch record is indexed under its directory's id", async () => { + seed(); + await runIndex(); + assert.deepEqual(records(), [ + { key: ["20260105", CHANNEL, N], id: N, slug: `${CHANNEL}/${N}`, cues: 3 }, + ]); +}); + +test("an index holding the native v-id re-keys it once", async () => { + // The index as a build before canonical ids left it. + withIndex((db) => { + const key: Key = ["20260105", CHANNEL, N]; + const old: Key = ["20260105", CHANNEL, `v${N}`]; + const sum = db("sums").get(key) as Summary; + db("sums").putSync(old, { ...sum, id: `v${N}`, slug: `${CHANNEL}/v${N}` }); + db("cues").putSync(old, db("cues").get(key)); + db("sums").removeSync(key); + db("cues").removeSync(key); + const mtimes = db("mtimes"); + for (const { key: pk, value } of mtimes.getRange()) { + mtimes.putSync(pk, { ...(value as object), indexKey: old }); + } + db("meta").removeSync("twitchIds"); + }); + assert.equal(records()[0].id, `v${N}`); + + const log = await runIndex(); + assert.ok(log.includes("Twitch ids v1: 1 record(s) indexed under the native v-id, re-keyed."), log.join("\n")); + assert.deepEqual(records(), [ + { key: ["20260105", CHANNEL, N], id: N, slug: `${CHANNEL}/${N}`, cues: 3 }, + ]); + + const again = await runIndex(); + assert.equal(again.some((l) => l.startsWith("Twitch ids")), false, again.join("\n")); +}); diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts @@ -12,7 +12,7 @@ import { parseTranscriptJson, } from "../lib/whisper"; import type { TranscriptOutputFormat } from "../lib/transcriptionApps"; -import { summarize, type RawMetadata } from "../lib/transcripts-server"; +import { normalizeSummaryId, summarize, type RawMetadata } from "../lib/transcripts-server"; import type { TranscriptDetail } from "../lib/transcripts"; import { transcriptCoverage, @@ -174,7 +174,9 @@ export async function readNormalizedTranscript( const raw = await readFile(cuesPath, "utf8"); const parsed = JSON.parse(raw) as NormalizedTranscript; if (typeof parsed.version !== "number") return null; - return parsed; + // Frozen when it was written: a Twitch record from before canonical ids + // reads with its canonical id (release 20 D2). + return normalizeSummaryId(parsed); } catch { return null; } diff --git a/common/controller/twitchId.test.ts b/common/controller/twitchId.test.ts @@ -0,0 +1,77 @@ +// One Twitch id (release 20 D2): the canonical `<n>` its directory carries, +// never yt-dlp's native `v<n>`, wherever a record is made or read back. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/twitchId.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { canonicalTwitchVideoId, extractVideoId } from "../lib/videoId"; +import { normalizeSummaryId, summarize, summaryIdStale } from "../lib/transcripts-server"; +import { readNormalizedTranscript } from "./normalizeTranscript"; + +const META = { + id: "v1234567890", + title: "Example stream", + upload_date: "20260101", + duration: 60, + webpage_url: "https://www.twitch.tv/videos/1234567890", + extractor_key: "TwitchVod", +}; + +test("canonicalTwitchVideoId drops only a v before digits", () => { + assert.equal(canonicalTwitchVideoId("v1234567890"), "1234567890"); + assert.equal(canonicalTwitchVideoId("1234567890"), "1234567890"); + assert.equal(canonicalTwitchVideoId("vod-clip-slug"), "vod-clip-slug"); + assert.equal(canonicalTwitchVideoId("v"), "v"); + // The same id the URL gives, which is what data/<id>/ is named. + assert.equal(extractVideoId(META.webpage_url), canonicalTwitchVideoId(META.id)); +}); + +test("summarize gives a Twitch record its directory's id", () => { + const s = summarize("example-twitch", "1234567890", META); + assert.equal(s.platform, "twitch"); + assert.equal(s.id, "1234567890"); + assert.equal(s.slug, "example-twitch/1234567890"); + assert.equal(summaryIdStale(s), false); +}); + +test("a summary from before canonical ids is normalized; other platforms are untouched", () => { + const old = { id: "v1234567890", slug: "example-twitch/v1234567890", channelSlug: "example-twitch", platform: "twitch" as const }; + assert.equal(summaryIdStale(old), true); + assert.deepEqual(normalizeSummaryId(old), { + id: "1234567890", + slug: "example-twitch/1234567890", + channelSlug: "example-twitch", + platform: "twitch", + }); + const yt = { id: "v12345678901", slug: "c/v12345678901", channelSlug: "c", platform: "youtube" as const }; + assert.equal(normalizeSummaryId(yt), yt); + assert.equal(summaryIdStale(yt), false); +}); + +test("a normalized transcript.cues.json written with the native id reads back canonical", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "twitch-cues-")); + try { + const file = path.join(dir, "transcript.cues.json"); + await writeFile( + file, + JSON.stringify({ + version: 1, + source: "vtt", + ...summarize("example-twitch", "1234567890", META), + id: "v1234567890", + slug: "example-twitch/v1234567890", + cues: [{ start: 0, end: 1, text: "hello" }], + }), + ); + const t = await readNormalizedTranscript(file); + assert.equal(t?.id, "1234567890"); + assert.equal(t?.slug, "example-twitch/1234567890"); + assert.equal(t?.cues?.length, 1); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -6,7 +6,7 @@ import { archiveOrgPlayableUrl } from "./archiveOrg"; import { bitchutePlayableUrl } from "./bitchute"; import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId"; import { parseWaybackUrl } from "./wayback"; -import { extractVideoId } from "./videoId"; +import { canonicalTwitchVideoId, extractVideoId } from "./videoId"; import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; import type { MediaType, VideoStat, VideoStatus } from "./stats"; import type { VideoState } from "./availability"; @@ -113,6 +113,25 @@ const PAGE_DECIDES: ReadonlyArray<Platform> = ["archiveorg", "bitchute"]; // the record's metadata (controller/buildIndex.ts), and so does every reader // of a normalized transcript.cues.json, whose summary was frozen when it was // written. Pure: the summary alone decides. +// THE RECORD ID AT THE BOUNDARY (release 20 D2): a summary written before +// Twitch ids were canonical (a normalized transcript.cues.json, an index +// record) carries the native `v<n>`; this gives it the canonical `<n>` and the +// slug that goes with it (lib/videoId.ts, canonicalTwitchVideoId). The same +// object when nothing changes. +export function normalizeSummaryId<T extends { id: string; slug: string; channelSlug: string; platform: Platform }>( + summary: T, +): T { + if (summary.platform !== "twitch") return summary; + const id = canonicalTwitchVideoId(summary.id); + if (id === summary.id) return summary; + return { ...summary, id, slug: `${summary.channelSlug}/${id}` }; +} + +// Whether a stored summary's id is one normalizeSummaryId would change. +export function summaryIdStale(summary: { id: string; platform: Platform }): boolean { + return summary.platform === "twitch" && canonicalTwitchVideoId(summary.id) !== summary.id; +} + export function platformLabelStale(summary: { platform?: Platform; webpageUrl?: string; @@ -157,7 +176,9 @@ export function summarize( ? (meta.webpage_url_basename ?? meta.id ?? videoDir) : platform === "archiveorg" ? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir) - : (meta.id ?? videoDir)); + : platform === "twitch" + ? canonicalTwitchVideoId(meta.id ?? videoDir) + : (meta.id ?? videoDir)); const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1]; return { slug: `${channelSlug}/${id}`, diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts @@ -12,6 +12,15 @@ import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId"; import { isJwPlayerHost, isWaybackHost, jwPlayerMediaId, parseWaybackUrl } from "./wayback"; +// A TWITCH VOD'S ONE ID (release 20 D2). yt-dlp's native id is `v<n>`; the +// canonical id — the URL's `/videos/<n>`, and so the name of the video's +// `data/<n>/` — is `<n>`. A record that carried the native id could not be +// joined to its own directory, so every record id is normalized here, once: +// a `v` followed by digits loses the `v`; anything else is returned as it is. +export function canonicalTwitchVideoId(id: string): string { + return /^v\d+$/.test(id) ? id.slice(1) : id; +} + export function extractVideoId(url: string): string | null { return extractVideoIdAt(url, 0); }