Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 6469058b22b8e3646f5954e8723f6892da7589aa
parent 945b20ccd5a1c32e3bd71d4a33dd698b85b6421e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 26 Sep 2026 16:37:35 -0400

common: metadata.history.json — every metadata.info.json rewrite whose bytes differ appends one normalized diff (content keys whole, counters [from, to], volatile keys by fingerprint), capped at 200; withMetadataHistory snapshots around a writer and records even when it throws

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/lib/metadataHistory-server.test.ts | 128+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/metadataHistory-server.ts | 131+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/metadataHistory.test.ts | 215+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/metadataHistory.ts | 372+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/sidecar-server.test.ts | 4+++-
5 files changed, 849 insertions(+), 1 deletion(-)

diff --git a/common/lib/metadataHistory-server.test.ts b/common/lib/metadataHistory-server.test.ts @@ -0,0 +1,128 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, readFile, readdir, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { + loadMetadataHistory, + metadataHistoryPath, + withMetadataHistory, +} from "./metadataHistory-server"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/metadataHistory-server.test.ts + +async function withDir(fn: (dir: string) => Promise<void>): Promise<void> { + const dir = await mkdtemp(path.join(os.tmpdir(), "metadata-history-")); + try { + await fn(dir); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +const INFO = "metadata.info.json"; +const meta = (title: string, view_count = 1) => + JSON.stringify({ id: "v1", title, view_count, formats: [{ url: `u-${title}` }] }); + +test("a rewrite that changed the file appends one entry", async () => { + await withDir(async (dir) => { + await writeFile(path.join(dir, INFO), meta("old", 10)); + const out = await withMetadataHistory( + dir, + { by: "prefetch", requestedBy: "umtool" }, + async () => { + await writeFile(path.join(dir, INFO), meta("new", 12)); + return 42; + }, + ); + assert.equal(out, 42); + const h = await loadMetadataHistory(dir); + assert.equal(h?.entries.length, 1); + const e = h!.entries[0]; + assert.equal(e.by, "prefetch"); + assert.equal(e.requestedBy, "umtool"); + assert.deepEqual(e.changed.title, { from: "old", to: "new" }); + assert.deepEqual(e.counters.view_count, [10, 12]); + assert.deepEqual(e.volatile, ["formats"]); + assert.ok(e.from.at, "the old file's mtime is recorded"); + + // A second rewrite appends, newest last. + await withMetadataHistory(dir, { by: "audio-check" }, async () => { + await writeFile(path.join(dir, INFO), meta("newer", 12)); + }); + const h2 = await loadMetadataHistory(dir); + assert.deepEqual( + h2?.entries.map((x) => x.by), + ["prefetch", "audio-check"], + ); + assert.equal("requestedBy" in h2!.entries[1], false); + // Written through the sidecar: indented JSON with a trailing newline, and + // no temp file left behind. + const raw = await readFile(metadataHistoryPath(dir), "utf8"); + assert.ok(raw.startsWith("{\n") && raw.endsWith("}\n")); + assert.deepEqual((await readdir(dir)).sort(), [ + "metadata.history.json", + "metadata.info.json", + ]); + }); +}); + +test("an identical rewrite records nothing", async () => { + await withDir(async (dir) => { + await writeFile(path.join(dir, INFO), meta("same")); + await withMetadataHistory(dir, { by: "prefetch" }, async () => { + await writeFile(path.join(dir, INFO), meta("same")); + }); + assert.equal(await loadMetadataHistory(dir), null); + assert.deepEqual(await readdir(dir), [INFO]); + }); +}); + +test("no prior file records nothing: a first write is not a rewrite", async () => { + await withDir(async (dir) => { + await withMetadataHistory(dir, { by: "prefetch" }, async () => { + await writeFile(path.join(dir, INFO), meta("first")); + }); + assert.equal(await loadMetadataHistory(dir), null); + assert.deepEqual(await readdir(dir), [INFO]); + }); +}); + +test("a run that throws still records the rewrite, and its error comes through", async () => { + await withDir(async (dir) => { + await writeFile(path.join(dir, INFO), meta("before")); + await assert.rejects( + withMetadataHistory(dir, { by: "download-one" }, async () => { + await writeFile(path.join(dir, INFO), meta("after")); + throw new Error("yt-dlp exited 1"); + }), + /yt-dlp exited 1/, + ); + const h = await loadMetadataHistory(dir); + assert.equal(h?.entries.length, 1); + assert.equal(h!.entries[0].by, "download-one"); + assert.deepEqual(h!.entries[0].changed.title, { from: "before", to: "after" }); + }); +}); + +test("a failure to record is logged, never thrown", async () => { + await withDir(async (dir) => { + await writeFile(path.join(dir, INFO), meta("a")); + // A DIRECTORY where the history file goes: the atomic rename onto it fails. + await import("node:fs/promises").then((fs) => + fs.mkdir(metadataHistoryPath(dir)), + ); + let log = ""; + const out = await withMetadataHistory( + dir, + { by: "prefetch", onLog: (s) => (log += s) }, + async () => { + await writeFile(path.join(dir, INFO), meta("b")); + return "ran"; + }, + ); + assert.equal(out, "ran"); + assert.match(log, /Could not record the metadata history/); + }); +}); diff --git a/common/lib/metadataHistory-server.ts b/common/lib/metadataHistory-server.ts @@ -0,0 +1,131 @@ +// THE METADATA HISTORY SIDECAR, and the wrap every metadata.info.json writer +// runs inside. Release 10 slice N; the rules (what is stored, what is only +// fingerprinted, when an entry is written at all) are in metadataHistory.ts. +// +// ONE FILE, `metadata.history.json`, beside the file it describes — not a +// directory. A new directory under data/<id>/ would need the `clips/` treatment +// in every video-dir enumerator (the snapshot, the storage view, the +// reconciler); a bounded file needs none. +// +// IT IS NOT IN discardPrefetchDir's ALLOW-LIST, deliberately +// (downloadOneManaged.ts, PREFETCH_OWN_FILES). A history exists only because a +// metadata.info.json was already in the directory when some pass rewrote it — +// the directory predates the pass now deciding whether to delete it, and the +// history is the one record of what the source used to say. So a title-filter +// rejection of such a directory keeps it and writes the outcome sidecar, which +// is the cheap direction of being wrong (one metadata stub) that the allow-list +// is built on. The case is narrow: a rejection discards its own prefetch dir, +// so this needs a directory left by an earlier pass that was NOT rejected (a +// failed download, say) and a filter that rejects it now. +// +// THE FILE yt-dlp WRITES IS STILL THE FILE. Nothing here changes what lands in +// metadata.info.json; it reads the old bytes before the spawn and the new ones +// after it, and appends what moved. +// +// SERVER-ONLY (node:fs). + +import path from "node:path"; +import { readFile, stat } from "node:fs/promises"; +import { withJsonFileLock } from "./jsonFile-server"; +import { sidecar, sidecarField } from "./sidecar-server"; +import { + METADATA_HISTORY_FILENAME, + appendMetadataHistoryEntry, + buildMetadataHistoryEntry, + coerceMetadataHistory, + type MetadataHistoryEntry, + type MetadataHistoryWriter, +} from "./metadataHistory"; + +const INFO_JSON = "metadata.info.json"; + +export const metadataHistorySidecar = sidecar( + METADATA_HISTORY_FILENAME, + sidecarField(coerceMetadataHistory), +); + +export const { path: metadataHistoryPath, load: loadMetadataHistory } = + metadataHistorySidecar; + +export type MetadataSnapshot = { text: Buffer; mtime: Date }; + +// The info json as it is on disk right now, or null when there is none (or it +// cannot be read — the history is best-effort and never the reason a download +// fails). +export async function snapshotMetadata( + videoDir: string, +): Promise<MetadataSnapshot | null> { + const file = path.join(videoDir, INFO_JSON); + try { + const [text, st] = await Promise.all([readFile(file), stat(file)]); + return { text, mtime: st.mtime }; + } catch { + return null; + } +} + +export type MetadataRewriteContext = { + by: MetadataHistoryWriter; + requestedBy?: string; +}; + +// Compare the file now against `before` and append an entry when it was +// rewritten. Returns the entry, or null when there was nothing to record: no +// old file (a first write is not a rewrite), no new file, or the same bytes. +// +// A READ-MODIFY-WRITE, so it holds the path's lock: two writers finishing at +// once (a prefetch and an audio-check pass on one video are sequential, but a +// second job on the same video is not) must not each append to the file they +// read and drop the other's entry. +export async function recordMetadataRewrite( + videoDir: string, + before: MetadataSnapshot | null, + ctx: MetadataRewriteContext, +): Promise<MetadataHistoryEntry | null> { + if (!before) return null; + const after = await snapshotMetadata(videoDir); + if (!after) return null; + const entry = buildMetadataHistoryEntry({ + before, + after, + at: new Date().toISOString(), + by: ctx.by, + ...(ctx.requestedBy ? { requestedBy: ctx.requestedBy } : {}), + }); + if (!entry) return null; + await withJsonFileLock(metadataHistorySidecar.path(videoDir), async () => { + const existing = await loadMetadataHistory(videoDir); + await metadataHistorySidecar.write( + videoDir, + appendMetadataHistoryEntry(existing, entry), + ); + }); + return entry; +} + +// Run one metadata writer with the history kept: snapshot before `run()`, +// compare and append after it — whether it succeeded OR threw, because a +// yt-dlp that fails late (a prefetch that wrote the file and then hit a +// subtitle error) has still rewritten it. `run`'s own result or error is +// returned unchanged; a failure to record is logged through `onLog` and +// swallowed. +export async function withMetadataHistory<T>( + videoDir: string, + ctx: MetadataRewriteContext & { onLog?: (line: string) => void }, + run: () => Promise<T>, +): Promise<T> { + const before = await snapshotMetadata(videoDir); + try { + return await run(); + } finally { + if (before) { + try { + await recordMetadataRewrite(videoDir, before, ctx); + } catch (err) { + ctx.onLog?.( + `Could not record the metadata history: ${(err as Error).message}\n`, + ); + } + } + } +} diff --git a/common/lib/metadataHistory.test.ts b/common/lib/metadataHistory.test.ts @@ -0,0 +1,215 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + METADATA_HISTORY_CAP, + METADATA_HISTORY_MAX_VALUE_BYTES, + appendMetadataHistoryEntry, + buildMetadataHistoryEntry, + coerceMetadataHistory, + diffMetadata, + isValueDigest, + normalizeForDiff, + type MetadataHistoryEntry, +} from "./metadataHistory"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/metadataHistory.test.ts + +// A small but real-shaped info json: content, counters, and the volatile keys +// that move on every fetch (signed format URLs, epoch). +function info(over: Record<string, unknown> = {}): Record<string, unknown> { + return { + id: "dbnS-cBgStY", + title: "The original title", + description: "First description.", + duration: 3600, + view_count: 100, + like_count: 7, + formats: [{ format_id: "18", url: "https://example/sig=aaa" }], + thumbnails: [{ url: "https://i.ytimg.com/a.jpg" }], + subtitles: {}, + automatic_captions: { en: [{ url: "https://example/cap=1" }] }, + epoch: 1_700_000_000, + _version: { version: "2026.09.01" }, + ...over, + }; +} + +const text = (o: unknown) => JSON.stringify(o); + +function entry( + before: Record<string, unknown>, + after: Record<string, unknown>, +): MetadataHistoryEntry | null { + return buildMetadataHistoryEntry({ + before: { text: text(before), mtime: new Date("2026-09-01T00:00:00.000Z") }, + after: { text: text(after) }, + at: "2026-09-26T12:00:00.000Z", + by: "prefetch", + }); +} + +test("a formats-only rewrite is an all-volatile entry: nothing changed, nothing stored", () => { + const e = entry( + info(), + info({ + formats: [{ format_id: "18", url: "https://example/sig=bbb" }], + epoch: 1_700_000_999, + }), + ); + assert.ok(e); + assert.deepEqual(e.changed, {}); + assert.deepEqual(e.added, {}); + assert.deepEqual(e.removed, {}); + assert.deepEqual(e.counters, {}); + assert.deepEqual(e.volatile, ["epoch", "formats"]); + // The volatile VALUES are never stored — only that they moved. + assert.ok(!JSON.stringify(e).includes("sig=bbb")); + assert.equal(e.by, "prefetch"); + assert.equal(e.from.at, "2026-09-01T00:00:00.000Z"); + assert.notEqual(e.from.sha256, e.to.sha256); + assert.equal(e.to.bytes, Buffer.byteLength(text(info({ + formats: [{ format_id: "18", url: "https://example/sig=bbb" }], + epoch: 1_700_000_999, + })))); + // Small: the operator asked to be told every rewrite, and this is what it costs. + assert.ok(JSON.stringify(e).length < 400, String(JSON.stringify(e).length)); +}); + +test("a title change is changed.title, both sides whole", () => { + const e = entry(info(), info({ title: "A new title" })); + assert.ok(e); + assert.deepEqual(e.changed, { + title: { from: "The original title", to: "A new title" }, + }); + assert.deepEqual(e.volatile, []); +}); + +test("a new caption language is changed.subtitles_langs, though subtitles itself is volatile", () => { + const e = entry( + info(), + info({ subtitles: { en: [{ url: "https://example/sub=1" }] } }), + ); + assert.ok(e); + assert.deepEqual(e.changed, { + subtitles_langs: { from: [], to: ["en"] }, + }); + assert.deepEqual(e.volatile, ["subtitles"]); +}); + +test("a caption map appearing where there was none is added.<x>_langs", () => { + const before = info(); + delete before.subtitles; + const e = entry(before, info({ subtitles: { de: [] } })); + assert.ok(e); + assert.deepEqual(e.added, { subtitles_langs: ["de"] }); +}); + +test("counters that moved land in counters only, [from, to]", () => { + const e = entry(info(), info({ view_count: 150, like_count: 7 })); + assert.ok(e); + assert.deepEqual(e.counters, { view_count: [100, 150] }); + assert.deepEqual(e.changed, {}); + // A counter the old file lacked reads from null. + const e2 = entry(info(), info({ comment_count: 3 })); + assert.deepEqual(e2?.counters, { comment_count: [null, 3] }); +}); + +test("added and removed content keys", () => { + const before = info({ availability: "public" }); + const after = info({ chapters: [{ title: "Intro", start_time: 0 }] }); + const e = entry(before, after); + assert.ok(e); + assert.deepEqual(e.added, { chapters: [{ title: "Intro", start_time: 0 }] }); + assert.deepEqual(e.removed, { availability: "public" }); +}); + +test("key order alone is not a change, at any depth", () => { + const d = diffMetadata( + { title: "t", chapters: [{ a: 1, b: 2 }], formats: [{ x: 1, y: 2 }] }, + { chapters: [{ b: 2, a: 1 }], formats: [{ y: 2, x: 1 }], title: "t" }, + ); + assert.deepEqual(d, { + changed: {}, + added: {}, + removed: {}, + counters: {}, + volatile: [], + }); +}); + +test("a stored value over 16 KB is replaced by its { sha256, bytes }", () => { + const long = "x".repeat(METADATA_HISTORY_MAX_VALUE_BYTES + 10); + const e = entry(info(), info({ description: long })); + assert.ok(e); + const { from, to } = e.changed.description as { from: unknown; to: unknown }; + assert.equal(from, "First description."); + assert.ok(isValueDigest(to), JSON.stringify(to).slice(0, 80)); + assert.equal((to as { bytes: number }).bytes, long.length + 2); // its JSON: quotes + // And a value at the cap is kept whole. + const atCap = "y".repeat(METADATA_HISTORY_MAX_VALUE_BYTES - 2); + const e2 = entry(info(), info({ description: atCap })); + assert.equal((e2?.changed.description as { to: unknown }).to, atCap); +}); + +test("byte-identical → no entry", () => { + assert.equal(entry(info(), info()), null); +}); + +test("an unparseable side records the rewrite with no diff", () => { + const e = buildMetadataHistoryEntry({ + before: { text: "{ torn" }, + after: { text: text(info()) }, + at: "2026-09-26T12:00:00.000Z", + by: "audio-check", + requestedBy: "mcp", + }); + assert.ok(e); + assert.deepEqual(e.unparseable, ["from"]); + assert.deepEqual(e.changed, {}); + assert.deepEqual(e.added, {}); + assert.equal(e.requestedBy, "mcp"); + assert.equal("at" in e.from, false); +}); + +test("the cap drops the oldest, newest stays last", () => { + let h = null as ReturnType<typeof appendMetadataHistoryEntry> | null; + const mk = (i: number) => + ({ + ...entry(info(), info({ title: `t${i}` }))!, + at: `2026-09-26T12:00:${String(i % 60).padStart(2, "0")}.000Z`, + }) as MetadataHistoryEntry; + for (let i = 0; i < METADATA_HISTORY_CAP + 5; i++) { + h = appendMetadataHistoryEntry(h, mk(i)); + } + assert.equal(h!.entries.length, METADATA_HISTORY_CAP); + assert.deepEqual( + (h!.entries[0].changed.title as { to: string }).to, + "t5", + ); + assert.deepEqual( + (h!.entries.at(-1)!.changed.title as { to: string }).to, + `t${METADATA_HISTORY_CAP + 4}`, + ); + // A smaller cap, as passed. + const small = appendMetadataHistoryEntry(h, mk(999), 3); + assert.equal(small.entries.length, 3); +}); + +test("normalizeForDiff splits content, counters and volatile fingerprints", () => { + const n = normalizeForDiff(info()); + assert.deepEqual(Object.keys(n.counters).sort(), ["like_count", "view_count"]); + assert.deepEqual(n.content.automatic_captions_langs, ["en"]); + assert.deepEqual(n.content.subtitles_langs, []); + assert.equal("formats" in n.content, false); + assert.match(n.volatile.formats, /^[0-9a-f]{64}$/); +}); + +test("coerceMetadataHistory keeps good entries and drops torn ones", () => { + const good = entry(info(), info({ title: "x" }))!; + assert.equal(coerceMetadataHistory(null), null); + assert.equal(coerceMetadataHistory({ entries: "no" }), null); + assert.deepEqual(coerceMetadataHistory({ entries: [good, { at: 1 }, null] }), { + entries: [good], + }); +}); diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts @@ -0,0 +1,372 @@ +// WHAT CHANGED EACH TIME yt-dlp REWROTE A VIDEO'S metadata.info.json. +// +// Release 10 slice N. Every managed download rewrites `data/<id>/metadata.info.json` +// (the metadata prefetch has no existence check, by design: it is what the +// filters decide on), so the file only ever says what the source says TODAY. A +// title that changed upstream, a description that was edited, a caption track +// that appeared — all of it was overwritten without a trace. The history keeps +// one entry per rewrite, beside the video, in `metadata.history.json`. +// +// PURE: no fs. The server half (metadataHistory-server.ts) snapshots the file +// around a yt-dlp spawn and hands both texts here, so every rule below is +// testable without a directory. +// +// ── WHAT IS STORED, AND WHAT IS ONLY FINGERPRINTED ─────────────────────────── +// +// A real info json is ~50 KB and ~83 % of it is `formats`: signed URLs that +// differ on EVERY fetch. Storing old versions whole would be 50 KB of noise per +// rewrite. So the keys fall into three sets: +// +// VOLATILE compared by fingerprint (sha256 of their canonical JSON) and never +// stored — an entry lists which of them moved, nothing more. The +// two caption maps ALSO contribute their language-key lists as +// content keys (`subtitles_langs`, `automatic_captions_langs`), so a +// caption track appearing IS a meaningful change. +// COUNTERS view/like/comment counts: they drift every fetch, and an entry +// records only the ones that moved, as [from, to]. +// CONTENT everything else (title, description, duration, availability, +// chapters, uploader, …): deep-compared on the raw value and stored +// WHOLE on both sides, because a description's from/to is exactly +// what the operator wants to read. A single value over 16 KB is +// replaced by its { sha256, bytes }. +// +// AN ENTRY IS APPENDED ON EVERY REWRITE whose bytes differ — even one where +// only the volatile keys moved. The operator wants to know it was rewritten; +// such an entry is ~250 bytes. Byte-identical → no entry. No old file → no +// entry (a first write is not a rewrite). + +import { createHash } from "node:crypto"; + +export const METADATA_HISTORY_FILENAME = "metadata.history.json"; + +// Newest last; the oldest are dropped past this. Bounded so the file lives with +// the video with no eviction pass of its own. +export const METADATA_HISTORY_CAP = 200; + +// A stored value whose JSON is larger than this is replaced by its digest. +export const METADATA_HISTORY_MAX_VALUE_BYTES = 16 * 1024; + +// Who rewrote the file: the three yt-dlp spawns that write it. +export const METADATA_HISTORY_WRITERS = [ + // downloadOneManaged's attempt 0, on every managed download. + "prefetch", + // The audio-checked primary, which re-extracts on purpose (its restarts + // would outlive the prefetch's pinned format URLs). + "audio-check", + // runYtdlp's legacy single-video `download-one-audio` job. + "download-one", +] as const; +export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number]; + +export const VOLATILE_KEYS: ReadonlySet<string> = new Set([ + "formats", + "requested_formats", + "requested_downloads", + "thumbnails", + "thumbnail", + "heatmap", + "automatic_captions", + "subtitles", + "epoch", + "_version", + "_format_sort_fields", + "url", + "http_headers", + "format", + "format_id", + "format_note", + "filesize", + "filesize_approx", + "tbr", + "abr", + "vbr", + "asr", + "acodec", + "vcodec", + "fps", + "width", + "height", + "resolution", + "aspect_ratio", + "dynamic_range", + "protocol", + "ext", + "audio_channels", + "container", + "downloader_options", + "quality", + "source_preference", + "has_drm", + "language_preference", +]); + +export const COUNTER_KEYS: ReadonlySet<string> = new Set([ + "view_count", + "like_count", + "dislike_count", + "comment_count", + "channel_follower_count", + "average_rating", + "repost_count", +]); + +// The volatile caption maps whose LANGUAGE LIST is content: the URLs inside +// them are signed and move every fetch, but a language appearing or vanishing +// is a real change to what the video offers. +const LANG_LIST_KEYS: Readonly<Record<string, string>> = { + subtitles: "subtitles_langs", + automatic_captions: "automatic_captions_langs", +}; + +// A file, identified. +export type FileFingerprint = { sha256: string; bytes: number }; + +// What stands in for a stored value over METADATA_HISTORY_MAX_VALUE_BYTES. +export type ValueDigest = { sha256: string; bytes: number }; + +export type MetadataHistoryEntry = { + // When the rewrite was recorded (after the spawn returned). + at: string; + by: MetadataHistoryWriter; + // Who asked for the download, when the caller knew: the saved-video origin's + // requester on a whole-recording fetch ("umtool", "mcp", …). + requestedBy?: string; + // The file before; `at` is its mtime, i.e. when THAT version was written. + from: FileFingerprint & { at?: string }; + to: FileFingerprint; + // Content keys present on both sides whose values differ. + changed: Record<string, { from: unknown; to: unknown }>; + // Content keys only the new file has / only the old one had. + added: Record<string, unknown>; + removed: Record<string, unknown>; + // Counters that moved, [from, to]; null for a side that lacked the key. + counters: Record<string, [unknown, unknown]>; + // Volatile keys whose fingerprint differs (present on one side only counts). + volatile: string[]; + // A side that was not a JSON object (a torn or foreign file). The diff is + // then left empty: comparing against `{}` would list every key as added or + // removed, which is a lie about the source. The fingerprints still say the + // file changed. + unparseable?: Array<"from" | "to">; +}; + +export type MetadataHistory = { entries: MetadataHistoryEntry[] }; + +type Json = unknown; +type JsonObject = Record<string, Json>; + +function isPlainObject(v: unknown): v is JsonObject { + return typeof v === "object" && v !== null && !Array.isArray(v); +} + +function sha256(text: string | Buffer): string { + return createHash("sha256").update(text).digest("hex"); +} + +// JSON with object keys sorted at every depth, so two files that hold the same +// value in a different key order compare equal. `undefined` (an absent key) +// has no JSON and stays undefined. +export function canonicalJson(v: Json): string | undefined { + if (v === undefined) return undefined; + return JSON.stringify(sortKeys(v)); +} + +function sortKeys(v: Json): Json { + if (Array.isArray(v)) return v.map(sortKeys); + if (isPlainObject(v)) { + const out: JsonObject = {}; + for (const k of Object.keys(v).sort()) out[k] = sortKeys(v[k]); + return out; + } + return v; +} + +function jsonEqual(a: Json, b: Json): boolean { + return canonicalJson(a) === canonicalJson(b); +} + +function fingerprint(v: Json): string { + return sha256(canonicalJson(v) ?? ""); +} + +// A value as it is STORED in an entry: itself, or its digest when its JSON is +// over the cap. Measured in UTF-8 bytes, the unit the file is. +export function guardStoredValue(v: Json): Json { + const text = JSON.stringify(v); + if (text === undefined) return null; + const bytes = Buffer.byteLength(text, "utf8"); + if (bytes <= METADATA_HISTORY_MAX_VALUE_BYTES) return v; + const digest: ValueDigest = { sha256: sha256(text), bytes }; + return digest; +} + +// Is this stored value a digest standing in for an oversized one? (A real +// metadata value with exactly these two keys and a 64-hex sha is not a +// plausible yt-dlp field.) +export function isValueDigest(v: unknown): v is ValueDigest { + if (!isPlainObject(v)) return false; + const keys = Object.keys(v); + return ( + keys.length === 2 && + typeof v.sha256 === "string" && + /^[0-9a-f]{64}$/.test(v.sha256) && + typeof v.bytes === "number" + ); +} + +export type NormalizedMetadata = { + content: JsonObject; + counters: JsonObject; + // Volatile key -> fingerprint of its value. + volatile: Record<string, string>; +}; + +// Split one parsed info json into the three sets above. +export function normalizeForDiff(meta: JsonObject): NormalizedMetadata { + const content: JsonObject = {}; + const counters: JsonObject = {}; + const volatile: Record<string, string> = {}; + for (const [key, value] of Object.entries(meta)) { + if (VOLATILE_KEYS.has(key)) { + volatile[key] = fingerprint(value); + const langKey = LANG_LIST_KEYS[key]; + if (langKey && isPlainObject(value)) { + content[langKey] = Object.keys(value).sort(); + } + } else if (COUNTER_KEYS.has(key)) { + counters[key] = value; + } else { + content[key] = value; + } + } + return { content, counters, volatile }; +} + +export type MetadataDiff = Pick< + MetadataHistoryEntry, + "changed" | "added" | "removed" | "counters" | "volatile" +>; + +// Union of both sides' keys: the new file's order first, then keys only the +// old one had — so an entry reads in the order the source writes its fields. +function unionKeys(a: object, b: object): string[] { + const out = Object.keys(b); + const seen = new Set(out); + for (const k of Object.keys(a)) if (!seen.has(k)) out.push(k); + return out; +} + +export function diffMetadata(oldMeta: JsonObject, newMeta: JsonObject): MetadataDiff { + const a = normalizeForDiff(oldMeta); + const b = normalizeForDiff(newMeta); + const changed: MetadataDiff["changed"] = {}; + const added: MetadataDiff["added"] = {}; + const removed: MetadataDiff["removed"] = {}; + for (const key of unionKeys(a.content, b.content)) { + const inOld = Object.hasOwn(a.content, key); + const inNew = Object.hasOwn(b.content, key); + if (inOld && inNew) { + if (!jsonEqual(a.content[key], b.content[key])) { + changed[key] = { + from: guardStoredValue(a.content[key]), + to: guardStoredValue(b.content[key]), + }; + } + } else if (inNew) { + added[key] = guardStoredValue(b.content[key]); + } else { + removed[key] = guardStoredValue(a.content[key]); + } + } + const counters: MetadataDiff["counters"] = {}; + for (const key of unionKeys(a.counters, b.counters)) { + const from = Object.hasOwn(a.counters, key) ? a.counters[key] : null; + const to = Object.hasOwn(b.counters, key) ? b.counters[key] : null; + if (!jsonEqual(from, to)) counters[key] = [from, to]; + } + const volatile = unionKeys(a.volatile, b.volatile) + .filter((k) => a.volatile[k] !== b.volatile[k]) + .sort(); + return { changed, added, removed, counters, volatile }; +} + +function parseObject(text: string | Buffer): JsonObject | null { + try { + const v: unknown = JSON.parse(typeof text === "string" ? text : text.toString("utf8")); + return isPlainObject(v) ? v : null; + } catch { + return null; + } +} + +// One rewrite, as an entry — or null when there is nothing to record (the new +// bytes are the old bytes). +export function buildMetadataHistoryEntry(input: { + before: { text: string | Buffer; mtime?: Date | string }; + after: { text: string | Buffer }; + at: string; + by: MetadataHistoryWriter; + requestedBy?: string; +}): MetadataHistoryEntry | null { + const fromSha = sha256(input.before.text); + const toSha = sha256(input.after.text); + if (fromSha === toSha) return null; + const mtime = input.before.mtime; + const from: MetadataHistoryEntry["from"] = { + sha256: fromSha, + bytes: Buffer.byteLength(input.before.text), + ...(mtime !== undefined + ? { at: typeof mtime === "string" ? mtime : mtime.toISOString() } + : {}), + }; + const to = { sha256: toSha, bytes: Buffer.byteLength(input.after.text) }; + const oldMeta = parseObject(input.before.text); + const newMeta = parseObject(input.after.text); + const unparseable: Array<"from" | "to"> = []; + if (!oldMeta) unparseable.push("from"); + if (!newMeta) unparseable.push("to"); + const diff: MetadataDiff = + oldMeta && newMeta + ? diffMetadata(oldMeta, newMeta) + : { changed: {}, added: {}, removed: {}, counters: {}, volatile: [] }; + return { + at: input.at, + by: input.by, + ...(input.requestedBy ? { requestedBy: input.requestedBy } : {}), + from, + to, + ...diff, + ...(unparseable.length > 0 ? { unparseable } : {}), + }; +} + +// Append, newest last, keeping at most `cap` (the oldest go first). +export function appendMetadataHistoryEntry( + history: MetadataHistory | null, + entry: MetadataHistoryEntry, + cap: number = METADATA_HISTORY_CAP, +): MetadataHistory { + const entries = [...(history?.entries ?? []), entry]; + return { entries: entries.slice(-Math.max(1, cap)) }; +} + +// The sidecar's shape check. An entry that is not recognisably one is dropped +// rather than failing the whole file: one torn entry must not hide the other +// 199 from the page. +export function coerceMetadataHistory(value: unknown): MetadataHistory | null { + if (!isPlainObject(value) || !Array.isArray(value.entries)) return null; + const entries = value.entries.filter( + (e): e is MetadataHistoryEntry => + isPlainObject(e) && + typeof e.at === "string" && + typeof e.by === "string" && + isPlainObject(e.from) && + isPlainObject(e.to) && + isPlainObject(e.changed) && + isPlainObject(e.added) && + isPlainObject(e.removed) && + isPlainObject(e.counters) && + Array.isArray(e.volatile), + ); + return { entries }; +} diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts @@ -9,6 +9,7 @@ import { SUB_FILE_RE } from "./videoStatus"; import "./attribution-server"; import "./diarization-server"; import "./digest-server"; +import "./metadataHistory-server"; import { availabilitySidecar, loadAvailability, @@ -40,7 +41,7 @@ async function scratch(): Promise<string> { return mkdtemp(path.join(os.tmpdir(), "sidecar-")); } -test("every declared sidecar filename escapes SUB_FILE_RE, and all nine are declared", () => { +test("every declared sidecar filename escapes SUB_FILE_RE, and all ten are declared", () => { assert.deepEqual([...SIDECAR_FILENAMES].sort(), [ "ai-digest.json", "ai-digest.overrides.json", @@ -50,6 +51,7 @@ test("every declared sidecar filename escapes SUB_FILE_RE, and all nine are decl "do-not-clean.json", "download-outcome.json", "exclude-truncated-check.json", + "metadata.history.json", "transcribe-outcome.json", ]); for (const name of SIDECAR_FILENAMES) {