Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 4737bb2a3cc663585d56e4e920a9ebc7a33a1f07
parent f8152a9c3ec6f66a0f04b8a1186fd51a9a0c6d5f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 29 Jul 2026 11:19:40 -0400

Digest cluster sharing was dead for 14.5% of the corpus — the half it was for

`digestBatch` looked the cluster plan up with `${channelSlug}/${directoryName}`.
The plan is keyed `${channelSlug}/${metadataId}`. Those are not the same string
for 11,175 of 77,106 videos, and they are not spread evenly: **7,870 of them are
`the-quartering-rumble`, where dir !== id for every single video** — i.e.
exactly the mirror set this file's own header cites as the win ("5,863
Quartering YouTube↔Rumble mirror pairs", "~11% of the sweep").

Every one of those lookups missed. A miss is silent and looks like success: the
mirror simply isn't recognised as a mirror, so it is generated from scratch and
the log says nothing. The same mistake sat on the other side of the pass —
`videoDirForSlug` built a path out of the id, so even a plan hit would have
opened a directory that does not exist and reported "no transcript".

The fix is to record what the detector already knows and was throwing away: it
keys its own scan by directory and holds both halves in `Entry`. So
`DuplicateVideoRef` gains `videoDir`, written ONLY when it differs from `id` (no
redundant field on the other 85%, and readers fall back to `id`, which is what
they all assumed before). The plan derives both directions from it, and
`planSlugForDir` is the one place the batch converts a directory back to a slug.

An old report with no `videoDir` keeps today's behaviour rather than resolving
to nothing — so this degrades to the status quo instead of failing.

Found while pricing the sweep, not by a test: nothing asserted on a corpus where
the two identifiers disagree. There is one now.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/digestBatch.ts | 9++++++++-
Acommon/controller/digestSharing.test.ts | 186+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/digestSharing.ts | 98+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Mcommon/controller/duplicateShorts.ts | 9+++++++--
Mcommon/lib/duplicates.ts | 11+++++++++++
5 files changed, 301 insertions(+), 12 deletions(-)

diff --git a/common/controller/digestBatch.ts b/common/controller/digestBatch.ts @@ -34,6 +34,7 @@ import { CUES_JSON_FILENAME } from "../lib/videoStatus"; import { digestVideo } from "./digestVideo"; import { buildDigestClusterPlan, + planSlugForDir, shareDigestToCluster, type DigestClusterPlan, } from "./digestSharing"; @@ -243,7 +244,12 @@ export async function runDigestBatch( const attempted = new Set<string>(); let cursor = 0; - const slugOf = (id: string): string => `${opts.channelSlug}/${id}`; + // `candidate.id` is a DIRECTORY name (it comes from readdir), and a slug is + // `${channelSlug}/${metadataId}`. Those disagree for 14.5% of the corpus — + // every Rumble re-upload — so resolving one to the other through the plan is + // what makes the cluster lookups below hit at all. See planSlugForDir. + const slugOf = (videoDir: string): string => + planSlugForDir(clusterPlan, opts.channelSlug, videoDir); const next = async (): Promise<Candidate | null> => { while (cursor < candidates.length) { @@ -342,6 +348,7 @@ export async function runDigestBatch( clusterId: role.clusterId, canonicalSlug: slugOf(candidate.id), mirrors: role.mirrors, + dirBySlug: clusterPlan?.dirBySlug, onLog: log, }); for (const o of outcomes) { diff --git a/common/controller/digestSharing.test.ts b/common/controller/digestSharing.test.ts @@ -0,0 +1,186 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import { + DUPLICATES_FILENAME, + DUPLICATE_REPORT_VERSION, + type DuplicateCluster, + type DuplicateVideoRef, +} from "../lib/duplicates"; +import { + buildDigestClusterPlan, + dirBySlugForCluster, + planSlugForDir, +} from "./digestSharing"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/digestSharing.test.ts + +async function withPaths(fn: (paths: Paths) => Promise<void>): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-digest-sharing-")); + const paths = { + transcriptsDir: dir, + channelsDir: path.join(dir, "channels"), + } as Paths; + await mkdir(paths.transcriptsDir, { recursive: true }); + try { + await fn(paths); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +function ref( + channelSlug: string, + id: string, + extra: Partial<DuplicateVideoRef> = {}, +): DuplicateVideoRef { + return { + slug: `${channelSlug}/${id}`, + channelSlug, + channel: channelSlug, + platform: "youtube", + id, + title: "t", + duration: 600, + uploadDate: "20240101", + hasTranscript: true, + ...extra, + }; +} + +async function writeReport( + paths: Paths, + clusters: DuplicateCluster[], +): Promise<void> { + await writeFile( + path.join(paths.transcriptsDir, DUPLICATES_FILENAME), + JSON.stringify({ + version: DUPLICATE_REPORT_VERSION, + generatedAt: new Date(0).toISOString(), + runConfig: { + thresholdSeconds: null, + durationToleranceSeconds: 2, + nearThreshold: 0.35, + containmentThreshold: 0.8, + shingleSize: 5, + }, + totals: { videosScanned: 2, clusters: clusters.length, videosInClusters: 2 }, + clusters, + }), + ); +} + +function cluster( + clusterId: string, + videoRefs: DuplicateVideoRef[], + extra: Partial<DuplicateCluster> = {}, +): DuplicateCluster { + return { + clusterId, + matchKind: "transcript-near", + score: 0.9, + contained: false, + durationBucket: 600, + crossPlatform: true, + crossChannel: true, + videoRefs, + canonicalSlug: videoRefs[0].slug, + ...extra, + }; +} + +// The bug this whole mapping exists for: a slug is `${channelSlug}/${id}` while +// the batch runner enumerates DIRECTORIES, and those disagree for 14.5% of the +// real corpus (every Rumble re-upload). Before the mapping, the lookup below +// missed and every such mirror was regenerated instead of shared. +test("plan maps a directory name to its slug when the two differ", async () => { + await withPaths(async (paths) => { + await writeReport(paths, [ + cluster("c1", [ + ref("chan-yt", "canon1"), + // The Rumble shape: on-disk dir `v1007ay`, metadata id `vxe1ae`. + ref("chan-rumble", "vxe1ae", { videoDir: "v1007ay" }), + ]), + ]); + const plan = await buildDigestClusterPlan(paths, { overrides: null }); + + assert.equal( + planSlugForDir(plan, "chan-rumble", "v1007ay"), + "chan-rumble/vxe1ae", + "the directory name must resolve to the metadata-id slug", + ); + assert.equal( + plan.bySlug.get(planSlugForDir(plan, "chan-rumble", "v1007ay"))?.kind, + "mirror", + "and that slug must then hit the plan as a mirror", + ); + assert.equal( + plan.dirBySlug.get("chan-rumble/vxe1ae"), + "v1007ay", + "the reverse direction is what lets the sharing pass open the mirror's files", + ); + }); +}); + +test("a member whose dir equals its id is absent from the mapping", async () => { + await withPaths(async (paths) => { + await writeReport(paths, [ + cluster("c1", [ref("chan-a", "same1"), ref("chan-b", "same2")]), + ]); + const plan = await buildDigestClusterPlan(paths, { overrides: null }); + assert.equal(plan.dirBySlug.size, 0); + assert.equal(plan.slugByDir.size, 0); + // The fallback is the old behaviour, and it is the correct one here. + assert.equal(planSlugForDir(plan, "chan-a", "same1"), "chan-a/same1"); + }); +}); + +// A report written before `videoDir` existed must not regress: it keeps the +// pre-mapping behaviour rather than resolving to nothing. +test("an old report with no videoDir falls back to the id", async () => { + await withPaths(async (paths) => { + await writeReport(paths, [ + cluster("c1", [ref("chan-a", "aaa"), ref("chan-b", "bbb")]), + ]); + const plan = await buildDigestClusterPlan(paths, { overrides: null }); + assert.equal(planSlugForDir(plan, "chan-b", "bbb"), "chan-b/bbb"); + assert.equal(planSlugForDir(null, "chan-b", "bbb"), "chan-b/bbb"); + }); +}); + +test("dirBySlugForCluster carries only the members that differ", () => { + const c = cluster("c1", [ + ref("chan-yt", "canon1"), + ref("chan-rumble", "vxe1ae", { videoDir: "v1007ay" }), + ref("chan-odysee", "plain", { videoDir: "plain" }), + ]); + const map = dirBySlugForCluster(c); + assert.deepEqual([...map], [["chan-rumble/vxe1ae", "v1007ay"]]); +}); + +// The directory mapping is recorded for EVERY cluster, including ones that +// share nothing — their members are still generated, and a caller holding the +// plan still needs a correct path for them. +test("an unshareable cluster still contributes its directory mapping", async () => { + await withPaths(async (paths) => { + await writeReport(paths, [ + cluster( + "c1", + [ + ref("chan-yt", "canon1"), + ref("chan-rumble", "vxe1ae", { videoDir: "v1007ay" }), + ], + // needsReview and unconfirmed → clusterMaySharePartial fails closed. + { needsReview: true }, + ), + ]); + const plan = await buildDigestClusterPlan(paths, { overrides: null }); + assert.equal(plan.bySlug.size, 0, "it shares nothing"); + assert.equal(plan.independentClusters, 1); + assert.equal(plan.dirBySlug.get("chan-rumble/vxe1ae"), "v1007ay"); + }); +}); diff --git a/common/controller/digestSharing.ts b/common/controller/digestSharing.ts @@ -46,6 +46,22 @@ export type DigestClusterPlan = { // `${channelSlug}/${id}` → role. Videos absent from the map are in no cluster // and are generated normally, which is the overwhelming majority. bySlug: Map<string, DigestClusterRole>; + // The two directions of the id ↔ on-disk-directory mapping, populated ONLY for + // cluster members whose directory name is not their metadata id. + // + // These exist because a slug is `${channelSlug}/${id}` and is not a path, + // while every caller that reaches this plan starts from one side or the other: + // the batch runner enumerates DIRECTORIES off disk, and the sharing pass has + // SLUGS from the report and has to open their files. 14.5% of the corpus has + // dir !== id, and it is not spread evenly — it is 100% of `the-quartering- + // rumble` (7,870 videos), i.e. exactly the mirror set cluster-sharing was + // written to exploit. Keying either lookup with the wrong half misses silently + // and simply regenerates the mirror, which is why this went unnoticed. + // + // Empty for a report written before `DuplicateVideoRef.videoDir` existed; + // lookups fall back to treating the id as the directory, the old behaviour. + slugByDir: Map<string, string>; // `${channelSlug}/${videoDir}` → slug + dirBySlug: Map<string, string>; // slug → videoDir (bare name, not a path) // Clusters that exist but share nothing (contained, or marked not-a-duplicate). // Their members are absent from bySlug and are each generated independently — // a clip is a different artifact and deserves its own digest. @@ -54,7 +70,25 @@ export type DigestClusterPlan = { }; export function emptyDigestClusterPlan(): DigestClusterPlan { - return { bySlug: new Map(), independentClusters: 0, clusters: 0 }; + return { + bySlug: new Map(), + slugByDir: new Map(), + dirBySlug: new Map(), + independentClusters: 0, + clusters: 0, + }; +} + +// The batch runner enumerates on-disk directories; the plan is keyed by slug. +// Falls back to the directory name as the id, which is correct for the ~85.5% +// where they agree and is what the code did before the mapping existed. +export function planSlugForDir( + plan: DigestClusterPlan | null | undefined, + channelSlug: string, + videoDir: string, +): string { + const key = `${channelSlug}/${videoDir}`; + return plan?.slugByDir.get(key) ?? key; } // Build the plan from the last detection run. Absent report → an empty plan, so @@ -71,6 +105,14 @@ export async function buildDigestClusterPlan( plan.clusters = report.clusters.length; for (const cluster of report.clusters) { + // Recorded for EVERY cluster, including the ones that share nothing: an + // independent cluster's members are still generated, and a caller that + // reaches the plan at all deserves a correct path for them. + for (const ref of cluster.videoRefs) { + if (!ref.videoDir || ref.videoDir === ref.id) continue; + plan.dirBySlug.set(ref.slug, ref.videoDir); + plan.slugByDir.set(`${ref.channelSlug}/${ref.videoDir}`, ref.slug); + } const canonicalSlug = resolveCanonicalSlug(cluster, overrides); // null → a human said "not a duplicate": every member stands alone. // The overrides also carry the `confirmed` flag that is the ONLY thing @@ -103,19 +145,31 @@ export async function buildDigestClusterPlan( return plan; } -function videoDirForSlug(paths: Paths, slug: string): string | null { +// A slug is `${channelSlug}/${id}`, and an id is NOT always the directory name — +// see DigestClusterPlan.slugByDir. `dirBySlug` carries the exceptions; without +// it this resolves the 14.5% of the corpus with dir !== id to a path that does +// not exist, which reads as "no transcript" and silently declines to share. +function videoDirForSlug( + paths: Paths, + slug: string, + dirBySlug?: ReadonlyMap<string, string>, +): string | null { const at = slug.indexOf("/"); if (at <= 0) return null; return path.join( paths.channelsDir, slug.slice(0, at), "data", - slug.slice(at + 1), + dirBySlug?.get(slug) ?? slug.slice(at + 1), ); } -async function readCues(paths: Paths, slug: string): Promise<Cue[] | null> { - const dir = videoDirForSlug(paths, slug); +async function readCues( + paths: Paths, + slug: string, + dirBySlug?: ReadonlyMap<string, string>, +): Promise<Cue[] | null> { + const dir = videoDirForSlug(paths, slug, dirBySlug); if (!dir) return null; const t = await readNormalizedTranscript(path.join(dir, CUES_JSON_FILENAME)); return t?.cues ?? null; @@ -133,6 +187,11 @@ export type ShareDigestOptions = { clusterId: string; canonicalSlug: string; mirrors: string[]; + // slug → on-disk directory, for members where the two differ. Take it from + // DigestClusterPlan.dirBySlug (or build it from the cluster's own refs). + // Omitting it is not an error, but every member with dir !== id will resolve + // to a nonexistent path and be reported "no-transcript". + dirBySlug?: ReadonlyMap<string, string>; toleranceSeconds?: number; onLog?: (msg: string) => void; }; @@ -146,17 +205,25 @@ export async function shareDigestToCluster( const log = opts.onLog ?? (() => {}); const tolerance = opts.toleranceSeconds ?? DEFAULT_ALIGNMENT_TOLERANCE_SECONDS; - const canonicalDir = videoDirForSlug(opts.paths, opts.canonicalSlug); + const canonicalDir = videoDirForSlug( + opts.paths, + opts.canonicalSlug, + opts.dirBySlug, + ); if (!canonicalDir) return []; const source: DigestRecord | null = await loadDigest(canonicalDir); if (!source) return []; - const canonicalCues = await readCues(opts.paths, opts.canonicalSlug); + const canonicalCues = await readCues( + opts.paths, + opts.canonicalSlug, + opts.dirBySlug, + ); if (!canonicalCues || canonicalCues.length === 0) return []; const sharedAt = new Date().toISOString(); const outcomes: ShareOutcome[] = []; for (const slug of opts.mirrors) { - const dir = videoDirForSlug(opts.paths, slug); + const dir = videoDirForSlug(opts.paths, slug, opts.dirBySlug); if (!dir) { outcomes.push({ slug, status: "failed", error: "unparseable slug" }); continue; @@ -173,7 +240,7 @@ export async function shareDigestToCluster( outcomes.push({ slug, status: "already-shared" }); continue; } - const cues = await readCues(opts.paths, slug); + const cues = await readCues(opts.paths, slug, opts.dirBySlug); if (!cues || cues.length === 0) { outcomes.push({ slug, status: "no-transcript" }); continue; @@ -237,6 +304,19 @@ export async function shareClusterFromCanonical( mirrors: cluster.videoRefs .map((r) => r.slug) .filter((slug) => slug !== canonicalSlug), + dirBySlug: dirBySlugForCluster(cluster), onLog: opts.onLog, }); } + +// The single-cluster equivalent of DigestClusterPlan.dirBySlug, for callers that +// hold one cluster rather than a whole plan. +export function dirBySlugForCluster( + cluster: DuplicateCluster, +): Map<string, string> { + const map = new Map<string, string>(); + for (const ref of cluster.videoRefs) { + if (ref.videoDir && ref.videoDir !== ref.id) map.set(ref.slug, ref.videoDir); + } + return map; +} diff --git a/common/controller/duplicateShorts.ts b/common/controller/duplicateShorts.ts @@ -536,7 +536,7 @@ export async function detectDuplicateShorts( const refs = slugs .map((slug) => toRef( - (bySlug.get(slug) as Entry).stat, + bySlug.get(slug) as Entry, hasTranscriptBySlug.get(slug) ?? false, ), ) @@ -1111,13 +1111,18 @@ function evalContainment( // Built from the participant index rather than a fingerprint: fingerprints are // released with their block, and VideoStat is the small half of what they held. // `offsetSeconds` / `aligned` are filled in afterwards by the alignment pass. -function toRef(s: VideoStat, hasTranscript: boolean): DuplicateVideoRef { +function toRef(e: Entry, hasTranscript: boolean): DuplicateVideoRef { + const s = e.stat; return { slug: s.slug, channelSlug: s.channelSlug, channel: s.channel, platform: s.platform, id: s.id, + // Only when it differs — see DuplicateVideoRef.videoDir. The detector is the + // one place that has both halves in hand, and dropping the directory here is + // what silently broke digest cluster-sharing for every Rumble mirror. + ...(e.videoDir !== s.id ? { videoDir: e.videoDir } : {}), title: s.title, duration: s.duration, uploadDate: s.uploadDate, diff --git a/common/lib/duplicates.ts b/common/lib/duplicates.ts @@ -66,6 +66,17 @@ export type DuplicateVideoRef = { channel: string; // display name platform: Platform; id: string; // canonical id (may differ from the on-disk dir name) + // The on-disk directory under `<channel>/data/`, recorded ONLY when it differs + // from `id` — which it does for 14.5% of the corpus (every Rumble re-upload: + // 7,870 of the 7,870 `the-quartering-rumble` videos alone). + // + // It is here because `slug` is `${channelSlug}/${id}` and is therefore NOT a + // path. Anything that resolves a member back to its files has to be told the + // difference, and the detector is the only place that cheaply knows it (it + // keys its own scan by directory). Omitted when dir === id so the shipped + // report does not grow a redundant field on the other 85%; readers fall back + // to `id`, which is what every reader assumed before this field existed. + videoDir?: string; title: string; duration: number; // seconds uploadDate: string; // YYYYMMDD