commit 4737bb2a3cc663585d56e4e920a9ebc7a33a1f07
parent f8152a9c3ec6f66a0f04b8a1186fd51a9a0c6d5f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 29 Jul 2026 11:19:40 -0400
Digest cluster sharing was dead for 14.5% of the corpus — the half it was for
`digestBatch` looked the cluster plan up with `${channelSlug}/${directoryName}`.
The plan is keyed `${channelSlug}/${metadataId}`. Those are not the same string
for 11,175 of 77,106 videos, and they are not spread evenly: **7,870 of them are
`the-quartering-rumble`, where dir !== id for every single video** — i.e.
exactly the mirror set this file's own header cites as the win ("5,863
Quartering YouTube↔Rumble mirror pairs", "~11% of the sweep").
Every one of those lookups missed. A miss is silent and looks like success: the
mirror simply isn't recognised as a mirror, so it is generated from scratch and
the log says nothing. The same mistake sat on the other side of the pass —
`videoDirForSlug` built a path out of the id, so even a plan hit would have
opened a directory that does not exist and reported "no transcript".
The fix is to record what the detector already knows and was throwing away: it
keys its own scan by directory and holds both halves in `Entry`. So
`DuplicateVideoRef` gains `videoDir`, written ONLY when it differs from `id` (no
redundant field on the other 85%, and readers fall back to `id`, which is what
they all assumed before). The plan derives both directions from it, and
`planSlugForDir` is the one place the batch converts a directory back to a slug.
An old report with no `videoDir` keeps today's behaviour rather than resolving
to nothing — so this degrades to the status quo instead of failing.
Found while pricing the sweep, not by a test: nothing asserted on a corpus where
the two identifiers disagree. There is one now.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
5 files changed, 301 insertions(+), 12 deletions(-)
diff --git a/common/controller/digestBatch.ts b/common/controller/digestBatch.ts
@@ -34,6 +34,7 @@ import { CUES_JSON_FILENAME } from "../lib/videoStatus";
import { digestVideo } from "./digestVideo";
import {
buildDigestClusterPlan,
+ planSlugForDir,
shareDigestToCluster,
type DigestClusterPlan,
} from "./digestSharing";
@@ -243,7 +244,12 @@ export async function runDigestBatch(
const attempted = new Set<string>();
let cursor = 0;
- const slugOf = (id: string): string => `${opts.channelSlug}/${id}`;
+ // `candidate.id` is a DIRECTORY name (it comes from readdir), and a slug is
+ // `${channelSlug}/${metadataId}`. Those disagree for 14.5% of the corpus —
+ // every Rumble re-upload — so resolving one to the other through the plan is
+ // what makes the cluster lookups below hit at all. See planSlugForDir.
+ const slugOf = (videoDir: string): string =>
+ planSlugForDir(clusterPlan, opts.channelSlug, videoDir);
const next = async (): Promise<Candidate | null> => {
while (cursor < candidates.length) {
@@ -342,6 +348,7 @@ export async function runDigestBatch(
clusterId: role.clusterId,
canonicalSlug: slugOf(candidate.id),
mirrors: role.mirrors,
+ dirBySlug: clusterPlan?.dirBySlug,
onLog: log,
});
for (const o of outcomes) {
diff --git a/common/controller/digestSharing.test.ts b/common/controller/digestSharing.test.ts
@@ -0,0 +1,186 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import {
+ DUPLICATES_FILENAME,
+ DUPLICATE_REPORT_VERSION,
+ type DuplicateCluster,
+ type DuplicateVideoRef,
+} from "../lib/duplicates";
+import {
+ buildDigestClusterPlan,
+ dirBySlugForCluster,
+ planSlugForDir,
+} from "./digestSharing";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/digestSharing.test.ts
+
+async function withPaths(fn: (paths: Paths) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-digest-sharing-"));
+ const paths = {
+ transcriptsDir: dir,
+ channelsDir: path.join(dir, "channels"),
+ } as Paths;
+ await mkdir(paths.transcriptsDir, { recursive: true });
+ try {
+ await fn(paths);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+function ref(
+ channelSlug: string,
+ id: string,
+ extra: Partial<DuplicateVideoRef> = {},
+): DuplicateVideoRef {
+ return {
+ slug: `${channelSlug}/${id}`,
+ channelSlug,
+ channel: channelSlug,
+ platform: "youtube",
+ id,
+ title: "t",
+ duration: 600,
+ uploadDate: "20240101",
+ hasTranscript: true,
+ ...extra,
+ };
+}
+
+async function writeReport(
+ paths: Paths,
+ clusters: DuplicateCluster[],
+): Promise<void> {
+ await writeFile(
+ path.join(paths.transcriptsDir, DUPLICATES_FILENAME),
+ JSON.stringify({
+ version: DUPLICATE_REPORT_VERSION,
+ generatedAt: new Date(0).toISOString(),
+ runConfig: {
+ thresholdSeconds: null,
+ durationToleranceSeconds: 2,
+ nearThreshold: 0.35,
+ containmentThreshold: 0.8,
+ shingleSize: 5,
+ },
+ totals: { videosScanned: 2, clusters: clusters.length, videosInClusters: 2 },
+ clusters,
+ }),
+ );
+}
+
+function cluster(
+ clusterId: string,
+ videoRefs: DuplicateVideoRef[],
+ extra: Partial<DuplicateCluster> = {},
+): DuplicateCluster {
+ return {
+ clusterId,
+ matchKind: "transcript-near",
+ score: 0.9,
+ contained: false,
+ durationBucket: 600,
+ crossPlatform: true,
+ crossChannel: true,
+ videoRefs,
+ canonicalSlug: videoRefs[0].slug,
+ ...extra,
+ };
+}
+
+// The bug this whole mapping exists for: a slug is `${channelSlug}/${id}` while
+// the batch runner enumerates DIRECTORIES, and those disagree for 14.5% of the
+// real corpus (every Rumble re-upload). Before the mapping, the lookup below
+// missed and every such mirror was regenerated instead of shared.
+test("plan maps a directory name to its slug when the two differ", async () => {
+ await withPaths(async (paths) => {
+ await writeReport(paths, [
+ cluster("c1", [
+ ref("chan-yt", "canon1"),
+ // The Rumble shape: on-disk dir `v1007ay`, metadata id `vxe1ae`.
+ ref("chan-rumble", "vxe1ae", { videoDir: "v1007ay" }),
+ ]),
+ ]);
+ const plan = await buildDigestClusterPlan(paths, { overrides: null });
+
+ assert.equal(
+ planSlugForDir(plan, "chan-rumble", "v1007ay"),
+ "chan-rumble/vxe1ae",
+ "the directory name must resolve to the metadata-id slug",
+ );
+ assert.equal(
+ plan.bySlug.get(planSlugForDir(plan, "chan-rumble", "v1007ay"))?.kind,
+ "mirror",
+ "and that slug must then hit the plan as a mirror",
+ );
+ assert.equal(
+ plan.dirBySlug.get("chan-rumble/vxe1ae"),
+ "v1007ay",
+ "the reverse direction is what lets the sharing pass open the mirror's files",
+ );
+ });
+});
+
+test("a member whose dir equals its id is absent from the mapping", async () => {
+ await withPaths(async (paths) => {
+ await writeReport(paths, [
+ cluster("c1", [ref("chan-a", "same1"), ref("chan-b", "same2")]),
+ ]);
+ const plan = await buildDigestClusterPlan(paths, { overrides: null });
+ assert.equal(plan.dirBySlug.size, 0);
+ assert.equal(plan.slugByDir.size, 0);
+ // The fallback is the old behaviour, and it is the correct one here.
+ assert.equal(planSlugForDir(plan, "chan-a", "same1"), "chan-a/same1");
+ });
+});
+
+// A report written before `videoDir` existed must not regress: it keeps the
+// pre-mapping behaviour rather than resolving to nothing.
+test("an old report with no videoDir falls back to the id", async () => {
+ await withPaths(async (paths) => {
+ await writeReport(paths, [
+ cluster("c1", [ref("chan-a", "aaa"), ref("chan-b", "bbb")]),
+ ]);
+ const plan = await buildDigestClusterPlan(paths, { overrides: null });
+ assert.equal(planSlugForDir(plan, "chan-b", "bbb"), "chan-b/bbb");
+ assert.equal(planSlugForDir(null, "chan-b", "bbb"), "chan-b/bbb");
+ });
+});
+
+test("dirBySlugForCluster carries only the members that differ", () => {
+ const c = cluster("c1", [
+ ref("chan-yt", "canon1"),
+ ref("chan-rumble", "vxe1ae", { videoDir: "v1007ay" }),
+ ref("chan-odysee", "plain", { videoDir: "plain" }),
+ ]);
+ const map = dirBySlugForCluster(c);
+ assert.deepEqual([...map], [["chan-rumble/vxe1ae", "v1007ay"]]);
+});
+
+// The directory mapping is recorded for EVERY cluster, including ones that
+// share nothing — their members are still generated, and a caller holding the
+// plan still needs a correct path for them.
+test("an unshareable cluster still contributes its directory mapping", async () => {
+ await withPaths(async (paths) => {
+ await writeReport(paths, [
+ cluster(
+ "c1",
+ [
+ ref("chan-yt", "canon1"),
+ ref("chan-rumble", "vxe1ae", { videoDir: "v1007ay" }),
+ ],
+ // needsReview and unconfirmed → clusterMaySharePartial fails closed.
+ { needsReview: true },
+ ),
+ ]);
+ const plan = await buildDigestClusterPlan(paths, { overrides: null });
+ assert.equal(plan.bySlug.size, 0, "it shares nothing");
+ assert.equal(plan.independentClusters, 1);
+ assert.equal(plan.dirBySlug.get("chan-rumble/vxe1ae"), "v1007ay");
+ });
+});
diff --git a/common/controller/digestSharing.ts b/common/controller/digestSharing.ts
@@ -46,6 +46,22 @@ export type DigestClusterPlan = {
// `${channelSlug}/${id}` → role. Videos absent from the map are in no cluster
// and are generated normally, which is the overwhelming majority.
bySlug: Map<string, DigestClusterRole>;
+ // The two directions of the id ↔ on-disk-directory mapping, populated ONLY for
+ // cluster members whose directory name is not their metadata id.
+ //
+ // These exist because a slug is `${channelSlug}/${id}` and is not a path,
+ // while every caller that reaches this plan starts from one side or the other:
+ // the batch runner enumerates DIRECTORIES off disk, and the sharing pass has
+ // SLUGS from the report and has to open their files. 14.5% of the corpus has
+ // dir !== id, and it is not spread evenly — it is 100% of `the-quartering-
+ // rumble` (7,870 videos), i.e. exactly the mirror set cluster-sharing was
+ // written to exploit. Keying either lookup with the wrong half misses silently
+ // and simply regenerates the mirror, which is why this went unnoticed.
+ //
+ // Empty for a report written before `DuplicateVideoRef.videoDir` existed;
+ // lookups fall back to treating the id as the directory, the old behaviour.
+ slugByDir: Map<string, string>; // `${channelSlug}/${videoDir}` → slug
+ dirBySlug: Map<string, string>; // slug → videoDir (bare name, not a path)
// Clusters that exist but share nothing (contained, or marked not-a-duplicate).
// Their members are absent from bySlug and are each generated independently —
// a clip is a different artifact and deserves its own digest.
@@ -54,7 +70,25 @@ export type DigestClusterPlan = {
};
export function emptyDigestClusterPlan(): DigestClusterPlan {
- return { bySlug: new Map(), independentClusters: 0, clusters: 0 };
+ return {
+ bySlug: new Map(),
+ slugByDir: new Map(),
+ dirBySlug: new Map(),
+ independentClusters: 0,
+ clusters: 0,
+ };
+}
+
+// The batch runner enumerates on-disk directories; the plan is keyed by slug.
+// Falls back to the directory name as the id, which is correct for the ~85.5%
+// where they agree and is what the code did before the mapping existed.
+export function planSlugForDir(
+ plan: DigestClusterPlan | null | undefined,
+ channelSlug: string,
+ videoDir: string,
+): string {
+ const key = `${channelSlug}/${videoDir}`;
+ return plan?.slugByDir.get(key) ?? key;
}
// Build the plan from the last detection run. Absent report → an empty plan, so
@@ -71,6 +105,14 @@ export async function buildDigestClusterPlan(
plan.clusters = report.clusters.length;
for (const cluster of report.clusters) {
+ // Recorded for EVERY cluster, including the ones that share nothing: an
+ // independent cluster's members are still generated, and a caller that
+ // reaches the plan at all deserves a correct path for them.
+ for (const ref of cluster.videoRefs) {
+ if (!ref.videoDir || ref.videoDir === ref.id) continue;
+ plan.dirBySlug.set(ref.slug, ref.videoDir);
+ plan.slugByDir.set(`${ref.channelSlug}/${ref.videoDir}`, ref.slug);
+ }
const canonicalSlug = resolveCanonicalSlug(cluster, overrides);
// null → a human said "not a duplicate": every member stands alone.
// The overrides also carry the `confirmed` flag that is the ONLY thing
@@ -103,19 +145,31 @@ export async function buildDigestClusterPlan(
return plan;
}
-function videoDirForSlug(paths: Paths, slug: string): string | null {
+// A slug is `${channelSlug}/${id}`, and an id is NOT always the directory name —
+// see DigestClusterPlan.slugByDir. `dirBySlug` carries the exceptions; without
+// it this resolves the 14.5% of the corpus with dir !== id to a path that does
+// not exist, which reads as "no transcript" and silently declines to share.
+function videoDirForSlug(
+ paths: Paths,
+ slug: string,
+ dirBySlug?: ReadonlyMap<string, string>,
+): string | null {
const at = slug.indexOf("/");
if (at <= 0) return null;
return path.join(
paths.channelsDir,
slug.slice(0, at),
"data",
- slug.slice(at + 1),
+ dirBySlug?.get(slug) ?? slug.slice(at + 1),
);
}
-async function readCues(paths: Paths, slug: string): Promise<Cue[] | null> {
- const dir = videoDirForSlug(paths, slug);
+async function readCues(
+ paths: Paths,
+ slug: string,
+ dirBySlug?: ReadonlyMap<string, string>,
+): Promise<Cue[] | null> {
+ const dir = videoDirForSlug(paths, slug, dirBySlug);
if (!dir) return null;
const t = await readNormalizedTranscript(path.join(dir, CUES_JSON_FILENAME));
return t?.cues ?? null;
@@ -133,6 +187,11 @@ export type ShareDigestOptions = {
clusterId: string;
canonicalSlug: string;
mirrors: string[];
+ // slug → on-disk directory, for members where the two differ. Take it from
+ // DigestClusterPlan.dirBySlug (or build it from the cluster's own refs).
+ // Omitting it is not an error, but every member with dir !== id will resolve
+ // to a nonexistent path and be reported "no-transcript".
+ dirBySlug?: ReadonlyMap<string, string>;
toleranceSeconds?: number;
onLog?: (msg: string) => void;
};
@@ -146,17 +205,25 @@ export async function shareDigestToCluster(
const log = opts.onLog ?? (() => {});
const tolerance =
opts.toleranceSeconds ?? DEFAULT_ALIGNMENT_TOLERANCE_SECONDS;
- const canonicalDir = videoDirForSlug(opts.paths, opts.canonicalSlug);
+ const canonicalDir = videoDirForSlug(
+ opts.paths,
+ opts.canonicalSlug,
+ opts.dirBySlug,
+ );
if (!canonicalDir) return [];
const source: DigestRecord | null = await loadDigest(canonicalDir);
if (!source) return [];
- const canonicalCues = await readCues(opts.paths, opts.canonicalSlug);
+ const canonicalCues = await readCues(
+ opts.paths,
+ opts.canonicalSlug,
+ opts.dirBySlug,
+ );
if (!canonicalCues || canonicalCues.length === 0) return [];
const sharedAt = new Date().toISOString();
const outcomes: ShareOutcome[] = [];
for (const slug of opts.mirrors) {
- const dir = videoDirForSlug(opts.paths, slug);
+ const dir = videoDirForSlug(opts.paths, slug, opts.dirBySlug);
if (!dir) {
outcomes.push({ slug, status: "failed", error: "unparseable slug" });
continue;
@@ -173,7 +240,7 @@ export async function shareDigestToCluster(
outcomes.push({ slug, status: "already-shared" });
continue;
}
- const cues = await readCues(opts.paths, slug);
+ const cues = await readCues(opts.paths, slug, opts.dirBySlug);
if (!cues || cues.length === 0) {
outcomes.push({ slug, status: "no-transcript" });
continue;
@@ -237,6 +304,19 @@ export async function shareClusterFromCanonical(
mirrors: cluster.videoRefs
.map((r) => r.slug)
.filter((slug) => slug !== canonicalSlug),
+ dirBySlug: dirBySlugForCluster(cluster),
onLog: opts.onLog,
});
}
+
+// The single-cluster equivalent of DigestClusterPlan.dirBySlug, for callers that
+// hold one cluster rather than a whole plan.
+export function dirBySlugForCluster(
+ cluster: DuplicateCluster,
+): Map<string, string> {
+ const map = new Map<string, string>();
+ for (const ref of cluster.videoRefs) {
+ if (ref.videoDir && ref.videoDir !== ref.id) map.set(ref.slug, ref.videoDir);
+ }
+ return map;
+}
diff --git a/common/controller/duplicateShorts.ts b/common/controller/duplicateShorts.ts
@@ -536,7 +536,7 @@ export async function detectDuplicateShorts(
const refs = slugs
.map((slug) =>
toRef(
- (bySlug.get(slug) as Entry).stat,
+ bySlug.get(slug) as Entry,
hasTranscriptBySlug.get(slug) ?? false,
),
)
@@ -1111,13 +1111,18 @@ function evalContainment(
// Built from the participant index rather than a fingerprint: fingerprints are
// released with their block, and VideoStat is the small half of what they held.
// `offsetSeconds` / `aligned` are filled in afterwards by the alignment pass.
-function toRef(s: VideoStat, hasTranscript: boolean): DuplicateVideoRef {
+function toRef(e: Entry, hasTranscript: boolean): DuplicateVideoRef {
+ const s = e.stat;
return {
slug: s.slug,
channelSlug: s.channelSlug,
channel: s.channel,
platform: s.platform,
id: s.id,
+ // Only when it differs — see DuplicateVideoRef.videoDir. The detector is the
+ // one place that has both halves in hand, and dropping the directory here is
+ // what silently broke digest cluster-sharing for every Rumble mirror.
+ ...(e.videoDir !== s.id ? { videoDir: e.videoDir } : {}),
title: s.title,
duration: s.duration,
uploadDate: s.uploadDate,
diff --git a/common/lib/duplicates.ts b/common/lib/duplicates.ts
@@ -66,6 +66,17 @@ export type DuplicateVideoRef = {
channel: string; // display name
platform: Platform;
id: string; // canonical id (may differ from the on-disk dir name)
+ // The on-disk directory under `<channel>/data/`, recorded ONLY when it differs
+ // from `id` — which it does for 14.5% of the corpus (every Rumble re-upload:
+ // 7,870 of the 7,870 `the-quartering-rumble` videos alone).
+ //
+ // It is here because `slug` is `${channelSlug}/${id}` and is therefore NOT a
+ // path. Anything that resolves a member back to its files has to be told the
+ // difference, and the detector is the only place that cheaply knows it (it
+ // keys its own scan by directory). Omitted when dir === id so the shipped
+ // report does not grow a redundant field on the other 85%; readers fall back
+ // to `id`, which is what every reader assumed before this field existed.
+ videoDir?: string;
title: string;
duration: number; // seconds
uploadDate: string; // YYYYMMDD