commit b2e8a0568f6d3987ae13346ac76a9662e5945fa4
parent 25962b84275c0ee1d87495c66a4f483b0fef1c21
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 4 Jun 2026 20:49:31 -0400
Duplicate detection
Diffstat:
19 files changed, 1608 insertions(+), 8 deletions(-)
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -14,9 +14,14 @@
// only that site's data. Checked-in static assets in public/ are left intact.
import path from "node:path";
-import { cp, mkdir, rm, readdir, access } from "node:fs/promises";
+import { cp, mkdir, rm, readdir, access, readFile, writeFile } from "node:fs/promises";
import { getPaths } from "../lib/paths";
import { getSite } from "../lib/site";
+import {
+ DUPLICATES_FILENAME,
+ filterClusterToChannels,
+ type DuplicateReport,
+} from "../lib/duplicates";
async function exists(p: string): Promise<boolean> {
try {
@@ -99,6 +104,32 @@ async function main(): Promise<void> {
await cp(templatesSrc, templatesDest);
}
+ // --- duplicate-shorts report (global → site-filtered) ---
+ // The detector writes one global duplicates.json over the whole channel pool;
+ // each site only serves its own channels, so filter clusters to the site's
+ // members (dropping out-of-site refs and clusters that fall below 2 members)
+ // before serving. Absent report → no file, and the page shows its empty state.
+ const dupSrc = path.join(paths.transcriptsDir, DUPLICATES_FILENAME);
+ const dupDest = path.join(paths.exportPublicDir, DUPLICATES_FILENAME);
+ await rm(dupDest, { force: true });
+ if (await exists(dupSrc)) {
+ const report = JSON.parse(await readFile(dupSrc, "utf8")) as DuplicateReport;
+ const memberSet = new Set(memberSlugs);
+ const clusters = report.clusters
+ .map((c) => filterClusterToChannels(c, memberSet))
+ .filter((c): c is NonNullable<typeof c> => c !== null);
+ const filtered: DuplicateReport = {
+ ...report,
+ totals: {
+ ...report.totals,
+ clusters: clusters.length,
+ videosInClusters: clusters.reduce((n, c) => n + c.videoRefs.length, 0),
+ },
+ clusters,
+ };
+ await writeFile(dupDest, JSON.stringify(filtered));
+ }
+
const channelDirs = (await readdir(paths.exportTranscriptsDir).catch(
() => [] as string[],
)).length;
diff --git a/common/bin/duplicate-shorts.ts b/common/bin/duplicate-shorts.ts
@@ -0,0 +1,33 @@
+#!/usr/bin/env tsx
+// On-demand duplicate-shorts detection. Runs AFTER build:index + build:stats
+// (it reads the cues + statsByPath those populate), so it is deliberately not
+// chained into build:data.
+//
+// Flags:
+// --threshold N duration cutoff in seconds (default 180)
+// --all-durations no length filter; enables containment matching
+// --near F Jaccard near-duplicate threshold (default 0.6)
+// --tolerance N duration bucketing window in seconds (default 2)
+import { getPaths } from "../lib/paths";
+import { detectDuplicateShorts } from "../controller/duplicateShorts";
+import { parseFlags } from "./_parseFlags";
+
+const flags = parseFlags(process.argv.slice(2));
+
+const thresholdSeconds =
+ flags["all-durations"] === "true"
+ ? null
+ : flags.threshold !== undefined
+ ? Number(flags.threshold)
+ : undefined;
+
+detectDuplicateShorts({
+ paths: getPaths(),
+ thresholdSeconds,
+ nearThreshold: flags.near !== undefined ? Number(flags.near) : undefined,
+ durationToleranceSeconds:
+ flags.tolerance !== undefined ? Number(flags.tolerance) : undefined,
+}).catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/common/controller/duplicateShorts.ts b/common/controller/duplicateShorts.ts
@@ -0,0 +1,521 @@
+// Cross-platform duplicate "shorts" detection — a global pass over every
+// channel's videos (NOT per-channel, since duplicates are cross-channel and
+// cross-platform by definition).
+//
+// Hybrid cascade:
+// Phase 1 (cheap) — pre-cluster candidates by rounded duration, the only
+// signal stable across platforms and re-titles. Consumes
+// the already-aggregated VideoStat index (LMDB statsByPath,
+// populated by buildStats) rather than re-walking dirs.
+// Phase 2 (confirm)— for each candidate pair, compare transcript content
+// (read from the LMDB `cues` sub-db that buildIndex
+// populates): exact text hash → 5-gram Jaccard →
+// (all-durations only) containment for "short is a clip of
+// a longer video". Falls back to a metadata-only match when
+// a transcript is missing/empty.
+//
+// Output is a single global transcripts/duplicates.json. Flag-only: nothing is
+// merged or deleted. Requires build:index (cues) + build:stats (statsByPath) to
+// have run first.
+
+import path from "node:path";
+import { mkdir, rename, writeFile, readFile } from "node:fs/promises";
+import { createHash } from "node:crypto";
+import { open } from "lmdb";
+import pLimit from "p-limit";
+import type { Paths } from "../lib/paths";
+import type { VideoStat } from "../lib/stats";
+import { STATS_SCHEMA_VERSION } from "../lib/stats";
+import type { Cue } from "../lib/vtt";
+import { parseWhisper } from "../lib/whisper";
+import { readVideoFiles, pickIndexTranscript } from "../lib/videoStatus";
+import {
+ DEFAULT_CONTAINMENT_THRESHOLD,
+ DEFAULT_DURATION_TOLERANCE_SECONDS,
+ DEFAULT_NEAR_THRESHOLD,
+ DEFAULT_SHINGLE_SIZE,
+ DEFAULT_SHORT_THRESHOLD_SECONDS,
+ DUPLICATES_FILENAME,
+ DUPLICATE_REPORT_VERSION,
+ UnionFind,
+ comparisonText,
+ containment,
+ jaccard,
+ shingles,
+ strongerMatch,
+ type DuplicateCluster,
+ type DuplicateMatchKind,
+ type DuplicateReport,
+ type DuplicateRunConfig,
+ type DuplicateVideoRef,
+} from "../lib/duplicates";
+
+type PathKey = [string, string];
+type IndexKey = [string, string, string];
+type StatsRecord = { metaMs: number; stat: VideoStat };
+
+export type DetectDuplicateShortsOptions = {
+ paths: Paths;
+ // Duration cutoff for what counts as a "short". `null` === all durations
+ // (no length filter; enables containment matching against long videos).
+ thresholdSeconds?: number | null;
+ durationToleranceSeconds?: number;
+ nearThreshold?: number;
+ containmentThreshold?: number;
+ shingleSize?: number;
+ onLog?: (msg: string) => void;
+ signal?: AbortSignal;
+};
+
+// Per-participant transcript fingerprint, computed once and reused across pairs.
+type Fingerprint = {
+ stat: VideoStat;
+ hasTranscript: boolean;
+ hash: string | null;
+ shingleSet: Set<string>;
+};
+
+type Entry = { stat: VideoStat; videoDir: string };
+
+type PairResult = {
+ a: string; // slug
+ b: string; // slug
+ kind: DuplicateMatchKind;
+ score: number | null;
+ contained: boolean;
+};
+
+export async function detectDuplicateShorts(
+ opts: DetectDuplicateShortsOptions,
+): Promise<DuplicateReport> {
+ const log = opts.onLog ?? ((m: string) => console.log(m));
+ const thresholdSeconds =
+ opts.thresholdSeconds === undefined
+ ? DEFAULT_SHORT_THRESHOLD_SECONDS
+ : opts.thresholdSeconds;
+ const W = opts.durationToleranceSeconds ?? DEFAULT_DURATION_TOLERANCE_SECONDS;
+ const nearThreshold = opts.nearThreshold ?? DEFAULT_NEAR_THRESHOLD;
+ const containmentThreshold =
+ opts.containmentThreshold ?? DEFAULT_CONTAINMENT_THRESHOLD;
+ const shingleSize = opts.shingleSize ?? DEFAULT_SHINGLE_SIZE;
+
+ const runConfig: DuplicateRunConfig = {
+ thresholdSeconds,
+ durationToleranceSeconds: W,
+ nearThreshold,
+ containmentThreshold,
+ shingleSize,
+ };
+
+ // ---- Open the index: statsByPath (Phase-1 input) + cues (Phase-2 input) ---
+ const root = open({ path: opts.paths.lmdbPath, maxDbs: 12, compression: true });
+ const statsByPath = root.openDB<StatsRecord, PathKey>({
+ name: "statsByPath",
+ encoding: "msgpack",
+ });
+ const cuesDb = root.openDB<Cue[], IndexKey>({
+ name: "cues",
+ encoding: "msgpack",
+ });
+ const meta = root.openDB<unknown, string>({
+ name: "statsMeta",
+ encoding: "msgpack",
+ });
+
+ const storedSchema = meta.get("schema") as number | undefined;
+ if (storedSchema !== STATS_SCHEMA_VERSION) {
+ log(
+ `Warning: stats cache schema ${storedSchema ?? "<none>"} != ${STATS_SCHEMA_VERSION}; run build:stats first for accurate results.`,
+ );
+ }
+
+ const all: Entry[] = [];
+ for (const { key, value } of statsByPath.getRange()) {
+ // statsByPath is keyed [channelSlug, videoDir]; keep videoDir so the
+ // Phase-2 fallback can locate the raw transcript file on disk.
+ const videoDir = (key as PathKey)[1];
+ all.push({ stat: (value as StatsRecord).stat, videoDir });
+ }
+
+ const videosScanned = all.length;
+ const isShort = (d: number) =>
+ d > 0 && (thresholdSeconds === null || d <= thresholdSeconds);
+ const includeContainment = thresholdSeconds === null;
+ // In all-durations mode every video participates (long videos are the
+ // containment targets); otherwise only shorts do.
+ const participants = all.filter((e) =>
+ thresholdSeconds === null ? e.stat.duration > 0 : isShort(e.stat.duration),
+ );
+ log(
+ `Scanned ${videosScanned} videos; ${participants.length} in scope ` +
+ `(threshold=${thresholdSeconds === null ? "all" : `${thresholdSeconds}s`}).`,
+ );
+
+ // ---- Phase 1: bucket by rounded duration ---------------------------------
+ const bucketOf = (d: number) => Math.round(d / W);
+ const buckets = new Map<number, Entry[]>();
+ for (const e of participants) {
+ const b = bucketOf(e.stat.duration);
+ const arr = buckets.get(b);
+ if (arr) arr.push(e);
+ else buckets.set(b, [e]);
+ }
+
+ // Candidate pairs. Same or adjacent duration bucket + relative-duration guard,
+ // or (all-durations mode) short/longer pairs eligible for containment.
+ const maxDelta = (d: number) => Math.max(W, Math.ceil(0.02 * d));
+ type Candidate = { a: Entry; b: Entry; containmentOnly: boolean };
+ const candidates: Candidate[] = [];
+ const sortedBuckets = [...buckets.keys()].sort((x, y) => x - y);
+ for (const b of sortedBuckets) {
+ const here = buckets.get(b) as Entry[];
+ const next = buckets.get(b + 1) ?? [];
+ for (let i = 0; i < here.length; i++) {
+ for (let j = i + 1; j < here.length; j++) {
+ if (durationsClose(here[i], here[j], maxDelta))
+ candidates.push({ a: here[i], b: here[j], containmentOnly: false });
+ }
+ }
+ for (const x of here) {
+ for (const y of next) {
+ if (durationsClose(x, y, maxDelta))
+ candidates.push({ a: x, b: y, containmentOnly: false });
+ }
+ }
+ }
+
+ // Containment candidates: a short paired with any meaningfully-longer video.
+ if (includeContainment) {
+ const shortsForContainment = participants.filter(
+ (e) =>
+ e.stat.duration > 0 &&
+ e.stat.duration <= DEFAULT_SHORT_THRESHOLD_SECONDS,
+ );
+ for (const s of shortsForContainment) {
+ for (const v of participants) {
+ if (v === s) continue;
+ if (v.stat.duration >= s.stat.duration * 1.5)
+ candidates.push({ a: s, b: v, containmentOnly: true });
+ }
+ }
+ }
+ log(
+ `Phase 1: ${buckets.size} duration buckets → ${candidates.length} candidate pairs.`,
+ );
+
+ // ---- Build fingerprints for every video in a candidate ------------------
+ // Primary source is the LMDB `cues` sub-db. But parseVtt only extracts text
+ // from YouTube karaoke-tagged cues, so plain-VTT transcripts (most non-YouTube
+ // captions, whisper-as-vtt, manual subs) store as zero cues and would look
+ // transcript-less. For those, fall back to reading the raw transcript file off
+ // disk and extracting plain text — for comparison only, not stored anywhere.
+ const needed = new Map<string, Entry>();
+ for (const c of candidates) {
+ needed.set(c.a.stat.slug, c.a);
+ needed.set(c.b.stat.slug, c.b);
+ }
+ const fingerprints = new Map<string, Fingerprint>();
+ const rawFallbacks: Entry[] = [];
+ for (const e of needed.values()) {
+ opts.signal?.throwIfAborted();
+ const s = e.stat;
+ const cues = cuesDb.get([s.uploadDate, s.channelSlug, s.id]);
+ if (cues && cues.length > 0) {
+ fingerprints.set(s.slug, fingerprintFrom(s, cues, shingleSize));
+ } else {
+ rawFallbacks.push(e);
+ }
+ }
+ await root.close();
+
+ let rawHits = 0;
+ if (rawFallbacks.length > 0) {
+ const limit = pLimit(16);
+ await Promise.all(
+ rawFallbacks.map((e) =>
+ limit(async () => {
+ opts.signal?.throwIfAborted();
+ const cues = await readRawCues(opts.paths.channelsDir, e);
+ if (cues && cues.length > 0) rawHits++;
+ fingerprints.set(e.stat.slug, fingerprintFrom(e.stat, cues, shingleSize));
+ }),
+ ),
+ );
+ }
+ log(
+ `Phase 2: fingerprinted ${fingerprints.size} candidate videos ` +
+ `(${rawFallbacks.length} missing LMDB cues; ${rawHits} recovered from raw transcripts).`,
+ );
+
+ // ---- Phase 2: evaluate each candidate pair -------------------------------
+ const matches: PairResult[] = [];
+ for (const c of candidates) {
+ opts.signal?.throwIfAborted();
+ const fa = fingerprints.get(c.a.stat.slug) as Fingerprint;
+ const fb = fingerprints.get(c.b.stat.slug) as Fingerprint;
+ const res = c.containmentOnly
+ ? evalContainment(fa, fb, containmentThreshold)
+ : evalSimilar(fa, fb, nearThreshold);
+ if (res) matches.push({ a: c.a.stat.slug, b: c.b.stat.slug, ...res });
+ }
+
+ // ---- Cluster matched pairs (union-find) ----------------------------------
+ const uf = new UnionFind();
+ for (const m of matches) uf.union(m.a, m.b);
+ const components = uf.groups().filter((g) => g.length >= 2);
+ const componentRoot = new Map<string, string>();
+ for (const g of components) for (const slug of g) componentRoot.set(slug, g[0]);
+
+ type Agg = {
+ matchKind: DuplicateMatchKind;
+ score: number | null;
+ contained: boolean;
+ };
+ const aggByRoot = new Map<string, Agg>();
+ for (const m of matches) {
+ const root2 = componentRoot.get(m.a);
+ if (!root2) continue;
+ const cur = aggByRoot.get(root2);
+ if (!cur) {
+ aggByRoot.set(root2, {
+ matchKind: m.kind,
+ score: m.score,
+ contained: m.contained,
+ });
+ } else {
+ cur.matchKind = strongerMatch(cur.matchKind, m.kind);
+ cur.score = maxScore(cur.score, m.score);
+ cur.contained = cur.contained || m.contained;
+ }
+ }
+
+ const clusters: DuplicateCluster[] = components.map((slugs) => {
+ const agg = aggByRoot.get(slugs[0]) as Agg;
+ const refs = slugs
+ .map((slug) => toRef(fingerprints.get(slug) as Fingerprint))
+ .sort((x, y) => x.slug.localeCompare(y.slug));
+ const platforms = new Set(refs.map((r) => r.platform));
+ const channels = new Set(refs.map((r) => r.channelSlug));
+ const minDuration = Math.min(...refs.map((r) => r.duration));
+ return {
+ clusterId: sha1(refs.map((r) => r.slug).join("\n")),
+ matchKind: agg.matchKind,
+ score: agg.score,
+ contained: agg.contained,
+ durationBucket: Math.round(minDuration),
+ crossPlatform: platforms.size > 1,
+ crossChannel: channels.size > 1,
+ videoRefs: refs,
+ };
+ });
+
+ const rank: Record<DuplicateMatchKind, number> = {
+ "transcript-exact": 2,
+ "transcript-near": 1,
+ };
+ clusters.sort(
+ (a, b) =>
+ rank[b.matchKind] - rank[a.matchKind] ||
+ b.videoRefs.length - a.videoRefs.length ||
+ (b.score ?? 0) - (a.score ?? 0) ||
+ a.clusterId.localeCompare(b.clusterId),
+ );
+
+ const videosInClusters = clusters.reduce((n, c) => n + c.videoRefs.length, 0);
+ const report: DuplicateReport = {
+ version: DUPLICATE_REPORT_VERSION,
+ generatedAt: new Date().toISOString(),
+ runConfig,
+ totals: { videosScanned, clusters: clusters.length, videosInClusters },
+ clusters,
+ };
+
+ const outPath = path.join(opts.paths.transcriptsDir, DUPLICATES_FILENAME);
+ await mkdir(path.dirname(outPath), { recursive: true });
+ const tmp = `${outPath}.tmp-${process.pid}`;
+ await writeFile(tmp, JSON.stringify(report));
+ await rename(tmp, outPath);
+ log(
+ `Done: ${clusters.length} duplicate cluster(s) over ${videosInClusters} video(s) → ${DUPLICATES_FILENAME}.`,
+ );
+ return report;
+}
+
+export async function readDuplicateReport(
+ paths: Paths,
+): Promise<DuplicateReport | null> {
+ try {
+ const raw = await readFile(
+ path.join(paths.transcriptsDir, DUPLICATES_FILENAME),
+ "utf8",
+ );
+ const parsed = JSON.parse(raw) as DuplicateReport;
+ if (typeof parsed.version !== "number") return null;
+ return parsed;
+ } catch {
+ return null;
+ }
+}
+
+// --- helpers ---------------------------------------------------------------
+
+function durationsClose(
+ a: Entry,
+ b: Entry,
+ maxDelta: (d: number) => number,
+): boolean {
+ const lo = Math.min(a.stat.duration, b.stat.duration);
+ return Math.abs(a.stat.duration - b.stat.duration) <= maxDelta(lo);
+}
+
+function fingerprintFrom(
+ stat: VideoStat,
+ cues: Cue[] | undefined | null,
+ shingleSize: number,
+): Fingerprint {
+ if (!cues || cues.length === 0) {
+ return { stat, hasTranscript: false, hash: null, shingleSet: new Set() };
+ }
+ const text = comparisonText(cues);
+ return {
+ stat,
+ hasTranscript: text.length > 0,
+ hash: text.length > 0 ? sha1(text) : null,
+ shingleSet: shingles(text, shingleSize),
+ };
+}
+
+// Fallback transcript read for videos whose LMDB cues are empty (plain VTT that
+// parseVtt skips because it has no karaoke timing tags). Reads the raw
+// transcript file and extracts plain text for comparison only — nothing is
+// persisted. Returns null when there's no transcript file.
+async function readRawCues(
+ channelsDir: string,
+ e: Entry,
+): Promise<Cue[] | null> {
+ const dir = path.join(channelsDir, e.stat.channelSlug, "data", e.videoDir);
+ let picked: ReturnType<typeof pickIndexTranscript>;
+ try {
+ picked = pickIndexTranscript(await readVideoFiles(dir));
+ } catch {
+ return null;
+ }
+ if (!picked) return null;
+ let raw: string;
+ try {
+ raw = await readFile(path.join(dir, picked.filename), "utf8");
+ } catch {
+ return null;
+ }
+ return picked.kind === "whisper" ? parseWhisper(raw) : lenientVttCues(raw);
+}
+
+// Lenient VTT text extraction: unlike parseVtt (which only keeps YouTube
+// karaoke-tagged lines), this collects text from every cue body, strips all
+// tags/entities, and consecutive-dedups rolling captions. Good enough for a
+// comparison fingerprint; set-based shingles absorb residual repetition.
+function lenientVttCues(src: string): Cue[] {
+ const lines = src.replace(/\r\n/g, "\n").split("\n");
+ const cues: Cue[] = [];
+ let i = 0;
+ while (i < lines.length) {
+ const m = lines[i].match(
+ /(\d{2}:\d{2}:\d{2}\.\d{3})\s+-->\s+(\d{2}:\d{2}:\d{2}\.\d{3})/,
+ );
+ if (!m) {
+ i++;
+ continue;
+ }
+ i++;
+ const body: string[] = [];
+ while (i < lines.length && lines[i].trim() !== "") {
+ body.push(lines[i]);
+ i++;
+ }
+ const text = stripVttMarkup(body.join(" "));
+ if (text) cues.push({ start: 0, end: 0, text });
+ }
+ const out: Cue[] = [];
+ for (const c of cues) {
+ if (out.length > 0 && out[out.length - 1].text === c.text) continue;
+ out.push(c);
+ }
+ return out;
+}
+
+function stripVttMarkup(s: string): string {
+ return s
+ .replace(/ /g, " ")
+ .replace(/&/g, "&")
+ .replace(/</g, "<")
+ .replace(/>/g, ">")
+ .replace(/'/g, "'")
+ .replace(/"/g, '"')
+ .replace(/<[^>]*>/g, "")
+ .replace(/\s+/g, " ")
+ .trim();
+}
+
+// Similar-duration pair: exact → near. Content agreement is required — a shared
+// duration alone is NOT evidence of duplication, so when either side lacks a
+// comparable transcript we return null rather than matching on metadata. (The
+// old metadata fallback collapsed every same-length video into one cluster.)
+function evalSimilar(
+ a: Fingerprint,
+ b: Fingerprint,
+ nearThreshold: number,
+): Omit<PairResult, "a" | "b"> | null {
+ if (!a.hasTranscript || !b.hasTranscript) return null;
+ if (a.hash && a.hash === b.hash) {
+ return { kind: "transcript-exact", score: 1, contained: false };
+ }
+ const j = jaccard(a.shingleSet, b.shingleSet);
+ if (j >= nearThreshold) {
+ return { kind: "transcript-near", score: round3(j), contained: false };
+ }
+ return null;
+}
+
+// Containment pair: the shorter transcript's shingles are largely a subset of
+// the longer one's. Requires both transcripts (no metadata fallback here).
+function evalContainment(
+ a: Fingerprint,
+ b: Fingerprint,
+ containmentThreshold: number,
+): Omit<PairResult, "a" | "b"> | null {
+ if (!a.hasTranscript || !b.hasTranscript) return null;
+ const c = containment(a.shingleSet, b.shingleSet);
+ if (c >= containmentThreshold) {
+ return { kind: "transcript-near", score: round3(c), contained: true };
+ }
+ return null;
+}
+
+function toRef(fp: Fingerprint): DuplicateVideoRef {
+ const s = fp.stat;
+ return {
+ slug: s.slug,
+ channelSlug: s.channelSlug,
+ channel: s.channel,
+ platform: s.platform,
+ id: s.id,
+ title: s.title,
+ duration: s.duration,
+ uploadDate: s.uploadDate,
+ hasTranscript: fp.hasTranscript,
+ };
+}
+
+function maxScore(a: number | null, b: number | null): number | null {
+ if (a === null) return b;
+ if (b === null) return a;
+ return Math.max(a, b);
+}
+
+function round3(n: number): number {
+ return Math.round(n * 1000) / 1000;
+}
+
+function sha1(s: string): string {
+ return createHash("sha1").update(s).digest("hex");
+}
diff --git a/common/lib/duplicates.ts b/common/lib/duplicates.ts
@@ -0,0 +1,213 @@
+// Shared types + tuning constants + pure helpers for cross-platform duplicate
+// shorts detection. The algorithm (a hybrid cascade: cheap duration-bucket
+// pre-clustering followed by transcript-content confirmation) lives in
+// controller/duplicateShorts.ts; everything pure and reusable lives here so it
+// can be unit-tested and imported by the editor without pulling in lmdb/fs.
+
+import type { Platform } from "./platform";
+import type { Cue } from "./vtt";
+
+export const DUPLICATE_REPORT_VERSION = 1;
+
+// Default duration cutoff (seconds) for what counts as a "short". One-off runs
+// can override this (higher threshold, or null for "all durations").
+export const DEFAULT_SHORT_THRESHOLD_SECONDS = 180;
+// Phase-1 bucketing window. Two videos within ~this many seconds of each other
+// land in a shared comparison set (see controller).
+export const DEFAULT_DURATION_TOLERANCE_SECONDS = 2;
+// Phase-2 near-duplicate threshold (5-gram Jaccard).
+export const DEFAULT_NEAR_THRESHOLD = 0.6;
+// Containment threshold for the "short is a clip of a longer video" case.
+export const DEFAULT_CONTAINMENT_THRESHOLD = 0.8;
+// Shingle (word n-gram) size for similarity. 5 tolerates word-level ASR
+// variance between whisper and auto-captions without over-matching on common
+// unigrams.
+export const DEFAULT_SHINGLE_SIZE = 5;
+
+// Only content-confirmed tiers form clusters. Duration coincidence alone is
+// never treated as a match (it produced enormous false clusters of unrelated
+// same-length videos).
+export type DuplicateMatchKind = "transcript-exact" | "transcript-near";
+
+export type DuplicateVideoRef = {
+ slug: string; // `${channelSlug}/${id}`
+ channelSlug: string;
+ channel: string; // display name
+ platform: Platform;
+ id: string; // canonical id (may differ from the on-disk dir name)
+ title: string;
+ duration: number; // seconds
+ uploadDate: string; // YYYYMMDD
+ hasTranscript: boolean;
+};
+
+export type DuplicateCluster = {
+ clusterId: string; // stable sha1 over sorted member slugs
+ matchKind: DuplicateMatchKind; // strongest tier present in the cluster
+ score: number | null; // strongest pair score; null for metadata-only
+ contained: boolean; // any member pair matched by containment (clip-of-longer)
+ durationBucket: number; // representative rounded duration (seconds)
+ crossPlatform: boolean; // members span more than one platform
+ crossChannel: boolean; // members span more than one channelSlug
+ videoRefs: DuplicateVideoRef[];
+};
+
+export type DuplicateRunConfig = {
+ thresholdSeconds: number | null; // null === all durations
+ durationToleranceSeconds: number;
+ nearThreshold: number; // Jaccard
+ containmentThreshold: number;
+ shingleSize: number;
+};
+
+export type DuplicateReport = {
+ version: number;
+ generatedAt: string;
+ runConfig: DuplicateRunConfig;
+ totals: {
+ videosScanned: number;
+ clusters: number;
+ videosInClusters: number;
+ };
+ clusters: DuplicateCluster[];
+};
+
+export const DUPLICATES_FILENAME = "duplicates.json";
+
+// ---------------------------------------------------------------------------
+// Pure helpers (no I/O) — exported for direct testing.
+// ---------------------------------------------------------------------------
+
+// Build a normalized comparison string from cues. Cues are already
+// tag-stripped, entity-decoded, whitespace-collapsed and consecutive-deduped by
+// parseVtt/parseWhisper, so this only lowercases and drops punctuation — the
+// two things auto-captions and whisper most often disagree on.
+export function comparisonText(cues: Cue[]): string {
+ return cues
+ .map((c) => c.text)
+ .join(" ")
+ .toLowerCase()
+ .replace(/[^\p{L}\p{N}\s]/gu, " ")
+ .replace(/\s+/g, " ")
+ .trim();
+}
+
+// k-word shingles (n-grams) as a set of space-joined token windows. Texts
+// shorter than k collapse to a single shingle of the whole text.
+export function shingles(text: string, k: number): Set<string> {
+ const set = new Set<string>();
+ if (!text) return set;
+ const words = text.split(" ");
+ if (words.length < k) {
+ set.add(words.join(" "));
+ return set;
+ }
+ for (let i = 0; i + k <= words.length; i++) {
+ set.add(words.slice(i, i + k).join(" "));
+ }
+ return set;
+}
+
+export function intersectionSize<T>(a: Set<T>, b: Set<T>): number {
+ const [small, large] = a.size <= b.size ? [a, b] : [b, a];
+ let n = 0;
+ for (const x of small) if (large.has(x)) n++;
+ return n;
+}
+
+// |A∩B| / |A∪B|.
+export function jaccard<T>(a: Set<T>, b: Set<T>): number {
+ if (a.size === 0 && b.size === 0) return 0;
+ const inter = intersectionSize(a, b);
+ return inter / (a.size + b.size - inter);
+}
+
+// |A∩B| / min(|A|,|B|): how much of the smaller set is contained in the larger.
+// Catches a short whose shingles are largely a subset of a longer video's.
+export function containment<T>(a: Set<T>, b: Set<T>): number {
+ const min = Math.min(a.size, b.size);
+ if (min === 0) return 0;
+ return intersectionSize(a, b) / min;
+}
+
+// Minimal union-find over string keys, used to merge transitively-related
+// duplicate pairs (A≈B, B≈C ⇒ {A,B,C}) into clusters.
+export class UnionFind {
+ private parent = new Map<string, string>();
+
+ add(x: string): void {
+ if (!this.parent.has(x)) this.parent.set(x, x);
+ }
+
+ find(x: string): string {
+ this.add(x);
+ let root = x;
+ while (this.parent.get(root) !== root) root = this.parent.get(root) as string;
+ // Path compression.
+ let cur = x;
+ while (this.parent.get(cur) !== root) {
+ const next = this.parent.get(cur) as string;
+ this.parent.set(cur, root);
+ cur = next;
+ }
+ return root;
+ }
+
+ union(a: string, b: string): void {
+ const ra = this.find(a);
+ const rb = this.find(b);
+ if (ra !== rb) this.parent.set(ra, rb);
+ }
+
+ // Connected components, keyed by representative root. Only includes keys that
+ // were add()ed or union()ed.
+ groups(): string[][] {
+ const out = new Map<string, string[]>();
+ for (const key of this.parent.keys()) {
+ const root = this.find(key);
+ const arr = out.get(root);
+ if (arr) arr.push(key);
+ else out.set(root, [key]);
+ }
+ return [...out.values()];
+ }
+}
+
+// Tier ranking so a cluster reports its strongest evidence.
+const MATCH_RANK: Record<DuplicateMatchKind, number> = {
+ "transcript-near": 1,
+ "transcript-exact": 2,
+};
+
+export function strongerMatch(
+ a: DuplicateMatchKind,
+ b: DuplicateMatchKind,
+): DuplicateMatchKind {
+ return MATCH_RANK[a] >= MATCH_RANK[b] ? a : b;
+}
+
+// Narrow a cluster to the channels a single site exposes. The detector runs
+// globally over the whole channel pool, but each deployed site carries only a
+// subset of channels, so members outside the site can't be opened there. We
+// keep only in-site members, drop the cluster entirely if fewer than two
+// remain (a lone video is not a visible duplicate), and recompute the
+// cross-platform / cross-channel flags over the survivors. matchKind / score /
+// contained / durationBucket are left as the detector reported them — they
+// describe the strongest pair in the full cluster, which is the best signal we
+// have without re-running similarity here. Returns null when the cluster does
+// not survive.
+export function filterClusterToChannels(
+ cluster: DuplicateCluster,
+ channelSlugs: Set<string>,
+): DuplicateCluster | null {
+ const videoRefs = cluster.videoRefs.filter((r) =>
+ channelSlugs.has(r.channelSlug),
+ );
+ if (videoRefs.length < 2) return null;
+ return {
+ ...cluster,
+ videoRefs,
+ crossPlatform: new Set(videoRefs.map((r) => r.platform)).size > 1,
+ crossChannel: new Set(videoRefs.map((r) => r.channelSlug)).size > 1,
+ };
+}
diff --git a/common/lib/stats.ts b/common/lib/stats.ts
@@ -2,7 +2,7 @@ import type { Platform } from "./platform";
// Bumping this invalidates the LMDB `statsByPath` incremental cache and forces
// a full re-extraction (e.g. when a new field is added below).
-export const STATS_SCHEMA_VERSION = 1;
+export const STATS_SCHEMA_VERSION = 2;
export const STATS_MANIFEST_VERSION = 1;
// Per-page byte cap for the served stats index. Static hosts limit individual
@@ -21,6 +21,7 @@ export type VideoStat = {
id: string;
channelSlug: string;
channel: string; // display name
+ title: string;
platform: Platform;
uploadDate: string; // YYYYMMDD
timestamp: number | null; // unix seconds
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -100,6 +100,7 @@ export function summarizeStats(
id: base.id,
channelSlug: base.channelSlug,
channel: base.channel,
+ title: base.title,
platform: base.platform,
uploadDate: base.uploadDate,
timestamp: numOrNull(meta.timestamp),
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Duplicate shorts detection on `/actionable`.** A new **Duplicate shorts** section finds the same short re-uploaded under a new id, re-titled, posted on another channel, or cross-posted to another platform. It runs a global, cross-platform pass: a cheap duration pre-cluster narrows candidates, then transcript content is compared (exact text hash → near-duplicate similarity → containment for "this short is a clip of a longer video"). Matches are **content-confirmed only** — a shared duration alone never clusters videos (that produced huge false clusters of unrelated same-length videos), and plain (non-karaoke) VTT captions that the index stores as zero cues are recovered by reading the raw transcript. Pick a scope — **Shorts only** (default ≤ 180s) or **All durations** (a heavier one-off that also surfaces clip-of-longer matches) — and click **Detect duplicates**; results list each cluster with its members (links), match strength, score, and cross-platform/cross-channel/clip-of-longer badges. It's **flag-only** — nothing is merged or deleted. Reads the cues + per-video stats written by **Build index** + **Build stats dataset**, so run those first; the report is written to `transcripts/duplicates.json` (also producible from the CLI via `pnpm --filter export run detect:duplicates [-- --all-durations]`).
- **Cleanup sections now estimate how much disk space they'll reclaim.** Both the `/actionable` cleanup rows and a channel's **Cleanup** stage show an estimated size next to each cleanup operation — the **Actionable** page gained an **Est. reclaim** column for "cleanable transcribed audio" and "extra audio formats", and the channel Cleanup stage shows an "Estimated space to reclaim: ~N" line under each of its two actions. The estimate is computed when a channel's report is generated (it sums the `audio.*` files each cleanup would delete — all audio for transcribed-audio cleanup, every non-target format for extra-format cleanup — and respects "do not clean"), so refresh a channel's report to populate it. Older reports without the figure show `~0 B` until refreshed.
- **Videos whose English captions only exist under a regional/auto code are now indexed.** YouTube occasionally serves a video's English subtitles only as `en-US`, `en-en-US`, or `en-orig` with no plain `en` track, so yt-dlp wrote e.g. `transcript.en-US.vtt` but never `transcript.en.vtt`. The index only recognized the literal `transcript.en.vtt`, so such a video looked untranscribed and never appeared in search. Build index now falls back to the best available English VTT (preferring `en`, then `en-orig`, then regional `en-US`/`en-GB`, then auto-translated `en-en-*`) while ignoring true translation tracks like `es-en-US`; whisper also treats these as already-transcribed. Re-run **Build index** to pick up affected videos already on disk. The video detail page now reflects the same fallback (it previously hardcoded `transcript.en.vtt`, so a regional-only video showed as untranscribed there).
- **Pick which subtitle track is a video's transcript.** The video page has a new **Transcript source** section listing every `transcript.<lang>.vtt` track, marking the current primary, with a **Set as transcript** button that promotes any track to the canonical `transcript.en.vtt` (the chosen track is copied, so the original stays and the choice is reversible — delete `transcript.en.vtt` to fall back to the automatic English pick, or pick another track to switch). Useful when the auto-picked track isn't the one you want, or when a video's only captions are a non-English track.
diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts
@@ -9,12 +9,19 @@ import {
} from "yt-dlp-transcript-common/controller/channelSnapshot";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand";
+import { detectDuplicateShorts } from "yt-dlp-transcript-common/controller/duplicateShorts";
export type RefreshAllResult = {
queued: string[];
skipped: { slug: string; reason: string }[];
};
+export type DuplicateScope = "shorts" | "all";
+
+export type RunDuplicateDetectionResult =
+ | { ok: true; clusters: number; videosInClusters: number }
+ | { ok: false; error: string };
+
async function drainStream(stream: ReadableStream<string>): Promise<void> {
const reader = stream.getReader();
try {
@@ -94,3 +101,36 @@ export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResu
revalidatePath("/");
return { queued, skipped };
}
+
+// Runs the global cross-platform duplicate-shorts pass. `scope: "all"` removes
+// the duration cutoff (a heavier one-off run that also surfaces clip-of-longer
+// containment). Reads the cues + statsByPath written by build:index/build:stats,
+// so a build must have run first for meaningful results.
+export async function runDuplicateDetectionAction(
+ scope: DuplicateScope = "shorts",
+): Promise<RunDuplicateDetectionResult> {
+ const paths = getPaths();
+ let clusters = 0;
+ let videosInClusters = 0;
+ const result = await runManagedFunction({
+ kind: "detect-duplicates",
+ // Empty queueKey: a local read-only scan, no reason to wait behind
+ // sync/download work (see refreshAllChannelSnapshotsAction).
+ queueKey: "",
+ paths,
+ fn: async (onLog, signal) => {
+ const report = await detectDuplicateShorts({
+ paths,
+ thresholdSeconds: scope === "all" ? null : undefined,
+ onLog,
+ signal,
+ });
+ clusters = report.totals.clusters;
+ videosInClusters = report.totals.videosInClusters;
+ },
+ });
+ if (!result.ok) return { ok: false, error: result.error };
+ await drainStream(result.stream);
+ revalidatePath("/actionable");
+ return { ok: true, clusters, videosInClusters };
+}
diff --git a/editor/app/actionable/components/RunDuplicateDetectionButton.tsx b/editor/app/actionable/components/RunDuplicateDetectionButton.tsx
@@ -0,0 +1,85 @@
+"use client";
+
+import { useState } from "react";
+import {
+ runDuplicateDetectionAction,
+ type DuplicateScope,
+ type RunDuplicateDetectionResult,
+} from "../actions";
+
+type Status =
+ | { kind: "idle" }
+ | { kind: "running" }
+ | { kind: "done"; result: RunDuplicateDetectionResult }
+ | { kind: "error"; message: string };
+
+export function RunDuplicateDetectionButton() {
+ const [scope, setScope] = useState<DuplicateScope>("shorts");
+ const [status, setStatus] = useState<Status>({ kind: "idle" });
+
+ async function handleClick() {
+ setStatus({ kind: "running" });
+ try {
+ const result = await runDuplicateDetectionAction(scope);
+ setStatus({ kind: "done", result });
+ } catch (e) {
+ setStatus({ kind: "error", message: (e as Error).message });
+ }
+ }
+
+ const running = status.kind === "running";
+ return (
+ <div className="flex items-center gap-2">
+ <label className="sr-only" htmlFor="duplicate-scope">
+ Detection scope
+ </label>
+ <select
+ id="duplicate-scope"
+ aria-label="duplicate detection scope"
+ value={scope}
+ onChange={(e) => setScope(e.target.value as DuplicateScope)}
+ disabled={running}
+ className="text-sm rounded-md border border-zinc-300 dark:border-zinc-700 bg-transparent px-2 py-2"
+ >
+ <option value="shorts">Shorts only</option>
+ <option value="all">All durations</option>
+ </select>
+ <button
+ type="button"
+ onClick={handleClick}
+ disabled={running}
+ aria-label="detect duplicate shorts"
+ className="px-3 py-2 rounded-md bg-zinc-900 dark:bg-zinc-100 text-zinc-100 dark:text-zinc-900 text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {running ? "Detecting…" : "Detect duplicates"}
+ </button>
+ {status.kind === "done" && status.result.ok && (
+ <span
+ aria-label="detect duplicate shorts result"
+ className="text-xs text-zinc-500"
+ >
+ {status.result.clusters} cluster(s) ·{" "}
+ {status.result.videosInClusters} video(s)
+ </span>
+ )}
+ {status.kind === "done" && !status.result.ok && (
+ <span
+ role="alert"
+ aria-label="detect duplicate shorts error"
+ className="text-xs text-red-700 dark:text-red-400"
+ >
+ {status.result.error}
+ </span>
+ )}
+ {status.kind === "error" && (
+ <span
+ role="alert"
+ aria-label="detect duplicate shorts error"
+ className="text-xs text-red-700 dark:text-red-400"
+ >
+ {status.message}
+ </span>
+ )}
+ </div>
+ );
+}
diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts
@@ -8,6 +8,8 @@ import {
readChannelSnapshot,
type ChannelSnapshot,
} from "yt-dlp-transcript-common/controller/channelSnapshot";
+import { readDuplicateReport } from "yt-dlp-transcript-common/controller/duplicateShorts";
+import type { DuplicateReport } from "yt-dlp-transcript-common/lib/duplicates";
export type ActionableRow = {
channel: ChannelStat;
@@ -21,6 +23,7 @@ export type ActionableSummary = {
cleanTranscribedAudio: ActionableRow[];
cleanExtraFormats: ActionableRow[];
staleOrMissing: ActionableRow[];
+ duplicates: DuplicateReport | null;
};
export function isStaleOrMissing(row: ActionableRow): boolean {
@@ -83,12 +86,15 @@ export async function loadActionableSummary(
paths: Paths,
): Promise<ActionableSummary> {
const channels = await listChannels(paths);
- const rows: ActionableRow[] = await Promise.all(
- channels.map(async (channel) => ({
- channel,
- snapshot: await readChannelSnapshot(paths, channel.slug),
- })),
- );
+ const [rows, duplicates] = await Promise.all([
+ Promise.all(
+ channels.map(async (channel) => ({
+ channel,
+ snapshot: await readChannelSnapshot(paths, channel.slug),
+ })),
+ ),
+ readDuplicateReport(paths),
+ ]);
const undownloaded = rows
.filter((r) => actionableUndownloadedCount(r) > 0)
@@ -128,5 +134,6 @@ export async function loadActionableSummary(
cleanTranscribedAudio,
cleanExtraFormats,
staleOrMissing,
+ duplicates,
};
}
diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx
@@ -14,6 +14,11 @@ import {
} from "./lib/loadActionable";
import { InlineActionButton } from "./components/InlineActionButton";
import { RefreshAllReportsButton } from "./components/RefreshAllReportsButton";
+import { RunDuplicateDetectionButton } from "./components/RunDuplicateDetectionButton";
+import type {
+ DuplicateCluster,
+ DuplicateReport,
+} from "yt-dlp-transcript-common/lib/duplicates";
export const dynamic = "force-dynamic";
@@ -164,10 +169,107 @@ export default async function ActionablePage() {
<Section key={config.id} config={config} rows={rows} />
))
)}
+ <DuplicatesSection report={summary.duplicates} />
</div>
);
}
+function DuplicatesSection({ report }: { report: DuplicateReport | null }) {
+ const clusters = report?.clusters ?? [];
+ return (
+ <section aria-label="duplicate-shorts" className="flex flex-col gap-2">
+ <div className="flex items-center justify-between flex-wrap gap-2">
+ <h2 className="text-lg font-semibold">Duplicate shorts</h2>
+ <RunDuplicateDetectionButton />
+ </div>
+ <p className="text-sm text-zinc-500">
+ Cross-platform, cross-channel duplicate detection (metadata pre-cluster →
+ transcript comparison). Flag-only — review and act manually.
+ {report ? (
+ <>
+ {" "}
+ Last run{" "}
+ <time dateTime={report.generatedAt}>
+ {new Date(report.generatedAt).toLocaleString()}
+ </time>{" "}
+ ·{" "}
+ {report.runConfig.thresholdSeconds === null
+ ? "all durations"
+ : `≤ ${report.runConfig.thresholdSeconds}s`}
+ .
+ </>
+ ) : (
+ " Never run."
+ )}
+ </p>
+ {clusters.length === 0 ? (
+ <p
+ aria-label="duplicate-shorts empty"
+ className="text-sm text-zinc-500 border border-dashed border-zinc-300 dark:border-zinc-700 rounded p-4"
+ >
+ {report ? "No duplicate clusters found." : "Run detection to scan."}
+ </p>
+ ) : (
+ <ul className="flex flex-col gap-3">
+ {clusters.map((cluster) => (
+ <DuplicateClusterCard key={cluster.clusterId} cluster={cluster} />
+ ))}
+ </ul>
+ )}
+ </section>
+ );
+}
+
+function DuplicateClusterCard({ cluster }: { cluster: DuplicateCluster }) {
+ const matchLabel: Record<DuplicateCluster["matchKind"], string> = {
+ "transcript-exact": "exact transcript",
+ "transcript-near": "near transcript",
+ };
+ return (
+ <li
+ aria-label={`duplicate cluster ${cluster.clusterId}`}
+ className="border border-zinc-200 dark:border-zinc-800 rounded-md p-3 flex flex-col gap-2"
+ >
+ <div className="flex items-center gap-2 flex-wrap text-xs">
+ <Badge>{matchLabel[cluster.matchKind]}</Badge>
+ {cluster.score !== null && (
+ <Badge>score {cluster.score.toFixed(2)}</Badge>
+ )}
+ {cluster.contained && <Badge>clip-of-longer</Badge>}
+ {cluster.crossPlatform && <Badge>cross-platform</Badge>}
+ {cluster.crossChannel && <Badge>cross-channel</Badge>}
+ <span className="text-zinc-500">
+ {cluster.videoRefs.length} videos · ~{cluster.durationBucket}s
+ </span>
+ </div>
+ <ul className="flex flex-col gap-1">
+ {cluster.videoRefs.map((ref) => (
+ <li key={ref.slug} className="text-sm flex items-baseline gap-2 flex-wrap">
+ <Link
+ href={`/channels/${ref.channelSlug}/videos/${ref.id}`}
+ className="underline hover:text-zinc-900 dark:hover:text-zinc-100"
+ >
+ {ref.title || ref.slug}
+ </Link>
+ <span className="text-xs text-zinc-500 font-mono">
+ {ref.platform} · {ref.channelSlug} · {ref.uploadDate}
+ {ref.hasTranscript ? "" : " · no transcript"}
+ </span>
+ </li>
+ ))}
+ </ul>
+ </li>
+ );
+}
+
+function Badge({ children }: { children: React.ReactNode }) {
+ return (
+ <span className="inline-flex items-center rounded-full bg-zinc-100 dark:bg-zinc-800 px-2 py-0.5 text-zinc-700 dark:text-zinc-300">
+ {children}
+ </span>
+ );
+}
+
function Section({
config,
rows,
diff --git a/editor/e2e/duplicate-shorts.spec.ts b/editor/e2e/duplicate-shorts.spec.ts
@@ -0,0 +1,278 @@
+import { mkdir, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import { readJson, resetData, resolvePath, writeSite } from "./helpers";
+
+// Cross-platform duplicate shorts detection. Seeds four channels with shorts
+// whose transcripts are near-identical across platforms/channels, plus a unique
+// short, a metadata-only pair (one missing its transcript), and a long video
+// that contains a short's transcript (for the all-durations containment case).
+// Drives Build index + Build stats, then runs detection from /actionable and
+// asserts on transcripts/duplicates.json.
+
+type DuplicateRef = { slug: string; channelSlug: string; platform: string };
+type DuplicateCluster = {
+ matchKind: string;
+ score: number | null;
+ contained: boolean;
+ crossPlatform: boolean;
+ crossChannel: boolean;
+ videoRefs: DuplicateRef[];
+};
+type DuplicateReport = {
+ runConfig: { thresholdSeconds: number | null };
+ totals: { clusters: number };
+ clusters: DuplicateCluster[];
+};
+
+const BASE_WORDS =
+ "alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima " +
+ "mike november oscar papa quebec romeo sierra tango uniform victor whiskey " +
+ "xray yankee zulu one two three four";
+
+function variant(replaceIndex: number, word: string): string {
+ const words = BASE_WORDS.split(" ");
+ words[replaceIndex] = word;
+ return words.join(" ");
+}
+
+const UNIQUE_WORDS =
+ "completely different words nothing in common here lorem ipsum dolor sit amet " +
+ "consectetur adipiscing elit sed do eiusmod tempor incididunt ut labore et " +
+ "dolore magna aliqua enim ad minim veniam quis";
+
+const PLAIN_WORDS =
+ "plain vtt caption track without any karaoke timing tags just sentences " +
+ "spoken across several distinct cue blocks one after another here";
+
+const PLAIN_OTHER =
+ "an entirely unrelated plain caption discussing gardening compost soil and " +
+ "seedlings with nothing whatsoever in common with the other clip";
+
+// parseVtt only extracts cue text from lines carrying YouTube's inline timing
+// tags (e.g. `<00:00:01.000>`), so synthesize an auto-caption-style cue whose
+// karaoke line carries every word.
+function vtt(text: string): string {
+ const words = text.split(" ");
+ const tagged = words
+ .map((w, i) =>
+ i === 0 ? w : `<00:00:0${i % 10}.000><c> ${w}</c>`,
+ )
+ .join("");
+ return [
+ "WEBVTT",
+ "Kind: captions",
+ "Language: en",
+ "",
+ "00:00:00.000 --> 00:10:00.000 align:start position:0%",
+ tagged,
+ "",
+ ].join("\n");
+}
+
+function stamp(sec: number): string {
+ const h = String(Math.floor(sec / 3600)).padStart(2, "0");
+ const m = String(Math.floor((sec % 3600) / 60)).padStart(2, "0");
+ const s = String(sec % 60).padStart(2, "0");
+ return `${h}:${m}:${s}.000`;
+}
+
+// Plain (non-karaoke) VTT — one short cue per few words, no inline timing tags.
+// parseVtt produces zero cues for this, so detection must recover the text via
+// its raw-transcript fallback.
+function plainVtt(text: string): string {
+ const words = text.split(" ");
+ const out = ["WEBVTT", ""];
+ let t = 0;
+ for (let i = 0; i < words.length; i += 6) {
+ out.push(`${stamp(t)} --> ${stamp(t + 3)}`, words.slice(i, i + 6).join(" "), "");
+ t += 3;
+ }
+ return out.join("\n");
+}
+
+type Seed = {
+ channel: string;
+ id: string;
+ platform: "youtube" | "rumble";
+ duration: number;
+ transcript: string | null;
+ plain?: boolean; // emit plain (non-karaoke) VTT
+};
+
+const SEEDS: Seed[] = [
+ // Near-identical shorts across two channels and two platforms.
+ { channel: "yt-a", id: "shorta", platform: "youtube", duration: 150, transcript: BASE_WORDS },
+ { channel: "yt-b", id: "shortb", platform: "youtube", duration: 151, transcript: variant(5, "foxtrotx") },
+ { channel: "rumble-c", id: "rumc", platform: "rumble", duration: 150, transcript: variant(25, "zulux") },
+ // A unique short sharing the duration bucket but unrelated content.
+ { channel: "yt-a", id: "unique", platform: "youtube", duration: 150, transcript: UNIQUE_WORDS },
+ // Same duration, one is missing its transcript entirely. Must NOT cluster:
+ // duration coincidence alone is no longer treated as a duplicate.
+ { channel: "yt-a", id: "metayes", platform: "youtube", duration: 90, transcript: BASE_WORDS },
+ { channel: "yt-a", id: "metano", platform: "youtube", duration: 90, transcript: null },
+ // Plain-VTT near-duplicates (zero karaoke cues): must still cluster via the
+ // raw-transcript fallback.
+ { channel: "yt-a", id: "plaina", platform: "youtube", duration: 120, transcript: PLAIN_WORDS, plain: true },
+ { channel: "yt-b", id: "plainb", platform: "youtube", duration: 120, transcript: PLAIN_WORDS, plain: true },
+ // Same duration as the plain pair but unrelated content: must NOT be pulled in.
+ { channel: "yt-a", id: "plainx", platform: "youtube", duration: 120, transcript: PLAIN_OTHER, plain: true },
+ // Long video that contains shorta's transcript (containment / all-durations).
+ { channel: "yt-d", id: "longvid", platform: "youtube", duration: 600, transcript: `${BASE_WORDS} the rest of this much longer recording continues well beyond the clip` },
+];
+
+const PLATFORM_KEY: Record<Seed["platform"], string> = {
+ youtube: "Youtube",
+ rumble: "Rumble",
+};
+
+async function seed(): Promise<void> {
+ const channels = new Set(SEEDS.map((s) => s.channel));
+ for (const channel of channels) {
+ const dir = resolvePath(`test-transcripts/channels/${channel}`);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ `${dir}/config.json`,
+ JSON.stringify({ handling: "transcribe", name: channel }),
+ );
+ }
+ for (const s of SEEDS) {
+ const dir = resolvePath(`test-transcripts/channels/${s.channel}/data/${s.id}`);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ `${dir}/metadata.info.json`,
+ JSON.stringify({
+ id: s.id,
+ title: `Title ${s.id}`,
+ channel: s.channel,
+ upload_date: "20240101",
+ duration: s.duration,
+ extractor_key: PLATFORM_KEY[s.platform],
+ webpage_url:
+ s.platform === "rumble"
+ ? `https://rumble.com/${s.id}`
+ : `https://www.youtube.com/watch?v=${s.id}`,
+ }),
+ );
+ if (s.transcript !== null) {
+ await writeFile(
+ `${dir}/transcript.en.vtt`,
+ s.plain ? plainVtt(s.transcript) : vtt(s.transcript),
+ );
+ }
+ }
+}
+
+async function buildData(page: import("@playwright/test").Page): Promise<void> {
+ await page.goto("/build");
+ await page.getByRole("button", { name: "Build index" }).click();
+ await expect(page.getByLabel("Build index output")).toContainText("Done", {
+ timeout: 30_000,
+ });
+ await page.getByRole("button", { name: "Build stats dataset" }).click();
+ await expect(page.getByLabel("Build stats dataset output")).toContainText(
+ "Stats built",
+ { timeout: 30_000 },
+ );
+}
+
+function clusterWith(report: DuplicateReport, slug: string): DuplicateCluster | undefined {
+ return report.clusters.find((c) => c.videoRefs.some((r) => r.slug === slug));
+}
+
+test("clusters cross-platform near-duplicate shorts (incl. plain VTT) and ignores duration-only coincidences", async ({
+ page,
+}) => {
+ await resetData(null);
+ await seed();
+ await writeSite("testsite", {
+ channels: [...new Set(SEEDS.map((s) => s.channel))].map((slug) => ({
+ slug,
+ groupId: "default",
+ })),
+ });
+ await buildData(page);
+
+ await page.goto("/actionable");
+ await page.getByLabel("duplicate detection scope").selectOption("shorts");
+ await page.getByRole("button", { name: "detect duplicate shorts" }).click();
+ await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({
+ timeout: 30_000,
+ });
+
+ const report = await readJson<DuplicateReport>("test-transcripts/duplicates.json");
+ expect(report.runConfig.thresholdSeconds).toBe(180);
+
+ // The three near-identical shorts cluster together, across platform + channel.
+ const near = clusterWith(report, "yt-a/shorta");
+ expect(near).toBeTruthy();
+ expect(near!.matchKind).toBe("transcript-near");
+ expect(near!.crossPlatform).toBe(true);
+ expect(near!.crossChannel).toBe(true);
+ expect(near!.videoRefs.map((r) => r.slug).sort()).toEqual([
+ "rumble-c/rumc",
+ "yt-a/shorta",
+ "yt-b/shortb",
+ ]);
+
+ // The unique short is not part of any cluster.
+ expect(clusterWith(report, "yt-a/unique")).toBeUndefined();
+
+ // Duration coincidence alone is NOT a duplicate: the transcript-less video and
+ // its unrelated same-duration neighbour must not cluster.
+ expect(clusterWith(report, "yt-a/metano")).toBeUndefined();
+ expect(clusterWith(report, "yt-a/metayes")).toBeUndefined();
+
+ // Plain (non-karaoke) VTT yields zero parseVtt cues, but the raw-transcript
+ // fallback recovers the text so the near-identical plain pair still clusters.
+ const plain = clusterWith(report, "yt-a/plaina");
+ expect(plain).toBeTruthy();
+ expect(plain!.matchKind).toBe("transcript-exact");
+ expect(plain!.videoRefs.map((r) => r.slug).sort()).toEqual([
+ "yt-a/plaina",
+ "yt-b/plainb",
+ ]);
+ // ...and the unrelated same-duration plain video stays out of that cluster.
+ expect(plain!.videoRefs.map((r) => r.slug)).not.toContain("yt-a/plainx");
+ expect(clusterWith(report, "yt-a/plainx")).toBeUndefined();
+
+ // Every reported cluster is content-confirmed (no metadata-only matches).
+ for (const c of report.clusters) {
+ expect(["transcript-exact", "transcript-near"]).toContain(c.matchKind);
+ expect(c.score).not.toBeNull();
+ }
+
+ // The detected cluster renders on the page after a refresh.
+ await page.reload();
+ await expect(page.getByLabel("duplicate-shorts")).toContainText(
+ "cross-platform",
+ );
+});
+
+test("all-durations run pulls a long video into the short's cluster via containment", async ({
+ page,
+}) => {
+ await resetData(null);
+ await seed();
+ await writeSite("testsite", {
+ channels: [...new Set(SEEDS.map((s) => s.channel))].map((slug) => ({
+ slug,
+ groupId: "default",
+ })),
+ });
+ await buildData(page);
+
+ await page.goto("/actionable");
+ await page.getByLabel("duplicate detection scope").selectOption("all");
+ await page.getByRole("button", { name: "detect duplicate shorts" }).click();
+ await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({
+ timeout: 30_000,
+ });
+
+ const report = await readJson<DuplicateReport>("test-transcripts/duplicates.json");
+ expect(report.runConfig.thresholdSeconds).toBeNull();
+
+ const cluster = clusterWith(report, "yt-a/shorta");
+ expect(cluster).toBeTruthy();
+ expect(cluster!.contained).toBe(true);
+ expect(cluster!.videoRefs.map((r) => r.slug)).toContain("yt-d/longvid");
+});
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,5 +1,8 @@
# Changelog
+## [Unreleased]
+- **Duplicates page.** A new **Duplicates** tab in the primary nav lists shorts whose transcripts match across channels and platforms — re-uploads, mirrors, and cross-posts of the same clip — grouped into clusters (strongest match first). Each cluster shows how it matched (exact or near transcript, similarity score, clip-of-longer) and whether it spans multiple channels or platforms; clicking any member plays it in the transcript modal. The list is scoped to the channels this site exposes, so every entry is openable here. Sites with no detection run yet show an empty state.
+
## [0.3.3] - 2026-06-01
- **Videos whose English captions only existed under a regional/auto code now appear.** A handful of videos had English subtitles only under codes like `en-US` or `en-en-US` (no plain `en`), which the index didn't recognize — so they were missing from the site even though they had a transcript. These now show up in browse and search like any other transcribed video.
diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx
@@ -0,0 +1,125 @@
+"use client";
+
+import { useEffect, useState } from "react";
+import { usePlayer } from "yt-dlp-transcript-common/components/PlayerProvider";
+import type {
+ DuplicateCluster,
+ DuplicateReport,
+} from "yt-dlp-transcript-common/lib/duplicates";
+
+type LoadState =
+ | { status: "loading" }
+ | { status: "ready"; report: DuplicateReport | null };
+
+export function DuplicatesClient() {
+ const [state, setState] = useState<LoadState>({ status: "loading" });
+
+ // The site-filtered report is baked into the static build at /duplicates.json
+ // (see common/bin/compose-site.ts). A missing file (404) or parse failure
+ // means no detection has been run for this site — fall back to the empty
+ // state rather than an error.
+ useEffect(() => {
+ let alive = true;
+ fetch("/duplicates.json")
+ .then((r) => (r.ok ? r.json() : null))
+ .then((report: DuplicateReport | null) => {
+ if (alive) setState({ status: "ready", report });
+ })
+ .catch(() => {
+ if (alive) setState({ status: "ready", report: null });
+ });
+ return () => {
+ alive = false;
+ };
+ }, []);
+
+ if (state.status === "loading") {
+ return <p className="text-sm text-zinc-500">Loading duplicates…</p>;
+ }
+
+ const report = state.report;
+ const clusters = report?.clusters ?? [];
+
+ return (
+ <section aria-label="duplicate-shorts" className="flex flex-col gap-3">
+ {report && (
+ <p className="text-xs text-zinc-500">
+ {report.runConfig.thresholdSeconds === null
+ ? "all durations"
+ : `shorts ≤${report.runConfig.thresholdSeconds}s`}
+ {" · "}
+ {report.totals.clusters}{" "}
+ {report.totals.clusters === 1 ? "cluster" : "clusters"}
+ {" · "}
+ <span title={report.generatedAt}>
+ generated {new Date(report.generatedAt).toLocaleDateString()}
+ </span>
+ </p>
+ )}
+ {clusters.length === 0 ? (
+ <p className="text-sm text-zinc-500 border border-dashed border-zinc-300 dark:border-zinc-700 rounded p-4">
+ No duplicate shorts have been detected for this site yet.
+ </p>
+ ) : (
+ <ul className="flex flex-col gap-3">
+ {clusters.map((cluster) => (
+ <DuplicateClusterCard key={cluster.clusterId} cluster={cluster} />
+ ))}
+ </ul>
+ )}
+ </section>
+ );
+}
+
+const MATCH_LABEL: Record<DuplicateCluster["matchKind"], string> = {
+ "transcript-exact": "exact transcript",
+ "transcript-near": "near transcript",
+};
+
+function DuplicateClusterCard({ cluster }: { cluster: DuplicateCluster }) {
+ const { openTranscript } = usePlayer();
+ return (
+ <li
+ aria-label={`duplicate cluster ${cluster.clusterId}`}
+ className="border border-zinc-200 dark:border-zinc-800 rounded-md p-3 flex flex-col gap-2"
+ >
+ <div className="flex items-center gap-2 flex-wrap text-xs">
+ <Badge>{MATCH_LABEL[cluster.matchKind]}</Badge>
+ {cluster.score !== null && <Badge>score {cluster.score.toFixed(2)}</Badge>}
+ {cluster.contained && <Badge>clip-of-longer</Badge>}
+ {cluster.crossPlatform && <Badge>cross-platform</Badge>}
+ {cluster.crossChannel && <Badge>cross-channel</Badge>}
+ <span className="text-zinc-500">
+ {cluster.videoRefs.length} videos · ~{cluster.durationBucket}s
+ </span>
+ </div>
+ <ul className="flex flex-col gap-1">
+ {cluster.videoRefs.map((ref) => (
+ <li
+ key={ref.slug}
+ className="text-sm flex items-baseline gap-2 flex-wrap"
+ >
+ <button
+ type="button"
+ onClick={() => openTranscript(ref.slug)}
+ className="underline text-left hover:text-zinc-900 dark:hover:text-zinc-100"
+ >
+ {ref.title || ref.slug}
+ </button>
+ <span className="text-xs text-zinc-500 font-mono">
+ {ref.platform} · {ref.channelSlug} · {ref.uploadDate}
+ </span>
+ </li>
+ ))}
+ </ul>
+ </li>
+ );
+}
+
+function Badge({ children }: { children: React.ReactNode }) {
+ return (
+ <span className="inline-flex items-center rounded-full bg-zinc-100 dark:bg-zinc-800 px-2 py-0.5 text-zinc-700 dark:text-zinc-300">
+ {children}
+ </span>
+ );
+}
diff --git a/export/app/duplicates/page.tsx b/export/app/duplicates/page.tsx
@@ -0,0 +1,25 @@
+import type { Metadata } from "next";
+import { PlayerProvider } from "yt-dlp-transcript-common/components/PlayerProvider";
+import TranscriptModal from "yt-dlp-transcript-common/components/TranscriptModal";
+import { DuplicatesClient } from "./DuplicatesClient";
+
+export const metadata: Metadata = { title: "Duplicates" };
+
+export default function DuplicatesPage() {
+ return (
+ <PlayerProvider>
+ <div className="flex flex-col gap-4">
+ <div>
+ <h1 className="text-2xl font-semibold">Duplicate shorts</h1>
+ <p className="mt-1 text-sm text-zinc-500">
+ Shorts whose transcripts match across channels and platforms —
+ re-uploads, mirrors, and cross-posts of the same clip. Click any
+ member to play it.
+ </p>
+ </div>
+ <DuplicatesClient />
+ </div>
+ <TranscriptModal />
+ </PlayerProvider>
+ );
+}
diff --git a/export/app/layout.tsx b/export/app/layout.tsx
@@ -59,6 +59,12 @@ export default async function RootLayout({
>
Charts
</Link>
+ <Link
+ href="/duplicates"
+ className="text-zinc-900 dark:text-zinc-100 hover:text-blue-600 dark:hover:text-blue-400"
+ >
+ Duplicates
+ </Link>
</nav>
</div>
<nav className="text-sm shrink-0">
diff --git a/export/e2e/duplicates.spec.ts b/export/e2e/duplicates.spec.ts
@@ -0,0 +1,71 @@
+import { expect, test, type Page } from "@playwright/test";
+import { expectModalOpen, installRoutes } from "./helpers";
+import { duplicatesReport } from "./fixtures/data";
+
+// Duplicates page e2e. The site-filtered /duplicates.json is route-mocked (it
+// is produced at build time by compose-site.ts); the page renders cluster
+// cards and each member opens the real transcript modal via the mocked
+// transcript routes installed by installRoutes.
+async function installDuplicatesRoute(
+ page: Page,
+ report: unknown | null,
+): Promise<void> {
+ await page.route("**/duplicates.json", async (route) => {
+ if (report === null) {
+ await route.fulfill({ status: 404, contentType: "application/json", body: "{}" });
+ return;
+ }
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(report),
+ });
+ });
+}
+
+test.describe("duplicates", () => {
+ test("renders a cluster and opens a member in the transcript modal", async ({
+ page,
+ }) => {
+ await installRoutes(page);
+ await installDuplicatesRoute(page, duplicatesReport());
+
+ await page.goto("/duplicates");
+
+ // Run-config / staleness header.
+ await expect(page.getByText("shorts ≤180s")).toBeVisible();
+
+ // Cluster card with its evidence badges.
+ const card = page.getByLabel(/^duplicate cluster /);
+ await expect(card).toBeVisible();
+ await expect(card.getByText("near transcript")).toBeVisible();
+ await expect(card.getByText("score 0.87")).toBeVisible();
+ await expect(card.getByText("cross-platform")).toBeVisible();
+ await expect(card.getByText("cross-channel")).toBeVisible();
+ await expect(card.getByText("2 videos · ~150s")).toBeVisible();
+
+ // Both members render as clickable links.
+ await expect(
+ page.getByRole("button", { name: "How to X", exact: true }),
+ ).toBeVisible();
+
+ // Clicking a member opens the transcript modal for that video.
+ await page
+ .getByRole("button", { name: "How to X (reup)", exact: true })
+ .click();
+ await expectModalOpen(page);
+ });
+
+ test("shows the empty state when no report has been built", async ({
+ page,
+ }) => {
+ await installRoutes(page);
+ await installDuplicatesRoute(page, null);
+
+ await page.goto("/duplicates");
+
+ await expect(
+ page.getByText("No duplicate shorts have been detected for this site yet."),
+ ).toBeVisible();
+ });
+});
diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts
@@ -302,6 +302,62 @@ export function subsPage() {
];
}
+// ─── Duplicate-shorts feature fixtures ───
+// One site-filtered cluster of two members whose slugs resolve via the mocked
+// transcript routes, so a member click opens the real transcript modal. The
+// cross-platform / cross-channel flags and score are display-only here (the
+// page renders whatever the composed report contains).
+export const DUP_CLUSTER_ID = "dup-cluster-fixture";
+
+function dupRef(
+ id: string,
+ title: string,
+ platform: "youtube" | "rumble",
+ uploadDate: string,
+) {
+ return {
+ slug: slug(id),
+ channelSlug: CHANNEL_SLUG,
+ channel: CHANNEL,
+ platform,
+ id,
+ title,
+ duration: 150,
+ uploadDate,
+ hasTranscript: true,
+ };
+}
+
+export function duplicatesReport() {
+ return {
+ version: 1,
+ generatedAt: "2026-06-04T12:00:00.000Z",
+ runConfig: {
+ thresholdSeconds: 180,
+ durationToleranceSeconds: 2,
+ nearThreshold: 0.6,
+ containmentThreshold: 0.8,
+ shingleSize: 5,
+ },
+ totals: { videosScanned: 3, clusters: 1, videosInClusters: 2 },
+ clusters: [
+ {
+ clusterId: DUP_CLUSTER_ID,
+ matchKind: "transcript-near" as const,
+ score: 0.87,
+ contained: false,
+ durationBucket: 150,
+ crossPlatform: true,
+ crossChannel: true,
+ videoRefs: [
+ dupRef(VIDEO_TRANSCRIPT_ONLY, "How to X", "youtube", "20240101"),
+ dupRef(VIDEO_CHAT_SMALL, "How to X (reup)", "rumble", "20240102"),
+ ],
+ },
+ ],
+ };
+}
+
export function buildFixtureSettings(filePath: string): void {
const settings = {
siteTitle: "Test Export",
diff --git a/export/package.json b/export/package.json
@@ -9,6 +9,7 @@
"build:stats": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/build-stats.ts",
"build:templates": "tsx ../common/bin/build-chart-templates.ts",
"build:data": "pnpm run build:index && pnpm run build:stats && pnpm run build:templates",
+ "detect:duplicates": "NODE_OPTIONS=--max-old-space-size=8192 tsx ../common/bin/duplicate-shorts.ts",
"compose:site": "tsx ../common/bin/compose-site.ts",
"prebuild": "pnpm run build:data",
"build": "pnpm run compose:site && next build",