import { mkdir, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { buildIndex, readJson, resetData, resolvePath, writeSite, } from "./helpers"; // Cross-platform duplicate shorts detection. Seeds four channels with shorts // whose transcripts are near-identical across platforms/channels, plus a unique // short, a metadata-only pair (one missing its transcript), and a long video // that contains a short's transcript (for the all-durations containment case). // Drives Build index + Build stats, then runs detection from /review and // asserts on transcripts/duplicates.json. type DuplicateRef = { slug: string; channelSlug: string; platform: string; aligned?: boolean; offsetSeconds?: number | null; }; type DuplicateCluster = { clusterId: string; matchKind: string; score: number | null; contained: boolean; crossPlatform: boolean; crossChannel: boolean; needsReview?: boolean; canonicalSlug?: string; videoRefs: DuplicateRef[]; }; type DuplicateReport = { runConfig: { thresholdSeconds: number | null; blocking?: string }; totals: { clusters: number }; clusters: DuplicateCluster[]; }; const BASE_WORDS = "alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima " + "mike november oscar papa quebec romeo sierra tango uniform victor whiskey " + "xray yankee zulu one two three four"; function variant(replaceIndex: number, word: string): string { const words = BASE_WORDS.split(" "); words[replaceIndex] = word; return words.join(" "); } const UNIQUE_WORDS = "completely different words nothing in common here lorem ipsum dolor sit amet " + "consectetur adipiscing elit sed do eiusmod tempor incididunt ut labore et " + "dolore magna aliqua enim ad minim veniam quis"; const PLAIN_WORDS = "plain vtt caption track without any karaoke timing tags just sentences " + "spoken across several distinct cue blocks one after another here"; const PLAIN_OTHER = "an entirely unrelated plain caption discussing gardening compost soil and " + "seedlings with nothing whatsoever in common with the other clip"; // parseVtt only extracts cue text from lines carrying YouTube's inline timing // tags (e.g. `<00:00:01.000>`), so synthesize an auto-caption-style cue whose // karaoke line carries every word. function vtt(text: string): string { const words = text.split(" "); const tagged = words .map((w, i) => i === 0 ? w : `<00:00:0${i % 10}.000> ${w}`, ) .join(""); return [ "WEBVTT", "Kind: captions", "Language: en", "", "00:00:00.000 --> 00:10:00.000 align:start position:0%", tagged, "", ].join("\n"); } function stamp(sec: number): string { const h = String(Math.floor(sec / 3600)).padStart(2, "0"); const m = String(Math.floor((sec % 3600) / 60)).padStart(2, "0"); const s = String(sec % 60).padStart(2, "0"); return `${h}:${m}:${s}.000`; } // Plain (non-karaoke) VTT — one short cue per few words, no inline timing tags. // parseVtt produces zero cues for this, so detection must recover the text via // its raw-transcript fallback. function plainVtt(text: string): string { const words = text.split(" "); const out = ["WEBVTT", ""]; let t = 0; for (let i = 0; i < words.length; i += 6) { out.push(`${stamp(t)} --> ${stamp(t + 3)}`, words.slice(i, i + 6).join(" "), ""); t += 3; } return out.join("\n"); } type Seed = { channel: string; id: string; platform: "youtube" | "rumble"; duration: number; transcript: string | null; plain?: boolean; // emit plain (non-karaoke) VTT // Defaults to a per-id unique title, so the duration-blocking seeds below // never accidentally block by title too. The title-blocking seeds set it. title?: string; }; const SEEDS: Seed[] = [ // Near-identical shorts across two channels and two platforms. { channel: "yt-a", id: "shorta", platform: "youtube", duration: 150, transcript: BASE_WORDS }, { channel: "yt-b", id: "shortb", platform: "youtube", duration: 151, transcript: variant(5, "foxtrotx") }, { channel: "rumble-c", id: "rumc", platform: "rumble", duration: 150, transcript: variant(25, "zulux") }, // A unique short sharing the duration bucket but unrelated content. { channel: "yt-a", id: "unique", platform: "youtube", duration: 150, transcript: UNIQUE_WORDS }, // Same duration, one is missing its transcript entirely. Must NOT cluster: // duration coincidence alone is no longer treated as a duplicate. { channel: "yt-a", id: "metayes", platform: "youtube", duration: 90, transcript: BASE_WORDS }, { channel: "yt-a", id: "metano", platform: "youtube", duration: 90, transcript: null }, // Plain-VTT near-duplicates (zero karaoke cues): must still cluster via the // raw-transcript fallback. { channel: "yt-a", id: "plaina", platform: "youtube", duration: 120, transcript: PLAIN_WORDS, plain: true }, { channel: "yt-b", id: "plainb", platform: "youtube", duration: 120, transcript: PLAIN_WORDS, plain: true }, // Same duration as the plain pair but unrelated content: must NOT be pulled in. { channel: "yt-a", id: "plainx", platform: "youtube", duration: 120, transcript: PLAIN_OTHER, plain: true }, // Long video that contains shorta's transcript (containment / all-durations). { channel: "yt-d", id: "longvid", platform: "youtube", duration: 600, transcript: `${BASE_WORDS} the rest of this much longer recording continues well beyond the clip` }, ]; const PLATFORM_KEY: Record = { youtube: "Youtube", rumble: "Rumble", }; async function seed(seeds: Seed[] = SEEDS): Promise { const channels = new Set(seeds.map((s) => s.channel)); for (const channel of channels) { const dir = resolvePath(`test-transcripts/channels/${channel}`); await mkdir(dir, { recursive: true }); await writeFile( `${dir}/config.json`, JSON.stringify({ handling: "transcribe", name: channel }), ); } for (const s of seeds) { const dir = resolvePath(`test-transcripts/channels/${s.channel}/data/${s.id}`); await mkdir(dir, { recursive: true }); await writeFile( `${dir}/metadata.info.json`, JSON.stringify({ id: s.id, title: s.title ?? `Title ${s.id}`, channel: s.channel, upload_date: "20240101", duration: s.duration, extractor_key: PLATFORM_KEY[s.platform], webpage_url: s.platform === "rumble" ? `https://rumble.com/${s.id}` : `https://www.youtube.com/watch?v=${s.id}`, }), ); if (s.transcript !== null) { await writeFile( `${dir}/transcript.en.vtt`, s.plain ? plainVtt(s.transcript) : vtt(s.transcript), ); } } } // The index AND the stats: one stage since release 18 ("Build stats dataset" // is gone — the index update builds the stats datasets too). async function buildData(page: import("@playwright/test").Page): Promise { await buildIndex(page); } function clusterWith(report: DuplicateReport, slug: string): DuplicateCluster | undefined { return report.clusters.find((c) => c.videoRefs.some((r) => r.slug === slug)); } test("clusters cross-platform near-duplicate shorts (incl. plain VTT) and ignores duration-only coincidences", async ({ page, }) => { await resetData(null); await seed(); await writeSite("testsite", { channels: [...new Set(SEEDS.map((s) => s.channel))].map((slug) => ({ slug, groupId: "default", })), }); await buildData(page); await page.goto("/review"); await page.getByLabel("duplicate detection scope").selectOption("shorts"); await page.getByRole("button", { name: "detect duplicate shorts" }).click(); await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({ timeout: 30_000, }); const report = await readJson("test-transcripts/duplicates.json"); expect(report.runConfig.thresholdSeconds).toBe(180); // The three near-identical shorts cluster together, across platform + channel. const near = clusterWith(report, "yt-a/shorta"); expect(near).toBeTruthy(); expect(near!.matchKind).toBe("transcript-near"); expect(near!.crossPlatform).toBe(true); expect(near!.crossChannel).toBe(true); expect(near!.videoRefs.map((r) => r.slug).sort()).toEqual([ "rumble-c/rumc", "yt-a/shorta", "yt-b/shortb", ]); // The unique short is not part of any cluster. expect(clusterWith(report, "yt-a/unique")).toBeUndefined(); // Duration coincidence alone is NOT a duplicate: the transcript-less video and // its unrelated same-duration neighbour must not cluster. expect(clusterWith(report, "yt-a/metano")).toBeUndefined(); expect(clusterWith(report, "yt-a/metayes")).toBeUndefined(); // Plain (non-karaoke) VTT yields zero parseVtt cues, but the raw-transcript // fallback recovers the text so the near-identical plain pair still clusters. const plain = clusterWith(report, "yt-a/plaina"); expect(plain).toBeTruthy(); expect(plain!.matchKind).toBe("transcript-exact"); expect(plain!.videoRefs.map((r) => r.slug).sort()).toEqual([ "yt-a/plaina", "yt-b/plainb", ]); // ...and the unrelated same-duration plain video stays out of that cluster. expect(plain!.videoRefs.map((r) => r.slug)).not.toContain("yt-a/plainx"); expect(clusterWith(report, "yt-a/plainx")).toBeUndefined(); // Every cluster that would SHIP is content-confirmed. (The original invariant // was "every reported cluster", which was true when duration coincidence // produced nothing at all. Title + near-identical runtime is a far stronger // claim and now does produce clusters — but they are quarantined behind // needsReview, never auto-share, and never reach a built site, so the // guarantee is kept exactly where it matters. These seeds have distinct titles // anyway, so nothing here is a suspect.) for (const c of report.clusters) { if (c.needsReview) { expect(c.matchKind).toBe("title-duration"); continue; } expect(["transcript-exact", "transcript-near"]).toContain(c.matchKind); expect(c.score).not.toBeNull(); } expect(report.clusters.filter((c) => c.needsReview)).toHaveLength(0); // The detected cluster renders on the page after a refresh. await page.reload(); await expect(page.getByLabel("duplicate-shorts")).toContainText( "cross-platform", ); }); test("all-durations run pulls a long video into the short's cluster via containment", async ({ page, }) => { await resetData(null); await seed(); await writeSite("testsite", { channels: [...new Set(SEEDS.map((s) => s.channel))].map((slug) => ({ slug, groupId: "default", })), }); await buildData(page); await page.goto("/review"); await page.getByLabel("duplicate detection scope").selectOption("all"); await page.getByRole("button", { name: "detect duplicate shorts" }).click(); await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({ timeout: 30_000, }); const report = await readJson("test-transcripts/duplicates.json"); expect(report.runConfig.thresholdSeconds).toBeNull(); const cluster = clusterWith(report, "yt-a/shorta"); expect(cluster).toBeTruthy(); expect(cluster!.contained).toBe(true); expect(cluster!.videoRefs.map((r) => r.slug)).toContain("yt-d/longvid"); }); // --------------------------------------------------------------------------- // Title blocking: the pre-filter proposes, the transcript disposes // --------------------------------------------------------------------------- // Three same-title, compatible-runtime pairs that differ only in what the // TRANSCRIPTS say. The point of the whole design is that the pre-filter treats // all three identically and the content cascade then splits them three ways. const TITLE_SEEDS: Seed[] = [ // (1) Same title, same content → CONFIRMED. A cross-platform mirror. { channel: "t-a", id: "mirror1", platform: "youtube", duration: 300, transcript: BASE_WORDS, title: "The Weekly Roundup" }, { channel: "t-b", id: "mirror2", platform: "rumble", duration: 303, transcript: variant(5, "foxtrotx"), title: "The Weekly Roundup" }, // (2) Same title, same runtime, DIFFERENT content → REJECTED. Two episodes of // a daily show. This is the case that makes nominating aggressively safe. { channel: "t-a", id: "epis1", platform: "youtube", duration: 240, transcript: BASE_WORDS, title: "Daily Show Recap" }, { channel: "t-b", id: "epis2", platform: "rumble", duration: 241, transcript: UNIQUE_WORDS, title: "Daily Show Recap" }, // (3) Same title, same runtime, one side has NO transcript → untestable, so a // needsReview SUSPECT rather than a match or a silent drop. { channel: "t-a", id: "susp1", platform: "youtube", duration: 200, transcript: BASE_WORDS, title: "Archive Upload" }, { channel: "t-b", id: "susp2", platform: "rumble", duration: 200, transcript: null, title: "Archive Upload" }, // (4) Same title but a genuinely different cut (2x the runtime) → not even // nominated, so the durations gate is doing its job. { channel: "t-a", id: "cut1", platform: "youtube", duration: 300, transcript: BASE_WORDS, title: "Extended Interview" }, { channel: "t-b", id: "cut2", platform: "rumble", duration: 700, transcript: BASE_WORDS, title: "Extended Interview" }, ]; test("title blocking confirms matching content, rejects differing content, and quarantines the untestable", async ({ page, }) => { await resetData(null); await seed(TITLE_SEEDS); await writeSite("testsite", { channels: [...new Set(TITLE_SEEDS.map((s) => s.channel))].map((slug) => ({ slug, groupId: "default", })), }); await buildData(page); // Scope "all" runs corpus-wide, where the default blocking strategy is title. await page.goto("/review"); await page.getByLabel("duplicate detection scope").selectOption("all"); await page.getByRole("button", { name: "detect duplicate shorts" }).click(); await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({ timeout: 30_000, }); const report = await readJson("test-transcripts/duplicates.json"); expect(report.runConfig.blocking).toBe("title"); // (1) CONFIRMED — the content agreed, so this is a real cluster that may share. const mirror = clusterWith(report, "t-a/mirror1"); expect(mirror).toBeTruthy(); expect(mirror!.needsReview).toBeFalsy(); expect(["transcript-exact", "transcript-near"]).toContain(mirror!.matchKind); expect(mirror!.videoRefs.map((r) => r.slug).sort()).toEqual([ "t-a/mirror1", "t-b/mirror2", ]); // Alignment is measured and persisted for confirmed clusters — without it the // viewer's "jump to this moment" has nothing honest to key off. expect(mirror!.canonicalSlug).toBeTruthy(); for (const ref of mirror!.videoRefs) { expect(typeof ref.aligned).toBe("boolean"); } // (2) REJECTED — same title, same runtime, different words. No cluster at all. expect(clusterWith(report, "t-a/epis1")).toBeUndefined(); expect(clusterWith(report, "t-b/epis2")).toBeUndefined(); // (3) SUSPECT — nothing could compare the content, so it is a review item. const suspect = clusterWith(report, "t-a/susp1"); expect(suspect).toBeTruthy(); expect(suspect!.needsReview).toBe(true); expect(suspect!.matchKind).toBe("title-duration"); expect(suspect!.score).toBeNull(); expect(suspect!.videoRefs.map((r) => r.slug).sort()).toEqual([ "t-a/susp1", "t-b/susp2", ]); // (4) NOT NOMINATED — a 300s and a 700s video are a different cut, not a // mirror, however identical their titles. expect(clusterWith(report, "t-a/cut1")).toBeUndefined(); expect(clusterWith(report, "t-b/cut2")).toBeUndefined(); // The suspect is visibly flagged in the editor's review list. await page.reload(); const card = page.getByLabel(`duplicate cluster ${suspect!.clusterId}`); await expect(card).toContainText("needs review"); await expect(card).toContainText("title + runtime"); });