import { mkdir, writeFile } from "node:fs/promises";
import { test, expect } from "@playwright/test";
import {
buildIndex,
readJson,
resetData,
resolvePath,
writeSite,
} from "./helpers";
// Cross-platform duplicate shorts detection. Seeds four channels with shorts
// whose transcripts are near-identical across platforms/channels, plus a unique
// short, a metadata-only pair (one missing its transcript), and a long video
// that contains a short's transcript (for the all-durations containment case).
// Drives Build index + Build stats, then runs detection from /review and
// asserts on transcripts/duplicates.json.
type DuplicateRef = {
slug: string;
channelSlug: string;
platform: string;
aligned?: boolean;
offsetSeconds?: number | null;
};
type DuplicateCluster = {
clusterId: string;
matchKind: string;
score: number | null;
contained: boolean;
crossPlatform: boolean;
crossChannel: boolean;
needsReview?: boolean;
canonicalSlug?: string;
videoRefs: DuplicateRef[];
};
type DuplicateReport = {
runConfig: { thresholdSeconds: number | null; blocking?: string };
totals: { clusters: number };
clusters: DuplicateCluster[];
};
const BASE_WORDS =
"alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima " +
"mike november oscar papa quebec romeo sierra tango uniform victor whiskey " +
"xray yankee zulu one two three four";
function variant(replaceIndex: number, word: string): string {
const words = BASE_WORDS.split(" ");
words[replaceIndex] = word;
return words.join(" ");
}
const UNIQUE_WORDS =
"completely different words nothing in common here lorem ipsum dolor sit amet " +
"consectetur adipiscing elit sed do eiusmod tempor incididunt ut labore et " +
"dolore magna aliqua enim ad minim veniam quis";
const PLAIN_WORDS =
"plain vtt caption track without any karaoke timing tags just sentences " +
"spoken across several distinct cue blocks one after another here";
const PLAIN_OTHER =
"an entirely unrelated plain caption discussing gardening compost soil and " +
"seedlings with nothing whatsoever in common with the other clip";
// parseVtt only extracts cue text from lines carrying YouTube's inline timing
// tags (e.g. `<00:00:01.000>`), so synthesize an auto-caption-style cue whose
// karaoke line carries every word.
function vtt(text: string): string {
const words = text.split(" ");
const tagged = words
.map((w, i) =>
i === 0 ? w : `<00:00:0${i % 10}.000> ${w}`,
)
.join("");
return [
"WEBVTT",
"Kind: captions",
"Language: en",
"",
"00:00:00.000 --> 00:10:00.000 align:start position:0%",
tagged,
"",
].join("\n");
}
function stamp(sec: number): string {
const h = String(Math.floor(sec / 3600)).padStart(2, "0");
const m = String(Math.floor((sec % 3600) / 60)).padStart(2, "0");
const s = String(sec % 60).padStart(2, "0");
return `${h}:${m}:${s}.000`;
}
// Plain (non-karaoke) VTT — one short cue per few words, no inline timing tags.
// parseVtt produces zero cues for this, so detection must recover the text via
// its raw-transcript fallback.
function plainVtt(text: string): string {
const words = text.split(" ");
const out = ["WEBVTT", ""];
let t = 0;
for (let i = 0; i < words.length; i += 6) {
out.push(`${stamp(t)} --> ${stamp(t + 3)}`, words.slice(i, i + 6).join(" "), "");
t += 3;
}
return out.join("\n");
}
type Seed = {
channel: string;
id: string;
platform: "youtube" | "rumble";
duration: number;
transcript: string | null;
plain?: boolean; // emit plain (non-karaoke) VTT
// Defaults to a per-id unique title, so the duration-blocking seeds below
// never accidentally block by title too. The title-blocking seeds set it.
title?: string;
};
const SEEDS: Seed[] = [
// Near-identical shorts across two channels and two platforms.
{ channel: "yt-a", id: "shorta", platform: "youtube", duration: 150, transcript: BASE_WORDS },
{ channel: "yt-b", id: "shortb", platform: "youtube", duration: 151, transcript: variant(5, "foxtrotx") },
{ channel: "rumble-c", id: "rumc", platform: "rumble", duration: 150, transcript: variant(25, "zulux") },
// A unique short sharing the duration bucket but unrelated content.
{ channel: "yt-a", id: "unique", platform: "youtube", duration: 150, transcript: UNIQUE_WORDS },
// Same duration, one is missing its transcript entirely. Must NOT cluster:
// duration coincidence alone is no longer treated as a duplicate.
{ channel: "yt-a", id: "metayes", platform: "youtube", duration: 90, transcript: BASE_WORDS },
{ channel: "yt-a", id: "metano", platform: "youtube", duration: 90, transcript: null },
// Plain-VTT near-duplicates (zero karaoke cues): must still cluster via the
// raw-transcript fallback.
{ channel: "yt-a", id: "plaina", platform: "youtube", duration: 120, transcript: PLAIN_WORDS, plain: true },
{ channel: "yt-b", id: "plainb", platform: "youtube", duration: 120, transcript: PLAIN_WORDS, plain: true },
// Same duration as the plain pair but unrelated content: must NOT be pulled in.
{ channel: "yt-a", id: "plainx", platform: "youtube", duration: 120, transcript: PLAIN_OTHER, plain: true },
// Long video that contains shorta's transcript (containment / all-durations).
{ channel: "yt-d", id: "longvid", platform: "youtube", duration: 600, transcript: `${BASE_WORDS} the rest of this much longer recording continues well beyond the clip` },
];
const PLATFORM_KEY: Record = {
youtube: "Youtube",
rumble: "Rumble",
};
async function seed(seeds: Seed[] = SEEDS): Promise {
const channels = new Set(seeds.map((s) => s.channel));
for (const channel of channels) {
const dir = resolvePath(`test-transcripts/channels/${channel}`);
await mkdir(dir, { recursive: true });
await writeFile(
`${dir}/config.json`,
JSON.stringify({ handling: "transcribe", name: channel }),
);
}
for (const s of seeds) {
const dir = resolvePath(`test-transcripts/channels/${s.channel}/data/${s.id}`);
await mkdir(dir, { recursive: true });
await writeFile(
`${dir}/metadata.info.json`,
JSON.stringify({
id: s.id,
title: s.title ?? `Title ${s.id}`,
channel: s.channel,
upload_date: "20240101",
duration: s.duration,
extractor_key: PLATFORM_KEY[s.platform],
webpage_url:
s.platform === "rumble"
? `https://rumble.com/${s.id}`
: `https://www.youtube.com/watch?v=${s.id}`,
}),
);
if (s.transcript !== null) {
await writeFile(
`${dir}/transcript.en.vtt`,
s.plain ? plainVtt(s.transcript) : vtt(s.transcript),
);
}
}
}
// The index AND the stats: one stage since release 18 ("Build stats dataset"
// is gone — the index update builds the stats datasets too).
async function buildData(page: import("@playwright/test").Page): Promise {
await buildIndex(page);
}
function clusterWith(report: DuplicateReport, slug: string): DuplicateCluster | undefined {
return report.clusters.find((c) => c.videoRefs.some((r) => r.slug === slug));
}
test("clusters cross-platform near-duplicate shorts (incl. plain VTT) and ignores duration-only coincidences", async ({
page,
}) => {
await resetData(null);
await seed();
await writeSite("testsite", {
channels: [...new Set(SEEDS.map((s) => s.channel))].map((slug) => ({
slug,
groupId: "default",
})),
});
await buildData(page);
await page.goto("/review");
await page.getByLabel("duplicate detection scope").selectOption("shorts");
await page.getByRole("button", { name: "detect duplicate shorts" }).click();
await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({
timeout: 30_000,
});
const report = await readJson("test-transcripts/duplicates.json");
expect(report.runConfig.thresholdSeconds).toBe(180);
// The three near-identical shorts cluster together, across platform + channel.
const near = clusterWith(report, "yt-a/shorta");
expect(near).toBeTruthy();
expect(near!.matchKind).toBe("transcript-near");
expect(near!.crossPlatform).toBe(true);
expect(near!.crossChannel).toBe(true);
expect(near!.videoRefs.map((r) => r.slug).sort()).toEqual([
"rumble-c/rumc",
"yt-a/shorta",
"yt-b/shortb",
]);
// The unique short is not part of any cluster.
expect(clusterWith(report, "yt-a/unique")).toBeUndefined();
// Duration coincidence alone is NOT a duplicate: the transcript-less video and
// its unrelated same-duration neighbour must not cluster.
expect(clusterWith(report, "yt-a/metano")).toBeUndefined();
expect(clusterWith(report, "yt-a/metayes")).toBeUndefined();
// Plain (non-karaoke) VTT yields zero parseVtt cues, but the raw-transcript
// fallback recovers the text so the near-identical plain pair still clusters.
const plain = clusterWith(report, "yt-a/plaina");
expect(plain).toBeTruthy();
expect(plain!.matchKind).toBe("transcript-exact");
expect(plain!.videoRefs.map((r) => r.slug).sort()).toEqual([
"yt-a/plaina",
"yt-b/plainb",
]);
// ...and the unrelated same-duration plain video stays out of that cluster.
expect(plain!.videoRefs.map((r) => r.slug)).not.toContain("yt-a/plainx");
expect(clusterWith(report, "yt-a/plainx")).toBeUndefined();
// Every cluster that would SHIP is content-confirmed. (The original invariant
// was "every reported cluster", which was true when duration coincidence
// produced nothing at all. Title + near-identical runtime is a far stronger
// claim and now does produce clusters — but they are quarantined behind
// needsReview, never auto-share, and never reach a built site, so the
// guarantee is kept exactly where it matters. These seeds have distinct titles
// anyway, so nothing here is a suspect.)
for (const c of report.clusters) {
if (c.needsReview) {
expect(c.matchKind).toBe("title-duration");
continue;
}
expect(["transcript-exact", "transcript-near"]).toContain(c.matchKind);
expect(c.score).not.toBeNull();
}
expect(report.clusters.filter((c) => c.needsReview)).toHaveLength(0);
// The detected cluster renders on the page after a refresh.
await page.reload();
await expect(page.getByLabel("duplicate-shorts")).toContainText(
"cross-platform",
);
});
test("all-durations run pulls a long video into the short's cluster via containment", async ({
page,
}) => {
await resetData(null);
await seed();
await writeSite("testsite", {
channels: [...new Set(SEEDS.map((s) => s.channel))].map((slug) => ({
slug,
groupId: "default",
})),
});
await buildData(page);
await page.goto("/review");
await page.getByLabel("duplicate detection scope").selectOption("all");
await page.getByRole("button", { name: "detect duplicate shorts" }).click();
await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({
timeout: 30_000,
});
const report = await readJson("test-transcripts/duplicates.json");
expect(report.runConfig.thresholdSeconds).toBeNull();
const cluster = clusterWith(report, "yt-a/shorta");
expect(cluster).toBeTruthy();
expect(cluster!.contained).toBe(true);
expect(cluster!.videoRefs.map((r) => r.slug)).toContain("yt-d/longvid");
});
// ---------------------------------------------------------------------------
// Title blocking: the pre-filter proposes, the transcript disposes
// ---------------------------------------------------------------------------
// Three same-title, compatible-runtime pairs that differ only in what the
// TRANSCRIPTS say. The point of the whole design is that the pre-filter treats
// all three identically and the content cascade then splits them three ways.
const TITLE_SEEDS: Seed[] = [
// (1) Same title, same content → CONFIRMED. A cross-platform mirror.
{ channel: "t-a", id: "mirror1", platform: "youtube", duration: 300, transcript: BASE_WORDS, title: "The Weekly Roundup" },
{ channel: "t-b", id: "mirror2", platform: "rumble", duration: 303, transcript: variant(5, "foxtrotx"), title: "The Weekly Roundup" },
// (2) Same title, same runtime, DIFFERENT content → REJECTED. Two episodes of
// a daily show. This is the case that makes nominating aggressively safe.
{ channel: "t-a", id: "epis1", platform: "youtube", duration: 240, transcript: BASE_WORDS, title: "Daily Show Recap" },
{ channel: "t-b", id: "epis2", platform: "rumble", duration: 241, transcript: UNIQUE_WORDS, title: "Daily Show Recap" },
// (3) Same title, same runtime, one side has NO transcript → untestable, so a
// needsReview SUSPECT rather than a match or a silent drop.
{ channel: "t-a", id: "susp1", platform: "youtube", duration: 200, transcript: BASE_WORDS, title: "Archive Upload" },
{ channel: "t-b", id: "susp2", platform: "rumble", duration: 200, transcript: null, title: "Archive Upload" },
// (4) Same title but a genuinely different cut (2x the runtime) → not even
// nominated, so the durations gate is doing its job.
{ channel: "t-a", id: "cut1", platform: "youtube", duration: 300, transcript: BASE_WORDS, title: "Extended Interview" },
{ channel: "t-b", id: "cut2", platform: "rumble", duration: 700, transcript: BASE_WORDS, title: "Extended Interview" },
];
test("title blocking confirms matching content, rejects differing content, and quarantines the untestable", async ({
page,
}) => {
await resetData(null);
await seed(TITLE_SEEDS);
await writeSite("testsite", {
channels: [...new Set(TITLE_SEEDS.map((s) => s.channel))].map((slug) => ({
slug,
groupId: "default",
})),
});
await buildData(page);
// Scope "all" runs corpus-wide, where the default blocking strategy is title.
await page.goto("/review");
await page.getByLabel("duplicate detection scope").selectOption("all");
await page.getByRole("button", { name: "detect duplicate shorts" }).click();
await expect(page.getByLabel("detect duplicate shorts result")).toBeVisible({
timeout: 30_000,
});
const report = await readJson("test-transcripts/duplicates.json");
expect(report.runConfig.blocking).toBe("title");
// (1) CONFIRMED — the content agreed, so this is a real cluster that may share.
const mirror = clusterWith(report, "t-a/mirror1");
expect(mirror).toBeTruthy();
expect(mirror!.needsReview).toBeFalsy();
expect(["transcript-exact", "transcript-near"]).toContain(mirror!.matchKind);
expect(mirror!.videoRefs.map((r) => r.slug).sort()).toEqual([
"t-a/mirror1",
"t-b/mirror2",
]);
// Alignment is measured and persisted for confirmed clusters — without it the
// viewer's "jump to this moment" has nothing honest to key off.
expect(mirror!.canonicalSlug).toBeTruthy();
for (const ref of mirror!.videoRefs) {
expect(typeof ref.aligned).toBe("boolean");
}
// (2) REJECTED — same title, same runtime, different words. No cluster at all.
expect(clusterWith(report, "t-a/epis1")).toBeUndefined();
expect(clusterWith(report, "t-b/epis2")).toBeUndefined();
// (3) SUSPECT — nothing could compare the content, so it is a review item.
const suspect = clusterWith(report, "t-a/susp1");
expect(suspect).toBeTruthy();
expect(suspect!.needsReview).toBe(true);
expect(suspect!.matchKind).toBe("title-duration");
expect(suspect!.score).toBeNull();
expect(suspect!.videoRefs.map((r) => r.slug).sort()).toEqual([
"t-a/susp1",
"t-b/susp2",
]);
// (4) NOT NOMINATED — a 300s and a 700s video are a different cut, not a
// mirror, however identical their titles.
expect(clusterWith(report, "t-a/cut1")).toBeUndefined();
expect(clusterWith(report, "t-b/cut2")).toBeUndefined();
// The suspect is visibly flagged in the editor's review list.
await page.reload();
const card = page.getByLabel(`duplicate cluster ${suspect!.clusterId}`);
await expect(card).toContainText("needs review");
await expect(card).toContainText("title + runtime");
});