import { test } from "node:test"; import assert from "node:assert/strict"; import { DEFAULT_TITLE_DURATION_RATIO, TITLE_DURATION_MIN_TOLERANCE_SECONDS, clusterIsPublishable, clusterMaySharePartial, durationsCompatible, filterClusterToChannels, isClusterReviewed, measureAlignment, normalizeTitleKey, pickCanonicalSlug, resolveCanonicalSlug, sanitizeDuplicateOverrides, strongerMatch, type DuplicateCluster, type DuplicateVideoRef, } from "./duplicates"; import type { Cue } from "./vtt"; function ref(over: Partial = {}): DuplicateVideoRef { return { slug: "chan/vid", channelSlug: "chan", channel: "Chan", platform: "youtube", id: "vid", title: "A video", duration: 600, uploadDate: "20250101", hasTranscript: true, ...over, }; } function cluster( refs: DuplicateVideoRef[], over: Partial = {}, ): DuplicateCluster { return { clusterId: "cluster1", matchKind: "transcript-exact", score: 1, contained: false, durationBucket: 600, crossPlatform: true, crossChannel: false, videoRefs: refs, ...over, }; } // --------------------------------------------------------------------------- // Canonical selection // --------------------------------------------------------------------------- test("a member with a transcript beats one without", () => { const c = cluster([ ref({ slug: "a/1", hasTranscript: false, duration: 900 }), ref({ slug: "b/2", hasTranscript: true, duration: 600 }), ]); assert.equal(pickCanonicalSlug(c), "b/2"); }); test("the longest (most complete) member wins next", () => { // Load-bearing: a mirror that cut the intro would place every shared chapter // wrong, so the fullest artifact owns the digest. const c = cluster([ ref({ slug: "a/1", duration: 600 }), ref({ slug: "b/2", duration: 640 }), ]); assert.equal(pickCanonicalSlug(c), "b/2"); }); test("the preferred platform breaks a duration tie", () => { const c = cluster([ ref({ slug: "a/1", platform: "rumble" }), ref({ slug: "b/2", platform: "youtube" }), ]); assert.equal(pickCanonicalSlug(c), "b/2"); }); test("the earliest upload breaks a platform tie", () => { const c = cluster([ ref({ slug: "a/1", uploadDate: "20250301" }), ref({ slug: "b/2", uploadDate: "20250101" }), ]); assert.equal(pickCanonicalSlug(c), "b/2"); }); test("selection is deterministic when everything ties", () => { const refs = [ref({ slug: "b/2" }), ref({ slug: "a/1" })]; assert.equal(pickCanonicalSlug(cluster(refs)), "a/1"); assert.equal(pickCanonicalSlug(cluster([...refs].reverse())), "a/1"); }); test("a human canonical override WINS over the rule", () => { const c = cluster([ ref({ slug: "a/1", duration: 600 }), ref({ slug: "b/2", duration: 900 }), ]); assert.equal(pickCanonicalSlug(c), "b/2", "the rule prefers the longer one"); const overrides = sanitizeDuplicateOverrides({ clusters: { cluster1: { canonicalSlug: "a/1" } }, }); assert.equal(resolveCanonicalSlug(c, overrides), "a/1"); }); test("an override naming a non-member is ignored, not obeyed", () => { const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })]); const overrides = sanitizeDuplicateOverrides({ clusters: { cluster1: { canonicalSlug: "someone/else" } }, }); assert.equal(resolveCanonicalSlug(c, overrides), pickCanonicalSlug(c)); }); test("notDuplicate suppresses the cluster entirely (nothing is shared)", () => { const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })]); const overrides = sanitizeDuplicateOverrides({ clusters: { cluster1: { notDuplicate: true } }, }); assert.equal(resolveCanonicalSlug(c, overrides), null); }); test("a recorded canonicalSlug on the report is honored over the rule", () => { const c = cluster( [ref({ slug: "a/1", duration: 600 }), ref({ slug: "b/2", duration: 900 })], { canonicalSlug: "a/1" }, ); assert.equal(resolveCanonicalSlug(c, null), "a/1"); }); test("isClusterReviewed distinguishes a decision from a bare stub", () => { const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })]); assert.equal(isClusterReviewed(c, null), false); assert.equal( isClusterReviewed( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { canonicalSlug: "a/1" } } }), ), true, ); // A decidedAt-only entry is dropped by the sanitizer, so it never reads as // reviewed. assert.equal( isClusterReviewed( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { decidedAt: "now" } } }), ), false, ); }); test("sanitizeDuplicateOverrides drops malformed entries rather than throwing", () => { const o = sanitizeDuplicateOverrides({ clusters: { good: { canonicalSlug: "a/1" }, blank: { canonicalSlug: " " }, junk: 42, arrayish: [], }, }); assert.deepEqual(Object.keys(o.clusters), ["good"]); }); test("sanitizeDuplicateOverrides survives a non-object file", () => { assert.deepEqual(sanitizeDuplicateOverrides(null).clusters, {}); assert.deepEqual(sanitizeDuplicateOverrides([1, 2, 3]).clusters, {}); }); // `confirmed` is the ONLY thing that lets a title+duration suspect ship or share // derived work. If the sanitizer dropped it, every human confirmation would be // silently discarded on the next read and the cluster would quietly revert to // "unreviewed" — a failure that leaves no trace anywhere. test("sanitizeDuplicateOverrides preserves confirmed", () => { const o = sanitizeDuplicateOverrides({ clusters: { c1: { confirmed: true }, c2: { confirmed: "yes" } }, }); assert.equal(o.clusters.c1?.confirmed, true); // Only a real boolean true counts; a truthy string is not a decision. assert.equal(o.clusters.c2, undefined); }); test("isClusterReviewed accepts a confirmed-only entry", () => { // A confirmation with no canonical override IS a decision — it is the whole // point of the review queue — so it must clear the cluster off the worklist. const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })]); assert.equal( isClusterReviewed( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { confirmed: true } } }), ), true, ); }); // --------------------------------------------------------------------------- // Suspects: needsReview gates sharing until a human confirms // --------------------------------------------------------------------------- test("a needsReview cluster shares nothing without a confirmation", () => { // Title + near-identical runtime is a suspicion, not a comparison. Two // episodes of a daily show can share both and be entirely different material, // and a shared digest would then describe the wrong video convincingly. const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })], { matchKind: "title-duration", score: null, needsReview: true, }); assert.equal(clusterMaySharePartial(c, null), false); assert.equal( clusterMaySharePartial( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { canonicalSlug: "a/1" } } }), ), false, "picking a canonical member is not the same as confirming the duplicate", ); }); test("confirmed unblocks a needsReview cluster", () => { const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })], { matchKind: "title-duration", score: null, needsReview: true, }); assert.equal( clusterMaySharePartial( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { confirmed: true } } }), ), true, ); }); test("confirmed does NOT unblock a contained (clip-of-longer) cluster", () => { // A clip is a different artifact from the recording it was cut from, however // confident a human is that they are related. const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })], { contained: true, }); assert.equal( clusterMaySharePartial( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { confirmed: true } } }), ), false, ); }); // --------------------------------------------------------------------------- // What reaches a built site // --------------------------------------------------------------------------- test("an unconfirmed suspect does NOT ship to a built site", () => { // compose-site applies this. A suspect asserts a relationship no machine and // no human has checked, so it stays an internal review queue. const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })], { matchKind: "title-duration", score: null, needsReview: true, }); assert.equal(clusterIsPublishable(c, null), false); assert.equal( clusterIsPublishable( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { canonicalSlug: "a/1" } } }), ), false, ); assert.equal( clusterIsPublishable( c, sanitizeDuplicateOverrides({ clusters: { cluster1: { confirmed: true } } }), ), true, ); }); test("a content-confirmed cluster ships with no overrides at all", () => { const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })]); assert.equal(clusterIsPublishable(c, null), true); }); test("a contained cluster ships even though it shares nothing", () => { // The two predicates deliberately disagree here: "this is a clip of that" is a // real relationship worth showing a viewer, and simultaneously a reason never // to copy the longer video's derived work onto the clip. const c = cluster([ref({ slug: "a/1" }), ref({ slug: "b/2" })], { contained: true, }); assert.equal(clusterIsPublishable(c, null), true); assert.equal(clusterMaySharePartial(c, null), false); }); // --------------------------------------------------------------------------- // Match tiers // --------------------------------------------------------------------------- test("strongerMatch ranks exact > near > title-duration", () => { assert.equal(strongerMatch("transcript-near", "transcript-exact"), "transcript-exact"); assert.equal(strongerMatch("transcript-exact", "transcript-near"), "transcript-exact"); assert.equal(strongerMatch("title-duration", "transcript-near"), "transcript-near"); assert.equal(strongerMatch("transcript-near", "title-duration"), "transcript-near"); assert.equal(strongerMatch("title-duration", "title-duration"), "title-duration"); // Load-bearing for needsReview: a cluster is only a suspect when EVERY pair in // it was untestable, so one content-confirmed pair must dominate. assert.equal(strongerMatch("title-duration", "transcript-exact"), "transcript-exact"); }); // --------------------------------------------------------------------------- // The title blocking key // --------------------------------------------------------------------------- test("normalizeTitleKey folds case, punctuation and diacritics", () => { assert.equal(normalizeTitleKey("Pokémon: The First Movie!"), "pokemon the first movie"); assert.equal(normalizeTitleKey(" Multiple spaces "), "multiple spaces"); assert.equal( normalizeTitleKey("The Show — Episode 4"), normalizeTitleKey("the show episode 4"), ); }); test("normalizeTitleKey drops platform re-upload suffixes", () => { // These are what a mirror ADDS to an otherwise identical title, so folding // them is what makes the mirror block with its original. const base = normalizeTitleKey("Weekly Roundup"); assert.equal(normalizeTitleKey("Weekly Roundup (reupload)"), base); assert.equal(normalizeTitleKey("Weekly Roundup [mirror]"), base); assert.equal(normalizeTitleKey("Weekly Roundup #shorts"), base); }); test("normalizeTitleKey does NOT merge a series", () => { // The whole reason the key is exact rather than fuzzy: a looser key merges // consecutive episodes, and every such merge is a false cluster a human then // has to reject. assert.notEqual(normalizeTitleKey("Episode 12"), normalizeTitleKey("Episode 13")); assert.notEqual( normalizeTitleKey("Morning Show Jan 4"), normalizeTitleKey("Morning Show Jan 5"), ); }); test("durationsCompatible allows a re-encode's drift but not a different cut", () => { // 2% of the longer runtime, floored at 2s. assert.equal(durationsCompatible(3600, 3620), true, "20s on an hour is 0.6%"); assert.equal(durationsCompatible(3600, 3800), false, "200s on an hour is 5.6%"); // The floor keeps sub-second rounding from splitting two short clips: 2% of // 30s is 0.6s, which nothing survives. assert.equal(durationsCompatible(30, 31), true); assert.equal(durationsCompatible(30, 40), false); assert.equal( durationsCompatible(1000, 1000 + 1000 * DEFAULT_TITLE_DURATION_RATIO), true, "exactly at the ratio is compatible", ); assert.equal( durationsCompatible(100, 100 + TITLE_DURATION_MIN_TOLERANCE_SECONDS), true, "exactly at the floor is compatible", ); }); test("durationsCompatible refuses a missing or zero duration", () => { // An unknown runtime is not a match — treating 0 as "close to 0" would block // every metadata-less video together. assert.equal(durationsCompatible(0, 0), false); assert.equal(durationsCompatible(600, 0), false); assert.equal(durationsCompatible(-5, -5), false); }); // --------------------------------------------------------------------------- // Narrowing a global cluster to one site's channels // --------------------------------------------------------------------------- test("filterClusterToChannels keeps in-site members and recomputes the flags", () => { const c = cluster( [ ref({ slug: "a/1", channelSlug: "a", platform: "youtube" }), ref({ slug: "b/2", channelSlug: "b", platform: "rumble" }), ref({ slug: "c/3", channelSlug: "c", platform: "odysee" }), ], { crossPlatform: true, crossChannel: true }, ); const out = filterClusterToChannels(c, new Set(["a", "b"])); assert.ok(out); assert.deepEqual(out.videoRefs.map((r) => r.slug), ["a/1", "b/2"]); assert.equal(out.crossPlatform, true); assert.equal(out.crossChannel, true); }); test("filterClusterToChannels drops a cluster that falls below two members", () => { // One video is not a visible duplicate — there is nothing to switch to. const c = cluster([ ref({ slug: "a/1", channelSlug: "a" }), ref({ slug: "b/2", channelSlug: "b" }), ]); assert.equal(filterClusterToChannels(c, new Set(["a"])), null); assert.equal(filterClusterToChannels(c, new Set(["z"])), null); }); test("filterClusterToChannels clears cross-* flags the survivors no longer earn", () => { const c = cluster( [ ref({ slug: "a/1", channelSlug: "a", platform: "youtube" }), ref({ slug: "a/2", channelSlug: "a", platform: "youtube" }), ref({ slug: "b/3", channelSlug: "b", platform: "rumble" }), ], { crossPlatform: true, crossChannel: true }, ); const out = filterClusterToChannels(c, new Set(["a"])); assert.ok(out); assert.equal(out.crossPlatform, false); assert.equal(out.crossChannel, false); }); test("filterClusterToChannels carries needsReview and the alignment fields through", () => { // The site-narrowing step must not launder a suspect into a shipped cluster, // and it must not lose the per-member alignment the viewer's jump depends on. const c = cluster( [ ref({ slug: "a/1", channelSlug: "a", aligned: true, offsetSeconds: 0 }), ref({ slug: "b/2", channelSlug: "b", aligned: false, offsetSeconds: 41.5 }), ref({ slug: "c/3", channelSlug: "c" }), ], { matchKind: "title-duration", score: null, needsReview: true, canonicalSlug: "a/1" }, ); const out = filterClusterToChannels(c, new Set(["a", "b"])); assert.ok(out); assert.equal(out.needsReview, true); assert.equal(out.canonicalSlug, "a/1"); assert.equal(out.videoRefs[0].aligned, true); assert.equal(out.videoRefs[1].aligned, false); assert.equal(out.videoRefs[1].offsetSeconds, 41.5); }); test("filterClusterToChannels leaves matchKind and score as the detector reported them", () => { // Intentional and documented: they describe the strongest pair in the FULL // cluster. Recomputing them here would change the meaning of clusters already // shipped, and there is no similarity data at this point to recompute from. const c = cluster( [ ref({ slug: "a/1", channelSlug: "a" }), ref({ slug: "b/2", channelSlug: "b" }), ref({ slug: "c/3", channelSlug: "c" }), ], { matchKind: "transcript-exact", score: 1 }, ); const out = filterClusterToChannels(c, new Set(["a", "b"])); assert.ok(out); assert.equal(out.matchKind, "transcript-exact"); assert.equal(out.score, 1); }); // --------------------------------------------------------------------------- // The alignment gate // --------------------------------------------------------------------------- // A synthetic transcript whose sentences do NOT share long phrases, so the // 8-word anchors are unambiguous — which is what the gate requires and what a // real transcript mostly provides. (An earlier fixture repeated the same nine // words in every cue; the gate correctly refused to measure it, which is how the // ambiguous-anchor rule got written.) const VOCAB = [ "filing", "deadline", "jury", "selection", "witness", "testimony", "exhibit", "objection", "sustained", "overruled", "docket", "motion", "dismissal", "appeal", "verdict", "sentencing", "transcript", "counsel", "recess", "subpoena", "affidavit", "discovery", "deposition", "settlement", "mediation", "injunction", "damages", "liability", "negligence", "statute", "precedent", "jurisdiction", "venue", "indictment", "arraignment", "plea", "bail", "custody", "warrant", "evidence", ]; function makeCues(shiftSeconds = 0, count = 40): Cue[] { return Array.from({ length: count }, (_, i) => { // Each cue draws a distinct rotation of the vocabulary, so no 8-word window // recurs anywhere in the transcript. const words = Array.from( { length: 12 }, (_, j) => VOCAB[(i * 7 + j * 3) % VOCAB.length], ); return { start: i * 15 + shiftSeconds, end: i * 15 + 14 + shiftSeconds, text: `${words.join(" ")} marker${i}`, }; }); } test("identical timings are aligned at a zero offset", () => { const a = makeCues(0); const result = measureAlignment(a, makeCues(0)); assert.equal(result.aligned, true); assert.equal(result.maxOffsetSeconds, 0); assert.equal(result.matchedAnchors, result.totalAnchors); }); test("a mirror with a shifted intro is REFUSED", () => { // The failure mode that looks like success: identical text, every chapter // placed 40s wrong. Content similarity cannot see this — shingles are a set. const result = measureAlignment(makeCues(0), makeCues(40)); assert.equal(result.aligned, false); assert.equal(result.reason, "offset-exceeded"); assert.ok(result.maxOffsetSeconds >= 39 && result.maxOffsetSeconds <= 41); }); test("a sub-tolerance shift is still aligned", () => { const result = measureAlignment(makeCues(0), makeCues(2)); assert.equal(result.aligned, true); }); test("the tolerance is configurable", () => { assert.equal( measureAlignment(makeCues(0), makeCues(8), { toleranceSeconds: 10 }).aligned, true, ); assert.equal( measureAlignment(makeCues(0), makeCues(8), { toleranceSeconds: 3 }).aligned, false, ); }); test("unrelated transcripts are refused for want of anchors", () => { const other: Cue[] = Array.from({ length: 40 }, (_, i) => ({ start: i * 15, end: i * 15 + 14, text: `zebra${i} quartz${i} lantern${i} beacon${i} pumice${i} sorrel${i} thicket${i} vellum${i} wicket${i}`, })); const result = measureAlignment(makeCues(0), other); assert.equal(result.aligned, false); assert.equal(result.reason, "too-few-anchors"); }); test("a mirror aligned at the start but drifting mid-way is refused", () => { // Sampling SEVERAL anchors is what catches an ad break inserted in the middle. const canonical = makeCues(0, 40); const drifting = canonical.map((c, i) => ({ ...c, start: i < 20 ? c.start : c.start + 45, end: i < 20 ? c.end : c.end + 45, })); const result = measureAlignment(canonical, drifting); assert.equal(result.aligned, false); assert.equal(result.reason, "offset-exceeded"); }); test("an empty transcript is refused, not treated as aligned", () => { assert.equal(measureAlignment([], makeCues(0)).aligned, false); assert.equal(measureAlignment(makeCues(0), []).reason, "empty-transcript"); }); test("a contained cluster (clip of a longer video) never shares", () => { // Correct even at a perfect zero offset: a clip is a different artifact, and // the longer video's chapters describe material it does not contain. assert.equal( clusterMaySharePartial(cluster([ref()], { contained: true })), false, ); assert.equal( clusterMaySharePartial(cluster([ref()], { contained: false })), true, ); });