// Shared types + tuning constants + pure helpers for cross-platform duplicate // shorts detection. The algorithm (a hybrid cascade: cheap duration-bucket // pre-clustering followed by transcript-content confirmation) lives in // controller/duplicateShorts.ts; everything pure and reusable lives here so it // can be unit-tested and imported by the editor without pulling in lmdb/fs. import type { Platform } from "./platform"; import type { Cue } from "./vtt"; export const DUPLICATE_REPORT_VERSION = 1; // Default duration cutoff (seconds) for what counts as a "short". One-off runs // can override this (higher threshold, or null for "all durations"). export const DEFAULT_SHORT_THRESHOLD_SECONDS = 180; // Phase-1 bucketing window. Two videos within ~this many seconds of each other // land in a shared comparison set (see controller). export const DEFAULT_DURATION_TOLERANCE_SECONDS = 2; // Phase-2 near-duplicate threshold (5-gram Jaccard). // // 0.35, lowered from 0.6 on measurement (2026-07-29). 0.6 was tuned for // same-engine text and structurally failed the case detection exists to catch: // the two sides of a cross-platform mirror are transcribed by DIFFERENT ASR // engines, and a 5-gram Jaccard is unforgiving of word-level disagreement, so // two transcripts of the same audio land at ~0.35–0.60 rather than ≥ 0.6. // // Bracketed corpus-wide at 0.6 / 0.45 / 0.35 / 0.25 on identical inputs. Each // step down is a strict superset — zero videos are lost — and the returns fall // off a cliff: 0.6→0.45 adds 4,264 clusters, 0.45→0.35 adds 324, 0.35→0.25 adds // 59. The marginal band was read, not just counted: of the 4,264 admitted at // 0.45, 96.2% are byte-identical titles, 99.5% cross-platform, and 3 (0.07%) // are same-channel. The 324 admitted at 0.35 are the same shape (97.2% // byte-identical titles) and the riskiest 11 were inspected individually — all // same recording, same runtime to the second, mirrored platform. // // 0.45 is where the RECALL knee is, if a more conservative value is ever // wanted; 0.35 is chosen because its marginal band is still clean and the // report's job is to surface real mirrors to readers. 0.25 is the flat tail. // // This is NOT a meaningful digest-sweep cost lever, whatever the sharing code's // header says: cluster members are 19% of the corpus by video count but only // 4.3% of its AUDIO-HOURS (mirrors skew short, long-form VODs are unclustered), // so 0.6→0.35 moves the sweep from 80.0 to 78.8 days. It is a publishing change // — it is what the archive asserts to readers — and it is justified on that. // See bin/digest-plan.ts for the measurement. export const DEFAULT_NEAR_THRESHOLD = 0.35; // Containment threshold for the "short is a clip of a longer video" case. export const DEFAULT_CONTAINMENT_THRESHOLD = 0.8; // Shingle (word n-gram) size for similarity. 5 tolerates word-level ASR // variance between whisper and auto-captions without over-matching on common // unigrams. export const DEFAULT_SHINGLE_SIZE = 5; // Match tiers, weakest first. // // `title-duration` is a SUSPECT, not a confirmation. Duration coincidence alone // was never a match — it produced enormous false clusters of unrelated // same-length videos — but the same title AND a near-identical runtime is a // different claim entirely, and it is the only signal available when one side // has no transcript to compare. Such a cluster is flagged `needsReview` and // shares nothing until a human confirms it (see clusterMaySharePartial). export type DuplicateMatchKind = | "title-duration" | "transcript-near" | "transcript-exact"; // Two videos with the same normalized title are candidates when their runtimes // agree within this fraction of the longer one. A mirror re-encode drifts by a // second or two; 2% also absorbs an ad-break difference on a long video without // admitting a genuinely different cut. export const DEFAULT_TITLE_DURATION_RATIO = 0.02; // ...but never demand tighter than this, so two 30-second clips are not split by // sub-second rounding. export const TITLE_DURATION_MIN_TOLERANCE_SECONDS = 2; // A title group larger than this is a FORMAT, not a title — "live stream", // "untitled", a daily show's date-less name. Pairing inside it is quadratic and // the matches would be noise, so the group is skipped and reported rather than // silently truncated. export const MAX_TITLE_GROUP_SIZE = 40; // The same treatment for a pathological DURATION block. Round numbers attract // videos — a corpus can hold thousands of videos that are exactly 60s — and one // such block is quadratic on its own. Blocks over this are skipped and reported, // for the same reason: a silent truncation reads as "covered everything". // Generous, because the proven shorts-mode run has legitimately large buckets // (34,915 shorts over ~90 two-second buckets) and must keep behaving as it does. export const MAX_DURATION_BLOCK_SIZE = 2000; export type DuplicateVideoRef = { slug: string; // `${channelSlug}/${id}` channelSlug: string; channel: string; // display name platform: Platform; id: string; // canonical id (may differ from the on-disk dir name) // The on-disk directory under `/data/`, recorded ONLY when it differs // from `id` — which it does for 14.5% of the corpus (every Rumble re-upload: // 7,870 of the 7,870 `the-quartering-rumble` videos alone). // // It is here because `slug` is `${channelSlug}/${id}` and is therefore NOT a // path. Anything that resolves a member back to its files has to be told the // difference, and the detector is the only place that cheaply knows it (it // keys its own scan by directory). Omitted when dir === id so the shipped // report does not grow a redundant field on the other 85%; readers fall back // to `id`, which is what every reader assumed before this field existed. videoDir?: string; title: string; duration: number; // seconds uploadDate: string; // YYYYMMDD hasTranscript: boolean; // Timing alignment against the cluster's canonical member, measured by // measureAlignment() at detection time (confirmed clusters only — see the // detector). Both are optional: reports written before these fields existed // lack them, and absent must be read as "not measured", i.e. NOT aligned. // // `offsetSeconds` is the largest |offset| observed across the matched anchors, // not a signed shift — the same quantity writeSharedDigest already records. // It is informational; `aligned` is the gate. Anything that seeks INTO a // sibling (the search-result duplicate badge, a shared digest's chapters) must // key off `aligned`, because a mirror with a longer intro matches on text at // shifted times and would otherwise land in the wrong place while looking // perfectly plausible. offsetSeconds?: number | null; aligned?: boolean; }; export type DuplicateCluster = { clusterId: string; // stable sha1 over sorted member slugs matchKind: DuplicateMatchKind; // strongest tier present in the cluster score: number | null; // strongest pair score; null for metadata-only contained: boolean; // any member pair matched by containment (clip-of-longer) durationBucket: number; // representative rounded duration (seconds) crossPlatform: boolean; // members span more than one platform crossChannel: boolean; // members span more than one channelSlug // True when the cluster's strongest evidence is title+duration only — nothing // compared the actual content. It is offered for human review and shares no // derived work until confirmed. Optional: reports written before this field // existed lack it, and absent means "content-confirmed", which is what those // reports only ever contained. needsReview?: boolean; videoRefs: DuplicateVideoRef[]; // The member that OWNS derived work for this cluster: the one an AI digest is // generated for, and the one aligned mirrors copy it from. Chosen by // pickCanonicalSlug() at detection time and human-overridable afterwards // (see DuplicateOverrides). Optional: reports written before this field // existed lack it, so readers fall back to pickCanonicalSlug(). canonicalSlug?: string; }; // How candidate pairs are NOMINATED. A blocking strategy never decides anything // on its own — it only proposes pairs that the transcript cascade then confirms // or rejects (see the controller's evalBlocked). The choice is therefore about // recall and cost, not correctness. // // "duration" — same/adjacent rounded-duration bucket. The only strategy that // catches a RE-TITLED mirror. Corpus-viable only since the // streaming rewrite; still much the more expensive of the two. // "title" — exact normalized-title groups, paired when the runtimes agree. // Effectively linear, and it is what actually finds cross-platform // re-uploads, which keep their name. // "both" — the union, deduped. export type DuplicateBlocking = "duration" | "title" | "both"; export type DuplicateRunConfig = { thresholdSeconds: number | null; // null === all durations durationToleranceSeconds: number; nearThreshold: number; // Jaccard containmentThreshold: number; shingleSize: number; // Optional: reports written before blocking was configurable lack it, and // those were all duration-blocked. blocking?: DuplicateBlocking; }; export type DuplicateReport = { version: number; generatedAt: string; runConfig: DuplicateRunConfig; totals: { videosScanned: number; clusters: number; videosInClusters: number; }; clusters: DuplicateCluster[]; }; export const DUPLICATES_FILENAME = "duplicates.json"; // --------------------------------------------------------------------------- // Human review: canonical choice + not-a-duplicate, kept OUT of the report // --------------------------------------------------------------------------- // duplicates.json is regenerated wholesale by every detection run, so a human // decision recorded in it would be destroyed on the next run. It therefore lives // in a sibling override file — the same separation, for the same reason, as // ai-digest.overrides.json vs ai-digest.json. export const DUPLICATE_OVERRIDES_FILENAME = "duplicates.overrides.json"; export const DUPLICATE_OVERRIDES_VERSION = 1; export type DuplicateClusterOverride = { // Operator's choice of canonical member (a `${channelSlug}/${id}` slug). Wins // over the rule in pickCanonicalSlug. Ignored when the slug is not a member. canonicalSlug?: string; // "These are not the same video." Suppresses the cluster entirely: it stops // being offered for review AND stops sharing derived work. The detector will // keep finding it (content really is similar), which is exactly why the // decision has to be recorded outside the report. notDuplicate?: boolean; // "I looked, and these really are the same video." The positive counterpart of // notDuplicate, and the ONLY thing that lets a title-duration cluster share // derived work. Content-confirmed clusters do not need it. confirmed?: boolean; decidedAt?: string; note?: string; }; export type DuplicateOverrides = { version: number; // Keyed by clusterId — a stable sha1 over the sorted member slugs, so the key // survives re-detection as long as the membership does. A cluster that GAINS a // member gets a new id and returns to review, which is the honest behavior: // the canonical choice was made over a different set of videos. clusters: Record; }; export function emptyDuplicateOverrides(): DuplicateOverrides { return { version: DUPLICATE_OVERRIDES_VERSION, clusters: {} }; } // Coerce a raw overrides file, dropping ill-typed entries rather than throwing — // one hand-edit typo must not break the duplicates page. export function sanitizeDuplicateOverrides(value: unknown): DuplicateOverrides { const out = emptyDuplicateOverrides(); if (!value || typeof value !== "object") return out; const raw = (value as { clusters?: unknown }).clusters; if (!raw || typeof raw !== "object" || Array.isArray(raw)) return out; for (const [clusterId, entry] of Object.entries(raw as Record)) { if (!entry || typeof entry !== "object") continue; const e = entry as Record; const override: DuplicateClusterOverride = {}; if (typeof e.canonicalSlug === "string" && e.canonicalSlug.trim()) { override.canonicalSlug = e.canonicalSlug.trim(); } if (e.notDuplicate === true) override.notDuplicate = true; if (e.confirmed === true) override.confirmed = true; if (typeof e.decidedAt === "string") override.decidedAt = e.decidedAt; if (typeof e.note === "string" && e.note.trim()) override.note = e.note.trim(); if (Object.keys(override).length === 0) continue; out.clusters[clusterId] = override; } return out; } // --------------------------------------------------------------------------- // Canonical member selection // --------------------------------------------------------------------------- // Platform preference for the canonical member, most-preferred first. YouTube // leads because its videos carry the richest metadata and the most reliable // caption tracks, so a digest generated there is the best one to share. const PLATFORM_PREFERENCE: ReadonlyArray = [ "youtube", "rumble", "odysee", "kick", "twitch", "bitchute", ]; function platformRank(platform: string): number { const i = PLATFORM_PREFERENCE.indexOf(platform); return i < 0 ? PLATFORM_PREFERENCE.length : i; } // Pick the cluster member that should own derived work, by rule. Ordered by // what actually makes a shared digest good: // 1. has a transcript — you cannot digest a video without one // 2. longest duration — the most complete artifact; a mirror that cuts // the intro would place every shared chapter wrong // 3. preferred platform — richest metadata // 4. earliest upload — the original, where the same content appears twice // 5. slug — a total order, so the choice is deterministic // Deterministic and pure: no Date.now(), no Math.random(), so re-detection over // unchanged inputs picks the same member. export function pickCanonicalSlug(cluster: DuplicateCluster): string { const refs = cluster.videoRefs; if (refs.length === 0) return ""; const best = refs.reduce((a, b) => (canonicalBetter(b, a) ? b : a)); return best.slug; } function canonicalBetter( candidate: DuplicateVideoRef, incumbent: DuplicateVideoRef, ): boolean { if (candidate.hasTranscript !== incumbent.hasTranscript) { return candidate.hasTranscript; } if (candidate.duration !== incumbent.duration) { return candidate.duration > incumbent.duration; } const pc = platformRank(candidate.platform); const pi = platformRank(incumbent.platform); if (pc !== pi) return pc < pi; if (candidate.uploadDate !== incumbent.uploadDate) { // Empty upload dates sort last so a dated member wins over an undated one. if (!candidate.uploadDate) return false; if (!incumbent.uploadDate) return true; return candidate.uploadDate < incumbent.uploadDate; } return candidate.slug.localeCompare(incumbent.slug) < 0; } // The effective canonical member: the human choice when it names a real member, // otherwise the recorded one, otherwise the rule. Returns null for a cluster a // human marked not-a-duplicate — such a cluster shares nothing. export function resolveCanonicalSlug( cluster: DuplicateCluster, overrides?: DuplicateOverrides | null, ): string | null { const override = overrides?.clusters[cluster.clusterId]; if (override?.notDuplicate) return null; const members = new Set(cluster.videoRefs.map((r) => r.slug)); if (override?.canonicalSlug && members.has(override.canonicalSlug)) { return override.canonicalSlug; } if (cluster.canonicalSlug && members.has(cluster.canonicalSlug)) { return cluster.canonicalSlug; } return pickCanonicalSlug(cluster) || null; } // Whether a human has recorded a decision for this cluster. Drives the // /review "awaiting review" list — a cluster with no decision is work. export function isClusterReviewed( cluster: DuplicateCluster, overrides?: DuplicateOverrides | null, ): boolean { const o = overrides?.clusters[cluster.clusterId]; if (!o) return false; return ( o.notDuplicate === true || o.confirmed === true || Boolean(o.canonicalSlug) ); } // --------------------------------------------------------------------------- // The timestamp-alignment gate — the correctness crux of digest sharing // --------------------------------------------------------------------------- // Content similarity does NOT imply timing alignment, and this is the failure // mode that looks like success: a mirror with a 40-second-longer intro has // matching text at shifted times, so a shared digest places EVERY chapter wrong // while looking perfectly plausible. Nothing in the 5-gram Jaccard cascade // notices, because shingles are a set — order and position are discarded. // // So before sharing we measure it directly: sample several anchor phrases spread // through the canonical transcript, find where each occurs in the mirror, and // require every offset to be near zero. // Max |offset| (seconds) at any anchor for two transcripts to count as aligned. // Tight on purpose: a chapter placed 5s early still lands in the right sentence, // 15s does not. export const DEFAULT_ALIGNMENT_TOLERANCE_SECONDS = 5; // How many anchors to sample. Several, spread out, because a mirror can share a // start time and then diverge at an ad break in the middle. export const DEFAULT_ALIGNMENT_ANCHORS = 5; // An anchor must be this many words long to be a reliable locator; shorter // phrases recur. const ANCHOR_WORDS = 8; // How far forward to look for a USABLE anchor when the phrase at the sampled // position is ambiguous (occurs more than once on either side). Real transcripts // repeat themselves — intros, catchphrases, ad reads — so without this the gate // refuses perfectly aligned mirrors and the sharing optimisation never fires. const ANCHOR_SEARCH_WORDS = 400; export type AlignmentResult = { aligned: boolean; // Largest |offset| observed across the matched anchors, in seconds. maxOffsetSeconds: number; // How many anchors were located in the other transcript. Too few and the // measurement is not trustworthy, so `aligned` is false regardless of offset. matchedAnchors: number; totalAnchors: number; reason?: | "too-few-anchors" | "offset-exceeded" | "empty-transcript" | "contained"; }; function normalizeWords(cues: Cue[]): { word: string; start: number }[] { const out: { word: string; start: number }[] = []; for (const cue of cues) { const words = cue.text .toLowerCase() .replace(/[^\p{L}\p{N}\s]/gu, " ") .split(/\s+/) .filter(Boolean); for (const word of words) out.push({ word, start: cue.start }); } return out; } // Word n-gram -> { first start time, occurrence count }. The count is what makes // an ambiguous phrase skippable instead of silently mismatched. function buildAnchorIndex( words: { word: string; start: number }[], ): Map { const index = new Map(); for (let i = 0; i + ANCHOR_WORDS <= words.length; i++) { const key = words .slice(i, i + ANCHOR_WORDS) .map((w) => w.word) .join(" "); const existing = index.get(key); if (existing) existing.count++; else index.set(key, { start: words[i].start, count: 1 }); } return index; } // Measure timing alignment between two cue lists. Pure, so the gate is directly // unit-testable (and it IS tested: a shifted-intro mirror must fail). export function measureAlignment( canonical: Cue[], other: Cue[], opts: { toleranceSeconds?: number; anchors?: number; } = {}, ): AlignmentResult { const tolerance = opts.toleranceSeconds ?? DEFAULT_ALIGNMENT_TOLERANCE_SECONDS; const anchorCount = Math.max(1, opts.anchors ?? DEFAULT_ALIGNMENT_ANCHORS); const a = normalizeWords(canonical); const b = normalizeWords(other); if (a.length < ANCHOR_WORDS || b.length < ANCHOR_WORDS) { return { aligned: false, maxOffsetSeconds: Infinity, matchedAnchors: 0, totalAnchors: 0, reason: "empty-transcript", }; } // Index BOTH sides' word n-grams, counting occurrences — not just recording the // first. A phrase that occurs twice is not a locator: matching it to whichever // copy came first would report a bogus offset, which fails in both directions // (a false refusal wastes a generation; a false match shares a misplaced // digest). So an ambiguous phrase is skipped rather than guessed at. const indexB = buildAnchorIndex(b); const indexA = buildAnchorIndex(a); // Anchors spread evenly through the canonical transcript, skipping the very // start and end (intros/outros are exactly where mirrors differ). const usable = a.length - ANCHOR_WORDS; const positions: number[] = []; for (let k = 1; k <= anchorCount; k++) { positions.push(Math.floor((usable * k) / (anchorCount + 1))); } let matched = 0; let maxOffset = 0; const usedTargets = new Set(); for (const pos of positions) { // Walk forward from the sampled position until a phrase is unique on BOTH // sides. Bounded, so a pathologically repetitive stretch just yields no // anchor here rather than scanning the whole transcript. const limit = Math.min(usable, pos + ANCHOR_SEARCH_WORDS); for (let i = pos; i <= limit; i++) { const key = a .slice(i, i + ANCHOR_WORDS) .map((w) => w.word) .join(" "); if ((indexA.get(key)?.count ?? 0) !== 1) continue; const hit = indexB.get(key); if (!hit || hit.count !== 1) continue; // Don't let two sampled positions collapse onto the same anchor — that // would report "2 anchors matched" from one measurement. if (usedTargets.has(hit.start)) break; usedTargets.add(hit.start); matched++; maxOffset = Math.max(maxOffset, Math.abs(hit.start - a[i].start)); break; } } // Require a majority of anchors to be located: a mirror we can only match in // one place is not a mirror we can trust timings from. const enough = matched >= Math.ceil(positions.length / 2); if (!enough) { return { aligned: false, maxOffsetSeconds: matched === 0 ? Infinity : maxOffset, matchedAnchors: matched, totalAnchors: positions.length, reason: "too-few-anchors", }; } if (maxOffset > tolerance) { return { aligned: false, maxOffsetSeconds: maxOffset, matchedAnchors: matched, totalAnchors: positions.length, reason: "offset-exceeded", }; } return { aligned: true, maxOffsetSeconds: maxOffset, matchedAnchors: matched, totalAnchors: positions.length, }; } // Whether a cluster may share derived work AT ALL, before any timing is measured. // A `contained` cluster matched by containment — one member is a CLIP of a longer // video, not a mirror of it. A clip is a different artifact: the longer video's // chapters describe material the clip does not contain, so sharing wholesale // would be wrong even at a perfect zero offset. // // A `needsReview` cluster (title+duration only) shares nothing either, until a // human records `confirmed: true`. The alignment gate would still protect the // TIMING, but nothing here has compared the CONTENT — two episodes of a daily // show can share a title and a runtime and be entirely different material, and // a shared digest would then describe the wrong video convincingly. export function clusterMaySharePartial( cluster: DuplicateCluster, overrides?: DuplicateOverrides | null, ): boolean { if (cluster.contained) return false; if (!cluster.needsReview) return true; return overrides?.clusters[cluster.clusterId]?.confirmed === true; } // Whether a cluster may be SHIPPED to a built site at all. // // Distinct from clusterMaySharePartial, which asks whether derived work may flow // between members. The two disagree in both directions, on purpose: // // a `contained` cluster SHIPS (a clip of a longer video is a real, useful // relationship for a viewer to see) but SHARES NOTHING (the longer video's // chapters describe material the clip does not contain); // // an unconfirmed `needsReview` cluster does NEITHER — nothing compared its // members' content, so asserting the relationship to a viewer would be // claiming something no machine and no human has actually checked. It stays an // internal review queue until someone records `confirmed`. // // Fails closed: no overrides means no confirmations means no suspects ship. export function clusterIsPublishable( cluster: DuplicateCluster, overrides?: DuplicateOverrides | null, ): boolean { if (!cluster.needsReview) return true; return overrides?.clusters[cluster.clusterId]?.confirmed === true; } // The blocking key for title-based candidate generation. // // This is what makes corpus-wide detection tractable. The alternative the first // implementation used — pairing every short with every longer video — is // quadratic and measured at 498 MILLION pairs on this corpus. Grouping by an // exact normalized title instead is one pass and a hash lookup, and it is the // signal that actually finds cross-platform mirrors, which are re-uploads of the // same file under the same name. // // Deliberately conservative: lowercase, strip punctuation and diacritics, drop a // few platform suffixes, collapse whitespace. It does NOT stem, fuzzy-match or // drop stopwords — a looser key merges a series ("Episode 12" vs "Episode 13") // and every such merge is a false cluster a human then has to reject. const TITLE_NOISE_RE = /\s*(?:#shorts?|\(official(?: video| audio)?\)|\[official\]|\|\s*full episode|\(full episode\)|\(reupload\)|\[reupload\]|\(mirror\)|\[mirror\])\s*/gi; export function normalizeTitleKey(title: string): string { return title .normalize("NFKD") // Strip combining marks so "Pokémon" and "Pokemon" block together. .replace(/[\u0300-\u036f]/g, "") .toLowerCase() .replace(TITLE_NOISE_RE, " ") .replace(/[^\p{L}\p{N}]+/gu, " ") .trim(); } // Are two same-titled videos close enough in runtime to be candidates? export function durationsCompatible( a: number, b: number, ratio = DEFAULT_TITLE_DURATION_RATIO, ): boolean { if (!(a > 0) || !(b > 0)) return false; const tolerance = Math.max( TITLE_DURATION_MIN_TOLERANCE_SECONDS, Math.max(a, b) * ratio, ); return Math.abs(a - b) <= tolerance; } // --------------------------------------------------------------------------- // Pure helpers (no I/O) — exported for direct testing. // --------------------------------------------------------------------------- // Build a normalized comparison string from cues. Cues are already // tag-stripped, entity-decoded, whitespace-collapsed and consecutive-deduped by // parseVtt/parseWhisper, so this only lowercases and drops punctuation — the // two things auto-captions and whisper most often disagree on. export function comparisonText(cues: Cue[]): string { return cues .map((c) => c.text) .join(" ") .toLowerCase() .replace(/[^\p{L}\p{N}\s]/gu, " ") .replace(/\s+/g, " ") .trim(); } // k-word shingles (n-grams) as a set of space-joined token windows. Texts // shorter than k collapse to a single shingle of the whole text. export function shingles(text: string, k: number): Set { const set = new Set(); if (!text) return set; const words = text.split(" "); if (words.length < k) { set.add(words.join(" ")); return set; } for (let i = 0; i + k <= words.length; i++) { set.add(words.slice(i, i + k).join(" ")); } return set; } export function intersectionSize(a: Set, b: Set): number { const [small, large] = a.size <= b.size ? [a, b] : [b, a]; let n = 0; for (const x of small) if (large.has(x)) n++; return n; } // |A∩B| / |A∪B|. export function jaccard(a: Set, b: Set): number { if (a.size === 0 && b.size === 0) return 0; const inter = intersectionSize(a, b); return inter / (a.size + b.size - inter); } // |A∩B| / min(|A|,|B|): how much of the smaller set is contained in the larger. // Catches a short whose shingles are largely a subset of a longer video's. export function containment(a: Set, b: Set): number { const min = Math.min(a.size, b.size); if (min === 0) return 0; return intersectionSize(a, b) / min; } // Minimal union-find over string keys, used to merge transitively-related // duplicate pairs (A≈B, B≈C ⇒ {A,B,C}) into clusters. export class UnionFind { private parent = new Map(); add(x: string): void { if (!this.parent.has(x)) this.parent.set(x, x); } find(x: string): string { this.add(x); let root = x; while (this.parent.get(root) !== root) root = this.parent.get(root) as string; // Path compression. let cur = x; while (this.parent.get(cur) !== root) { const next = this.parent.get(cur) as string; this.parent.set(cur, root); cur = next; } return root; } union(a: string, b: string): void { const ra = this.find(a); const rb = this.find(b); if (ra !== rb) this.parent.set(ra, rb); } // Connected components, keyed by representative root. Only includes keys that // were add()ed or union()ed. groups(): string[][] { const out = new Map(); for (const key of this.parent.keys()) { const root = this.find(key); const arr = out.get(root); if (arr) arr.push(key); else out.set(root, [key]); } return [...out.values()]; } } // Tier ranking so a cluster reports its strongest evidence. const MATCH_RANK: Record = { "title-duration": 0, "transcript-near": 1, "transcript-exact": 2, }; export function strongerMatch( a: DuplicateMatchKind, b: DuplicateMatchKind, ): DuplicateMatchKind { return MATCH_RANK[a] >= MATCH_RANK[b] ? a : b; } // Narrow a cluster to the channels a single site exposes. The detector runs // globally over the whole channel pool, but each deployed site carries only a // subset of channels, so members outside the site can't be opened there. We // keep only in-site members, drop the cluster entirely if fewer than two // remain (a lone video is not a visible duplicate), and recompute the // cross-platform / cross-channel flags over the survivors. matchKind / score / // contained / durationBucket are left as the detector reported them — they // describe the strongest pair in the full cluster, which is the best signal we // have without re-running similarity here. Returns null when the cluster does // not survive. export function filterClusterToChannels( cluster: DuplicateCluster, channelSlugs: Set, ): DuplicateCluster | null { const videoRefs = cluster.videoRefs.filter((r) => channelSlugs.has(r.channelSlug), ); if (videoRefs.length < 2) return null; return { ...cluster, videoRefs, crossPlatform: new Set(videoRefs.map((r) => r.platform)).size > 1, crossChannel: new Set(videoRefs.map((r) => r.channelSlug)).size > 1, }; }