commit 61c73a90386f9392a233c152a99c2dead10c6ff0
parent b2e8a0568f6d3987ae13346ac76a9662e5945fa4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 5 Jun 2026 18:32:50 -0400
filters for duplicates page
Diffstat:
5 files changed, 540 insertions(+), 58 deletions(-)
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,7 +1,7 @@
# Changelog
-## [Unreleased]
-- **Duplicates page.** A new **Duplicates** tab in the primary nav lists shorts whose transcripts match across channels and platforms — re-uploads, mirrors, and cross-posts of the same clip — grouped into clusters (strongest match first). Each cluster shows how it matched (exact or near transcript, similarity score, clip-of-longer) and whether it spans multiple channels or platforms; clicking any member plays it in the transcript modal. The list is scoped to the channels this site exposes, so every entry is openable here. Sites with no detection run yet show an empty state.
+## [0.3.5] - 2026-06-04
+- **Duplicates page.** A new **Duplicates** tab in the primary nav lists shorts whose transcripts match across channels and platforms — re-uploads, mirrors, and cross-posts of the same clip — grouped into clusters (strongest match first). Each cluster is laid out like a group of search results: a header bar shows how it matched (exact or near transcript, similarity score, clip-of-longer) and whether it spans multiple channels or platforms — or is a **same-channel** re-upload — and every member is a search-hit-style row with its title, channel name, platform and upload date, plus the original video's URL on its own line so it's obvious the members are genuinely different videos. Clicking a member's title plays it in the transcript modal. The page header also counts how many clusters contain same-channel duplicates. A **Filters** section (styled like the search page's) scopes the list by platform — **YouTube only by default**, with Rumble/Odysee/Twitch toggleable — and by match type, relationship (same-channel / cross-channel / cross-platform / clip-of-longer), channel, and a title/channel text search; platform and channel toggles drop members on de-selected platforms/channels and hide any cluster left with fewer than two members. The list is scoped to the channels this site exposes, so every entry is openable here. Sites with no detection run yet show an empty state.
## [0.3.3] - 2026-06-01
- **Videos whose English captions only existed under a regional/auto code now appear.** A handful of videos had English subtitles only under codes like `en-US` or `en-en-US` (no plain `en`), which the index didn't recognize — so they were missing from the site even though they had a transcript. These now show up in browse and search like any other transcribed video.
diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx
@@ -1,18 +1,83 @@
"use client";
-import { useEffect, useState } from "react";
+import { useEffect, useMemo, useState } from "react";
import { usePlayer } from "yt-dlp-transcript-common/components/PlayerProvider";
+import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache";
+import { formatDate } from "yt-dlp-transcript-common/lib/format";
import type {
DuplicateCluster,
DuplicateReport,
+ DuplicateVideoRef,
} from "yt-dlp-transcript-common/lib/duplicates";
+type Detail = { title: string; webpageUrl: string };
+type DetailMap = ReadonlyMap<string, Detail>;
+
type LoadState =
| { status: "loading" }
| { status: "ready"; report: DuplicateReport | null };
+const PLATFORM_LABEL: Record<string, string> = {
+ youtube: "YouTube",
+ rumble: "Rumble",
+ odysee: "Odysee",
+ twitch: "Twitch",
+};
+
+const MATCH_LABEL: Record<DuplicateCluster["matchKind"], string> = {
+ "transcript-exact": "exact transcript",
+ "transcript-near": "near transcript",
+};
+
+const MATCH_KINDS: DuplicateCluster["matchKind"][] = [
+ "transcript-exact",
+ "transcript-near",
+];
+
+const RELATIONSHIPS = [
+ "same-channel",
+ "cross-channel",
+ "cross-platform",
+ "clip-of-longer",
+] as const;
+type Relationship = (typeof RELATIONSHIPS)[number];
+
+function hasSameChannelDup(cluster: DuplicateCluster): boolean {
+ return (
+ new Set(cluster.videoRefs.map((r) => r.channelSlug)).size <
+ cluster.videoRefs.length
+ );
+}
+
+// Keep only members passing `keep`, recomputing the cross-* flags over the
+// survivors. A cluster with fewer than two members left is no longer a visible
+// duplicate, so it is dropped (returns null).
+function filterClusterMembers(
+ cluster: DuplicateCluster,
+ keep: (ref: DuplicateVideoRef) => boolean,
+): DuplicateCluster | null {
+ const videoRefs = cluster.videoRefs.filter(keep);
+ if (videoRefs.length < 2) return null;
+ return {
+ ...cluster,
+ videoRefs,
+ crossPlatform: new Set(videoRefs.map((r) => r.platform)).size > 1,
+ crossChannel: new Set(videoRefs.map((r) => r.channelSlug)).size > 1,
+ };
+}
+
+function relationshipsOf(cluster: DuplicateCluster): Set<Relationship> {
+ const s = new Set<Relationship>();
+ if (hasSameChannelDup(cluster)) s.add("same-channel");
+ if (cluster.crossChannel) s.add("cross-channel");
+ if (cluster.crossPlatform) s.add("cross-platform");
+ if (cluster.contained) s.add("clip-of-longer");
+ return s;
+}
+
export function DuplicatesClient() {
const [state, setState] = useState<LoadState>({ status: "loading" });
+ const [details, setDetails] = useState<DetailMap>(new Map());
// The site-filtered report is baked into the static build at /duplicates.json
// (see common/bin/compose-site.ts). A missing file (404) or parse failure
@@ -33,13 +98,132 @@ export function DuplicatesClient() {
};
}, []);
- if (state.status === "loading") {
+ const report = state.status === "ready" ? state.report : null;
+
+ // Resolve every member's real title + source URL up front (the report's
+ // stored titles are unreliable — VideoStat carries none). The same cached
+ // fetch the modal uses, deduped per channel page, so opening a member is
+ // instant afterwards and title-based filtering has data to work with.
+ useEffect(() => {
+ if (!report) return;
+ const slugs = [
+ ...new Set(report.clusters.flatMap((c) => c.videoRefs.map((r) => r.slug))),
+ ];
+ let alive = true;
+ Promise.allSettled(
+ slugs.map((slug) =>
+ fetchTranscript(slug).then(
+ (d) => [slug, { title: d.title, webpageUrl: d.webpageUrl }] as const,
+ ),
+ ),
+ ).then((results) => {
+ if (!alive) return;
+ const m = new Map<string, Detail>();
+ for (const r of results) {
+ if (r.status === "fulfilled") m.set(r.value[0], r.value[1]);
+ }
+ setDetails(m);
+ });
+ return () => {
+ alive = false;
+ };
+ }, [report]);
+
+ return (
+ <DuplicatesView
+ loading={state.status === "loading"}
+ report={report}
+ details={details}
+ />
+ );
+}
+
+function DuplicatesView({
+ loading,
+ report,
+ details,
+}: {
+ loading: boolean;
+ report: DuplicateReport | null;
+ details: DetailMap;
+}) {
+ const clusters = useMemo(() => report?.clusters ?? [], [report]);
+
+ // Platforms / channels present across the report, for the toggle lists.
+ const presentPlatforms = useMemo(
+ () => [...new Set(clusters.flatMap((c) => c.videoRefs.map((r) => r.platform)))],
+ [clusters],
+ );
+ const presentChannels = useMemo(() => {
+ const m = new Map<string, string>();
+ for (const c of clusters)
+ for (const r of c.videoRefs) m.set(r.channelSlug, r.channel || r.channelSlug);
+ return [...m.entries()].map(([slug, name]) => ({ slug, name }));
+ }, [clusters]);
+
+ // Filter state. `null` means "untouched" → fall back to the default derived
+ // below: YouTube on (others off), all channels on. The first toggle
+ // materializes the set from that default.
+ const [platforms, setPlatforms] = useState<Set<string> | null>(null);
+ const [channels, setChannels] = useState<Set<string> | null>(null);
+ const [matchKinds, setMatchKinds] = useState<Set<string>>(
+ () => new Set(MATCH_KINDS),
+ );
+ const [relationships, setRelationships] = useState<Set<Relationship>>(
+ () => new Set(),
+ );
+ const [query, setQuery] = useState("");
+
+ const platformSet = useMemo(
+ () =>
+ platforms ??
+ new Set(
+ presentPlatforms.includes("youtube") ? ["youtube"] : presentPlatforms,
+ ),
+ [platforms, presentPlatforms],
+ );
+ const channelSet = useMemo(
+ () => channels ?? new Set(presentChannels.map((c) => c.slug)),
+ [channels, presentChannels],
+ );
+ const q = query.trim().toLowerCase();
+
+ const visible = useMemo(() => {
+ return clusters
+ .map((c) =>
+ filterClusterMembers(
+ c,
+ (r) => platformSet.has(r.platform) && channelSet.has(r.channelSlug),
+ ),
+ )
+ .filter((c): c is DuplicateCluster => c !== null)
+ .filter((c) => matchKinds.has(c.matchKind))
+ .filter((c) => {
+ if (relationships.size === 0) return true;
+ const rel = relationshipsOf(c);
+ return [...relationships].some((r) => rel.has(r));
+ })
+ .filter((c) => {
+ if (!q) return true;
+ return c.videoRefs.some((r) => {
+ const title = details.get(r.slug)?.title || r.title || r.slug;
+ return (
+ title.toLowerCase().includes(q) ||
+ (r.channel || r.channelSlug).toLowerCase().includes(q)
+ );
+ });
+ });
+ }, [clusters, platformSet, channelSet, matchKinds, relationships, q, details]);
+
+ const sameChannelCount = useMemo(
+ () => visible.filter(hasSameChannelDup).length,
+ [visible],
+ );
+
+ if (loading) {
return <p className="text-sm text-zinc-500">Loading duplicates…</p>;
}
- const report = state.report;
- const clusters = report?.clusters ?? [];
-
return (
<section aria-label="duplicate-shorts" className="flex flex-col gap-3">
{report && (
@@ -48,22 +232,51 @@ export function DuplicatesClient() {
? "all durations"
: `shorts ≤${report.runConfig.thresholdSeconds}s`}
{" · "}
- {report.totals.clusters}{" "}
- {report.totals.clusters === 1 ? "cluster" : "clusters"}
+ {visible.length} {visible.length === 1 ? "cluster" : "clusters"}
+ {" · "}
+ {sameChannelCount} with same-channel duplicates
{" · "}
<span title={report.generatedAt}>
generated {new Date(report.generatedAt).toLocaleDateString()}
</span>
</p>
)}
+
+ {report && clusters.length > 0 && (
+ <FilterPanel
+ query={query}
+ onQuery={setQuery}
+ presentPlatforms={presentPlatforms}
+ platforms={platformSet}
+ onTogglePlatform={(p) => setPlatforms(toggled(platformSet, p))}
+ matchKinds={matchKinds}
+ onToggleMatch={(k) => setMatchKinds(toggled(matchKinds, k))}
+ relationships={relationships}
+ onToggleRelationship={(r) =>
+ setRelationships(toggled(relationships, r))
+ }
+ presentChannels={presentChannels}
+ channels={channelSet}
+ onToggleChannel={(c) => setChannels(toggled(channelSet, c))}
+ />
+ )}
+
{clusters.length === 0 ? (
<p className="text-sm text-zinc-500 border border-dashed border-zinc-300 dark:border-zinc-700 rounded p-4">
No duplicate shorts have been detected for this site yet.
</p>
+ ) : visible.length === 0 ? (
+ <p className="text-sm text-zinc-500 border border-dashed border-zinc-300 dark:border-zinc-700 rounded p-4">
+ No duplicates match the current filters.
+ </p>
) : (
<ul className="flex flex-col gap-3">
- {clusters.map((cluster) => (
- <DuplicateClusterCard key={cluster.clusterId} cluster={cluster} />
+ {visible.map((cluster) => (
+ <DuplicateClusterCard
+ key={cluster.clusterId}
+ cluster={cluster}
+ details={details}
+ />
))}
</ul>
)}
@@ -71,51 +284,234 @@ export function DuplicatesClient() {
);
}
-const MATCH_LABEL: Record<DuplicateCluster["matchKind"], string> = {
- "transcript-exact": "exact transcript",
- "transcript-near": "near transcript",
-};
+function toggled<T>(set: Set<T>, value: T): Set<T> {
+ const next = new Set(set);
+ if (next.has(value)) next.delete(value);
+ else next.add(value);
+ return next;
+}
-function DuplicateClusterCard({ cluster }: { cluster: DuplicateCluster }) {
- const { openTranscript } = usePlayer();
+function FilterPanel({
+ query,
+ onQuery,
+ presentPlatforms,
+ platforms,
+ onTogglePlatform,
+ matchKinds,
+ onToggleMatch,
+ relationships,
+ onToggleRelationship,
+ presentChannels,
+ channels,
+ onToggleChannel,
+}: {
+ query: string;
+ onQuery: (v: string) => void;
+ presentPlatforms: string[];
+ platforms: Set<string>;
+ onTogglePlatform: (p: string) => void;
+ matchKinds: Set<string>;
+ onToggleMatch: (k: string) => void;
+ relationships: Set<Relationship>;
+ onToggleRelationship: (r: Relationship) => void;
+ presentChannels: { slug: string; name: string }[];
+ channels: Set<string>;
+ onToggleChannel: (c: string) => void;
+}) {
+ const platformLabel = (n: number) => `platform${n === 1 ? "" : "s"}`;
+ return (
+ <details open className="text-sm text-zinc-600 dark:text-zinc-400">
+ <summary className="cursor-pointer select-none flex flex-wrap items-baseline gap-x-3 gap-y-1">
+ <span className="text-xs uppercase tracking-wide text-zinc-500">
+ Filters
+ </span>
+ <span className="text-xs text-zinc-500">
+ {platforms.size} of {presentPlatforms.length}{" "}
+ {platformLabel(presentPlatforms.length)}
+ </span>
+ </summary>
+ <div className="mt-2 flex flex-wrap items-center gap-x-4 gap-y-2">
+ <label className="flex items-center gap-1.5 select-none">
+ <span className="text-xs uppercase tracking-wide text-zinc-500">
+ Search
+ </span>
+ <input
+ type="search"
+ aria-label="filter duplicates"
+ placeholder="title or channel…"
+ value={query}
+ onChange={(e) => onQuery(e.target.value)}
+ className="rounded-md border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm"
+ />
+ </label>
+ <FilterRow label="Platforms">
+ {presentPlatforms.map((p) => (
+ <Check
+ key={p}
+ checked={platforms.has(p)}
+ onChange={() => onTogglePlatform(p)}
+ >
+ {PLATFORM_LABEL[p] ?? p}
+ </Check>
+ ))}
+ </FilterRow>
+ <FilterRow label="Match">
+ {MATCH_KINDS.map((k) => (
+ <Check
+ key={k}
+ checked={matchKinds.has(k)}
+ onChange={() => onToggleMatch(k)}
+ >
+ {MATCH_LABEL[k]}
+ </Check>
+ ))}
+ </FilterRow>
+ <FilterRow label="Relationship">
+ {RELATIONSHIPS.map((r) => (
+ <Check
+ key={r}
+ checked={relationships.has(r)}
+ onChange={() => onToggleRelationship(r)}
+ >
+ {r}
+ </Check>
+ ))}
+ </FilterRow>
+ {presentChannels.length > 1 && (
+ <FilterRow label="Channels">
+ {presentChannels.map((c) => (
+ <Check
+ key={c.slug}
+ checked={channels.has(c.slug)}
+ onChange={() => onToggleChannel(c.slug)}
+ >
+ {c.name}
+ </Check>
+ ))}
+ </FilterRow>
+ )}
+ </div>
+ </details>
+ );
+}
+
+function FilterRow({
+ label,
+ children,
+}: {
+ label: string;
+ children: React.ReactNode;
+}) {
+ return (
+ <div className="flex flex-wrap items-center gap-x-3 gap-y-1">
+ <span className="text-xs uppercase tracking-wide text-zinc-500">
+ {label}
+ </span>
+ {children}
+ </div>
+ );
+}
+
+function Check({
+ checked,
+ onChange,
+ children,
+}: {
+ checked: boolean;
+ onChange: () => void;
+ children: React.ReactNode;
+}) {
+ return (
+ <label className="flex items-center gap-1.5 select-none">
+ <input
+ type="checkbox"
+ checked={checked}
+ onChange={onChange}
+ className="accent-blue-600"
+ />
+ {children}
+ </label>
+ );
+}
+
+// Mirrors the search ResultCard chrome (bordered rounded card, tinted header
+// bar, divided rows) so duplicate clusters read like a familiar group of
+// search hits — the header bar carries the duplicate evidence instead of a
+// search query, and each member row is a hit-style title + channel·date meta.
+function DuplicateClusterCard({
+ cluster,
+ details,
+}: {
+ cluster: DuplicateCluster;
+ details: DetailMap;
+}) {
return (
<li
aria-label={`duplicate cluster ${cluster.clusterId}`}
- className="border border-zinc-200 dark:border-zinc-800 rounded-md p-3 flex flex-col gap-2"
+ className="border border-zinc-200 dark:border-zinc-800 rounded-lg overflow-hidden bg-white dark:bg-zinc-900"
>
- <div className="flex items-center gap-2 flex-wrap text-xs">
+ <div className="flex items-center gap-2 flex-wrap px-4 py-1.5 bg-zinc-50 dark:bg-zinc-900/60 border-b border-zinc-200 dark:border-zinc-800 text-xs">
<Badge>{MATCH_LABEL[cluster.matchKind]}</Badge>
{cluster.score !== null && <Badge>score {cluster.score.toFixed(2)}</Badge>}
{cluster.contained && <Badge>clip-of-longer</Badge>}
{cluster.crossPlatform && <Badge>cross-platform</Badge>}
{cluster.crossChannel && <Badge>cross-channel</Badge>}
- <span className="text-zinc-500">
+ {hasSameChannelDup(cluster) && <Badge>same-channel</Badge>}
+ <span className="ml-auto text-zinc-500 shrink-0">
{cluster.videoRefs.length} videos · ~{cluster.durationBucket}s
</span>
</div>
- <ul className="flex flex-col gap-1">
+ <ul className="flex flex-col divide-y divide-zinc-200 dark:divide-zinc-800">
{cluster.videoRefs.map((ref) => (
- <li
- key={ref.slug}
- className="text-sm flex items-baseline gap-2 flex-wrap"
- >
- <button
- type="button"
- onClick={() => openTranscript(ref.slug)}
- className="underline text-left hover:text-zinc-900 dark:hover:text-zinc-100"
- >
- {ref.title || ref.slug}
- </button>
- <span className="text-xs text-zinc-500 font-mono">
- {ref.platform} · {ref.channelSlug} · {ref.uploadDate}
- </span>
- </li>
+ <MemberRow key={ref.slug} member={ref} detail={details.get(ref.slug)} />
))}
</ul>
</li>
);
}
+// The title is the primary clickable (opens the modal); the channel·date meta
+// sits alongside it (search-hit style), and the original-source URL sits on its
+// own row below so it's obvious the members are genuinely different videos.
+function MemberRow({
+ member,
+ detail,
+}: {
+ member: DuplicateVideoRef;
+ detail: Detail | undefined;
+}) {
+ const { openTranscript } = usePlayer();
+ const title = detail?.title || member.title || member.slug;
+ return (
+ <li className="flex flex-col gap-0.5 px-4 py-2 hover:bg-zinc-100 dark:hover:bg-zinc-800">
+ <div className="flex items-baseline gap-2">
+ <button
+ type="button"
+ onClick={() => openTranscript(member.slug)}
+ className="font-medium truncate flex-1 min-w-0 text-left hover:underline"
+ >
+ {title}
+ </button>
+ <span className="text-xs text-zinc-500 shrink-0">
+ {member.channel || member.channelSlug} · {member.platform} ·{" "}
+ {formatDate(member.uploadDate)}
+ </span>
+ </div>
+ {detail?.webpageUrl && (
+ <a
+ href={detail.webpageUrl}
+ target="_blank"
+ rel="noopener noreferrer"
+ className="text-xs text-blue-600 dark:text-blue-400 hover:underline truncate font-mono"
+ aria-label={`Open "${title}" on ${member.platform}`}
+ >
+ {detail.webpageUrl}
+ </a>
+ )}
+ </li>
+ );
+}
+
function Badge({ children }: { children: React.ReactNode }) {
return (
<span className="inline-flex items-center rounded-full bg-zinc-100 dark:bg-zinc-800 px-2 py-0.5 text-zinc-700 dark:text-zinc-300">
diff --git a/export/e2e/duplicates.spec.ts b/export/e2e/duplicates.spec.ts
@@ -4,8 +4,8 @@ import { duplicatesReport } from "./fixtures/data";
// Duplicates page e2e. The site-filtered /duplicates.json is route-mocked (it
// is produced at build time by compose-site.ts); the page renders cluster
-// cards and each member opens the real transcript modal via the mocked
-// transcript routes installed by installRoutes.
+// cards, fetches member titles/URLs via the mocked transcript routes, and
+// applies the platform / match / relationship / channel / text filters.
async function installDuplicatesRoute(
page: Page,
report: unknown | null,
@@ -24,7 +24,7 @@ async function installDuplicatesRoute(
}
test.describe("duplicates", () => {
- test("renders a cluster and opens a member in the transcript modal", async ({
+ test("scopes to YouTube by default and renders cluster evidence", async ({
page,
}) => {
await installRoutes(page);
@@ -32,30 +32,92 @@ test.describe("duplicates", () => {
await page.goto("/duplicates");
- // Run-config / staleness header.
+ // YouTube is on by default; other platforms start off.
+ await expect(page.getByRole("checkbox", { name: "YouTube" })).toBeChecked();
+ await expect(page.getByRole("checkbox", { name: "Rumble" })).not.toBeChecked();
+
+ // Header counts reflect the YouTube-scoped view: both clusters survive
+ // (each has ≥2 youtube members), one of which is a same-channel re-upload.
await expect(page.getByText("shorts ≤180s")).toBeVisible();
+ await expect(page.getByText("2 clusters")).toBeVisible();
+ await expect(page.getByText("1 with same-channel duplicates")).toBeVisible();
- // Cluster card with its evidence badges.
- const card = page.getByLabel(/^duplicate cluster /);
- await expect(card).toBeVisible();
- await expect(card.getByText("near transcript")).toBeVisible();
- await expect(card.getByText("score 0.87")).toBeVisible();
- await expect(card.getByText("cross-platform")).toBeVisible();
- await expect(card.getByText("cross-channel")).toBeVisible();
- await expect(card.getByText("2 videos · ~150s")).toBeVisible();
+ const sameChannelCard = page.getByLabel(/^duplicate cluster dup-cluster/);
+ // Real titles are fetched from transcript data for the resolvable members.
+ await expect(
+ sameChannelCard.getByRole("button", {
+ name: "Transcript only — no chat",
+ exact: true,
+ }),
+ ).toBeVisible();
+ await expect(
+ sameChannelCard.getByRole("button", { name: "Small live chat", exact: true }),
+ ).toBeVisible();
+ // The rumble member is filtered out by the default YouTube scope, so the
+ // cluster is no longer cross-platform.
+ await expect(
+ sameChannelCard.getByText("Large live chat"),
+ ).toHaveCount(0);
+ await expect(sameChannelCard.getByText("same-channel")).toBeVisible();
+ await expect(sameChannelCard.getByText("cross-platform")).toHaveCount(0);
- // Both members render as clickable links.
+ // Channel name + formatted date + original-source URL on its own row.
await expect(
- page.getByRole("button", { name: "How to X", exact: true }),
+ sameChannelCard.getByText("Test Channel · youtube · 2024-01-01"),
).toBeVisible();
+ await expect(
+ sameChannelCard.getByRole("link", {
+ name: 'Open "Transcript only — no chat" on youtube',
+ }),
+ ).toHaveAttribute("href", "https://example.com/vid-transcript-only");
- // Clicking a member opens the transcript modal for that video.
+ // Clicking a member's title opens the transcript modal for that video.
await page
- .getByRole("button", { name: "How to X (reup)", exact: true })
+ .getByRole("button", { name: "Small live chat", exact: true })
.click();
await expectModalOpen(page);
});
+ test("toggling a platform reveals its members and the cross-platform flag", async ({
+ page,
+ }) => {
+ await installRoutes(page);
+ await installDuplicatesRoute(page, duplicatesReport());
+ await page.goto("/duplicates");
+
+ const card = page.getByLabel(/^duplicate cluster dup-cluster/);
+ await expect(card.getByText("Large live chat")).toHaveCount(0);
+
+ await page.getByRole("checkbox", { name: "Rumble" }).check();
+
+ await expect(card.getByText("Large live chat")).toBeVisible();
+ await expect(card.getByText("cross-platform")).toBeVisible();
+ });
+
+ test("text search and match-type filters narrow the list", async ({ page }) => {
+ await installRoutes(page);
+ await installDuplicatesRoute(page, duplicatesReport());
+ await page.goto("/duplicates");
+
+ await expect(page.getByText("2 clusters")).toBeVisible();
+
+ // Text search keeps only clusters with a matching member title/channel.
+ await page.getByLabel("filter duplicates").fill("garden");
+ await expect(page.getByText("1 cluster")).toBeVisible();
+ await expect(page.getByText("Gardening basics", { exact: true })).toBeVisible();
+ await expect(
+ page.getByRole("button", { name: "Small live chat", exact: true }),
+ ).toHaveCount(0);
+
+ // Clearing search restores both, then a match-type toggle narrows by tier.
+ await page.getByLabel("filter duplicates").fill("");
+ await expect(page.getByText("2 clusters")).toBeVisible();
+ await page.getByRole("checkbox", { name: "near transcript" }).uncheck();
+ // Only the exact-transcript (gardening) cluster remains.
+ await expect(page.getByText("1 cluster")).toBeVisible();
+ await expect(page.getByText("Gardening basics", { exact: true })).toBeVisible();
+ });
+
test("shows the empty state when no report has been built", async ({
page,
}) => {
diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts
@@ -310,15 +310,17 @@ export function subsPage() {
export const DUP_CLUSTER_ID = "dup-cluster-fixture";
function dupRef(
+ channelSlug: string,
+ channel: string,
id: string,
title: string,
platform: "youtube" | "rumble",
uploadDate: string,
) {
return {
- slug: slug(id),
- channelSlug: CHANNEL_SLUG,
- channel: CHANNEL,
+ slug: `${channelSlug}/${id}`,
+ channelSlug,
+ channel,
platform,
id,
title,
@@ -328,6 +330,12 @@ function dupRef(
};
}
+// Two clusters exercising the filter UI:
+// • a same-channel re-upload that also has a rumble mirror (so toggling
+// platforms changes its members and cross-platform flag), with members whose
+// slugs resolve via the mocked transcript routes (titles + source links);
+// • a cross-channel exact match on youtube-only with synthetic slugs (titles
+// fall back to the report's stored title; no source link).
export function duplicatesReport() {
return {
version: 1,
@@ -339,7 +347,7 @@ export function duplicatesReport() {
containmentThreshold: 0.8,
shingleSize: 5,
},
- totals: { videosScanned: 3, clusters: 1, videosInClusters: 2 },
+ totals: { videosScanned: 5, clusters: 2, videosInClusters: 5 },
clusters: [
{
clusterId: DUP_CLUSTER_ID,
@@ -348,10 +356,24 @@ export function duplicatesReport() {
contained: false,
durationBucket: 150,
crossPlatform: true,
+ crossChannel: false,
+ videoRefs: [
+ dupRef(CHANNEL_SLUG, CHANNEL, VIDEO_TRANSCRIPT_ONLY, "How to X", "youtube", "20240101"),
+ dupRef(CHANNEL_SLUG, CHANNEL, VIDEO_CHAT_SMALL, "How to X (reup)", "youtube", "20240102"),
+ dupRef(CHANNEL_SLUG, CHANNEL, VIDEO_CHAT_LARGE, "How to X (rumble)", "rumble", "20240103"),
+ ],
+ },
+ {
+ clusterId: "dup-cross-channel-fixture",
+ matchKind: "transcript-exact" as const,
+ score: 1,
+ contained: false,
+ durationBucket: 120,
+ crossPlatform: false,
crossChannel: true,
videoRefs: [
- dupRef(VIDEO_TRANSCRIPT_ONLY, "How to X", "youtube", "20240101"),
- dupRef(VIDEO_CHAT_SMALL, "How to X (reup)", "rumble", "20240102"),
+ dupRef("garden-a", "Garden A", "g1", "Gardening basics", "youtube", "20240201"),
+ dupRef("garden-b", "Garden B", "g2", "Gardening basics mirror", "youtube", "20240202"),
],
},
],
diff --git a/export/public/duplicates.json b/export/public/duplicates.json
@@ -0,0 +1 @@
+{"version":1,"generatedAt":"2026-06-05T02:48:49.807Z","runConfig":{"thresholdSeconds":180,"durationToleranceSeconds":2,"nearThreshold":0.6,"containmentThreshold":0.8,"shingleSize":5},"totals":{"videosScanned":34915,"clusters":0,"videosInClusters":0},"clusters":[]}
+\ No newline at end of file