import { writeFileSync } from "node:fs"; import type { PublishedTags } from "yt-dlp-transcript-common/lib/curatedTags"; export const CHANNEL = "Test Channel"; export const CHANNEL_SLUG = "test-channel"; // A social channel + its posts, the parallel corpus to the video fixtures // above. Kept on its own channel so channel-selection behaviour is separable. export const POST_CHANNEL = "Test Social"; export const POST_CHANNEL_SLUG = "test-social"; export const POST_ROOT_ID = "post-root-1"; export const POST_REPLY_ID = "post-reply-1"; export const POST_OTHER_ID = "post-other-1"; export const POST_DELETED_ID = "post-deleted-1"; export const VIDEO_TRANSCRIPT_ONLY = "vid-transcript-only"; export const VIDEO_CHAT_SMALL = "vid-chat-small"; export const VIDEO_CHAT_LARGE = "vid-chat-large"; type Cue = { start: number; end: number; text: string }; function slug(id: string) { return `${CHANNEL_SLUG}/${id}`; } // Curated tags (common/lib/curatedTags.ts) carried by the fixture summaries. // Two of the three videos are tagged, one is not — the untagged one is the // control for "omitted when empty", which is how a real untagged corpus ships. export const TAG_COLLAB = "eva-collab"; export const TAG_TOPIC = "eva-topic"; function makeSummary(id: string, title: string, curatedTags?: string[]) { return { slug: slug(id), id, channelSlug: CHANNEL_SLUG, title, uploadDate: "20260101", date: "2026-01-01", duration: "5:00", channel: CHANNEL, isLivestream: false, ageRestricted: false, isDeleted: false, isUnlisted: false, platform: "rumble" as const, webpageUrl: `https://example.com/${id}`, ...(curatedTags && curatedTags.length > 0 ? { curatedTags } : {}), }; } function transcriptCues(text: string): Cue[] { return [ { start: 5, end: 9, text: `${text} — alpha line` }, { start: 50, end: 54, text: `${text} — beta line` }, { start: 100, end: 104, text: `${text} — gamma line` }, ]; } function smallChatCues(): Cue[] { const out: Cue[] = []; for (let i = 0; i < 30; i++) { const t = 5 + i * 2; out.push({ start: t, end: t + 2, text: `@user${i % 5}: chat message number ${i}`, }); } return out; } function largeChatCues(): Cue[] { const out: Cue[] = []; const TOTAL = 5000; for (let i = 0; i < TOTAL; i++) { const t = (i * 3600) / TOTAL; const author = `user${i % 50}`; const len = 20 + ((i * 7) % 120); const body = "x".repeat(len) + ` (#${i})`; out.push({ start: t, end: t + 5, text: `@${author}: ${body}` }); } return out; } export function summaries() { return [ makeSummary(VIDEO_TRANSCRIPT_ONLY, "Transcript only — no chat"), makeSummary(VIDEO_CHAT_SMALL, "Small live chat", [TAG_COLLAB]), makeSummary(VIDEO_CHAT_LARGE, "Large live chat", [TAG_COLLAB, TAG_TOPIC]), ]; } // The published /tags.json for this fixture site, in the shape compose-site // writes: counts are per site and must agree with summaries() above — // eva-collab on two videos, eva-topic on one, both on the one channel. A site // with no shippable tag writes no file at all, so a 404 is a valid empty state // (installRoutes leaves it 404 by default; a spec that wants chips calls // installTagRoutes, the same way duplicates.json works). export function tagsJson(): PublishedTags { return { version: 1, tags: [ { id: TAG_COLLAB, label: "Collab", group: "eva", groupLabel: "Eva", color: "#b48ead", order: 1, count: 2, channels: { [CHANNEL_SLUG]: 2 }, }, { id: TAG_TOPIC, label: "Discussed", group: "eva", groupLabel: "Eva", order: 2, count: 1, channels: { [CHANNEL_SLUG]: 1 }, }, ], }; } export function summariesManifest() { const list = summaries(); return { version: 2, totalCount: list.length, pageSize: 1000, pageCount: 1, generatedAt: new Date().toISOString(), channels: [{ name: CHANNEL, count: list.length }], }; } // ─── Charts feature fixtures ─── function makeStat( id: string, uploadDate: string, viewCount: number, duration: number, // Acquisition dates default to the same month (2026-05) for all fixtures, so a // chart binned by Downloaded/Transcribed collapses to a single category even // though the three videos were uploaded in three different months. downloadedDate: string | null = "20260515", transcribedDate: string | null = "20260516", ) { return { slug: slug(id), id, channelSlug: CHANNEL_SLUG, channel: CHANNEL, platform: "rumble" as const, uploadDate, downloadedDate, transcribedDate, timestamp: Date.parse(`${uploadDate.slice(0, 4)}-${uploadDate.slice(4, 6)}-${uploadDate.slice(6, 8)}`) / 1000, duration, viewCount, likeCount: Math.round(viewCount / 10), commentCount: Math.round(viewCount / 50), channelFollowerCount: 12345, categories: ["Entertainment"], tags: ["news"], language: "en", isLivestream: false, mediaType: "video" as const, status: "available" as const, hasTranscript: true, cueCount: 10, }; } // Three videos across three months so time-binned charts have >1 category. // Distinct mediaType/status/tags so those breakdowns have >1 bucket. export function statsPage() { return [ makeStat(VIDEO_TRANSCRIPT_ONLY, "20260101", 1000, 300), { ...makeStat(VIDEO_CHAT_SMALL, "20260201", 2000, 600), mediaType: "livestream" as const }, { ...makeStat(VIDEO_CHAT_LARGE, "20260301", 3000, 1200), mediaType: "short" as const, status: "deleted" as const, tags: ["news", "gaming"], }, ]; } export function statsManifest() { const list = statsPage(); return { version: 1, generatedAt: new Date().toISOString(), totalCount: list.length, pageCount: 1, maxPageBytes: 20 * 1024 * 1024, channels: [{ slug: CHANNEL_SLUG, name: CHANNEL, count: list.length }], }; } export function chartTemplates() { return { version: 1, defaultDashboard: { version: 1, title: "Overview", filters: {}, charts: [ { id: "fixture-uploads", title: "Uploads per month", type: "bar", x: { kind: "time", bin: "month" }, y: { agg: "count" }, groupBy: "none", cumulative: false, filters: {}, }, { id: "fixture-views", title: "Total views per month", type: "line", x: { kind: "time", bin: "month" }, y: { agg: "sum", field: "viewCount" }, groupBy: "none", cumulative: false, filters: {}, }, ], }, gallery: [ { id: "overview", title: "Overview", templates: [ { id: "g-top-channels", title: "Top channels by video count", config: { id: "g-top-channels", title: "Top channels by video count", type: "bar", x: { kind: "category", field: "channel" }, y: { agg: "count" }, groupBy: "none", cumulative: false, filters: {}, }, }, ], }, { id: "search", title: "Search examples", templates: [ { id: "g-mentions", title: "Mentions of a term over time", config: { id: "g-mentions", title: "Mentions over time", type: "line", x: { kind: "time", bin: "month" }, y: { agg: "count" }, groupBy: "none", cumulative: false, filters: {}, // Empty term → adding it should open the editor. search: { queryTree: '{"k":"g","o":"AND","c":[{"k":"l","q":"","s":"transcripts"}]}', metric: "totalHits", }, }, }, ], }, ], }; } export function transcriptPage() { return [ { ...makeSummary(VIDEO_TRANSCRIPT_ONLY, "Transcript only — no chat"), duration: 300, // "zebra" appears only here, and only in the description (not in any // title or transcript) so a description-scope search isolates it. description: "Filmed on location with a zebra in the background.", tags: ["news"], cues: transcriptCues("transcript-only video"), // An ALTERNATE English track (lib/captionTracks.ts): the uploaded // captions, whose words differ from the primary's. "zeppelin" is said // only here, at 200 s — what transcript-tracks.spec.ts searches for. track: "en-orig", altTracks: [ { track: "en", cues: [{ start: 200, end: 204, text: "uploaded words about a zeppelin" }], }, ], }, { ...makeSummary(VIDEO_CHAT_SMALL, "Small live chat"), duration: 300, description: "A quiet stream with a small audience.", tags: ["news"], cues: transcriptCues("small chat video"), }, { ...makeSummary(VIDEO_CHAT_LARGE, "Large live chat"), duration: 3600, description: "A busy stream with a large audience.", // "gaming" appears only as a tag here — isolates the tags scope. tags: ["news", "gaming"], cues: transcriptCues("large chat video"), }, ]; } // ─── AI-digest fixtures ─── // The derived layer is SPARSE: only VIDEO_TRANSCRIPT_ONLY carries a digest, so // the same fixture set covers both "has a digest" and "does not" without a // second channel. VIDEO_CHAT_SMALL deliberately has none — that is what proves // the Digest control stays hidden rather than becoming a dead end. // // Chapter starts are chosen to sit ON the transcript cue starts above (5 / 50 / // 100), which is what the real parser guarantees by snapping to cue boundaries. export function channelDigestsManifest() { return { version: 1, channelSlug: CHANNEL_SLUG, pageCount: 1, maxPageBytes: 8388608, generatedAt: new Date().toISOString(), slugToPage: { [VIDEO_TRANSCRIPT_ONLY]: 0 }, pageHashes: ["fixture-hash-0"], }; } export function digestPage() { return [ { slug: slug(VIDEO_TRANSCRIPT_ONLY), id: VIDEO_TRANSCRIPT_ONLY, generatedAt: "2026-07-20T00:00:00.000Z", chapters: [ { id: "c5", start: 5, clock: "00:00:05", title: "Opening remarks", decidedBy: "ai" as const, }, { id: "c50", start: 50, clock: "00:00:50", title: "The main argument", decidedBy: "ai" as const, }, { id: "c100", start: 100, clock: "00:01:40", // A human-corrected title, so the "edited" marker has something to // render and the ai/human distinction is exercised. title: "Corrected by hand", decidedBy: "human" as const, }, ], tags: [ { id: "tnews", tag: "news", decidedBy: "ai" as const }, { id: "tpolicy", tag: "policy", decidedBy: "ai" as const }, ], provenance: { chapters: { appId: "ollama-direct", model: "qwen2.5:7b", lane: "local-gpu" as const, generatedAt: "2026-07-20T00:00:00.000Z", promptVersion: 2, contextHash: "", }, }, }, ]; } // The same digest, marked as borrowed from a duplicate cluster's canonical // member. Used to prove a shared digest is presented AS shared — the failure // mode here looks like success, so it needs its own assertion. export function borrowedDigestPage() { const [entry] = digestPage(); return [ { ...entry, derivedFrom: { slug: "other-channel/vid-canonical", clusterId: "cluster-1", sharedAt: "2026-07-21T00:00:00.000Z", offsetSeconds: 0.4, }, }, ]; } export function channelTranscriptsManifest() { return { version: 1, channelSlug: CHANNEL_SLUG, pageCount: 1, maxPageBytes: 8388608, generatedAt: new Date().toISOString(), slugToPage: { [VIDEO_TRANSCRIPT_ONLY]: 0, [VIDEO_CHAT_SMALL]: 0, [VIDEO_CHAT_LARGE]: 0, }, }; } export function subsManifest() { return { version: 3, channels: [ { name: CHANNEL, slug: CHANNEL_SLUG, videoCount: 2, tracks: ["live_chat"], liveChatCount: 2, }, ], totalCount: 2, liveChatTotalCount: 2, generatedAt: new Date().toISOString(), }; } export function channelSubsManifest() { return { version: 1, channelSlug: CHANNEL_SLUG, pageCount: 1, maxPageBytes: 8388608, generatedAt: new Date().toISOString(), tracks: ["live_chat"], slugToPage: { [VIDEO_CHAT_SMALL]: 0, [VIDEO_CHAT_LARGE]: 0, }, }; } export function subsPage() { return [ { ...makeSummary(VIDEO_CHAT_SMALL, "Small live chat"), isLivestream: true, tracks: { live_chat: smallChatCues() }, }, { ...makeSummary(VIDEO_CHAT_LARGE, "Large live chat"), isLivestream: true, tracks: { live_chat: largeChatCues() }, }, ]; } // ─── Social-post corpus fixtures ─── // The posts have words of their own ("kappa", "sigma", "omega"), in no video's // cues, title or chat. Since release 16 a plain query reads posts by default // ("Search in": Transcripts and Posts ticked), and when the posts said "alpha" // and "gamma" like every video's cues, every spec that searches those words // for its own reasons got two post cards it was not about. A query that reads // both corpora names a word from each (posts-search.spec, search-in.spec). function makePost( id: string, text: string, extra: Record = {}, ) { return { id, slug: `${POST_CHANNEL_SLUG}/${id}`, channelSlug: POST_CHANNEL_SLUG, author: "tester.bsky.social", authorName: "Tester", createdAt: "2026-02-03T10:00:00.000Z", uploadDate: "20260203", text, url: `https://bsky.app/profile/tester.bsky.social/post/${id}`, platform: "bluesky" as const, isReply: false, isRepost: false, links: [] as string[], threadId: id, ...extra, }; } export function postsManifest() { return { version: 1, channels: [ { name: POST_CHANNEL, slug: POST_CHANNEL_SLUG, postCount: 4, platform: "bluesky" as const, }, ], totalCount: 4, generatedAt: new Date().toISOString(), }; } export function channelPostsManifest() { return { version: 1, channelSlug: POST_CHANNEL_SLUG, pageCount: 1, maxPageBytes: 8388608, generatedAt: new Date().toISOString(), slugToPage: { [POST_ROOT_ID]: 0, [POST_REPLY_ID]: 0, [POST_OTHER_ID]: 0, [POST_DELETED_ID]: 0, }, }; } export function postsPage() { return [ makePost(POST_ROOT_ID, "a post about kappa things", { links: ["https://example.com/linked"], engagement: { likes: 12, reposts: 3, replies: 1 }, }), makePost(POST_REPLY_ID, "replying about kappa again", { isReply: true, threadId: POST_ROOT_ID, replyTo: { platform: "bluesky", id: POST_ROOT_ID }, createdAt: "2026-02-03T11:00:00.000Z", }), makePost(POST_DELETED_ID, "a deleted sigma post", { isDeleted: true, availability: "deleted", availabilityCheckedAt: "2026-02-04T00:00:00.000Z", createdAt: "2026-02-03T12:00:00.000Z", }), makePost(POST_OTHER_ID, "an unrelated omega post", { createdAt: "2026-02-02T09:00:00.000Z", uploadDate: "20260202", }), ]; } // ─── Duplicate-shorts feature fixtures ─── // One site-filtered cluster of two members whose slugs resolve via the mocked // transcript routes, so a member click opens the real transcript modal. The // cross-platform / cross-channel flags and score are display-only here (the // page renders whatever the composed report contains). export const DUP_CLUSTER_ID = "dup-cluster-fixture"; function dupRef( channelSlug: string, channel: string, id: string, title: string, platform: "youtube" | "rumble", uploadDate: string, ) { return { slug: `${channelSlug}/${id}`, channelSlug, channel, platform, id, title, duration: 150, uploadDate, hasTranscript: true, }; } // Two clusters exercising the filter UI: // • a same-channel re-upload that also has a rumble mirror (so toggling // platforms changes its members and cross-platform flag), with members whose // slugs resolve via the mocked transcript routes (titles + source links); // • a cross-channel exact match on youtube-only with synthetic slugs (titles // fall back to the report's stored title; no source link). export function duplicatesReport() { return { version: 1, generatedAt: "2026-06-04T12:00:00.000Z", runConfig: { thresholdSeconds: 180, durationToleranceSeconds: 2, nearThreshold: 0.6, containmentThreshold: 0.8, shingleSize: 5, }, totals: { videosScanned: 5, clusters: 2, videosInClusters: 5 }, clusters: [ { clusterId: DUP_CLUSTER_ID, matchKind: "transcript-near" as const, score: 0.87, contained: false, durationBucket: 150, crossPlatform: true, crossChannel: false, videoRefs: [ dupRef(CHANNEL_SLUG, CHANNEL, VIDEO_TRANSCRIPT_ONLY, "How to X", "youtube", "20240101"), dupRef(CHANNEL_SLUG, CHANNEL, VIDEO_CHAT_SMALL, "How to X (reup)", "youtube", "20240102"), dupRef(CHANNEL_SLUG, CHANNEL, VIDEO_CHAT_LARGE, "How to X (rumble)", "rumble", "20240103"), ], }, { clusterId: "dup-cross-channel-fixture", matchKind: "transcript-exact" as const, score: 1, contained: false, durationBucket: 120, crossPlatform: false, crossChannel: true, videoRefs: [ dupRef("garden-a", "Garden A", "g1", "Gardening basics", "youtube", "20240201"), dupRef("garden-b", "Garden B", "g2", "Gardening basics mirror", "youtube", "20240202"), ], }, ], }; } export function buildFixtureSettings(filePath: string): void { const settings = { siteTitle: "Test Export", siteDescription: "E2E fixture", headerTitle: "Test Export", homeTagline: "", maxTranscriptPageBytes: 8388608, transcribeBin: "whisper-cli", transcribeModel: "/dev/null", transcribeArgs: [], cookiesFromBrowser: "", sleepBetweenDownloadsSeconds: 0, inlineTranscribeOnFallback: false, groups: [], defaultGroupId: "", socialLinks: [], }; writeFileSync(filePath, JSON.stringify(settings, null, 2)); }