Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c5cbd692de8e8ef1822098fe30e6f686fbe722cb
parent f20c42c95d703949265935c7a0af2a58f5a4bd63
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 14 May 2026 22:09:16 -0400

subs, delete filter

Diffstat:
A.dockerignore | 29+++++++++++++++++++++++++++++
M.gitignore | 1+
ADockerfile.test | 25+++++++++++++++++++++++++
MREADME.md | 26+++++++++++++++++++++++++-
Mcommon/components/TranscriptSearch.tsx | 0
Mcommon/components/searchPipeline.ts | 165+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/components/subsCache.ts | 99+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/components/urlState.ts | 22++++++++++++++++++++++
Mcommon/controller/buildIndex.ts | 274+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----
Mcommon/lib/channelConfig.ts | 2++
Acommon/lib/liveChat.ts | 112+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/manifest.ts | 32++++++++++++++++++++++++++++++++
Mcommon/lib/paths.ts | 2++
Acommon/lib/subs.ts | 19+++++++++++++++++++
Mcommon/lib/videoStatus.ts | 33+++++++++++++++++++++++++++++++++
Mcommon/ytdlp/runYtdlp.ts | 173+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/stages/DownloadStage.tsx | 38++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/pipelineActions.ts | 21++++++++++++++++++++-
Aeditor/e2e/export-search.spec.ts | 310+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/package.json | 1+
Meditor/playwright.config.ts | 25+++++++++++++++++++------
Aexport/public/subs/manifest.json | 2++
Mpackage.json | 3++-
Ascripts/run-sharded-e2e.mjs | 116+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
24 files changed, 1508 insertions(+), 22 deletions(-)

diff --git a/.dockerignore b/.dockerignore @@ -0,0 +1,29 @@ +**/node_modules +**/.next +**/out +**/test-results +**/playwright-report +**/blob-report +**/*.tsbuildinfo + +editor/test-transcripts/ +editor/test-settings.json + +transcripts/ +export/public/summaries/ +export/public/transcripts/ +export/public/code.tar.gz +export/public/code.tar.xz +export/public/transcripts.tar.gz +export/public/transcripts.tar.xz + +.git +.github +.claude +.vercel +.env* +*.pem +.DS_Store + +Dockerfile.test +.dockerignore diff --git a/.gitignore b/.gitignore @@ -59,3 +59,4 @@ yarn-error.log* /editor/test-settings.json /editor/playwright-report/ /editor/test-results/ +/editor/blob-report/ diff --git a/Dockerfile.test b/Dockerfile.test @@ -0,0 +1,25 @@ +FROM mcr.microsoft.com/playwright:v1.59.1-noble + +ENV CI=true \ + E2E_MODE=start \ + PNPM_HOME=/pnpm \ + PATH=/pnpm:$PATH + +RUN corepack enable && corepack prepare pnpm@9.15.4 --activate + +WORKDIR /repo + +COPY pnpm-lock.yaml pnpm-workspace.yaml package.json ./ +COPY common/package.json common/package.json +COPY editor/package.json editor/package.json +COPY export/package.json export/package.json + +RUN pnpm install --frozen-lockfile + +COPY . . + +RUN pnpm --filter editor exec next build + +WORKDIR /repo/editor + +CMD ["pnpm", "exec", "playwright", "test"] diff --git a/README.md b/README.md @@ -15,7 +15,8 @@ pnpm install pnpm dev:editor # editor at http://localhost:3001 pnpm build # build the static site under export/out/ pnpm start:export # serve export/out/ at http://localhost:3000 -pnpm e2e # run the cypress suite (editor) +pnpm e2e # run the editor's Playwright suite (sequential) +pnpm e2e:sharded # same suite, sharded across N Docker containers in parallel ``` ## Editor UI @@ -52,6 +53,29 @@ The editor's pipeline panel and `common/ytdlp/runYtdlp.ts` support three modes p YouTube channels (`handling: "youtube"`) use `--write-auto-subs --skip-download`. Transcribe channels (`handling: "transcribe"`) download audio only (`-f bestaudio -x --audio-format <m4a|mp3|opus>`); whisper-cpp transcribes them later via the whisper panel. +## End-to-end tests + +Playwright covers the editor UI from `editor/e2e/`. Two ways to run it: + +- **`pnpm e2e`** — sequential run on the host. Uses `pnpm dev:test` (Next.js dev mode) on port 3001. Simple, no Docker, but slow. +- **`pnpm e2e:sharded`** — containerized run that splits the suite across N shards in parallel using Playwright's `--shard=i/N` mechanism. Each shard runs in its own copy of a test-specific Docker image (`Dockerfile.test`) that bakes the monorepo plus a prebuilt editor (`next start`). After all shards finish, `playwright merge-reports` recombines the per-shard blob reports into a single `editor/playwright-report/index.html`. + + The npm script first rebuilds the image (Docker layer cache covers unchanged deps, so subsequent rebuilds are ~seconds), then runs `scripts/run-sharded-e2e.mjs` to orchestrate the shards. + + Shard count defaults to `min(max(2, cpus/2), 8)`. Override with `SHARDS=N pnpm e2e:sharded` or `node scripts/run-sharded-e2e.mjs --shards N`. + +### Wall-time comparison + +Measured on this machine (8 CPUs, 90 tests, 4 shards): + +| Mode | Wall time | Notes | +| --- | --- | --- | +| `pnpm e2e` (sequential) | **104 s** | one worker, `next dev` | +| `pnpm e2e:sharded` (image cached) | **42 s** | 4 shards × `next start`, ~2.5× speedup | +| `pnpm e2e:sharded` (cold image build) | ~117 s | one-off ~75 s `docker build` + ~42 s test run | + +The cold build is only paid the first time or after dependency changes; the steady-state run is the second row. + ## CLI shims The same controllers used by the editor are exposed as terminal shims under `common/bin/`: diff --git a/common/components/TranscriptSearch.tsx b/common/components/TranscriptSearch.tsx Binary files differ. diff --git a/common/components/searchPipeline.ts b/common/components/searchPipeline.ts @@ -1,8 +1,11 @@ import { fetchTranscript } from "./transcriptCache"; +import { fetchSubs } from "./subsCache"; import type { DisplaySummary } from "../lib/transcripts"; export type Hit = { start: number; text: string }; +export type SubsHit = { track: string; start: number; text: string }; + export type PipelineUpdate = { hitsBySlug: Record<string, Hit[]>; totalHits: number; @@ -249,3 +252,165 @@ export function filterSlugs( for (const t of summaries) if (passes(t)) slugs.push(t.slug); return slugs; } + +export type SubsPipelineUpdate = { + hitsBySlug: Record<string, SubsHit[]>; + totalHits: number; + processed: number; + totalToProcess: number; + capped: boolean; + done: boolean; +}; + +type SubsPipelineConfig = { + slugs: string[]; + query: string; + useRegex: boolean; + regex: RegExp | null; + // Tracks to exclude. Empty set means "search all tracks". + excludedTracks: Set<string>; + initialHitLimit: number; + concurrency: number; + flushIntervalMs: number; + emit: (update: SubsPipelineUpdate) => void; +}; + +export function createSubsSearchPipeline( + config: SubsPipelineConfig, +): PipelineController { + const { + slugs, + query, + useRegex, + regex, + excludedTracks, + initialHitLimit, + concurrency, + flushIntervalMs, + emit, + } = config; + + let cancelled = false; + let idx = 0; + let completed = 0; + let totalSoFar = 0; + let hitLimit = initialHitLimit; + let activeWorkers = 0; + let done = false; + const localHits: Record<string, SubsHit[]> = {}; + let flushTimer: number | null = null; + + const pushUpdate = (overrides: Partial<SubsPipelineUpdate> = {}) => { + emit({ + hitsBySlug: { ...localHits }, + totalHits: totalSoFar, + processed: completed, + totalToProcess: slugs.length, + capped: totalSoFar >= hitLimit && idx < slugs.length, + done, + ...overrides, + }); + }; + + const scheduleFlush = () => { + if (flushTimer !== null || cancelled) return; + flushTimer = window.setTimeout(() => { + flushTimer = null; + if (cancelled) return; + pushUpdate(); + }, flushIntervalMs); + }; + + const finalize = () => { + if (done || cancelled) return; + done = true; + if (flushTimer !== null) { + window.clearTimeout(flushTimer); + flushTimer = null; + } + pushUpdate(); + }; + + const worker = async () => { + activeWorkers++; + try { + while (!cancelled) { + if (totalSoFar >= hitLimit) return; + if (idx >= slugs.length) return; + const my = idx++; + const slug = slugs[my]; + try { + const detail = await fetchSubs(slug); + if (cancelled) return; + if (totalSoFar < hitLimit) { + const slugHits: SubsHit[] = []; + for (const [track, cueList] of Object.entries(detail.tracks)) { + if (excludedTracks.has(track)) continue; + if (totalSoFar + slugHits.length >= hitLimit) break; + const remaining = + hitLimit - totalSoFar - slugHits.length; + const trackHits = findHitsInCues( + cueList, + query, + useRegex, + regex, + remaining, + ); + for (const h of trackHits) { + slugHits.push({ track, start: h.start, text: h.text }); + } + } + if (slugHits.length > 0) { + // Order hits by start time so live_chat and language tracks + // interleave naturally instead of being grouped per-track. + slugHits.sort((a, b) => a.start - b.start); + localHits[slug] = slugHits; + totalSoFar += slugHits.length; + } + } + } catch { + // ignore per-video failures + } + completed++; + scheduleFlush(); + } + } finally { + activeWorkers--; + if (activeWorkers === 0 && !cancelled) finalize(); + } + }; + + const ensureWorkers = () => { + if (cancelled || done) return; + if (totalSoFar >= hitLimit) return; + if (idx >= slugs.length) return; + const needed = Math.min( + concurrency - activeWorkers, + slugs.length - idx, + ); + for (let i = 0; i < needed; i++) worker(); + }; + + pushUpdate(); + ensureWorkers(); + + return { + cancel() { + cancelled = true; + if (flushTimer !== null) { + window.clearTimeout(flushTimer); + flushTimer = null; + } + }, + setHitLimit(limit: number) { + if (cancelled) return; + if (limit <= hitLimit) return; + hitLimit = limit; + if (done) { + done = false; + pushUpdate(); + } + ensureWorkers(); + }, + }; +} diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts @@ -0,0 +1,99 @@ +"use client"; + +import { useQueries, useQuery } from "@tanstack/react-query"; +import type { SubsDetail } from "../lib/subs"; +import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest"; +import { subsPageFileName } from "../lib/manifest"; + +async function fetchJson<T>(url: string): Promise<T> { + const r = await fetch(url); + if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`); + return (await r.json()) as T; +} + +const resolved = new Map<string, SubsDetail>(); +const inFlight = new Map<string, Promise<SubsDetail>>(); +const channelManifests = new Map<string, Promise<ChannelSubsManifest>>(); +const pagePromises = new Map<string, Promise<SubsDetail[]>>(); + +export function fetchSubs(slug: string): Promise<SubsDetail> { + const hit = resolved.get(slug); + if (hit) return Promise.resolve(hit); + const flying = inFlight.get(slug); + if (flying) return flying; + const p = load(slug).then((detail) => { + resolved.set(slug, detail); + inFlight.delete(slug); + return detail; + }); + p.catch(() => inFlight.delete(slug)); + inFlight.set(slug, p); + return p; +} + +function fetchChannelSubsManifest( + channelSlug: string, +): Promise<ChannelSubsManifest> { + let p = channelManifests.get(channelSlug); + if (!p) { + p = fetchJson<ChannelSubsManifest>(`/subs/${channelSlug}/manifest.json`); + p.catch(() => channelManifests.delete(channelSlug)); + channelManifests.set(channelSlug, p); + } + return p; +} + +function fetchPage( + channelSlug: string, + pageIndex: number, +): Promise<SubsDetail[]> { + const key = `${channelSlug}:${pageIndex}`; + let p = pagePromises.get(key); + if (!p) { + p = fetchJson<SubsDetail[]>( + `/subs/${channelSlug}/${subsPageFileName(pageIndex)}`, + ); + p.catch(() => pagePromises.delete(key)); + pagePromises.set(key, p); + } + return p; +} + +async function load(slug: string): Promise<SubsDetail> { + const slashIdx = slug.indexOf("/"); + if (slashIdx < 0) throw new Error(`Malformed subs slug: ${slug}`); + const channelSlug = slug.slice(0, slashIdx); + const videoId = slug.slice(slashIdx + 1); + const manifest = await fetchChannelSubsManifest(channelSlug); + const pageIndex = manifest.slugToPage[videoId]; + if (pageIndex === undefined) + throw new Error(`Unknown subs slug: ${slug}`); + const page = await fetchPage(channelSlug, pageIndex); + let found: SubsDetail | undefined; + for (const entry of page) { + if (entry.slug === slug) found = entry; + resolved.set(entry.slug, entry); + } + if (!found) throw new Error(`Subs ${slug} missing from page ${pageIndex}`); + return found; +} + +export function useSubsManifest() { + return useQuery<SubsManifest>({ + queryKey: ["subs-manifest"], + queryFn: () => fetchJson<SubsManifest>("/subs/manifest.json"), + }); +} + +// Loads each per-channel subs manifest so we can enumerate the full set of +// sub-having slugs without fetching cue pages eagerly. Returns one entry per +// channel listed in the cross-channel manifest. Errors are absorbed per +// channel so a missing manifest doesn't break the whole UI. +export function useChannelSubsManifests(channelSlugs: string[]) { + return useQueries({ + queries: channelSlugs.map((channelSlug) => ({ + queryKey: ["channel-subs-manifest", channelSlug], + queryFn: () => fetchChannelSubsManifest(channelSlug), + })), + }); +} diff --git a/common/components/urlState.ts b/common/components/urlState.ts @@ -2,6 +2,8 @@ import { useMemo, useSyncExternalStore } from "react"; +export type SearchMode = "transcripts" | "subs"; + export type UrlParams = { q: string; re: boolean; @@ -14,6 +16,12 @@ export type UrlParams = { nar: boolean; nav: boolean; nd: boolean; + mode: SearchMode; + // Subs-mode track exclusion filter; empty array means "all tracks" (no + // exclusions). Mirrors the channel-exclusion convention. Persisted on the + // URL as repeated `tk=` params (e.g. ?m=subs&tk=live_chat to hide live + // chat results). + tracks: string[]; }; const listeners = new Set<() => void>(); @@ -39,6 +47,8 @@ function parse(search: string): UrlParams { const p = new URLSearchParams(search); const tRaw = p.get("t"); const t = tRaw !== null && tRaw !== "" ? Number(tRaw) : null; + const modeRaw = p.get("m"); + const mode: SearchMode = modeRaw === "subs" ? "subs" : "transcripts"; return { q: p.get("q") ?? "", re: p.get("re") === "1", @@ -51,6 +61,8 @@ function parse(search: string): UrlParams { nar: p.get("nar") === "1", nav: p.get("nav") === "1", nd: p.get("nd") === "1", + mode, + tracks: p.getAll("tk"), }; } @@ -71,6 +83,8 @@ type Patch = Partial<{ nar: boolean; nav: boolean; nd: boolean; + mode: SearchMode; + tracks: string[]; }>; export function writeUrlParams(patch: Patch) { @@ -119,6 +133,14 @@ export function writeUrlParams(patch: Patch) { if (patch.nd) params.set("nd", "1"); else params.delete("nd"); } + if (patch.mode !== undefined) { + if (patch.mode === "subs") params.set("m", "subs"); + else params.delete("m"); + } + if (patch.tracks !== undefined) { + params.delete("tk"); + for (const v of patch.tracks) params.append("tk", v); + } const qs = params.toString(); const next = `${window.location.pathname}${qs ? `?${qs}` : ""}`; if (next === window.location.pathname + window.location.search) return; diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -24,6 +24,7 @@ import type { Dirent } from "node:fs"; import { open } from "lmdb"; import { parseVtt, type Cue } from "../lib/vtt"; import { parseWhisper } from "../lib/whisper"; +import { parseLiveChat } from "../lib/liveChat"; import { summarize, toDisplaySummary, @@ -38,17 +39,24 @@ import type { TranscriptDetail, DisplaySummary, } from "../lib/transcripts"; +import type { StoredSubs, SubsDetail } from "../lib/subs"; import type { Manifest, ChannelEntry, ChannelTranscriptsManifest, + ChannelSubsManifest, + SubsChannelEntry, + SubsManifest, } from "../lib/manifest"; import { MANIFEST_VERSION, SUMMARIES_PAGE_SIZE, TRANSCRIPTS_MANIFEST_VERSION, + SUBS_CHANNEL_MANIFEST_VERSION, + SUBS_MANIFEST_VERSION, pageFileName, transcriptPageFileName, + subsPageFileName, } from "../lib/manifest"; import { getSettings } from "../lib/settings"; import { @@ -59,12 +67,14 @@ import { import type { Paths } from "../lib/paths"; import { pickIndexTranscript, + readSubTracks, readVideoFiles, type IndexTranscript, + type SubTrack, } from "../lib/videoStatus"; import { loadAvailability } from "../lib/availability-server"; -const SCHEMA_VERSION = 6; +const SCHEMA_VERSION = 7; type IndexKey = [string, string, string]; type ChannelKey = [string, string, string]; @@ -74,6 +84,7 @@ type PathKey = [string, string]; type MtimeRecord = { metaMs: number; transcriptMs: number | null; + subsMs: number | null; indexKey: IndexKey; }; @@ -97,6 +108,8 @@ type LiveEntry = { transcriptPath: string; transcriptMs: number | null; transcriptKind: IndexTranscript["kind"] | null; + subTracks: SubTrack[]; + subsMs: number | null; }; async function exists(p: string): Promise<boolean> { @@ -178,6 +191,16 @@ async function scanSource( transcriptMs = null; } } + const subTracks = await readSubTracks(fullVideoDir); + let subsMs: number | null = null; + for (const t of subTracks) { + try { + const ms = (await stat(path.join(fullVideoDir, t.filename))).mtimeMs; + if (subsMs === null || ms > subsMs) subsMs = ms; + } catch { + // ignore + } + } live.push({ channelSlug: ch.name, handling: cfg.handling, @@ -188,6 +211,8 @@ async function scanSource( transcriptPath, transcriptMs, transcriptKind: picked?.kind ?? null, + subTracks, + subsMs, }); } } @@ -241,15 +266,17 @@ export async function buildIndex({ const dbPath = paths.lmdbPath; const summariesDir = paths.exportSummariesDir; const transcriptsOutDir = paths.exportTranscriptsDir; + const subsOutDir = paths.exportSubsDir; const manifestPath = path.join(summariesDir, "manifest.json"); await mkdir(path.dirname(dbPath), { recursive: true }); await mkdir(summariesDir, { recursive: true }); await mkdir(transcriptsOutDir, { recursive: true }); + await mkdir(subsOutDir, { recursive: true }); const root = open({ path: dbPath, - maxDbs: 8, + maxDbs: 12, compression: true, }); const sums = root.openDB<TranscriptSummary, IndexKey>({ @@ -260,6 +287,10 @@ export async function buildIndex({ name: "cues", encoding: "msgpack", }); + const subs = root.openDB<StoredSubs, IndexKey>({ + name: "subs", + encoding: "msgpack", + }); const mtimes = root.openDB<MtimeRecord, PathKey>({ name: "mtimes", encoding: "msgpack", @@ -272,6 +303,10 @@ export async function buildIndex({ name: "pageHashes", encoding: "msgpack", }); + const subPageHashes = root.openDB<PageHashRecord, PageHashKey>({ + name: "subPageHashes", + encoding: "msgpack", + }); const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack", @@ -285,9 +320,11 @@ export async function buildIndex({ ); await sums.clearAsync(); await cues.clearAsync(); + await subs.clearAsync(); await mtimes.clearAsync(); await byChannel.clearAsync(); await pageHashes.clearAsync(); + await subPageHashes.clearAsync(); await meta.put("schema", SCHEMA_VERSION); } @@ -311,7 +348,8 @@ export async function buildIndex({ added.push(s); } else if ( prev.metaMs !== s.metaMs || - prev.transcriptMs !== s.transcriptMs + prev.transcriptMs !== s.transcriptMs || + (prev.subsMs ?? null) !== s.subsMs ) { changed.push(s); } @@ -473,16 +511,39 @@ export async function buildIndex({ if (prev && !indexKeysEqual(prev.indexKey, indexKey)) { sums.remove(prev.indexKey); cues.remove(prev.indexKey); + subs.remove(prev.indexKey); byChannel.remove(indexToChannelKey(prev.indexKey)); } sums.put(indexKey, summary); if (cueList) cues.put(indexKey, cueList); else cues.remove(indexKey); + + const parsedSubs: StoredSubs = []; + for (const t of s.subTracks) { + try { + const raw = await readFile( + path.join(path.dirname(s.metaPath), t.filename), + "utf8", + ); + const trackCues = + t.track === "live_chat" ? parseLiveChat(raw) : parseVtt(raw); + if (trackCues.length > 0) { + parsedSubs.push({ track: t.track, cues: trackCues }); + } + } catch { + // Skip unreadable / malformed sub tracks; the rest of the + // video's data is still useful. + } + } + if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs); + else subs.remove(indexKey); + byChannel.put(indexToChannelKey(indexKey), 1); mtimes.put(pk, { metaMs: s.metaMs, transcriptMs: s.transcriptMs, + subsMs: s.subsMs, indexKey, }); } catch (err) { @@ -499,12 +560,14 @@ export async function buildIndex({ for (const { pathKey, indexKey } of removed) { sums.remove(indexKey); cues.remove(indexKey); + subs.remove(indexKey); byChannel.remove(indexToChannelKey(indexKey)); mtimes.remove(pathKey); } await sums.flushed; await cues.flushed; + await subs.flushed; await byChannel.flushed; await mtimes.flushed; @@ -646,19 +709,11 @@ export async function buildIndex({ `Transcript pages: ${pagesWritten} written, ${pagesSkipped} unchanged, ${pagesDeleted} stale removed.`, ); - const pageSize = SUMMARIES_PAGE_SIZE; - const channelCounts = new Map<string, number>(); - for (const cfg of channelConfigs.values()) { - if (cfg.name) channelCounts.set(cfg.name, 0); - } - let pageIndex = 0; - let buffer: DisplaySummary[] = []; - let total = 0; - // Build a lookup from indexKey → isDeleted by reading each video's // availability.json sidecar. Keyed off mtimes, which holds the canonical // (channelSlug, videoDir) → indexKey mapping; only "deleted" entries get - // recorded so lookup defaults to "not deleted". + // recorded so lookup defaults to "not deleted". Built once and reused by + // both the sub-page loop and the cross-channel summaries loop. const deletedByIndexKey = new Set<string>(); for (const { key, value } of mtimes.getRange()) { const pk = key as PathKey; @@ -670,6 +725,199 @@ export async function buildIndex({ } } + let subsPagesWritten = 0; + let subsPagesSkipped = 0; + let subsPagesDeleted = 0; + const subsChannelEntries: SubsChannelEntry[] = []; + let subsTotalCount = 0; + + const writeSubsPage = async ( + channelSlug: string, + idx: number, + entries: PendingEntry[], + ): Promise<void> => { + const body = `[${entries.map((e) => e.encoded).join(",")}]`; + const hash = createHash("sha1").update(body).digest("hex"); + const prev = subPageHashes.get([channelSlug, idx]); + const outPath = path.join(subsOutDir, channelSlug, subsPageFileName(idx)); + if (prev?.hash === hash && (await exists(outPath))) { + subsPagesSkipped++; + return; + } + const sizeBytes = Buffer.byteLength(body, "utf8"); + const tmp = `${outPath}.tmp-${process.pid}`; + await writeFile(tmp, body); + await rename(tmp, outPath); + subPageHashes.put([channelSlug, idx], { + hash, + entryCount: entries.length, + sizeBytes, + }); + subsPagesWritten++; + }; + + for (const channelSlug of Array.from(channelConfigs.keys()).sort()) { + const cfg = channelConfigs.get(channelSlug)!; + const subsChannelDir = path.join(subsOutDir, channelSlug); + let subPageIdx = 0; + let subBuffer: PendingEntry[] = []; + let subPayloadBytes = 0; + const subSlugToPage: Record<string, number> = {}; + const tracksInChannel = new Set<string>(); + let videoCount = 0; + + for (const { key } of byChannel.getRange({ + start: [channelSlug], + end: [channelSlug, "￿"], + })) { + const ck = key as ChannelKey; + if (ck[0] !== channelSlug) continue; + const indexKey: IndexKey = [ck[1], ck[0], ck[2]]; + const stored = subs.get(indexKey); + if (!stored || stored.length === 0) continue; + const summary = sums.get(indexKey); + if (!summary) continue; + const isDeleted = deletedByIndexKey.has( + `${indexKey[0]}\x00${indexKey[1]}\x00${indexKey[2]}`, + ); + const display = toDisplaySummary(summary, { isDeleted }); + const tracks: Record<string, Cue[]> = {}; + for (const t of stored) { + tracks[t.track] = t.cues; + tracksInChannel.add(t.track); + } + const detail: SubsDetail = { ...display, tracks }; + const encoded = JSON.stringify(detail); + const entryBytes = Buffer.byteLength(encoded, "utf8"); + const commaBytes = subBuffer.length === 0 ? 0 : 1; + const delta = entryBytes + commaBytes; + + if ( + subBuffer.length > 0 && + subPayloadBytes + delta + 2 > maxTranscriptPageBytes + ) { + await mkdir(subsChannelDir, { recursive: true }); + await writeSubsPage(channelSlug, subPageIdx, subBuffer); + subPageIdx++; + subBuffer = []; + subPayloadBytes = 0; + } + + if (entryBytes + 2 > maxTranscriptPageBytes) { + log( + ` ${summary.slug}: sub entry (${entryBytes} bytes) exceeds page budget; emitting solo page.`, + ); + } + + subSlugToPage[summary.id] = subPageIdx; + subBuffer.push({ encoded, id: summary.id }); + subPayloadBytes += delta; + videoCount++; + } + + if (subBuffer.length > 0) { + await mkdir(subsChannelDir, { recursive: true }); + await writeSubsPage(channelSlug, subPageIdx, subBuffer); + subPageIdx++; + } + + if (videoCount === 0) { + // No subs for this channel — remove any stale dir + hashes. + await rm(subsChannelDir, { recursive: true, force: true }); + for (const { key } of subPageHashes.getRange({ + start: [channelSlug], + end: [channelSlug, Number.MAX_SAFE_INTEGER], + })) { + subPageHashes.remove(key as PageHashKey); + } + continue; + } + + const subPageCount = subPageIdx; + const subKeep = new Set<string>(["manifest.json"]); + for (let i = 0; i < subPageCount; i++) subKeep.add(subsPageFileName(i)); + const subExisting = await readdir(subsChannelDir).catch( + () => [] as string[], + ); + for (const name of subExisting) { + if (subKeep.has(name)) continue; + await rm(path.join(subsChannelDir, name), { force: true }); + subsPagesDeleted++; + } + for (const { key } of subPageHashes.getRange({ + start: [channelSlug, subPageCount], + end: [channelSlug, Number.MAX_SAFE_INTEGER], + })) { + subPageHashes.remove(key as PageHashKey); + } + + const tracksList = Array.from(tracksInChannel).sort(); + const channelSubsManifest: ChannelSubsManifest = { + version: SUBS_CHANNEL_MANIFEST_VERSION, + channelSlug, + pageCount: subPageCount, + maxPageBytes: maxTranscriptPageBytes, + generatedAt, + tracks: tracksList, + slugToPage: subSlugToPage, + }; + await writeJsonAtomic( + path.join(subsChannelDir, "manifest.json"), + channelSubsManifest, + ); + subsChannelEntries.push({ + name: cfg.name ?? channelSlug, + slug: channelSlug, + videoCount, + tracks: tracksList, + }); + subsTotalCount += videoCount; + } + + await subPageHashes.flushed; + + // Top-level cleanup: drop sub dirs for channels that no longer exist. + const topSubsEntries = await readdir(subsOutDir, { + withFileTypes: true, + }).catch(() => [] as Dirent[]); + const subsChannelSlugSet = new Set(subsChannelEntries.map((e) => e.slug)); + for (const e of topSubsEntries) { + if (e.isDirectory()) { + if (!subsChannelSlugSet.has(e.name)) { + await rm(path.join(subsOutDir, e.name), { + recursive: true, + force: true, + }); + } + } else if (e.isFile() && e.name !== "manifest.json") { + await rm(path.join(subsOutDir, e.name), { force: true }); + } + } + + const subsManifestOut: SubsManifest = { + version: SUBS_MANIFEST_VERSION, + channels: subsChannelEntries.sort((a, b) => a.name.localeCompare(b.name)), + totalCount: subsTotalCount, + generatedAt, + }; + await writeJsonAtomic( + path.join(subsOutDir, "manifest.json"), + subsManifestOut, + ); + + log( + `Sub pages: ${subsPagesWritten} written, ${subsPagesSkipped} unchanged, ${subsPagesDeleted} stale removed across ${subsChannelEntries.length} channels (${subsTotalCount} sub videos).`, + ); + + const pageSize = SUMMARIES_PAGE_SIZE; + const channelCounts = new Map<string, number>(); + for (const cfg of channelConfigs.values()) { + if (cfg.name) channelCounts.set(cfg.name, 0); + } + let pageIndex = 0; + let buffer: DisplaySummary[] = []; + let total = 0; + const flushPage = async () => { if (buffer.length === 0) return; const p = path.join(summariesDir, pageFileName(pageIndex)); diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts @@ -13,6 +13,7 @@ export type ChannelConfig = { keepSourceVideo?: boolean; cookiesFromBrowser?: string; ytdlpExtraArgs?: string[]; + subLangs?: string; lastSyncedAt?: string; lastFullDownloadAt?: string; }; @@ -60,6 +61,7 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null { ) { config.ytdlpExtraArgs = r.ytdlpExtraArgs as string[]; } + if (typeof r.subLangs === "string") config.subLangs = r.subLangs; if (typeof r.lastSyncedAt === "string") config.lastSyncedAt = r.lastSyncedAt; if (typeof r.lastFullDownloadAt === "string") { config.lastFullDownloadAt = r.lastFullDownloadAt; diff --git a/common/lib/liveChat.ts b/common/lib/liveChat.ts @@ -0,0 +1,112 @@ +import type { Cue } from "./vtt"; + +// yt-dlp writes live_chat as JSON-lines (one continuation action per line), +// extracted from YouTube's youtube_live_chat_replay protocol. Each line wraps +// a `replayChatItemAction` whose `actions[]` contain `addChatItemAction.item` +// entries with one of several renderers (`liveChatTextMessageRenderer`, +// `liveChatPaidMessageRenderer`, `liveChatMembershipItemRenderer`, etc.). +// +// We emit one Cue per addChatItemAction that has a renderable text body, with +// `start` derived from `videoOffsetTimeMsec` and `text` formatted as +// `<author>: <message>`. Non-text events (ticker pings without an associated +// message, viewer-engagement nags) are skipped. + +type Run = { text?: string; emoji?: { emojiId?: string; shortcuts?: string[] } }; +type MessageBody = { simpleText?: string; runs?: Run[] }; +type AuthorBody = { simpleText?: string }; +type ChatRenderer = { + message?: MessageBody; + headerSubtext?: MessageBody; + text?: MessageBody; + authorName?: AuthorBody; +}; +type ChatItem = Record<string, ChatRenderer | undefined>; +type AddAction = { item?: ChatItem }; +type ChatAction = { addChatItemAction?: AddAction }; +type ReplayAction = { + actions?: ChatAction[]; + videoOffsetTimeMsec?: string; +}; +type ReplayEnvelope = { + replayChatItemAction?: ReplayAction; + videoOffsetTimeMsec?: string; +}; + +const TEXT_RENDERERS = [ + "liveChatTextMessageRenderer", + "liveChatPaidMessageRenderer", + "liveChatPaidStickerRenderer", + "liveChatMembershipItemRenderer", + "liveChatSponsorshipsGiftPurchaseAnnouncementRenderer", + "liveChatSponsorshipsGiftRedemptionAnnouncementRenderer", +]; + +function renderMessage(body: MessageBody | undefined): string { + if (!body) return ""; + if (typeof body.simpleText === "string") return body.simpleText; + if (!Array.isArray(body.runs)) return ""; + const parts: string[] = []; + for (const run of body.runs) { + if (typeof run.text === "string") { + parts.push(run.text); + } else if (run.emoji?.shortcuts?.length) { + parts.push(run.emoji.shortcuts[0]); + } + } + return parts.join(""); +} + +function authorOf(renderer: ChatRenderer): string { + return renderer.authorName?.simpleText ?? ""; +} + +function textOf(renderer: ChatRenderer): string { + // Sponsorship gift announcements use `headerSubtext`; sticker/paid use + // `message`; membership pings use `headerSubtext` too. Fall back through + // the known fields. + return ( + renderMessage(renderer.message) || + renderMessage(renderer.headerSubtext) || + renderMessage(renderer.text) + ); +} + +export function parseLiveChat(src: string): Cue[] { + const lines = src.split("\n"); + const cues: Cue[] = []; + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed) continue; + let envelope: ReplayEnvelope; + try { + envelope = JSON.parse(trimmed) as ReplayEnvelope; + } catch { + continue; + } + const replay = envelope.replayChatItemAction; + const offsetRaw = + replay?.videoOffsetTimeMsec ?? envelope.videoOffsetTimeMsec; + const offsetMs = offsetRaw ? Number(offsetRaw) : NaN; + if (!Number.isFinite(offsetMs)) continue; + const start = offsetMs / 1000; + for (const action of replay?.actions ?? []) { + const item = action.addChatItemAction?.item; + if (!item) continue; + for (const rendererKey of TEXT_RENDERERS) { + const renderer = item[rendererKey]; + if (!renderer) continue; + const text = textOf(renderer).trim(); + if (!text) continue; + const author = authorOf(renderer); + cues.push({ + start, + end: start + 5, + text: author ? `${author}: ${text}` : text, + }); + break; + } + } + } + cues.sort((a, b) => a.start - b.start); + return cues; +} diff --git a/common/lib/manifest.ts b/common/lib/manifest.ts @@ -33,3 +33,35 @@ export type ChannelTranscriptsManifest = { generatedAt: string; slugToPage: Record<string, number>; }; + +export const SUBS_MANIFEST_VERSION = 1; + +export type SubsChannelEntry = { + name: string; + slug: string; + videoCount: number; + tracks: string[]; +}; + +export type SubsManifest = { + version: number; + channels: SubsChannelEntry[]; + totalCount: number; + generatedAt: string; +}; + +export const SUBS_CHANNEL_MANIFEST_VERSION = 1; + +export type ChannelSubsManifest = { + version: number; + channelSlug: string; + pageCount: number; + maxPageBytes: number; + generatedAt: string; + tracks: string[]; + slugToPage: Record<string, number>; +}; + +export function subsPageFileName(index: number): string { + return `page-${String(index).padStart(4, "0")}.json`; +} diff --git a/common/lib/paths.ts b/common/lib/paths.ts @@ -12,6 +12,7 @@ export type Paths = { exportPublicDir: string; exportSummariesDir: string; exportTranscriptsDir: string; + exportSubsDir: string; settingsFile: string; ytdlpBin: string; whisperBin: string; @@ -40,6 +41,7 @@ export function getPaths(): Paths { exportPublicDir, exportSummariesDir: path.join(exportPublicDir, "summaries"), exportTranscriptsDir: path.join(exportPublicDir, "transcripts"), + exportSubsDir: path.join(exportPublicDir, "subs"), settingsFile: process.env.SETTINGS_FILE ?? path.join(monorepoRoot, "settings.json"), ytdlpBin: process.env.YTDLP_BIN ?? "yt-dlp", diff --git a/common/lib/subs.ts b/common/lib/subs.ts @@ -0,0 +1,19 @@ +import type { Cue } from "./vtt"; +import type { DisplaySummary } from "./transcripts"; + +export type SubTrackCues = { + track: string; + cues: Cue[]; +}; + +// Stored in LMDB under the `subs` DB, keyed by [uploadDate, channelSlug, id]. +export type StoredSubs = SubTrackCues[]; + +// One entry per video in /public/subs/<channelSlug>/page-NNNN.json. The +// display fields mirror DisplaySummary so the subs search UI doesn't need a +// second metadata round-trip; cues are inlined per-track. +export type SubsDetail = DisplaySummary & { + tracks: Record<string, Cue[]>; +}; + +export type SubsPage = SubsDetail[]; diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -19,6 +19,24 @@ export const WHISPER_FILENAME = "transcript.json"; export const CUES_JSON_FILENAME = "transcript.cues.json"; export const META_FILENAME = "metadata.info.json"; +export type SubTrack = { + // "live_chat" | language code like "es", "en-orig", etc. + track: string; + filename: string; + ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3"; +}; + +// Treat any transcript.<x>.<y> file as a sub track when it isn't one of the +// primary transcript outputs (transcript.en.vtt, transcript.json) or a +// derived/auxiliary file (transcript.cues.json). Live chat lands as +// transcript.live_chat.json; non-en languages as transcript.<lang>.vtt. +const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/; +const SUB_EXT_VALUES = ["vtt", "json", "json3", "srv1", "srv2", "srv3"] as const; +type SubExt = (typeof SUB_EXT_VALUES)[number]; +function isSubExt(value: string): value is SubExt { + return (SUB_EXT_VALUES as readonly string[]).includes(value); +} + // Whisper "empty transcription" outputs include systeminfo/model/params/result // keys around an empty `transcription: []`, so they can be up to ~620B in // practice. Real transcripts observed start at ~2.9KB. A 4KB cutoff lets us @@ -69,6 +87,21 @@ export async function readVideoFiles( }; } +export async function readSubTracks(videoDir: string): Promise<SubTrack[]> { + const entries = await readdir(videoDir).catch(() => [] as string[]); + const tracks: SubTrack[] = []; + for (const entry of entries) { + if (entry === VTT_FILENAME || entry === WHISPER_FILENAME) continue; + if (entry === CUES_JSON_FILENAME) continue; + const m = entry.match(SUB_FILE_RE); + if (!m) continue; + const ext = m[2].toLowerCase(); + if (!isSubExt(ext)) continue; + tracks.push({ track: m[1], filename: entry, ext }); + } + return tracks; +} + export function pickIndexTranscript(files: VideoFiles): IndexTranscript | null { if (files.hasWhisper) return { kind: "whisper", filename: WHISPER_FILENAME }; if (files.hasYtVtt) return { kind: "vtt", filename: VTT_FILENAME }; diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -23,6 +23,7 @@ export type YtdlpMode = | "store-playlist" | "download-from-playlist" | "download-missing" + | "download-missing-subs" | "download-one-audio" | "sync"; @@ -66,6 +67,9 @@ export async function runYtdlp(opts: RunYtdlpOpts): Promise<void> { case "download-missing": await downloadMissing(opts); return; + case "download-missing-subs": + await downloadMissingSubs(opts); + return; case "download-one-audio": await downloadOneAudio(opts); return; @@ -92,6 +96,9 @@ function handlingArgs(config: ChannelConfig): string[] { if (config.handling === "youtube") { return [ "--write-auto-subs", + "--write-subs", + "--sub-langs", + config.subLangs ?? "en.*,live_chat", "--write-info-json", "--skip-download", "-t", @@ -336,6 +343,172 @@ async function downloadMissing(opts: RunYtdlpOpts): Promise<void> { await safeBackfillAvailability(opts); } +// yt-dlp's --sub-langs supports a comma-separated list with shell-glob style +// wildcards (e.g. `en.*`, `all`) and exclusions (`-fr`). Replicate just enough +// of that here to decide whether a given track key from a video's metadata +// (subtitles + automatic_captions) is one we'd expect on disk after running +// yt-dlp with the channel's configured `subLangs`. +function compileSubLangMatcher(spec: string): (track: string) => boolean { + const include: string[] = []; + const exclude: string[] = []; + for (const raw of spec.split(",")) { + const trimmed = raw.trim(); + if (!trimmed) continue; + if (trimmed.startsWith("-")) exclude.push(trimmed.slice(1)); + else include.push(trimmed); + } + const matches = (track: string, pattern: string): boolean => { + if (pattern === "all") return true; + if (pattern.includes("*")) { + const re = new RegExp( + "^" + + pattern + .replace(/[.+?^${}()|[\]\\]/g, "\\$&") + .replace(/\\\*/g, ".*") + + "$", + ); + return re.test(track); + } + return pattern === track; + }; + return (track: string) => { + if (exclude.some((p) => matches(track, p))) return false; + return include.some((p) => matches(track, p)); + }; +} + +const SUB_EXT_PATTERN = "(?:vtt|json|json3|srv1|srv2|srv3)"; + +function trackOnDisk(entries: string[], track: string): boolean { + const escaped = track.replace(/[.+?^${}()|[\]\\]/g, "\\$&"); + const re = new RegExp(`^transcript\\.${escaped}\\.${SUB_EXT_PATTERN}$`); + return entries.some((e) => re.test(e)); +} + +async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { + const root = channelRoot(opts); + const dataDir = path.join(root, "data"); + const playlistPath = path.join(root, "playlist"); + const toFetchPath = path.join(root, "playlist.tofetch-subs"); + + let playlistText: string; + try { + playlistText = await readFile(playlistPath, "utf8"); + } catch { + throw new Error( + `No saved playlist at ${playlistPath}. Run "Store playlist" first.`, + ); + } + const urls = playlistText + .split("\n") + .map((s) => s.trim()) + .filter(Boolean); + + const subLangs = opts.channelConfig.subLangs ?? "en.*,live_chat"; + const matchesSubLang = compileSubLangMatcher(subLangs); + + const rumbleIndex = urls.some(isRumbleUrl) + ? await buildRumbleSlugIndex(dataDir) + : null; + const tofetch: string[] = []; + let upToDate = 0; + let unidentifiable = 0; + let noMetadata = 0; + let noExpectedTracks = 0; + + for (const url of urls) { + const slug = extractVideoId(url); + if (!slug) { + unidentifiable++; + continue; + } + const dirId = + rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug; + if (!dirId) { + // No corresponding data dir yet — skip; download-from-playlist / + // download-missing should pull the video and its metadata first. + noMetadata++; + continue; + } + const videoDir = path.join(dataDir, dirId); + let metaRaw: string; + try { + metaRaw = await readFile( + path.join(videoDir, "metadata.info.json"), + "utf8", + ); + } catch { + noMetadata++; + continue; + } + let parsedMeta: { + subtitles?: Record<string, unknown>; + automatic_captions?: Record<string, unknown>; + }; + try { + parsedMeta = JSON.parse(metaRaw); + } catch { + noMetadata++; + continue; + } + const advertised = new Set<string>(); + for (const k of Object.keys(parsedMeta.subtitles ?? {})) advertised.add(k); + for (const k of Object.keys(parsedMeta.automatic_captions ?? {})) { + advertised.add(k); + } + const expected = Array.from(advertised).filter(matchesSubLang); + if (expected.length === 0) { + noExpectedTracks++; + continue; + } + const entries = await readdir(videoDir).catch(() => [] as string[]); + const missing = expected.filter((t) => !trackOnDisk(entries, t)); + if (missing.length === 0) { + upToDate++; + continue; + } + tofetch.push(url); + } + + opts.onLog( + `Prefilter: ${tofetch.length} videos missing subs, ${upToDate} already up to date` + + (noExpectedTracks + ? `, ${noExpectedTracks} with no tracks matching ${JSON.stringify(subLangs)}` + : "") + + (noMetadata ? `, ${noMetadata} without metadata (skipped)` : "") + + (unidentifiable ? `, ${unidentifiable} unidentifiable URLs` : "") + + `.\n`, + ); + + if (tofetch.length === 0) { + await rm(toFetchPath, { force: true }); + opts.onLog("Nothing to fetch.\n"); + return; + } + + await writeFile(toFetchPath, tofetch.join("\n") + "\n"); + + const args: string[] = [ + "--ignore-config", + "--restrict-filenames", + ...OUTPUT_ARGS, + "--write-auto-subs", + "--write-subs", + "--sub-langs", + subLangs, + "--skip-download", + "--no-write-info-json", + "--no-overwrites", + ...(opts.abortOnError === false ? [] : ["--abort-on-error"]), + "-a", + "playlist.tofetch-subs", + ...configArgs(opts.channelConfig), + ]; + await runChildAndStream(opts, root, args); + + await rm(toFetchPath, { force: true }); +} + async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> { if (!opts.singleVideoUrl) { throw new Error("download-one-audio requires singleVideoUrl"); diff --git a/editor/app/channels/[slug]/components/stages/DownloadStage.tsx b/editor/app/channels/[slug]/components/stages/DownloadStage.tsx @@ -12,6 +12,7 @@ import { cancelJobAction } from "../../../../jobs/actions"; import { downloadAction, downloadMissingAction, + downloadMissingSubsAction, } from "../../pipelineActions"; import { VideoIdList } from "../VideoIdList"; @@ -36,9 +37,11 @@ export function DownloadStage({ }: Props) { const [downloadQueue, setDownloadQueue] = useState(defaultQueueKey); const [missingQueue, setMissingQueue] = useState(defaultQueueKey); + const [subsQueue, setSubsQueue] = useState(defaultQueueKey); const [ignoreArchive, setIgnoreArchive] = useState(false); const [downloadAbortOnError, setDownloadAbortOnError] = useState(true); const [missingAbortOnError, setMissingAbortOnError] = useState(true); + const [subsAbortOnError, setSubsAbortOnError] = useState(false); const [missingShardTotal, setMissingShardTotal] = useState( missingShard ? String(missingShard.totalShards) : "", ); @@ -156,6 +159,41 @@ export function DownloadStage({ } /> </div> + <div className="flex flex-col gap-2"> + <Heading + title="Download missing subs" + desc="Backfill sub tracks (live_chat, alt-language captions) for already-downloaded videos. Reads each video's metadata.info.json for advertised tracks, then re-invokes yt-dlp with --skip-download for any whose expected sub file is missing on disk. Uses the channel's subLangs config (defaults to en.*,live_chat)." + /> + <label className="flex items-center gap-2 text-sm"> + <input + type="checkbox" + checked={subsAbortOnError} + onChange={(e) => setSubsAbortOnError(e.target.checked)} + aria-label="abort on error for Download missing subs" + /> + Abort on error + <span className="text-xs text-zinc-500"> + (stop the run on the first per-video failure) + </span> + </label> + <StreamActionLog + trigger={() => + downloadMissingSubsAction(slug, subsQueue, subsAbortOnError) + } + cancelAction={cancelJobAction} + buttonLabel="Download missing subs" + runningLabel="Downloading subs…" + extraControls={ + <QueueControl + value={subsQueue} + onChange={setSubsQueue} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel="Download missing subs" + /> + } + /> + </div> <NoTranscriptList slug={slug} ids={noTranscriptIds} /> </div> ); diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -20,7 +20,12 @@ function defaultQueueKey(config: ChannelConfig): string { async function runPipelineAction( slug: string, - mode: "store-playlist" | "download-from-playlist" | "sync" | "download-missing", + mode: + | "store-playlist" + | "download-from-playlist" + | "sync" + | "download-missing" + | "download-missing-subs", kind: string, queueKey?: string, options?: { @@ -107,3 +112,17 @@ export async function syncAction( ): Promise<StreamActionResult> { return runPipelineAction(slug, "sync", "sync", queueKey); } + +export async function downloadMissingSubsAction( + slug: string, + queueKey?: string, + abortOnError?: boolean, +): Promise<StreamActionResult> { + return runPipelineAction( + slug, + "download-missing-subs", + "download-missing-subs", + queueKey, + { abortOnError }, + ); +} diff --git a/editor/e2e/export-search.spec.ts b/editor/e2e/export-search.spec.ts @@ -0,0 +1,310 @@ +import { test, expect, type Page } from "@playwright/test"; + +const EXPORT_BASE = `http://localhost:${process.env.EXPORT_PORT ?? 3000}`; + +type Summary = { + slug: string; + id: string; + channelSlug: string; + title: string; + uploadDate: string; + date: string; + duration: string; + channel: string; + isLivestream: boolean; + ageRestricted: boolean; + isDeleted: boolean; + platform: "youtube" | "rumble" | "odysee"; + webpageUrl: string; +}; + +const CHANNEL = "Test Channel"; +const CHANNEL_SLUG = "test-channel"; + +const summaries: Summary[] = [ + // Title contains "platypus", transcript contains "platypus" + makeSummary("v-both", "Platypus facts and figures", false), + // Title contains "platypus", transcript does NOT + makeSummary("v-title-only", "All about the platypus", false), + // Title does not contain "platypus", transcript does + makeSummary("v-cue-only", "An ordinary mammal", false), + // Deleted, title contains "platypus" + makeSummary("v-deleted-title", "Deleted platypus chronicles", true), + // Deleted, title does NOT contain "platypus", transcript contains it + makeSummary("v-deleted-cue", "Vanished video on aquatic life", true), +]; + +function makeSummary(id: string, title: string, isDeleted: boolean): Summary { + return { + slug: `${CHANNEL_SLUG}/${id}`, + id, + channelSlug: CHANNEL_SLUG, + title, + uploadDate: "20260101", + date: "2026-01-01", + duration: "5:00", + channel: CHANNEL, + isLivestream: false, + ageRestricted: false, + isDeleted, + platform: "youtube", + webpageUrl: `https://example.com/${id}`, + }; +} + +const transcripts: Record<string, { cues: { start: number; text: string }[] }> = + { + "v-both": { cues: [{ start: 12, text: "the platypus has a beak" }] }, + "v-title-only": { + cues: [{ start: 0, text: "nothing matching here at all" }], + }, + "v-cue-only": { + cues: [{ start: 30, text: "an ordinary platypus appears" }], + }, + "v-deleted-title": { + cues: [{ start: 5, text: "no relevant content in this transcript" }], + }, + "v-deleted-cue": { + cues: [{ start: 45, text: "the platypus swims silently" }], + }, + }; + +async function installFixtureRoutes(page: Page) { + const manifest = { + version: 1, + totalCount: summaries.length, + pageSize: 1000, + pageCount: 1, + generatedAt: new Date().toISOString(), + channels: [{ name: CHANNEL, count: summaries.length }], + }; + + await page.route("**/summaries/manifest.json", async (route) => { + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify(manifest), + }); + }); + + await page.route("**/summaries/page-0000.json", async (route) => { + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify(summaries), + }); + }); + + await page.route("**/transcripts/**/manifest.json", async (route) => { + const url = new URL(route.request().url()); + const channelSlug = url.pathname.split("/").slice(-2, -1)[0]; + if (channelSlug !== CHANNEL_SLUG) { + await route.fulfill({ status: 404, body: "" }); + return; + } + const slugToPage: Record<string, number> = {}; + for (const s of summaries) slugToPage[s.id] = 0; + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify({ version: 1, pageSize: 1000, slugToPage }), + }); + }); + + await page.route("**/transcripts/**/page-*.json", async (route) => { + // The on-disk shape is an array of TranscriptDetail entries, each with + // its own `slug`. transcriptCache iterates and matches by slug. + const body = summaries.map((s) => ({ + slug: s.slug, + id: s.id, + cues: transcripts[s.id].cues, + })); + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify(body), + }); + }); + + // Subs manifest empty so subs mode never tries to fetch anything. + await page.route("**/subs/manifest.json", async (route) => { + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify({ + version: 1, + channels: [], + totalCount: 0, + generatedAt: new Date().toISOString(), + }), + }); + }); +} + +async function search(page: Page, query: string) { + const input = page.getByPlaceholder("Search transcripts..."); + await input.click(); + await input.fill(query); + await expect(input).toHaveValue(query); + await input.press("Enter"); + // commitSearch writes ?q=… to the URL; wait for it so subsequent assertions + // run against the post-search render rather than the no-query placeholder. + await page.waitForURL(/[?&]q=/); +} + +async function waitForHydration(page: Page) { + // The "Regex" checkbox is only rendered after React hydrates the + // TranscriptSearch client component, so it's a reliable hydration probe. + await page.getByRole("checkbox", { name: "Regex" }).waitFor(); +} + +test.describe("export TranscriptSearch — title matching + filter layout", () => { + test.beforeEach(async ({ page }) => { + await installFixtureRoutes(page); + await page.goto(EXPORT_BASE); + await waitForHydration(page); + }); + + test("subs tab is hidden when the subs manifest reports no subs", async ({ + page, + }) => { + // The search-mode tablist is not rendered at all when there are no subs + // to switch to. (Next.js dev-tools may render unrelated tablists; scope + // the search by aria-label.) + await expect( + page.getByRole("tablist", { name: "Search mode" }), + ).toHaveCount(0); + await expect(page.getByRole("tab", { name: /^Subs/ })).toHaveCount(0); + }); + + test("subs tab appears when the subs manifest has at least one channel", async ({ + page, + }) => { + // Re-route the manifest with non-zero totalCount, then reload so the + // updated response is fetched. + await page.unroute("**/subs/manifest.json"); + await page.route("**/subs/manifest.json", async (route) => { + await route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify({ + version: 1, + channels: [{ name: CHANNEL, slug: CHANNEL_SLUG, videoCount: 1, tracks: ["en"] }], + totalCount: 1, + generatedAt: new Date().toISOString(), + }), + }); + }); + await page.reload(); + await waitForHydration(page); + + await expect( + page.getByRole("tab", { name: /^Subs/ }), + ).toBeVisible(); + await expect( + page.getByRole("tab", { name: "Transcripts" }), + ).toBeVisible(); + }); + + test("filter checkboxes are grouped under Type / Audience / Availability labels", async ({ + page, + }) => { + // Trigger search so the result area renders alongside the filter row. + await search(page, "platypus"); + + await expect(page.getByText("Type", { exact: true })).toBeVisible(); + await expect(page.getByText("Audience", { exact: true })).toBeVisible(); + await expect( + page.getByText("Availability", { exact: true }), + ).toBeVisible(); + + // Each labeled group contains its expected checkboxes. + await expect( + page.getByRole("checkbox", { name: "Videos" }), + ).toBeChecked(); + await expect( + page.getByRole("checkbox", { name: "Available" }), + ).toBeChecked(); + await expect( + page.getByRole("checkbox", { name: "Deleted" }), + ).toBeChecked(); + }); + + test("title matches are surfaced as a 'title' badge row above transcript hits", async ({ + page, + }) => { + await search(page, "platypus"); + + // Group headers include the channel + date suffix; match those specifically + // so we don't collide with the title-row buttons. + const both = page.getByRole("button", { + name: /Platypus facts and figures.*Test Channel/, + }); + await expect(both).toBeVisible(); + + const titleOnly = page.getByRole("button", { + name: /All about the platypus.*Test Channel/, + }); + await expect(titleOnly).toBeVisible(); + + // 'title' badge appears once per title-matched video. With deleted included + // by default, that's v-both, v-title-only, v-deleted-title. + const titleBadges = page.getByText("title", { exact: true }); + await expect(titleBadges).toHaveCount(3); + }); + + test("'Deleted' unchecked hides deleted videos from both title and transcript matches", async ({ + page, + }) => { + await search(page, "platypus"); + // Sanity: by default, deleted videos are included. + await expect( + page.getByRole("button", { + name: /Deleted platypus chronicles.*Test Channel/, + }), + ).toBeVisible(); + + await page.getByRole("checkbox", { name: "Deleted" }).uncheck(); + await page.getByPlaceholder("Search transcripts...").press("Enter"); + + await expect( + page.getByRole("button", { name: /Deleted platypus chronicles/ }), + ).toHaveCount(0); + await expect( + page.getByRole("button", { name: /Vanished video on aquatic life/ }), + ).toHaveCount(0); + // Non-deleted matches remain. + await expect( + page.getByRole("button", { + name: /Platypus facts and figures.*Test Channel/, + }), + ).toBeVisible(); + }); + + test("'Available' unchecked narrows to deleted-only videos", async ({ + page, + }) => { + await search(page, "platypus"); + await page.getByRole("checkbox", { name: "Available" }).uncheck(); + await page.getByPlaceholder("Search transcripts...").press("Enter"); + + // Deleted videos remain. + await expect( + page.getByRole("button", { + name: /Deleted platypus chronicles.*Test Channel/, + }), + ).toBeVisible(); + await expect( + page.getByRole("button", { + name: /Vanished video on aquatic life.*Test Channel/, + }), + ).toBeVisible(); + // Non-deleted videos drop out. + await expect( + page.getByRole("button", { name: /Platypus facts and figures/ }), + ).toHaveCount(0); + await expect( + page.getByRole("button", { name: /All about the platypus/ }), + ).toHaveCount(0); + }); +}); diff --git a/editor/package.json b/editor/package.json @@ -6,6 +6,7 @@ "scripts": { "dev": "next dev --port 3001", "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs next dev --port 3001", + "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs next start --port 3001", "build": "next build", "start": "next start --port 3001", "lint": "eslint", diff --git a/editor/playwright.config.ts b/editor/playwright.config.ts @@ -2,6 +2,11 @@ import { defineConfig, devices } from "@playwright/test"; const PORT = Number(process.env.PORT ?? 3001); const baseURL = `http://localhost:${PORT}`; +const webServerCommand = + process.env.E2E_MODE === "start" ? "pnpm start:test" : "pnpm dev:test"; + +const EXPORT_PORT = Number(process.env.EXPORT_PORT ?? 3000); +const exportBaseURL = `http://localhost:${EXPORT_PORT}`; export default defineConfig({ testDir: "./e2e", @@ -11,12 +16,20 @@ export default defineConfig({ outputDir: "test-results/", fullyParallel: false, workers: 1, - webServer: { - command: "pnpm dev:test", - url: baseURL, - timeout: 120_000, - reuseExistingServer: !process.env.CI, - }, + webServer: [ + { + command: webServerCommand, + url: baseURL, + timeout: 120_000, + reuseExistingServer: !process.env.CI, + }, + { + command: `pnpm --filter export run dev --port ${EXPORT_PORT}`, + url: exportBaseURL, + timeout: 120_000, + reuseExistingServer: !process.env.CI, + }, + ], use: { baseURL, trace: "on-first-retry", diff --git a/export/public/subs/manifest.json b/export/public/subs/manifest.json @@ -0,0 +1 @@ +{"version":1,"channels":[],"totalCount":0,"generatedAt":"2026-05-15T00:29:24.351Z"} +\ No newline at end of file diff --git a/package.json b/package.json @@ -9,7 +9,8 @@ "build": "pnpm --filter export run build", "start:export": "pnpm --filter export run start", "dev:editor": "pnpm --filter editor run dev", - "e2e": "pnpm --filter editor run e2e-dev:headless", + "e2e": "pnpm --filter editor run e2e", + "e2e:sharded": "docker build -f Dockerfile.test -t yt-dlp-transcript-browser-e2e . && node scripts/run-sharded-e2e.mjs", "lint": "pnpm --filter export run lint" }, "devDependencies": { diff --git a/scripts/run-sharded-e2e.mjs b/scripts/run-sharded-e2e.mjs @@ -0,0 +1,116 @@ +#!/usr/bin/env node +import { spawn } from "node:child_process"; +import os from "node:os"; +import path from "node:path"; +import { mkdirSync } from "node:fs"; +import { fileURLToPath } from "node:url"; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, ".."); +const EDITOR_DIR = path.join(REPO_ROOT, "editor"); +const REPORT_DIR = path.join(EDITOR_DIR, "blob-report"); +const IMAGE = process.env.IMAGE ?? "yt-dlp-transcript-browser-e2e"; + +function parseShards() { + const argIdx = process.argv.indexOf("--shards"); + if (argIdx !== -1 && process.argv[argIdx + 1]) { + const n = Number(process.argv[argIdx + 1]); + if (Number.isInteger(n) && n >= 1) return n; + } + if (process.env.SHARDS) { + const n = Number(process.env.SHARDS); + if (Number.isInteger(n) && n >= 1) return n; + } + return Math.min(Math.max(2, Math.floor(os.cpus().length / 2)), 8); +} + +const SHARDS = parseShards(); + +function run(cmd, args, opts = {}) { + return new Promise((resolve, reject) => { + const child = spawn(cmd, args, { stdio: "inherit", ...opts }); + child.on("error", reject); + child.on("exit", (code) => resolve(code ?? 0)); + }); +} + +function runShard(i, n) { + const tag = `[shard ${i}/${n}]`; + return new Promise((resolve) => { + const args = [ + "run", + "--rm", + "--init", + "-e", "CI=true", + "-e", "E2E_MODE=start", + "-v", `${REPORT_DIR}:/repo/editor/blob-report`, + IMAGE, + "pnpm", "exec", "playwright", "test", + `--shard=${i}/${n}`, + "--reporter=blob", + ]; + const child = spawn("docker", args); + const prefix = (chunk) => { + const lines = chunk.toString("utf8").split("\n"); + const last = lines.pop(); + for (const line of lines) process.stdout.write(`${tag} ${line}\n`); + if (last) process.stdout.write(`${tag} ${last}`); + }; + child.stdout.on("data", prefix); + child.stderr.on("data", prefix); + child.on("error", (err) => { + process.stderr.write(`${tag} spawn error: ${err.message}\n`); + resolve(1); + }); + child.on("exit", (code) => resolve(code ?? 1)); + }); +} + +async function cleanReportDir() { + // Wipe blob-report via the test image so root-owned files from a prior run can be removed. + await run("docker", [ + "run", "--rm", + "-v", `${EDITOR_DIR}:/host-editor`, + IMAGE, + "rm", "-rf", "/host-editor/blob-report", + ]).catch(() => {}); + mkdirSync(REPORT_DIR, { recursive: true }); +} + +async function main() { + console.log(`Running ${SHARDS} shard(s) using image ${IMAGE}`); + const t0 = Date.now(); + + await cleanReportDir(); + + const shardPromises = []; + for (let i = 1; i <= SHARDS; i++) shardPromises.push(runShard(i, SHARDS)); + const codes = await Promise.all(shardPromises); + const tShards = Date.now(); + + console.log("\nShard exit codes:"); + codes.forEach((code, idx) => { + console.log(` shard ${idx + 1}/${SHARDS}: ${code === 0 ? "PASS" : `FAIL (${code})`}`); + }); + console.log(`Sharded test wall time: ${((tShards - t0) / 1000).toFixed(1)}s`); + + console.log("\nMerging blob reports..."); + const mergeCode = await run( + "pnpm", + ["--filter", "editor", "exec", "playwright", "merge-reports", "--reporter=html", "blob-report"], + { cwd: REPO_ROOT }, + ); + if (mergeCode !== 0) { + console.error(`merge-reports exited ${mergeCode}`); + } else { + console.log("HTML report: editor/playwright-report/index.html"); + } + + const failed = codes.some((c) => c !== 0); + process.exit(failed ? 1 : 0); +} + +main().catch((err) => { + console.error(err); + process.exit(1); +});