commit c5cbd692de8e8ef1822098fe30e6f686fbe722cb
parent f20c42c95d703949265935c7a0af2a58f5a4bd63
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 14 May 2026 22:09:16 -0400
subs, delete filter
Diffstat:
24 files changed, 1508 insertions(+), 22 deletions(-)
diff --git a/.dockerignore b/.dockerignore
@@ -0,0 +1,29 @@
+**/node_modules
+**/.next
+**/out
+**/test-results
+**/playwright-report
+**/blob-report
+**/*.tsbuildinfo
+
+editor/test-transcripts/
+editor/test-settings.json
+
+transcripts/
+export/public/summaries/
+export/public/transcripts/
+export/public/code.tar.gz
+export/public/code.tar.xz
+export/public/transcripts.tar.gz
+export/public/transcripts.tar.xz
+
+.git
+.github
+.claude
+.vercel
+.env*
+*.pem
+.DS_Store
+
+Dockerfile.test
+.dockerignore
diff --git a/.gitignore b/.gitignore
@@ -59,3 +59,4 @@ yarn-error.log*
/editor/test-settings.json
/editor/playwright-report/
/editor/test-results/
+/editor/blob-report/
diff --git a/Dockerfile.test b/Dockerfile.test
@@ -0,0 +1,25 @@
+FROM mcr.microsoft.com/playwright:v1.59.1-noble
+
+ENV CI=true \
+ E2E_MODE=start \
+ PNPM_HOME=/pnpm \
+ PATH=/pnpm:$PATH
+
+RUN corepack enable && corepack prepare pnpm@9.15.4 --activate
+
+WORKDIR /repo
+
+COPY pnpm-lock.yaml pnpm-workspace.yaml package.json ./
+COPY common/package.json common/package.json
+COPY editor/package.json editor/package.json
+COPY export/package.json export/package.json
+
+RUN pnpm install --frozen-lockfile
+
+COPY . .
+
+RUN pnpm --filter editor exec next build
+
+WORKDIR /repo/editor
+
+CMD ["pnpm", "exec", "playwright", "test"]
diff --git a/README.md b/README.md
@@ -15,7 +15,8 @@ pnpm install
pnpm dev:editor # editor at http://localhost:3001
pnpm build # build the static site under export/out/
pnpm start:export # serve export/out/ at http://localhost:3000
-pnpm e2e # run the cypress suite (editor)
+pnpm e2e # run the editor's Playwright suite (sequential)
+pnpm e2e:sharded # same suite, sharded across N Docker containers in parallel
```
## Editor UI
@@ -52,6 +53,29 @@ The editor's pipeline panel and `common/ytdlp/runYtdlp.ts` support three modes p
YouTube channels (`handling: "youtube"`) use `--write-auto-subs --skip-download`. Transcribe channels (`handling: "transcribe"`) download audio only (`-f bestaudio -x --audio-format <m4a|mp3|opus>`); whisper-cpp transcribes them later via the whisper panel.
+## End-to-end tests
+
+Playwright covers the editor UI from `editor/e2e/`. Two ways to run it:
+
+- **`pnpm e2e`** — sequential run on the host. Uses `pnpm dev:test` (Next.js dev mode) on port 3001. Simple, no Docker, but slow.
+- **`pnpm e2e:sharded`** — containerized run that splits the suite across N shards in parallel using Playwright's `--shard=i/N` mechanism. Each shard runs in its own copy of a test-specific Docker image (`Dockerfile.test`) that bakes the monorepo plus a prebuilt editor (`next start`). After all shards finish, `playwright merge-reports` recombines the per-shard blob reports into a single `editor/playwright-report/index.html`.
+
+ The npm script first rebuilds the image (Docker layer cache covers unchanged deps, so subsequent rebuilds are ~seconds), then runs `scripts/run-sharded-e2e.mjs` to orchestrate the shards.
+
+ Shard count defaults to `min(max(2, cpus/2), 8)`. Override with `SHARDS=N pnpm e2e:sharded` or `node scripts/run-sharded-e2e.mjs --shards N`.
+
+### Wall-time comparison
+
+Measured on this machine (8 CPUs, 90 tests, 4 shards):
+
+| Mode | Wall time | Notes |
+| --- | --- | --- |
+| `pnpm e2e` (sequential) | **104 s** | one worker, `next dev` |
+| `pnpm e2e:sharded` (image cached) | **42 s** | 4 shards × `next start`, ~2.5× speedup |
+| `pnpm e2e:sharded` (cold image build) | ~117 s | one-off ~75 s `docker build` + ~42 s test run |
+
+The cold build is only paid the first time or after dependency changes; the steady-state run is the second row.
+
## CLI shims
The same controllers used by the editor are exposed as terminal shims under `common/bin/`:
diff --git a/common/components/TranscriptSearch.tsx b/common/components/TranscriptSearch.tsx
Binary files differ.
diff --git a/common/components/searchPipeline.ts b/common/components/searchPipeline.ts
@@ -1,8 +1,11 @@
import { fetchTranscript } from "./transcriptCache";
+import { fetchSubs } from "./subsCache";
import type { DisplaySummary } from "../lib/transcripts";
export type Hit = { start: number; text: string };
+export type SubsHit = { track: string; start: number; text: string };
+
export type PipelineUpdate = {
hitsBySlug: Record<string, Hit[]>;
totalHits: number;
@@ -249,3 +252,165 @@ export function filterSlugs(
for (const t of summaries) if (passes(t)) slugs.push(t.slug);
return slugs;
}
+
+export type SubsPipelineUpdate = {
+ hitsBySlug: Record<string, SubsHit[]>;
+ totalHits: number;
+ processed: number;
+ totalToProcess: number;
+ capped: boolean;
+ done: boolean;
+};
+
+type SubsPipelineConfig = {
+ slugs: string[];
+ query: string;
+ useRegex: boolean;
+ regex: RegExp | null;
+ // Tracks to exclude. Empty set means "search all tracks".
+ excludedTracks: Set<string>;
+ initialHitLimit: number;
+ concurrency: number;
+ flushIntervalMs: number;
+ emit: (update: SubsPipelineUpdate) => void;
+};
+
+export function createSubsSearchPipeline(
+ config: SubsPipelineConfig,
+): PipelineController {
+ const {
+ slugs,
+ query,
+ useRegex,
+ regex,
+ excludedTracks,
+ initialHitLimit,
+ concurrency,
+ flushIntervalMs,
+ emit,
+ } = config;
+
+ let cancelled = false;
+ let idx = 0;
+ let completed = 0;
+ let totalSoFar = 0;
+ let hitLimit = initialHitLimit;
+ let activeWorkers = 0;
+ let done = false;
+ const localHits: Record<string, SubsHit[]> = {};
+ let flushTimer: number | null = null;
+
+ const pushUpdate = (overrides: Partial<SubsPipelineUpdate> = {}) => {
+ emit({
+ hitsBySlug: { ...localHits },
+ totalHits: totalSoFar,
+ processed: completed,
+ totalToProcess: slugs.length,
+ capped: totalSoFar >= hitLimit && idx < slugs.length,
+ done,
+ ...overrides,
+ });
+ };
+
+ const scheduleFlush = () => {
+ if (flushTimer !== null || cancelled) return;
+ flushTimer = window.setTimeout(() => {
+ flushTimer = null;
+ if (cancelled) return;
+ pushUpdate();
+ }, flushIntervalMs);
+ };
+
+ const finalize = () => {
+ if (done || cancelled) return;
+ done = true;
+ if (flushTimer !== null) {
+ window.clearTimeout(flushTimer);
+ flushTimer = null;
+ }
+ pushUpdate();
+ };
+
+ const worker = async () => {
+ activeWorkers++;
+ try {
+ while (!cancelled) {
+ if (totalSoFar >= hitLimit) return;
+ if (idx >= slugs.length) return;
+ const my = idx++;
+ const slug = slugs[my];
+ try {
+ const detail = await fetchSubs(slug);
+ if (cancelled) return;
+ if (totalSoFar < hitLimit) {
+ const slugHits: SubsHit[] = [];
+ for (const [track, cueList] of Object.entries(detail.tracks)) {
+ if (excludedTracks.has(track)) continue;
+ if (totalSoFar + slugHits.length >= hitLimit) break;
+ const remaining =
+ hitLimit - totalSoFar - slugHits.length;
+ const trackHits = findHitsInCues(
+ cueList,
+ query,
+ useRegex,
+ regex,
+ remaining,
+ );
+ for (const h of trackHits) {
+ slugHits.push({ track, start: h.start, text: h.text });
+ }
+ }
+ if (slugHits.length > 0) {
+ // Order hits by start time so live_chat and language tracks
+ // interleave naturally instead of being grouped per-track.
+ slugHits.sort((a, b) => a.start - b.start);
+ localHits[slug] = slugHits;
+ totalSoFar += slugHits.length;
+ }
+ }
+ } catch {
+ // ignore per-video failures
+ }
+ completed++;
+ scheduleFlush();
+ }
+ } finally {
+ activeWorkers--;
+ if (activeWorkers === 0 && !cancelled) finalize();
+ }
+ };
+
+ const ensureWorkers = () => {
+ if (cancelled || done) return;
+ if (totalSoFar >= hitLimit) return;
+ if (idx >= slugs.length) return;
+ const needed = Math.min(
+ concurrency - activeWorkers,
+ slugs.length - idx,
+ );
+ for (let i = 0; i < needed; i++) worker();
+ };
+
+ pushUpdate();
+ ensureWorkers();
+
+ return {
+ cancel() {
+ cancelled = true;
+ if (flushTimer !== null) {
+ window.clearTimeout(flushTimer);
+ flushTimer = null;
+ }
+ },
+ setHitLimit(limit: number) {
+ if (cancelled) return;
+ if (limit <= hitLimit) return;
+ hitLimit = limit;
+ if (done) {
+ done = false;
+ pushUpdate();
+ }
+ ensureWorkers();
+ },
+ };
+}
diff --git a/common/components/subsCache.ts b/common/components/subsCache.ts
@@ -0,0 +1,99 @@
+"use client";
+
+import { useQueries, useQuery } from "@tanstack/react-query";
+import type { SubsDetail } from "../lib/subs";
+import type { ChannelSubsManifest, SubsManifest } from "../lib/manifest";
+import { subsPageFileName } from "../lib/manifest";
+
+async function fetchJson<T>(url: string): Promise<T> {
+ const r = await fetch(url);
+ if (!r.ok) throw new Error(`Failed to fetch ${url}: ${r.status}`);
+ return (await r.json()) as T;
+}
+
+const resolved = new Map<string, SubsDetail>();
+const inFlight = new Map<string, Promise<SubsDetail>>();
+const channelManifests = new Map<string, Promise<ChannelSubsManifest>>();
+const pagePromises = new Map<string, Promise<SubsDetail[]>>();
+
+export function fetchSubs(slug: string): Promise<SubsDetail> {
+ const hit = resolved.get(slug);
+ if (hit) return Promise.resolve(hit);
+ const flying = inFlight.get(slug);
+ if (flying) return flying;
+ const p = load(slug).then((detail) => {
+ resolved.set(slug, detail);
+ inFlight.delete(slug);
+ return detail;
+ });
+ p.catch(() => inFlight.delete(slug));
+ inFlight.set(slug, p);
+ return p;
+}
+
+function fetchChannelSubsManifest(
+ channelSlug: string,
+): Promise<ChannelSubsManifest> {
+ let p = channelManifests.get(channelSlug);
+ if (!p) {
+ p = fetchJson<ChannelSubsManifest>(`/subs/${channelSlug}/manifest.json`);
+ p.catch(() => channelManifests.delete(channelSlug));
+ channelManifests.set(channelSlug, p);
+ }
+ return p;
+}
+
+function fetchPage(
+ channelSlug: string,
+ pageIndex: number,
+): Promise<SubsDetail[]> {
+ const key = `${channelSlug}:${pageIndex}`;
+ let p = pagePromises.get(key);
+ if (!p) {
+ p = fetchJson<SubsDetail[]>(
+ `/subs/${channelSlug}/${subsPageFileName(pageIndex)}`,
+ );
+ p.catch(() => pagePromises.delete(key));
+ pagePromises.set(key, p);
+ }
+ return p;
+}
+
+async function load(slug: string): Promise<SubsDetail> {
+ const slashIdx = slug.indexOf("/");
+ if (slashIdx < 0) throw new Error(`Malformed subs slug: ${slug}`);
+ const channelSlug = slug.slice(0, slashIdx);
+ const videoId = slug.slice(slashIdx + 1);
+ const manifest = await fetchChannelSubsManifest(channelSlug);
+ const pageIndex = manifest.slugToPage[videoId];
+ if (pageIndex === undefined)
+ throw new Error(`Unknown subs slug: ${slug}`);
+ const page = await fetchPage(channelSlug, pageIndex);
+ let found: SubsDetail | undefined;
+ for (const entry of page) {
+ if (entry.slug === slug) found = entry;
+ resolved.set(entry.slug, entry);
+ }
+ if (!found) throw new Error(`Subs ${slug} missing from page ${pageIndex}`);
+ return found;
+}
+
+export function useSubsManifest() {
+ return useQuery<SubsManifest>({
+ queryKey: ["subs-manifest"],
+ queryFn: () => fetchJson<SubsManifest>("/subs/manifest.json"),
+ });
+}
+
+// Loads each per-channel subs manifest so we can enumerate the full set of
+// sub-having slugs without fetching cue pages eagerly. Returns one entry per
+// channel listed in the cross-channel manifest. Errors are absorbed per
+// channel so a missing manifest doesn't break the whole UI.
+export function useChannelSubsManifests(channelSlugs: string[]) {
+ return useQueries({
+ queries: channelSlugs.map((channelSlug) => ({
+ queryKey: ["channel-subs-manifest", channelSlug],
+ queryFn: () => fetchChannelSubsManifest(channelSlug),
+ })),
+ });
+}
diff --git a/common/components/urlState.ts b/common/components/urlState.ts
@@ -2,6 +2,8 @@
import { useMemo, useSyncExternalStore } from "react";
+export type SearchMode = "transcripts" | "subs";
+
export type UrlParams = {
q: string;
re: boolean;
@@ -14,6 +16,12 @@ export type UrlParams = {
nar: boolean;
nav: boolean;
nd: boolean;
+ mode: SearchMode;
+ // Subs-mode track exclusion filter; empty array means "all tracks" (no
+ // exclusions). Mirrors the channel-exclusion convention. Persisted on the
+ // URL as repeated `tk=` params (e.g. ?m=subs&tk=live_chat to hide live
+ // chat results).
+ tracks: string[];
};
const listeners = new Set<() => void>();
@@ -39,6 +47,8 @@ function parse(search: string): UrlParams {
const p = new URLSearchParams(search);
const tRaw = p.get("t");
const t = tRaw !== null && tRaw !== "" ? Number(tRaw) : null;
+ const modeRaw = p.get("m");
+ const mode: SearchMode = modeRaw === "subs" ? "subs" : "transcripts";
return {
q: p.get("q") ?? "",
re: p.get("re") === "1",
@@ -51,6 +61,8 @@ function parse(search: string): UrlParams {
nar: p.get("nar") === "1",
nav: p.get("nav") === "1",
nd: p.get("nd") === "1",
+ mode,
+ tracks: p.getAll("tk"),
};
}
@@ -71,6 +83,8 @@ type Patch = Partial<{
nar: boolean;
nav: boolean;
nd: boolean;
+ mode: SearchMode;
+ tracks: string[];
}>;
export function writeUrlParams(patch: Patch) {
@@ -119,6 +133,14 @@ export function writeUrlParams(patch: Patch) {
if (patch.nd) params.set("nd", "1");
else params.delete("nd");
}
+ if (patch.mode !== undefined) {
+ if (patch.mode === "subs") params.set("m", "subs");
+ else params.delete("m");
+ }
+ if (patch.tracks !== undefined) {
+ params.delete("tk");
+ for (const v of patch.tracks) params.append("tk", v);
+ }
const qs = params.toString();
const next = `${window.location.pathname}${qs ? `?${qs}` : ""}`;
if (next === window.location.pathname + window.location.search) return;
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -24,6 +24,7 @@ import type { Dirent } from "node:fs";
import { open } from "lmdb";
import { parseVtt, type Cue } from "../lib/vtt";
import { parseWhisper } from "../lib/whisper";
+import { parseLiveChat } from "../lib/liveChat";
import {
summarize,
toDisplaySummary,
@@ -38,17 +39,24 @@ import type {
TranscriptDetail,
DisplaySummary,
} from "../lib/transcripts";
+import type { StoredSubs, SubsDetail } from "../lib/subs";
import type {
Manifest,
ChannelEntry,
ChannelTranscriptsManifest,
+ ChannelSubsManifest,
+ SubsChannelEntry,
+ SubsManifest,
} from "../lib/manifest";
import {
MANIFEST_VERSION,
SUMMARIES_PAGE_SIZE,
TRANSCRIPTS_MANIFEST_VERSION,
+ SUBS_CHANNEL_MANIFEST_VERSION,
+ SUBS_MANIFEST_VERSION,
pageFileName,
transcriptPageFileName,
+ subsPageFileName,
} from "../lib/manifest";
import { getSettings } from "../lib/settings";
import {
@@ -59,12 +67,14 @@ import {
import type { Paths } from "../lib/paths";
import {
pickIndexTranscript,
+ readSubTracks,
readVideoFiles,
type IndexTranscript,
+ type SubTrack,
} from "../lib/videoStatus";
import { loadAvailability } from "../lib/availability-server";
-const SCHEMA_VERSION = 6;
+const SCHEMA_VERSION = 7;
type IndexKey = [string, string, string];
type ChannelKey = [string, string, string];
@@ -74,6 +84,7 @@ type PathKey = [string, string];
type MtimeRecord = {
metaMs: number;
transcriptMs: number | null;
+ subsMs: number | null;
indexKey: IndexKey;
};
@@ -97,6 +108,8 @@ type LiveEntry = {
transcriptPath: string;
transcriptMs: number | null;
transcriptKind: IndexTranscript["kind"] | null;
+ subTracks: SubTrack[];
+ subsMs: number | null;
};
async function exists(p: string): Promise<boolean> {
@@ -178,6 +191,16 @@ async function scanSource(
transcriptMs = null;
}
}
+ const subTracks = await readSubTracks(fullVideoDir);
+ let subsMs: number | null = null;
+ for (const t of subTracks) {
+ try {
+ const ms = (await stat(path.join(fullVideoDir, t.filename))).mtimeMs;
+ if (subsMs === null || ms > subsMs) subsMs = ms;
+ } catch {
+ // ignore
+ }
+ }
live.push({
channelSlug: ch.name,
handling: cfg.handling,
@@ -188,6 +211,8 @@ async function scanSource(
transcriptPath,
transcriptMs,
transcriptKind: picked?.kind ?? null,
+ subTracks,
+ subsMs,
});
}
}
@@ -241,15 +266,17 @@ export async function buildIndex({
const dbPath = paths.lmdbPath;
const summariesDir = paths.exportSummariesDir;
const transcriptsOutDir = paths.exportTranscriptsDir;
+ const subsOutDir = paths.exportSubsDir;
const manifestPath = path.join(summariesDir, "manifest.json");
await mkdir(path.dirname(dbPath), { recursive: true });
await mkdir(summariesDir, { recursive: true });
await mkdir(transcriptsOutDir, { recursive: true });
+ await mkdir(subsOutDir, { recursive: true });
const root = open({
path: dbPath,
- maxDbs: 8,
+ maxDbs: 12,
compression: true,
});
const sums = root.openDB<TranscriptSummary, IndexKey>({
@@ -260,6 +287,10 @@ export async function buildIndex({
name: "cues",
encoding: "msgpack",
});
+ const subs = root.openDB<StoredSubs, IndexKey>({
+ name: "subs",
+ encoding: "msgpack",
+ });
const mtimes = root.openDB<MtimeRecord, PathKey>({
name: "mtimes",
encoding: "msgpack",
@@ -272,6 +303,10 @@ export async function buildIndex({
name: "pageHashes",
encoding: "msgpack",
});
+ const subPageHashes = root.openDB<PageHashRecord, PageHashKey>({
+ name: "subPageHashes",
+ encoding: "msgpack",
+ });
const meta = root.openDB<unknown, string>({
name: "meta",
encoding: "msgpack",
@@ -285,9 +320,11 @@ export async function buildIndex({
);
await sums.clearAsync();
await cues.clearAsync();
+ await subs.clearAsync();
await mtimes.clearAsync();
await byChannel.clearAsync();
await pageHashes.clearAsync();
+ await subPageHashes.clearAsync();
await meta.put("schema", SCHEMA_VERSION);
}
@@ -311,7 +348,8 @@ export async function buildIndex({
added.push(s);
} else if (
prev.metaMs !== s.metaMs ||
- prev.transcriptMs !== s.transcriptMs
+ prev.transcriptMs !== s.transcriptMs ||
+ (prev.subsMs ?? null) !== s.subsMs
) {
changed.push(s);
}
@@ -473,16 +511,39 @@ export async function buildIndex({
if (prev && !indexKeysEqual(prev.indexKey, indexKey)) {
sums.remove(prev.indexKey);
cues.remove(prev.indexKey);
+ subs.remove(prev.indexKey);
byChannel.remove(indexToChannelKey(prev.indexKey));
}
sums.put(indexKey, summary);
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
+
+ const parsedSubs: StoredSubs = [];
+ for (const t of s.subTracks) {
+ try {
+ const raw = await readFile(
+ path.join(path.dirname(s.metaPath), t.filename),
+ "utf8",
+ );
+ const trackCues =
+ t.track === "live_chat" ? parseLiveChat(raw) : parseVtt(raw);
+ if (trackCues.length > 0) {
+ parsedSubs.push({ track: t.track, cues: trackCues });
+ }
+ } catch {
+ // Skip unreadable / malformed sub tracks; the rest of the
+ // video's data is still useful.
+ }
+ }
+ if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs);
+ else subs.remove(indexKey);
+
byChannel.put(indexToChannelKey(indexKey), 1);
mtimes.put(pk, {
metaMs: s.metaMs,
transcriptMs: s.transcriptMs,
+ subsMs: s.subsMs,
indexKey,
});
} catch (err) {
@@ -499,12 +560,14 @@ export async function buildIndex({
for (const { pathKey, indexKey } of removed) {
sums.remove(indexKey);
cues.remove(indexKey);
+ subs.remove(indexKey);
byChannel.remove(indexToChannelKey(indexKey));
mtimes.remove(pathKey);
}
await sums.flushed;
await cues.flushed;
+ await subs.flushed;
await byChannel.flushed;
await mtimes.flushed;
@@ -646,19 +709,11 @@ export async function buildIndex({
`Transcript pages: ${pagesWritten} written, ${pagesSkipped} unchanged, ${pagesDeleted} stale removed.`,
);
- const pageSize = SUMMARIES_PAGE_SIZE;
- const channelCounts = new Map<string, number>();
- for (const cfg of channelConfigs.values()) {
- if (cfg.name) channelCounts.set(cfg.name, 0);
- }
- let pageIndex = 0;
- let buffer: DisplaySummary[] = [];
- let total = 0;
-
// Build a lookup from indexKey → isDeleted by reading each video's
// availability.json sidecar. Keyed off mtimes, which holds the canonical
// (channelSlug, videoDir) → indexKey mapping; only "deleted" entries get
- // recorded so lookup defaults to "not deleted".
+ // recorded so lookup defaults to "not deleted". Built once and reused by
+ // both the sub-page loop and the cross-channel summaries loop.
const deletedByIndexKey = new Set<string>();
for (const { key, value } of mtimes.getRange()) {
const pk = key as PathKey;
@@ -670,6 +725,199 @@ export async function buildIndex({
}
}
+ let subsPagesWritten = 0;
+ let subsPagesSkipped = 0;
+ let subsPagesDeleted = 0;
+ const subsChannelEntries: SubsChannelEntry[] = [];
+ let subsTotalCount = 0;
+
+ const writeSubsPage = async (
+ channelSlug: string,
+ idx: number,
+ entries: PendingEntry[],
+ ): Promise<void> => {
+ const body = `[${entries.map((e) => e.encoded).join(",")}]`;
+ const hash = createHash("sha1").update(body).digest("hex");
+ const prev = subPageHashes.get([channelSlug, idx]);
+ const outPath = path.join(subsOutDir, channelSlug, subsPageFileName(idx));
+ if (prev?.hash === hash && (await exists(outPath))) {
+ subsPagesSkipped++;
+ return;
+ }
+ const sizeBytes = Buffer.byteLength(body, "utf8");
+ const tmp = `${outPath}.tmp-${process.pid}`;
+ await writeFile(tmp, body);
+ await rename(tmp, outPath);
+ subPageHashes.put([channelSlug, idx], {
+ hash,
+ entryCount: entries.length,
+ sizeBytes,
+ });
+ subsPagesWritten++;
+ };
+
+ for (const channelSlug of Array.from(channelConfigs.keys()).sort()) {
+ const cfg = channelConfigs.get(channelSlug)!;
+ const subsChannelDir = path.join(subsOutDir, channelSlug);
+ let subPageIdx = 0;
+ let subBuffer: PendingEntry[] = [];
+ let subPayloadBytes = 0;
+ const subSlugToPage: Record<string, number> = {};
+ const tracksInChannel = new Set<string>();
+ let videoCount = 0;
+
+ for (const { key } of byChannel.getRange({
+ start: [channelSlug],
+ end: [channelSlug, ""],
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== channelSlug) continue;
+ const indexKey: IndexKey = [ck[1], ck[0], ck[2]];
+ const stored = subs.get(indexKey);
+ if (!stored || stored.length === 0) continue;
+ const summary = sums.get(indexKey);
+ if (!summary) continue;
+ const isDeleted = deletedByIndexKey.has(
+ `${indexKey[0]}\x00${indexKey[1]}\x00${indexKey[2]}`,
+ );
+ const display = toDisplaySummary(summary, { isDeleted });
+ const tracks: Record<string, Cue[]> = {};
+ for (const t of stored) {
+ tracks[t.track] = t.cues;
+ tracksInChannel.add(t.track);
+ }
+ const detail: SubsDetail = { ...display, tracks };
+ const encoded = JSON.stringify(detail);
+ const entryBytes = Buffer.byteLength(encoded, "utf8");
+ const commaBytes = subBuffer.length === 0 ? 0 : 1;
+ const delta = entryBytes + commaBytes;
+
+ if (
+ subBuffer.length > 0 &&
+ subPayloadBytes + delta + 2 > maxTranscriptPageBytes
+ ) {
+ await mkdir(subsChannelDir, { recursive: true });
+ await writeSubsPage(channelSlug, subPageIdx, subBuffer);
+ subPageIdx++;
+ subBuffer = [];
+ subPayloadBytes = 0;
+ }
+
+ if (entryBytes + 2 > maxTranscriptPageBytes) {
+ log(
+ ` ${summary.slug}: sub entry (${entryBytes} bytes) exceeds page budget; emitting solo page.`,
+ );
+ }
+
+ subSlugToPage[summary.id] = subPageIdx;
+ subBuffer.push({ encoded, id: summary.id });
+ subPayloadBytes += delta;
+ videoCount++;
+ }
+
+ if (subBuffer.length > 0) {
+ await mkdir(subsChannelDir, { recursive: true });
+ await writeSubsPage(channelSlug, subPageIdx, subBuffer);
+ subPageIdx++;
+ }
+
+ if (videoCount === 0) {
+ // No subs for this channel — remove any stale dir + hashes.
+ await rm(subsChannelDir, { recursive: true, force: true });
+ for (const { key } of subPageHashes.getRange({
+ start: [channelSlug],
+ end: [channelSlug, Number.MAX_SAFE_INTEGER],
+ })) {
+ subPageHashes.remove(key as PageHashKey);
+ }
+ continue;
+ }
+
+ const subPageCount = subPageIdx;
+ const subKeep = new Set<string>(["manifest.json"]);
+ for (let i = 0; i < subPageCount; i++) subKeep.add(subsPageFileName(i));
+ const subExisting = await readdir(subsChannelDir).catch(
+ () => [] as string[],
+ );
+ for (const name of subExisting) {
+ if (subKeep.has(name)) continue;
+ await rm(path.join(subsChannelDir, name), { force: true });
+ subsPagesDeleted++;
+ }
+ for (const { key } of subPageHashes.getRange({
+ start: [channelSlug, subPageCount],
+ end: [channelSlug, Number.MAX_SAFE_INTEGER],
+ })) {
+ subPageHashes.remove(key as PageHashKey);
+ }
+
+ const tracksList = Array.from(tracksInChannel).sort();
+ const channelSubsManifest: ChannelSubsManifest = {
+ version: SUBS_CHANNEL_MANIFEST_VERSION,
+ channelSlug,
+ pageCount: subPageCount,
+ maxPageBytes: maxTranscriptPageBytes,
+ generatedAt,
+ tracks: tracksList,
+ slugToPage: subSlugToPage,
+ };
+ await writeJsonAtomic(
+ path.join(subsChannelDir, "manifest.json"),
+ channelSubsManifest,
+ );
+ subsChannelEntries.push({
+ name: cfg.name ?? channelSlug,
+ slug: channelSlug,
+ videoCount,
+ tracks: tracksList,
+ });
+ subsTotalCount += videoCount;
+ }
+
+ await subPageHashes.flushed;
+
+ // Top-level cleanup: drop sub dirs for channels that no longer exist.
+ const topSubsEntries = await readdir(subsOutDir, {
+ withFileTypes: true,
+ }).catch(() => [] as Dirent[]);
+ const subsChannelSlugSet = new Set(subsChannelEntries.map((e) => e.slug));
+ for (const e of topSubsEntries) {
+ if (e.isDirectory()) {
+ if (!subsChannelSlugSet.has(e.name)) {
+ await rm(path.join(subsOutDir, e.name), {
+ recursive: true,
+ force: true,
+ });
+ }
+ } else if (e.isFile() && e.name !== "manifest.json") {
+ await rm(path.join(subsOutDir, e.name), { force: true });
+ }
+ }
+
+ const subsManifestOut: SubsManifest = {
+ version: SUBS_MANIFEST_VERSION,
+ channels: subsChannelEntries.sort((a, b) => a.name.localeCompare(b.name)),
+ totalCount: subsTotalCount,
+ generatedAt,
+ };
+ await writeJsonAtomic(
+ path.join(subsOutDir, "manifest.json"),
+ subsManifestOut,
+ );
+
+ log(
+ `Sub pages: ${subsPagesWritten} written, ${subsPagesSkipped} unchanged, ${subsPagesDeleted} stale removed across ${subsChannelEntries.length} channels (${subsTotalCount} sub videos).`,
+ );
+
+ const pageSize = SUMMARIES_PAGE_SIZE;
+ const channelCounts = new Map<string, number>();
+ for (const cfg of channelConfigs.values()) {
+ if (cfg.name) channelCounts.set(cfg.name, 0);
+ }
+ let pageIndex = 0;
+ let buffer: DisplaySummary[] = [];
+ let total = 0;
+
const flushPage = async () => {
if (buffer.length === 0) return;
const p = path.join(summariesDir, pageFileName(pageIndex));
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -13,6 +13,7 @@ export type ChannelConfig = {
keepSourceVideo?: boolean;
cookiesFromBrowser?: string;
ytdlpExtraArgs?: string[];
+ subLangs?: string;
lastSyncedAt?: string;
lastFullDownloadAt?: string;
};
@@ -60,6 +61,7 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null {
) {
config.ytdlpExtraArgs = r.ytdlpExtraArgs as string[];
}
+ if (typeof r.subLangs === "string") config.subLangs = r.subLangs;
if (typeof r.lastSyncedAt === "string") config.lastSyncedAt = r.lastSyncedAt;
if (typeof r.lastFullDownloadAt === "string") {
config.lastFullDownloadAt = r.lastFullDownloadAt;
diff --git a/common/lib/liveChat.ts b/common/lib/liveChat.ts
@@ -0,0 +1,112 @@
+import type { Cue } from "./vtt";
+
+// yt-dlp writes live_chat as JSON-lines (one continuation action per line),
+// extracted from YouTube's youtube_live_chat_replay protocol. Each line wraps
+// a `replayChatItemAction` whose `actions[]` contain `addChatItemAction.item`
+// entries with one of several renderers (`liveChatTextMessageRenderer`,
+// `liveChatPaidMessageRenderer`, `liveChatMembershipItemRenderer`, etc.).
+//
+// We emit one Cue per addChatItemAction that has a renderable text body, with
+// `start` derived from `videoOffsetTimeMsec` and `text` formatted as
+// `<author>: <message>`. Non-text events (ticker pings without an associated
+// message, viewer-engagement nags) are skipped.
+
+type Run = { text?: string; emoji?: { emojiId?: string; shortcuts?: string[] } };
+type MessageBody = { simpleText?: string; runs?: Run[] };
+type AuthorBody = { simpleText?: string };
+type ChatRenderer = {
+ message?: MessageBody;
+ headerSubtext?: MessageBody;
+ text?: MessageBody;
+ authorName?: AuthorBody;
+};
+type ChatItem = Record<string, ChatRenderer | undefined>;
+type AddAction = { item?: ChatItem };
+type ChatAction = { addChatItemAction?: AddAction };
+type ReplayAction = {
+ actions?: ChatAction[];
+ videoOffsetTimeMsec?: string;
+};
+type ReplayEnvelope = {
+ replayChatItemAction?: ReplayAction;
+ videoOffsetTimeMsec?: string;
+};
+
+const TEXT_RENDERERS = [
+ "liveChatTextMessageRenderer",
+ "liveChatPaidMessageRenderer",
+ "liveChatPaidStickerRenderer",
+ "liveChatMembershipItemRenderer",
+ "liveChatSponsorshipsGiftPurchaseAnnouncementRenderer",
+ "liveChatSponsorshipsGiftRedemptionAnnouncementRenderer",
+];
+
+function renderMessage(body: MessageBody | undefined): string {
+ if (!body) return "";
+ if (typeof body.simpleText === "string") return body.simpleText;
+ if (!Array.isArray(body.runs)) return "";
+ const parts: string[] = [];
+ for (const run of body.runs) {
+ if (typeof run.text === "string") {
+ parts.push(run.text);
+ } else if (run.emoji?.shortcuts?.length) {
+ parts.push(run.emoji.shortcuts[0]);
+ }
+ }
+ return parts.join("");
+}
+
+function authorOf(renderer: ChatRenderer): string {
+ return renderer.authorName?.simpleText ?? "";
+}
+
+function textOf(renderer: ChatRenderer): string {
+ // Sponsorship gift announcements use `headerSubtext`; sticker/paid use
+ // `message`; membership pings use `headerSubtext` too. Fall back through
+ // the known fields.
+ return (
+ renderMessage(renderer.message) ||
+ renderMessage(renderer.headerSubtext) ||
+ renderMessage(renderer.text)
+ );
+}
+
+export function parseLiveChat(src: string): Cue[] {
+ const lines = src.split("\n");
+ const cues: Cue[] = [];
+ for (const line of lines) {
+ const trimmed = line.trim();
+ if (!trimmed) continue;
+ let envelope: ReplayEnvelope;
+ try {
+ envelope = JSON.parse(trimmed) as ReplayEnvelope;
+ } catch {
+ continue;
+ }
+ const replay = envelope.replayChatItemAction;
+ const offsetRaw =
+ replay?.videoOffsetTimeMsec ?? envelope.videoOffsetTimeMsec;
+ const offsetMs = offsetRaw ? Number(offsetRaw) : NaN;
+ if (!Number.isFinite(offsetMs)) continue;
+ const start = offsetMs / 1000;
+ for (const action of replay?.actions ?? []) {
+ const item = action.addChatItemAction?.item;
+ if (!item) continue;
+ for (const rendererKey of TEXT_RENDERERS) {
+ const renderer = item[rendererKey];
+ if (!renderer) continue;
+ const text = textOf(renderer).trim();
+ if (!text) continue;
+ const author = authorOf(renderer);
+ cues.push({
+ start,
+ end: start + 5,
+ text: author ? `${author}: ${text}` : text,
+ });
+ break;
+ }
+ }
+ }
+ cues.sort((a, b) => a.start - b.start);
+ return cues;
+}
diff --git a/common/lib/manifest.ts b/common/lib/manifest.ts
@@ -33,3 +33,35 @@ export type ChannelTranscriptsManifest = {
generatedAt: string;
slugToPage: Record<string, number>;
};
+
+export const SUBS_MANIFEST_VERSION = 1;
+
+export type SubsChannelEntry = {
+ name: string;
+ slug: string;
+ videoCount: number;
+ tracks: string[];
+};
+
+export type SubsManifest = {
+ version: number;
+ channels: SubsChannelEntry[];
+ totalCount: number;
+ generatedAt: string;
+};
+
+export const SUBS_CHANNEL_MANIFEST_VERSION = 1;
+
+export type ChannelSubsManifest = {
+ version: number;
+ channelSlug: string;
+ pageCount: number;
+ maxPageBytes: number;
+ generatedAt: string;
+ tracks: string[];
+ slugToPage: Record<string, number>;
+};
+
+export function subsPageFileName(index: number): string {
+ return `page-${String(index).padStart(4, "0")}.json`;
+}
diff --git a/common/lib/paths.ts b/common/lib/paths.ts
@@ -12,6 +12,7 @@ export type Paths = {
exportPublicDir: string;
exportSummariesDir: string;
exportTranscriptsDir: string;
+ exportSubsDir: string;
settingsFile: string;
ytdlpBin: string;
whisperBin: string;
@@ -40,6 +41,7 @@ export function getPaths(): Paths {
exportPublicDir,
exportSummariesDir: path.join(exportPublicDir, "summaries"),
exportTranscriptsDir: path.join(exportPublicDir, "transcripts"),
+ exportSubsDir: path.join(exportPublicDir, "subs"),
settingsFile:
process.env.SETTINGS_FILE ?? path.join(monorepoRoot, "settings.json"),
ytdlpBin: process.env.YTDLP_BIN ?? "yt-dlp",
diff --git a/common/lib/subs.ts b/common/lib/subs.ts
@@ -0,0 +1,19 @@
+import type { Cue } from "./vtt";
+import type { DisplaySummary } from "./transcripts";
+
+export type SubTrackCues = {
+ track: string;
+ cues: Cue[];
+};
+
+// Stored in LMDB under the `subs` DB, keyed by [uploadDate, channelSlug, id].
+export type StoredSubs = SubTrackCues[];
+
+// One entry per video in /public/subs/<channelSlug>/page-NNNN.json. The
+// display fields mirror DisplaySummary so the subs search UI doesn't need a
+// second metadata round-trip; cues are inlined per-track.
+export type SubsDetail = DisplaySummary & {
+ tracks: Record<string, Cue[]>;
+};
+
+export type SubsPage = SubsDetail[];
diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts
@@ -19,6 +19,24 @@ export const WHISPER_FILENAME = "transcript.json";
export const CUES_JSON_FILENAME = "transcript.cues.json";
export const META_FILENAME = "metadata.info.json";
+export type SubTrack = {
+ // "live_chat" | language code like "es", "en-orig", etc.
+ track: string;
+ filename: string;
+ ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3";
+};
+
+// Treat any transcript.<x>.<y> file as a sub track when it isn't one of the
+// primary transcript outputs (transcript.en.vtt, transcript.json) or a
+// derived/auxiliary file (transcript.cues.json). Live chat lands as
+// transcript.live_chat.json; non-en languages as transcript.<lang>.vtt.
+const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/;
+const SUB_EXT_VALUES = ["vtt", "json", "json3", "srv1", "srv2", "srv3"] as const;
+type SubExt = (typeof SUB_EXT_VALUES)[number];
+function isSubExt(value: string): value is SubExt {
+ return (SUB_EXT_VALUES as readonly string[]).includes(value);
+}
+
// Whisper "empty transcription" outputs include systeminfo/model/params/result
// keys around an empty `transcription: []`, so they can be up to ~620B in
// practice. Real transcripts observed start at ~2.9KB. A 4KB cutoff lets us
@@ -69,6 +87,21 @@ export async function readVideoFiles(
};
}
+export async function readSubTracks(videoDir: string): Promise<SubTrack[]> {
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ const tracks: SubTrack[] = [];
+ for (const entry of entries) {
+ if (entry === VTT_FILENAME || entry === WHISPER_FILENAME) continue;
+ if (entry === CUES_JSON_FILENAME) continue;
+ const m = entry.match(SUB_FILE_RE);
+ if (!m) continue;
+ const ext = m[2].toLowerCase();
+ if (!isSubExt(ext)) continue;
+ tracks.push({ track: m[1], filename: entry, ext });
+ }
+ return tracks;
+}
+
export function pickIndexTranscript(files: VideoFiles): IndexTranscript | null {
if (files.hasWhisper) return { kind: "whisper", filename: WHISPER_FILENAME };
if (files.hasYtVtt) return { kind: "vtt", filename: VTT_FILENAME };
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -23,6 +23,7 @@ export type YtdlpMode =
| "store-playlist"
| "download-from-playlist"
| "download-missing"
+ | "download-missing-subs"
| "download-one-audio"
| "sync";
@@ -66,6 +67,9 @@ export async function runYtdlp(opts: RunYtdlpOpts): Promise<void> {
case "download-missing":
await downloadMissing(opts);
return;
+ case "download-missing-subs":
+ await downloadMissingSubs(opts);
+ return;
case "download-one-audio":
await downloadOneAudio(opts);
return;
@@ -92,6 +96,9 @@ function handlingArgs(config: ChannelConfig): string[] {
if (config.handling === "youtube") {
return [
"--write-auto-subs",
+ "--write-subs",
+ "--sub-langs",
+ config.subLangs ?? "en.*,live_chat",
"--write-info-json",
"--skip-download",
"-t",
@@ -336,6 +343,172 @@ async function downloadMissing(opts: RunYtdlpOpts): Promise<void> {
await safeBackfillAvailability(opts);
}
+// yt-dlp's --sub-langs supports a comma-separated list with shell-glob style
+// wildcards (e.g. `en.*`, `all`) and exclusions (`-fr`). Replicate just enough
+// of that here to decide whether a given track key from a video's metadata
+// (subtitles + automatic_captions) is one we'd expect on disk after running
+// yt-dlp with the channel's configured `subLangs`.
+function compileSubLangMatcher(spec: string): (track: string) => boolean {
+ const include: string[] = [];
+ const exclude: string[] = [];
+ for (const raw of spec.split(",")) {
+ const trimmed = raw.trim();
+ if (!trimmed) continue;
+ if (trimmed.startsWith("-")) exclude.push(trimmed.slice(1));
+ else include.push(trimmed);
+ }
+ const matches = (track: string, pattern: string): boolean => {
+ if (pattern === "all") return true;
+ if (pattern.includes("*")) {
+ const re = new RegExp(
+ "^" +
+ pattern
+ .replace(/[.+?^${}()|[\]\\]/g, "\\$&")
+ .replace(/\\\*/g, ".*") +
+ "$",
+ );
+ return re.test(track);
+ }
+ return pattern === track;
+ };
+ return (track: string) => {
+ if (exclude.some((p) => matches(track, p))) return false;
+ return include.some((p) => matches(track, p));
+ };
+}
+
+const SUB_EXT_PATTERN = "(?:vtt|json|json3|srv1|srv2|srv3)";
+
+function trackOnDisk(entries: string[], track: string): boolean {
+ const escaped = track.replace(/[.+?^${}()|[\]\\]/g, "\\$&");
+ const re = new RegExp(`^transcript\\.${escaped}\\.${SUB_EXT_PATTERN}$`);
+ return entries.some((e) => re.test(e));
+}
+
+async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> {
+ const root = channelRoot(opts);
+ const dataDir = path.join(root, "data");
+ const playlistPath = path.join(root, "playlist");
+ const toFetchPath = path.join(root, "playlist.tofetch-subs");
+
+ let playlistText: string;
+ try {
+ playlistText = await readFile(playlistPath, "utf8");
+ } catch {
+ throw new Error(
+ `No saved playlist at ${playlistPath}. Run "Store playlist" first.`,
+ );
+ }
+ const urls = playlistText
+ .split("\n")
+ .map((s) => s.trim())
+ .filter(Boolean);
+
+ const subLangs = opts.channelConfig.subLangs ?? "en.*,live_chat";
+ const matchesSubLang = compileSubLangMatcher(subLangs);
+
+ const rumbleIndex = urls.some(isRumbleUrl)
+ ? await buildRumbleSlugIndex(dataDir)
+ : null;
+ const tofetch: string[] = [];
+ let upToDate = 0;
+ let unidentifiable = 0;
+ let noMetadata = 0;
+ let noExpectedTracks = 0;
+
+ for (const url of urls) {
+ const slug = extractVideoId(url);
+ if (!slug) {
+ unidentifiable++;
+ continue;
+ }
+ const dirId =
+ rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug;
+ if (!dirId) {
+ // No corresponding data dir yet — skip; download-from-playlist /
+ // download-missing should pull the video and its metadata first.
+ noMetadata++;
+ continue;
+ }
+ const videoDir = path.join(dataDir, dirId);
+ let metaRaw: string;
+ try {
+ metaRaw = await readFile(
+ path.join(videoDir, "metadata.info.json"),
+ "utf8",
+ );
+ } catch {
+ noMetadata++;
+ continue;
+ }
+ let parsedMeta: {
+ subtitles?: Record<string, unknown>;
+ automatic_captions?: Record<string, unknown>;
+ };
+ try {
+ parsedMeta = JSON.parse(metaRaw);
+ } catch {
+ noMetadata++;
+ continue;
+ }
+ const advertised = new Set<string>();
+ for (const k of Object.keys(parsedMeta.subtitles ?? {})) advertised.add(k);
+ for (const k of Object.keys(parsedMeta.automatic_captions ?? {})) {
+ advertised.add(k);
+ }
+ const expected = Array.from(advertised).filter(matchesSubLang);
+ if (expected.length === 0) {
+ noExpectedTracks++;
+ continue;
+ }
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ const missing = expected.filter((t) => !trackOnDisk(entries, t));
+ if (missing.length === 0) {
+ upToDate++;
+ continue;
+ }
+ tofetch.push(url);
+ }
+
+ opts.onLog(
+ `Prefilter: ${tofetch.length} videos missing subs, ${upToDate} already up to date` +
+ (noExpectedTracks
+ ? `, ${noExpectedTracks} with no tracks matching ${JSON.stringify(subLangs)}`
+ : "") +
+ (noMetadata ? `, ${noMetadata} without metadata (skipped)` : "") +
+ (unidentifiable ? `, ${unidentifiable} unidentifiable URLs` : "") +
+ `.\n`,
+ );
+
+ if (tofetch.length === 0) {
+ await rm(toFetchPath, { force: true });
+ opts.onLog("Nothing to fetch.\n");
+ return;
+ }
+
+ await writeFile(toFetchPath, tofetch.join("\n") + "\n");
+
+ const args: string[] = [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...OUTPUT_ARGS,
+ "--write-auto-subs",
+ "--write-subs",
+ "--sub-langs",
+ subLangs,
+ "--skip-download",
+ "--no-write-info-json",
+ "--no-overwrites",
+ ...(opts.abortOnError === false ? [] : ["--abort-on-error"]),
+ "-a",
+ "playlist.tofetch-subs",
+ ...configArgs(opts.channelConfig),
+ ];
+ await runChildAndStream(opts, root, args);
+
+ await rm(toFetchPath, { force: true });
+}
+
async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> {
if (!opts.singleVideoUrl) {
throw new Error("download-one-audio requires singleVideoUrl");
diff --git a/editor/app/channels/[slug]/components/stages/DownloadStage.tsx b/editor/app/channels/[slug]/components/stages/DownloadStage.tsx
@@ -12,6 +12,7 @@ import { cancelJobAction } from "../../../../jobs/actions";
import {
downloadAction,
downloadMissingAction,
+ downloadMissingSubsAction,
} from "../../pipelineActions";
import { VideoIdList } from "../VideoIdList";
@@ -36,9 +37,11 @@ export function DownloadStage({
}: Props) {
const [downloadQueue, setDownloadQueue] = useState(defaultQueueKey);
const [missingQueue, setMissingQueue] = useState(defaultQueueKey);
+ const [subsQueue, setSubsQueue] = useState(defaultQueueKey);
const [ignoreArchive, setIgnoreArchive] = useState(false);
const [downloadAbortOnError, setDownloadAbortOnError] = useState(true);
const [missingAbortOnError, setMissingAbortOnError] = useState(true);
+ const [subsAbortOnError, setSubsAbortOnError] = useState(false);
const [missingShardTotal, setMissingShardTotal] = useState(
missingShard ? String(missingShard.totalShards) : "",
);
@@ -156,6 +159,41 @@ export function DownloadStage({
}
/>
</div>
+ <div className="flex flex-col gap-2">
+ <Heading
+ title="Download missing subs"
+ desc="Backfill sub tracks (live_chat, alt-language captions) for already-downloaded videos. Reads each video's metadata.info.json for advertised tracks, then re-invokes yt-dlp with --skip-download for any whose expected sub file is missing on disk. Uses the channel's subLangs config (defaults to en.*,live_chat)."
+ />
+ <label className="flex items-center gap-2 text-sm">
+ <input
+ type="checkbox"
+ checked={subsAbortOnError}
+ onChange={(e) => setSubsAbortOnError(e.target.checked)}
+ aria-label="abort on error for Download missing subs"
+ />
+ Abort on error
+ <span className="text-xs text-zinc-500">
+ (stop the run on the first per-video failure)
+ </span>
+ </label>
+ <StreamActionLog
+ trigger={() =>
+ downloadMissingSubsAction(slug, subsQueue, subsAbortOnError)
+ }
+ cancelAction={cancelJobAction}
+ buttonLabel="Download missing subs"
+ runningLabel="Downloading subs…"
+ extraControls={
+ <QueueControl
+ value={subsQueue}
+ onChange={setSubsQueue}
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ actionLabel="Download missing subs"
+ />
+ }
+ />
+ </div>
<NoTranscriptList slug={slug} ids={noTranscriptIds} />
</div>
);
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -20,7 +20,12 @@ function defaultQueueKey(config: ChannelConfig): string {
async function runPipelineAction(
slug: string,
- mode: "store-playlist" | "download-from-playlist" | "sync" | "download-missing",
+ mode:
+ | "store-playlist"
+ | "download-from-playlist"
+ | "sync"
+ | "download-missing"
+ | "download-missing-subs",
kind: string,
queueKey?: string,
options?: {
@@ -107,3 +112,17 @@ export async function syncAction(
): Promise<StreamActionResult> {
return runPipelineAction(slug, "sync", "sync", queueKey);
}
+
+export async function downloadMissingSubsAction(
+ slug: string,
+ queueKey?: string,
+ abortOnError?: boolean,
+): Promise<StreamActionResult> {
+ return runPipelineAction(
+ slug,
+ "download-missing-subs",
+ "download-missing-subs",
+ queueKey,
+ { abortOnError },
+ );
+}
diff --git a/editor/e2e/export-search.spec.ts b/editor/e2e/export-search.spec.ts
@@ -0,0 +1,310 @@
+import { test, expect, type Page } from "@playwright/test";
+
+const EXPORT_BASE = `http://localhost:${process.env.EXPORT_PORT ?? 3000}`;
+
+type Summary = {
+ slug: string;
+ id: string;
+ channelSlug: string;
+ title: string;
+ uploadDate: string;
+ date: string;
+ duration: string;
+ channel: string;
+ isLivestream: boolean;
+ ageRestricted: boolean;
+ isDeleted: boolean;
+ platform: "youtube" | "rumble" | "odysee";
+ webpageUrl: string;
+};
+
+const CHANNEL = "Test Channel";
+const CHANNEL_SLUG = "test-channel";
+
+const summaries: Summary[] = [
+ // Title contains "platypus", transcript contains "platypus"
+ makeSummary("v-both", "Platypus facts and figures", false),
+ // Title contains "platypus", transcript does NOT
+ makeSummary("v-title-only", "All about the platypus", false),
+ // Title does not contain "platypus", transcript does
+ makeSummary("v-cue-only", "An ordinary mammal", false),
+ // Deleted, title contains "platypus"
+ makeSummary("v-deleted-title", "Deleted platypus chronicles", true),
+ // Deleted, title does NOT contain "platypus", transcript contains it
+ makeSummary("v-deleted-cue", "Vanished video on aquatic life", true),
+];
+
+function makeSummary(id: string, title: string, isDeleted: boolean): Summary {
+ return {
+ slug: `${CHANNEL_SLUG}/${id}`,
+ id,
+ channelSlug: CHANNEL_SLUG,
+ title,
+ uploadDate: "20260101",
+ date: "2026-01-01",
+ duration: "5:00",
+ channel: CHANNEL,
+ isLivestream: false,
+ ageRestricted: false,
+ isDeleted,
+ platform: "youtube",
+ webpageUrl: `https://example.com/${id}`,
+ };
+}
+
+const transcripts: Record<string, { cues: { start: number; text: string }[] }> =
+ {
+ "v-both": { cues: [{ start: 12, text: "the platypus has a beak" }] },
+ "v-title-only": {
+ cues: [{ start: 0, text: "nothing matching here at all" }],
+ },
+ "v-cue-only": {
+ cues: [{ start: 30, text: "an ordinary platypus appears" }],
+ },
+ "v-deleted-title": {
+ cues: [{ start: 5, text: "no relevant content in this transcript" }],
+ },
+ "v-deleted-cue": {
+ cues: [{ start: 45, text: "the platypus swims silently" }],
+ },
+ };
+
+async function installFixtureRoutes(page: Page) {
+ const manifest = {
+ version: 1,
+ totalCount: summaries.length,
+ pageSize: 1000,
+ pageCount: 1,
+ generatedAt: new Date().toISOString(),
+ channels: [{ name: CHANNEL, count: summaries.length }],
+ };
+
+ await page.route("**/summaries/manifest.json", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(manifest),
+ });
+ });
+
+ await page.route("**/summaries/page-0000.json", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(summaries),
+ });
+ });
+
+ await page.route("**/transcripts/**/manifest.json", async (route) => {
+ const url = new URL(route.request().url());
+ const channelSlug = url.pathname.split("/").slice(-2, -1)[0];
+ if (channelSlug !== CHANNEL_SLUG) {
+ await route.fulfill({ status: 404, body: "" });
+ return;
+ }
+ const slugToPage: Record<string, number> = {};
+ for (const s of summaries) slugToPage[s.id] = 0;
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({ version: 1, pageSize: 1000, slugToPage }),
+ });
+ });
+
+ await page.route("**/transcripts/**/page-*.json", async (route) => {
+ // The on-disk shape is an array of TranscriptDetail entries, each with
+ // its own `slug`. transcriptCache iterates and matches by slug.
+ const body = summaries.map((s) => ({
+ slug: s.slug,
+ id: s.id,
+ cues: transcripts[s.id].cues,
+ }));
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(body),
+ });
+ });
+
+ // Subs manifest empty so subs mode never tries to fetch anything.
+ await page.route("**/subs/manifest.json", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({
+ version: 1,
+ channels: [],
+ totalCount: 0,
+ generatedAt: new Date().toISOString(),
+ }),
+ });
+ });
+}
+
+async function search(page: Page, query: string) {
+ const input = page.getByPlaceholder("Search transcripts...");
+ await input.click();
+ await input.fill(query);
+ await expect(input).toHaveValue(query);
+ await input.press("Enter");
+ // commitSearch writes ?q=… to the URL; wait for it so subsequent assertions
+ // run against the post-search render rather than the no-query placeholder.
+ await page.waitForURL(/[?&]q=/);
+}
+
+async function waitForHydration(page: Page) {
+ // The "Regex" checkbox is only rendered after React hydrates the
+ // TranscriptSearch client component, so it's a reliable hydration probe.
+ await page.getByRole("checkbox", { name: "Regex" }).waitFor();
+}
+
+test.describe("export TranscriptSearch — title matching + filter layout", () => {
+ test.beforeEach(async ({ page }) => {
+ await installFixtureRoutes(page);
+ await page.goto(EXPORT_BASE);
+ await waitForHydration(page);
+ });
+
+ test("subs tab is hidden when the subs manifest reports no subs", async ({
+ page,
+ }) => {
+ // The search-mode tablist is not rendered at all when there are no subs
+ // to switch to. (Next.js dev-tools may render unrelated tablists; scope
+ // the search by aria-label.)
+ await expect(
+ page.getByRole("tablist", { name: "Search mode" }),
+ ).toHaveCount(0);
+ await expect(page.getByRole("tab", { name: /^Subs/ })).toHaveCount(0);
+ });
+
+ test("subs tab appears when the subs manifest has at least one channel", async ({
+ page,
+ }) => {
+ // Re-route the manifest with non-zero totalCount, then reload so the
+ // updated response is fetched.
+ await page.unroute("**/subs/manifest.json");
+ await page.route("**/subs/manifest.json", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({
+ version: 1,
+ channels: [{ name: CHANNEL, slug: CHANNEL_SLUG, videoCount: 1, tracks: ["en"] }],
+ totalCount: 1,
+ generatedAt: new Date().toISOString(),
+ }),
+ });
+ });
+ await page.reload();
+ await waitForHydration(page);
+
+ await expect(
+ page.getByRole("tab", { name: /^Subs/ }),
+ ).toBeVisible();
+ await expect(
+ page.getByRole("tab", { name: "Transcripts" }),
+ ).toBeVisible();
+ });
+
+ test("filter checkboxes are grouped under Type / Audience / Availability labels", async ({
+ page,
+ }) => {
+ // Trigger search so the result area renders alongside the filter row.
+ await search(page, "platypus");
+
+ await expect(page.getByText("Type", { exact: true })).toBeVisible();
+ await expect(page.getByText("Audience", { exact: true })).toBeVisible();
+ await expect(
+ page.getByText("Availability", { exact: true }),
+ ).toBeVisible();
+
+ // Each labeled group contains its expected checkboxes.
+ await expect(
+ page.getByRole("checkbox", { name: "Videos" }),
+ ).toBeChecked();
+ await expect(
+ page.getByRole("checkbox", { name: "Available" }),
+ ).toBeChecked();
+ await expect(
+ page.getByRole("checkbox", { name: "Deleted" }),
+ ).toBeChecked();
+ });
+
+ test("title matches are surfaced as a 'title' badge row above transcript hits", async ({
+ page,
+ }) => {
+ await search(page, "platypus");
+
+ // Group headers include the channel + date suffix; match those specifically
+ // so we don't collide with the title-row buttons.
+ const both = page.getByRole("button", {
+ name: /Platypus facts and figures.*Test Channel/,
+ });
+ await expect(both).toBeVisible();
+
+ const titleOnly = page.getByRole("button", {
+ name: /All about the platypus.*Test Channel/,
+ });
+ await expect(titleOnly).toBeVisible();
+
+ // 'title' badge appears once per title-matched video. With deleted included
+ // by default, that's v-both, v-title-only, v-deleted-title.
+ const titleBadges = page.getByText("title", { exact: true });
+ await expect(titleBadges).toHaveCount(3);
+ });
+
+ test("'Deleted' unchecked hides deleted videos from both title and transcript matches", async ({
+ page,
+ }) => {
+ await search(page, "platypus");
+ // Sanity: by default, deleted videos are included.
+ await expect(
+ page.getByRole("button", {
+ name: /Deleted platypus chronicles.*Test Channel/,
+ }),
+ ).toBeVisible();
+
+ await page.getByRole("checkbox", { name: "Deleted" }).uncheck();
+ await page.getByPlaceholder("Search transcripts...").press("Enter");
+
+ await expect(
+ page.getByRole("button", { name: /Deleted platypus chronicles/ }),
+ ).toHaveCount(0);
+ await expect(
+ page.getByRole("button", { name: /Vanished video on aquatic life/ }),
+ ).toHaveCount(0);
+ // Non-deleted matches remain.
+ await expect(
+ page.getByRole("button", {
+ name: /Platypus facts and figures.*Test Channel/,
+ }),
+ ).toBeVisible();
+ });
+
+ test("'Available' unchecked narrows to deleted-only videos", async ({
+ page,
+ }) => {
+ await search(page, "platypus");
+ await page.getByRole("checkbox", { name: "Available" }).uncheck();
+ await page.getByPlaceholder("Search transcripts...").press("Enter");
+
+ // Deleted videos remain.
+ await expect(
+ page.getByRole("button", {
+ name: /Deleted platypus chronicles.*Test Channel/,
+ }),
+ ).toBeVisible();
+ await expect(
+ page.getByRole("button", {
+ name: /Vanished video on aquatic life.*Test Channel/,
+ }),
+ ).toBeVisible();
+ // Non-deleted videos drop out.
+ await expect(
+ page.getByRole("button", { name: /Platypus facts and figures/ }),
+ ).toHaveCount(0);
+ await expect(
+ page.getByRole("button", { name: /All about the platypus/ }),
+ ).toHaveCount(0);
+ });
+});
diff --git a/editor/package.json b/editor/package.json
@@ -6,6 +6,7 @@
"scripts": {
"dev": "next dev --port 3001",
"dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs next dev --port 3001",
+ "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs next start --port 3001",
"build": "next build",
"start": "next start --port 3001",
"lint": "eslint",
diff --git a/editor/playwright.config.ts b/editor/playwright.config.ts
@@ -2,6 +2,11 @@ import { defineConfig, devices } from "@playwright/test";
const PORT = Number(process.env.PORT ?? 3001);
const baseURL = `http://localhost:${PORT}`;
+const webServerCommand =
+ process.env.E2E_MODE === "start" ? "pnpm start:test" : "pnpm dev:test";
+
+const EXPORT_PORT = Number(process.env.EXPORT_PORT ?? 3000);
+const exportBaseURL = `http://localhost:${EXPORT_PORT}`;
export default defineConfig({
testDir: "./e2e",
@@ -11,12 +16,20 @@ export default defineConfig({
outputDir: "test-results/",
fullyParallel: false,
workers: 1,
- webServer: {
- command: "pnpm dev:test",
- url: baseURL,
- timeout: 120_000,
- reuseExistingServer: !process.env.CI,
- },
+ webServer: [
+ {
+ command: webServerCommand,
+ url: baseURL,
+ timeout: 120_000,
+ reuseExistingServer: !process.env.CI,
+ },
+ {
+ command: `pnpm --filter export run dev --port ${EXPORT_PORT}`,
+ url: exportBaseURL,
+ timeout: 120_000,
+ reuseExistingServer: !process.env.CI,
+ },
+ ],
use: {
baseURL,
trace: "on-first-retry",
diff --git a/export/public/subs/manifest.json b/export/public/subs/manifest.json
@@ -0,0 +1 @@
+{"version":1,"channels":[],"totalCount":0,"generatedAt":"2026-05-15T00:29:24.351Z"}
+\ No newline at end of file
diff --git a/package.json b/package.json
@@ -9,7 +9,8 @@
"build": "pnpm --filter export run build",
"start:export": "pnpm --filter export run start",
"dev:editor": "pnpm --filter editor run dev",
- "e2e": "pnpm --filter editor run e2e-dev:headless",
+ "e2e": "pnpm --filter editor run e2e",
+ "e2e:sharded": "docker build -f Dockerfile.test -t yt-dlp-transcript-browser-e2e . && node scripts/run-sharded-e2e.mjs",
"lint": "pnpm --filter export run lint"
},
"devDependencies": {
diff --git a/scripts/run-sharded-e2e.mjs b/scripts/run-sharded-e2e.mjs
@@ -0,0 +1,116 @@
+#!/usr/bin/env node
+import { spawn } from "node:child_process";
+import os from "node:os";
+import path from "node:path";
+import { mkdirSync } from "node:fs";
+import { fileURLToPath } from "node:url";
+
+const __dirname = path.dirname(fileURLToPath(import.meta.url));
+const REPO_ROOT = path.resolve(__dirname, "..");
+const EDITOR_DIR = path.join(REPO_ROOT, "editor");
+const REPORT_DIR = path.join(EDITOR_DIR, "blob-report");
+const IMAGE = process.env.IMAGE ?? "yt-dlp-transcript-browser-e2e";
+
+function parseShards() {
+ const argIdx = process.argv.indexOf("--shards");
+ if (argIdx !== -1 && process.argv[argIdx + 1]) {
+ const n = Number(process.argv[argIdx + 1]);
+ if (Number.isInteger(n) && n >= 1) return n;
+ }
+ if (process.env.SHARDS) {
+ const n = Number(process.env.SHARDS);
+ if (Number.isInteger(n) && n >= 1) return n;
+ }
+ return Math.min(Math.max(2, Math.floor(os.cpus().length / 2)), 8);
+}
+
+const SHARDS = parseShards();
+
+function run(cmd, args, opts = {}) {
+ return new Promise((resolve, reject) => {
+ const child = spawn(cmd, args, { stdio: "inherit", ...opts });
+ child.on("error", reject);
+ child.on("exit", (code) => resolve(code ?? 0));
+ });
+}
+
+function runShard(i, n) {
+ const tag = `[shard ${i}/${n}]`;
+ return new Promise((resolve) => {
+ const args = [
+ "run",
+ "--rm",
+ "--init",
+ "-e", "CI=true",
+ "-e", "E2E_MODE=start",
+ "-v", `${REPORT_DIR}:/repo/editor/blob-report`,
+ IMAGE,
+ "pnpm", "exec", "playwright", "test",
+ `--shard=${i}/${n}`,
+ "--reporter=blob",
+ ];
+ const child = spawn("docker", args);
+ const prefix = (chunk) => {
+ const lines = chunk.toString("utf8").split("\n");
+ const last = lines.pop();
+ for (const line of lines) process.stdout.write(`${tag} ${line}\n`);
+ if (last) process.stdout.write(`${tag} ${last}`);
+ };
+ child.stdout.on("data", prefix);
+ child.stderr.on("data", prefix);
+ child.on("error", (err) => {
+ process.stderr.write(`${tag} spawn error: ${err.message}\n`);
+ resolve(1);
+ });
+ child.on("exit", (code) => resolve(code ?? 1));
+ });
+}
+
+async function cleanReportDir() {
+ // Wipe blob-report via the test image so root-owned files from a prior run can be removed.
+ await run("docker", [
+ "run", "--rm",
+ "-v", `${EDITOR_DIR}:/host-editor`,
+ IMAGE,
+ "rm", "-rf", "/host-editor/blob-report",
+ ]).catch(() => {});
+ mkdirSync(REPORT_DIR, { recursive: true });
+}
+
+async function main() {
+ console.log(`Running ${SHARDS} shard(s) using image ${IMAGE}`);
+ const t0 = Date.now();
+
+ await cleanReportDir();
+
+ const shardPromises = [];
+ for (let i = 1; i <= SHARDS; i++) shardPromises.push(runShard(i, SHARDS));
+ const codes = await Promise.all(shardPromises);
+ const tShards = Date.now();
+
+ console.log("\nShard exit codes:");
+ codes.forEach((code, idx) => {
+ console.log(` shard ${idx + 1}/${SHARDS}: ${code === 0 ? "PASS" : `FAIL (${code})`}`);
+ });
+ console.log(`Sharded test wall time: ${((tShards - t0) / 1000).toFixed(1)}s`);
+
+ console.log("\nMerging blob reports...");
+ const mergeCode = await run(
+ "pnpm",
+ ["--filter", "editor", "exec", "playwright", "merge-reports", "--reporter=html", "blob-report"],
+ { cwd: REPO_ROOT },
+ );
+ if (mergeCode !== 0) {
+ console.error(`merge-reports exited ${mergeCode}`);
+ } else {
+ console.log("HTML report: editor/playwright-report/index.html");
+ }
+
+ const failed = codes.some((c) => c !== 0);
+ process.exit(failed ? 1 : 0);
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});