Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 25cabe6db14a345f9bda79fd08ddc5f1357d8cdf
parent 0e57df6e47a5270244ae1dcbb8c82167c4122296
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 13 Jun 2026 22:56:54 -0400

auto-refresh channel reports on a global debounce

Any action that changes report-relevant data (transcribe, transcode,
clean audio, availability checks, channel config edits, the direct
per-video file ops, and every download/transcription within a batch)
now marks its channel dirty and re-arms one shared debounce timer. When
activity settles, a global scheduler regenerates each dirty channel's
snapshot in parallel via the existing refresh-report job (deduped
against any refresh already running) and revalidates the affected pages.

- new common/jobs/snapshotScheduler.ts (globalThis singleton, dirty map,
  one timer, settings-driven window, registry-presence guard so a job
  abandoned across a reset can't bleed)
- central hook in runManagedFunction/runManagedCommand + per-sub-operation
  hook via ctx.recordTaskDone (incremental updates during a batch)
- explicit calls in the direct video-file actions and updateChannelAction
- pipeline inline regen removed in favor of the one uniform mechanism
- configurable window in Settings: Fast (~1s, default) / Balanced / Lazy
- reset wiring in the e2e invalidate-cache route
- new auto-report-refresh e2e spec; updated specs whose timing/job-row
  assumptions changed

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/jobs/drainStream.ts | 16++++++++++++++++
Acommon/jobs/snapshotScheduler.ts | 199+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/streamCommand.ts | 28+++++++++++++++++++++++++++-
Mcommon/lib/settings.ts | 33+++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/actionable/actions.ts | 13+------------
Meditor/app/api/test/invalidate-cache/route.ts | 4++++
Meditor/app/channels/[slug]/pipelineActions.ts | 12+++---------
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 6++++++
Meditor/app/channels/actions.ts | 4++++
Meditor/app/settings/actions.ts | 9+++++++++
Meditor/app/settings/components/SettingsForm.tsx | 20++++++++++++++++++++
Aeditor/e2e/auto-report-refresh.spec.ts | 70++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/jobs-batch-tasks-drain.spec.ts | 10+++++++++-
Meditor/e2e/sites-crud.spec.ts | 8++++++--
Meditor/e2e/skip-live.spec.ts | 19+++++++++++++++----
16 files changed, 423 insertions(+), 29 deletions(-)

diff --git a/common/jobs/drainStream.ts b/common/jobs/drainStream.ts @@ -0,0 +1,16 @@ +// Read a managed-job stream to completion, discarding chunks. Used when the +// caller doesn't render the log but needs to wait until the job's fn has +// finished (e.g. so a subsequent revalidate reads freshly-written files). +export async function drainStream( + stream: ReadableStream<string>, +): Promise<void> { + const reader = stream.getReader(); + try { + while (true) { + const { done } = await reader.read(); + if (done) return; + } + } finally { + reader.releaseLock(); + } +} diff --git a/common/jobs/snapshotScheduler.ts b/common/jobs/snapshotScheduler.ts @@ -0,0 +1,199 @@ +import type { Paths } from "../lib/paths"; +import { + DEFAULT_REPORT_DEBOUNCE_PRESET, + REPORT_DEBOUNCE_PRESETS, + getSettings, +} from "../lib/settings"; +import { + excludedDownloadIdSet, + generateChannelSnapshot, +} from "../controller/channelSnapshot"; +import { getRegistry } from "./registry"; +import { drainStream } from "./drainStream"; + +// Global, debounced "channel report" (snapshot) regeneration scheduler. +// +// Any action that changes a channel's report-relevant data calls +// `requestChannelSnapshot(paths, slug)` when it finishes. That marks the slug +// dirty and (re)arms a single shared timer. When the timer fires, every dirty +// channel's snapshot is regenerated in parallel via the existing +// `refresh-report` managed-job pattern, then the affected pages are revalidated. +// +// The debounce coalesces bursts of actions into one regen pass instead of +// rewriting snapshot.json after every individual operation. The window is read +// from global settings (REPORT_DEBOUNCE_PRESETS) at arm time. + +// Job kinds that must NOT trigger a regen. "refresh-report" is the regen itself +// — including it would make the central runManagedFunction hook re-arm the timer +// from inside the regen job, an infinite loop. "detect-duplicates" is a global +// read-only scan with no per-channel snapshot impact. +const NO_REGEN_KINDS = new Set<string>(["refresh-report", "detect-duplicates"]); + +export function shouldRequestSnapshot(kind: string): boolean { + return !NO_REGEN_KINDS.has(kind); +} + +type SchedulerState = { + // slug -> Paths to regen with. getPaths() is process-stable, so last-writer + // -wins is fine; the Map dedups repeat requests for the same channel. + dirty: Map<string, Paths>; + timer: ReturnType<typeof setTimeout> | null; + // When the current (uncleared) dirty batch started, for the max-wait cap. + firstDirtyAt: number | null; +}; + +declare global { + // eslint-disable-next-line no-var + var __yttSnapshotScheduler__: SchedulerState | undefined; +} + +function getState(): SchedulerState { + if (!globalThis.__yttSnapshotScheduler__) { + globalThis.__yttSnapshotScheduler__ = { + dirty: new Map(), + timer: null, + firstDirtyAt: null, + }; + } + return globalThis.__yttSnapshotScheduler__; +} + +function resolveWindow(): { debounceMs: number; maxWaitMs: number | null } { + try { + const preset = getSettings().reportDebouncePreset; + return REPORT_DEBOUNCE_PRESETS[preset] ?? REPORT_DEBOUNCE_PRESETS.fast; + } catch { + // getSettings reads the filesystem; fall back to the default window if it + // is unavailable for any reason. + return REPORT_DEBOUNCE_PRESETS[DEFAULT_REPORT_DEBOUNCE_PRESET]; + } +} + +// Mark a channel's report dirty and (re)arm the global debounce timer. +// Synchronous and non-throwing — safe to call from job completion handlers and +// request-scoped server actions without awaiting or try/catch. +export function requestChannelSnapshot(paths: Paths, slug: string): void { + if (!slug) return; + const state = getState(); + state.dirty.set(slug, paths); + if (state.firstDirtyAt === null) state.firstDirtyAt = Date.now(); + + const { debounceMs, maxWaitMs } = resolveWindow(); + if (state.timer) clearTimeout(state.timer); + const delay = + maxWaitMs === null + ? debounceMs + : Math.max( + 0, + Math.min(debounceMs, maxWaitMs - (Date.now() - state.firstDirtyAt)), + ); + state.timer = setTimeout(() => void fire(), delay); + // Never let a pending regen keep the process (or the e2e test runner) alive. + state.timer.unref?.(); +} + +async function fire(): Promise<void> { + const state = getState(); + // Snapshot and clear before the async work: any action that fires during + // regeneration re-marks the slug dirty and schedules a fresh pass (trailing + // edge), instead of being swallowed by this in-flight batch. + const batch = [...state.dirty.entries()]; + state.dirty.clear(); + state.timer = null; + state.firstDirtyAt = null; + if (batch.length === 0) return; + + try { + const { runManagedFunction } = await import("./streamCommand"); + + // Skip channels that already have a refresh-report job queued/running — + // a fresh snapshot is already on its way (mirrors + // refreshAllChannelSnapshotsAction's dedup). + const active = new Set( + getRegistry() + .list() + .filter( + (j) => + j.kind === "refresh-report" && + (j.status === "queued" || j.status === "running") && + j.channelSlug, + ) + .map((j) => j.channelSlug as string), + ); + + const regenerated: string[] = []; + const streams: ReadableStream<string>[] = []; + for (const [slug, paths] of batch) { + if (active.has(slug)) continue; + const result = await runManagedFunction({ + kind: "refresh-report", + // Empty queueKey: bypass queue serialization. Snapshot regen is a local + // filesystem scan, so it runs in parallel rather than waiting behind + // sync/download work (see registry.ts enqueue). + queueKey: "", + paths, + channelSlug: slug, + fn: async (onLog) => { + onLog(`Regenerating report for ${slug}…`); + const snap = await generateChannelSnapshot(paths, slug); + const excluded = excludedDownloadIdSet(snap); + const awaitingTranscription = excluded.size + ? snap.buckets.downloadedNoTranscript.filter( + (id) => !excluded.has(id), + ).length + : snap.buckets.downloadedNoTranscript.length; + onLog( + `Done. ${snap.totals.videos} videos · ` + + `${snap.undownloadedIds.length} undownloaded · ` + + `${awaitingTranscription} awaiting transcription.`, + ); + // No revalidatePath here — it's done once after all drains below. + }, + }); + if (!result.ok) continue; + regenerated.push(slug); + streams.push(result.stream); + } + + // Wait for every snapshot to finish writing before revalidating so the + // re-rendered pages read fresh counts. queueKey "" runs them in parallel, + // so this is ~the slowest snapshot, not the sum. + await Promise.all(streams.map(drainStream)); + + if (regenerated.length > 0) { + try { + // Imported lazily so non-Next consumers of common/jobs never resolve + // next/cache at module load. fire() only runs in the editor runtime. + const { revalidatePath } = await import("next/cache"); + for (const slug of regenerated) revalidatePath(`/channels/${slug}`); + revalidatePath("/channels"); + revalidatePath("/actionable"); + revalidatePath("/"); + } catch { + // revalidatePath outside a request/Next runtime — the snapshots are + // still written; the next render picks them up. + } + } + } catch { + // Backstop: a thrown fire() must never become an unhandledRejection. The + // per-channel refresh-report jobs handle their own errors/logging. + } +} + +// Fire any pending regen immediately and await it. For tests/diagnostics. +export async function flushChannelSnapshotsNow(): Promise<void> { + const state = getState(); + if (state.timer) { + clearTimeout(state.timer); + state.timer = null; + } + await fire(); +} + +// Clear the timer and drop all scheduler state. Called by the e2e cache-reset +// route so a pending regen can't fire against the next spec. +export function resetSnapshotScheduler(): void { + const existing = globalThis.__yttSnapshotScheduler__; + if (existing?.timer) clearTimeout(existing.timer); + globalThis.__yttSnapshotScheduler__ = undefined; +} diff --git a/common/jobs/streamCommand.ts b/common/jobs/streamCommand.ts @@ -11,6 +11,23 @@ import { } from "./registry"; import type { Paths } from "../lib/paths"; import { makeSafeController } from "../lib/safeStreamController"; +import { + requestChannelSnapshot, + shouldRequestSnapshot, +} from "./snapshotScheduler"; + +// Mark a job's channel report dirty so the debounced scheduler regenerates the +// snapshot — called both on each completed sub-operation and on the job's +// terminal state. Excludes the regen job kind itself (and other read-only +// kinds) to avoid an infinite loop. Skips jobs whose record is no longer in the +// registry: that means the registry was reset out from under a still-running +// job (e.g. the e2e cache-reset between specs), and a wiped job must not arm a +// regen that would bleed into unrelated work. Synchronous and non-throwing. +function requestSnapshotOnFinish(jobId: string, opts: CommonOpts): void { + if (!opts.channelSlug || !shouldRequestSnapshot(opts.kind)) return; + if (!getRegistry().get(jobId)) return; + requestChannelSnapshot(opts.paths, opts.channelSlug); +} export type StreamActionResult = | { ok: true; jobId: string; stream: ReadableStream<string> } @@ -162,6 +179,7 @@ export async function runManagedCommand( .finally(() => { fileStream?.end(); safe.safeClose(); + requestSnapshotOnFinish(id, opts); }); }; @@ -233,7 +251,14 @@ export async function runManagedFunction( addTask: (task) => registry.addTask(id, task), updateTask: (taskId, patch) => registry.updateTask(id, taskId, patch), removeTask: (taskId) => registry.removeTask(id, taskId), - recordTaskDone: (ms) => registry.recordTaskDuration(id, ms), + recordTaskDone: (ms) => { + registry.recordTaskDuration(id, ms); + // Refresh the report after EACH completed sub-operation (each video + // downloaded/transcribed in a batch), not only when the whole batch + // finishes — so a long batch updates incrementally. The global debounce + // coalesces sub-operations that finish close together. + requestSnapshotOnFinish(id, opts); + }, }; opts @@ -256,6 +281,7 @@ export async function runManagedFunction( .finally(() => { fileStream?.end(); safe.safeClose(); + requestSnapshotOnFinish(id, opts); }); }; diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -62,6 +62,10 @@ export type SiteSettings = { // video once the stream ends. Per-channel override available // (ChannelConfig.skipLiveDownloads). skipLiveDownloads: boolean; + // Debounce preset for the global snapshot scheduler: how long it waits after + // the last report-changing action before regenerating affected channel + // reports. See REPORT_DEBOUNCE_PRESETS. Default "fast" (~1s, no cap). + reportDebouncePreset: ReportDebouncePreset; // Default social links applied to every site that doesn't define its own. // A site inherits these unless its site.json carries an explicit // `socialLinks` array — see Site.socialLinks / resolveSocialLinks in @@ -82,6 +86,28 @@ export const SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS = 10; export const PARALLEL_TRANSCRIPTIONS_MAX = 16; export const PARALLEL_TRANSCRIPTIONS_DEFAULT = 2; +// Global snapshot-scheduler debounce presets. `debounceMs` is the quiet-period +// window after the last report-changing action; `maxWaitMs` caps the total +// delay under continuous activity (null = no cap, fire purely on the quiet +// period). Consumed by common/jobs/snapshotScheduler.ts and surfaced in the +// Settings form. +export type ReportDebouncePreset = "fast" | "balanced" | "lazy"; + +export const REPORT_DEBOUNCE_PRESETS: Record< + ReportDebouncePreset, + { debounceMs: number; maxWaitMs: number | null } +> = { + fast: { debounceMs: 1000, maxWaitMs: null }, + balanced: { debounceMs: 3000, maxWaitMs: 30000 }, + lazy: { debounceMs: 10000, maxWaitMs: 60000 }, +}; + +export const DEFAULT_REPORT_DEBOUNCE_PRESET: ReportDebouncePreset = "fast"; + +export function isReportDebouncePreset(v: unknown): v is ReportDebouncePreset { + return v === "fast" || v === "balanced" || v === "lazy"; +} + export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; @@ -99,6 +125,7 @@ function defaults(): SiteSettings { parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback: false, skipLiveDownloads: true, + reportDebouncePreset: DEFAULT_REPORT_DEBOUNCE_PRESET, socialLinks: [], }; } @@ -225,6 +252,9 @@ export function getSettings(): SiteSettings { if (typeof merged.skipLiveDownloads !== "boolean") { merged.skipLiveDownloads = true; } + if (!isReportDebouncePreset(merged.reportDebouncePreset)) { + merged.reportDebouncePreset = DEFAULT_REPORT_DEBOUNCE_PRESET; + } merged.socialLinks = parseSocialLinks(merged.socialLinks); return merged; } @@ -365,6 +395,9 @@ export async function writeSettings(next: SiteSettings): Promise<void> { ), inlineTranscribeOnFallback: next.inlineTranscribeOnFallback === true, skipLiveDownloads: next.skipLiveDownloads !== false, + reportDebouncePreset: isReportDebouncePreset(next.reportDebouncePreset) + ? next.reportDebouncePreset + : DEFAULT_REPORT_DEBOUNCE_PRESET, socialLinks, }; const tmp = `${file}.tmp-${process.pid}`; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Channel reports refresh themselves automatically, on a global debounce.** Any action that changes what a channel report (the per-channel snapshot powering the Channels list, `/actionable`, and the channel page) would say now regenerates that report on its own when it finishes — no more manually clicking **Refresh report** after transcribing, transcoding, cleaning audio, checking availability, editing channel config, or the per-video file operations (delete file, set primary transcript, delete dir, mark untranscribable, archive/do-not-clean). Every report-changing action marks its channel "dirty" and re-arms one shared debounce timer; when activity settles, the scheduler regenerates each dirty channel's snapshot in parallel (reusing the existing `refresh-report` job, deduped against any refresh already running) and revalidates the affected pages. This happens **after each completed sub-operation within a batch**, not only when the whole batch finishes — so a long download or transcription run updates its report incrementally as each video lands, rather than staying stale until the end. The debounce coalesces bursts — videos that finish within the same window collapse into a single regen pass instead of one rewrite per operation. The window is configurable in **Settings → Report refresh debounce**: **Fast** (~1s after activity settles, no cap — the default), **Balanced** (~3s, 30s max), or **Lazy** (~10s, 60s max). The download/sync pipeline's previous behavior of regenerating its report inline is removed in favor of this one uniform mechanism (so its report now lags by the debounce window — ~1s by default — rather than being written synchronously). The manual **Refresh report** / **Update all reports** buttons are unchanged. - **Parakeet transcriptions show a per-video ETA.** While a video is being transcribed with the parakeet app, its per-task progress bar on `/jobs/active` now shows an estimate of how long that single video has left (e.g. `segment 3/12 · ETA 6:10`), alongside the existing segment count and percent. The wrapper (`scripts/parakeet-stitch.mjs`) measures each segment's real transcription wall-time and projects the remaining time as **average time per completed segment × remaining segments**, emitting it on its progress lines; the parser surfaces that ETA in the task detail. It appears from the second segment onward (the first segment has no average to project from yet). This is distinct from the batch-level "~4:30 left" estimate across all videos. - **Each active sub-operation shows how long it has been running.** Every per-task bar on `/jobs/active` (each in-flight download or transcription) now leads with a live `m:ss` timer of its own elapsed wall-time, ticking once a second — e.g. `1:42 · 25% · segment 3/12 · ETA 6:10`. The start time comes from the job registry's per-task `startedAt`, so the timer survives page reloads and reflects the real runtime, not time-since-open. - **The single-video download pipeline reconstructs the source URL when a video has no `metadata.info.json`.** Running the per-video **Download** / **Audio + Whisper** action resolves the video's URL by reading its `metadata.info.json` (`webpage_url`), then by matching the canonical id against the channel's stored `playlist`. When a video directory exists but has neither — e.g. a manually-placed video, or a channel whose playlist was never stored — the pipeline previously gave up with "Could not determine the video URL". It now re-creates the URL from the video's canonical id (its directory name) and the channel's platform (from `config.platform`, falling back to detecting it from `config.url`), so the download proceeds. Reconstruction is a last resort — the exact URL from metadata or the playlist always wins when present — and only fires when the platform is known (no blind guess); note a reconstructed Odysee URL drops the channel-name prefix, so it may not resolve. diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts @@ -9,6 +9,7 @@ import { } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand"; +import { drainStream } from "yt-dlp-transcript-common/jobs/drainStream"; import { detectDuplicateShorts } from "yt-dlp-transcript-common/controller/duplicateShorts"; export type RefreshAllResult = { @@ -22,18 +23,6 @@ export type RunDuplicateDetectionResult = | { ok: true; clusters: number; videosInClusters: number } | { ok: false; error: string }; -async function drainStream(stream: ReadableStream<string>): Promise<void> { - const reader = stream.getReader(); - try { - while (true) { - const { done } = await reader.read(); - if (done) return; - } - } finally { - reader.releaseLock(); - } -} - export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResult> { const paths = getPaths(); const channels = await listChannels(paths); diff --git a/editor/app/api/test/invalidate-cache/route.ts b/editor/app/api/test/invalidate-cache/route.ts @@ -1,5 +1,6 @@ import { NextResponse } from "next/server"; import { revalidatePath } from "next/cache"; +import { resetSnapshotScheduler } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; export const dynamic = "force-dynamic"; @@ -14,6 +15,9 @@ export async function GET() { } function invalidate() { + // Clear any pending debounced snapshot regen FIRST (also clears its timer) so + // it can't fire against the about-to-be-reset registry mid-spec. + resetSnapshotScheduler(); // Reset the in-memory job registry too so tests don't observe stale jobs // from a prior spec in the same dev-server lifetime. // eslint-disable-next-line @typescript-eslint/no-explicit-any diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -15,7 +15,6 @@ import { readChannelConfig, readChannelStat, } from "yt-dlp-transcript-common/controller/channels"; -import { generateChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; @@ -107,14 +106,9 @@ async function runPipelineAction( setProgress, progressBaseline, }); - onLog("Regenerating channel report…"); - try { - await generateChannelSnapshot(paths, slug); - } catch (e) { - // A stale report is preferable to making a successful pipeline run - // look failed; surface the failure in the job log instead. - onLog(`[warn] report regen failed: ${(e as Error).message}`); - } + // The channel report (snapshot) is regenerated automatically after this + // job finishes, via the global debounced scheduler hooked into + // runManagedFunction's completion. See common/jobs/snapshotScheduler.ts. revalidatePath(`/channels/${slug}`); revalidatePath("/channels"); revalidatePath("/actionable"); diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -32,6 +32,7 @@ import { runManagedFunction, type StreamActionResult, } from "yt-dlp-transcript-common/jobs/streamCommand"; +import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks"; function videoQueueKey(config: ChannelConfig, override: string | undefined): string { @@ -275,6 +276,7 @@ export async function deleteVideoFileAction( } await rm(target, { force: true }); revalidatePath(`/channels/${slug}/videos/${videoId}`); + requestChannelSnapshot(getPaths(), slug); return { ok: true }; } @@ -317,6 +319,7 @@ export async function setPrimaryTranscriptAction( // refresh on the next snapshot regeneration (same as the other video actions). revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); + requestChannelSnapshot(getPaths(), slug); return { ok: true }; } @@ -343,6 +346,7 @@ export async function deleteVideoDirAction( } await rm(resolved, { recursive: true, force: true }); revalidatePath(`/channels/${slug}`); + requestChannelSnapshot(getPaths(), slug); redirect(`/channels/${slug}`); } @@ -382,6 +386,7 @@ export async function markVideoUntranscribableAction( await pruneFailedTranscriptions(failureListFile, new Set([videoId])); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); + requestChannelSnapshot(getPaths(), slug); return { ok: true }; } @@ -402,5 +407,6 @@ export async function toggleDoNotCleanAction( await setDoNotClean(videoDir, enabled); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); + requestChannelSnapshot(getPaths(), slug); return { ok: true }; } diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts @@ -13,6 +13,7 @@ import { writeChannelConfig, } from "yt-dlp-transcript-common/controller/channels"; import { generateChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; +import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { getSite, isValidSiteId, @@ -94,6 +95,9 @@ export async function updateChannelAction( for (const key of CHANNEL_FORM_FIELDS) delete merged[key]; Object.assign(merged, parsed.config); await writeChannelConfig(paths, slug, merged); + // Config changes (e.g. audioFormat / handling) feed snapshot buckets, so + // refresh the report through the global debounced scheduler. + requestChannelSnapshot(paths, slug); revalidatePath("/channels"); revalidatePath(`/channels/${slug}`); return undefined; diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -2,6 +2,8 @@ import { revalidatePath } from "next/cache"; import { + DEFAULT_REPORT_DEBOUNCE_PRESET, + isReportDebouncePreset, normalizeSocialSvg, parseSocialLinks, PARALLEL_TRANSCRIPTIONS_MAX, @@ -38,6 +40,12 @@ export async function saveSettingsAction( const inlineTranscribeOnFallback = formData.get("inlineTranscribeOnFallback") === "on"; const skipLiveDownloads = formData.get("skipLiveDownloads") === "on"; + const reportDebouncePresetRaw = String( + formData.get("reportDebouncePreset") ?? "", + ).trim(); + const reportDebouncePreset = isReportDebouncePreset(reportDebouncePresetRaw) + ? reportDebouncePresetRaw + : DEFAULT_REPORT_DEBOUNCE_PRESET; if (!adminTitle) return { ok: false, error: "Admin title is required" }; @@ -178,6 +186,7 @@ export async function saveSettingsAction( parallelTranscriptions: parallelParsed, inlineTranscribeOnFallback, skipLiveDownloads, + reportDebouncePreset, socialLinks, }; await writeSettings(next); diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -198,6 +198,26 @@ export function SettingsForm({ initial, apps }: Props) { </span> </span> </label> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Report refresh debounce</span> + <select + name="reportDebouncePreset" + defaultValue={initial.reportDebouncePreset} + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" + > + <option value="fast">Fast — ~1s after activity, no cap</option> + <option value="balanced"> + Balanced — ~3s after activity, 30s cap + </option> + <option value="lazy">Lazy — ~10s after activity, 60s cap</option> + </select> + <span className="text-xs text-zinc-500"> + After any action that changes a channel report (download, transcribe, + cleanup, file edits…), the report is regenerated automatically. This + controls how long the shared scheduler waits for activity to settle + before regenerating, so bursts of actions collapse into one pass. + </span> + </label> <fieldset className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded p-3"> <legend className="px-1 text-sm font-medium">Social links</legend> <p className="text-xs text-zinc-500"> diff --git a/editor/e2e/auto-report-refresh.spec.ts b/editor/e2e/auto-report-refresh.spec.ts @@ -0,0 +1,70 @@ +import { writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { readJson, resetData, resolvePath } from "./helpers"; + +// Verifies the global debounced snapshot scheduler: a direct (non-managed) +// action that changes report-relevant data regenerates the channel report on +// its own, without visiting the channel page or clicking "Refresh report". + +const SLUG = "test-transcribe"; +const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`; + +function dataRel(videoId: string, file: string): string { + return `test-transcripts/channels/${SLUG}/data/${videoId}/${file}`; +} + +async function seedTranscript(videoId: string): Promise<void> { + await writeFile( + resolvePath(dataRel(videoId, "transcript.json")), + '{"transcription":[]}\n', + ); +} + +type Snap = { + generatedAt: string; + buckets: { transcribedWithAudio?: string[] }; +}; + +test("a direct action auto-refreshes the channel report via the debounce", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedTranscript("vidA"); + await seedTranscript("vidB"); + await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {}); + + // Visiting the channel page generates the initial snapshot from disk: both + // transcribed-with-audio videos are cleanable. + await page.goto(`/channels/${SLUG}`); + const before = await readJson<Snap>(SNAPSHOT_REL); + expect(before.buckets.transcribedWithAudio).toContain("vidA"); + expect(before.buckets.transcribedWithAudio).toContain("vidB"); + + // Mark vidA "do not clean" from its video page. This is a direct action that + // does NOT regenerate the channel snapshot itself — only the debounced + // scheduler will. Stay off the channel page afterward so a page-load regen + // can't mask the scheduler. + await page.goto(`/channels/${SLUG}/videos/vidA`); + await page + .getByRole("button", { name: "mark video vidA do not clean" }) + .click(); + await expect(page.getByLabel("media archived")).toBeVisible(); + + // Within the Fast debounce window (~1s) the report regenerates on its own: + // vidA drops out of transcribedWithAudio while vidB stays, and generatedAt + // advances — all without ever revisiting the channel page. + await expect + .poll( + async () => { + const snap = await readJson<Snap>(SNAPSHOT_REL).catch(() => null); + if (!snap) return null; + return { + regenerated: snap.generatedAt !== before.generatedAt, + hasA: snap.buckets.transcribedWithAudio?.includes("vidA") ?? false, + hasB: snap.buckets.transcribedWithAudio?.includes("vidB") ?? false, + }; + }, + { timeout: 15_000, intervals: [200, 300, 500] }, + ) + .toEqual({ regenerated: true, hasA: false, hasB: true }); +}); diff --git a/editor/e2e/jobs-batch-tasks-drain.spec.ts b/editor/e2e/jobs-batch-tasks-drain.spec.ts @@ -196,8 +196,16 @@ test("hard Cancel during a drain ends the job cancelled without stream errors", await section.getByRole("button", { name: /^Cancel$/ }).click(); // The job ends cancelled (hard cancel wins over the in-progress drain). + // Target the whisper-all row specifically: finishing the batch now also + // queues an automatic `refresh-report` job for the same channel (the global + // debounced report refresh), which would otherwise be matched by a bare + // slug filter. await page.goto("/jobs"); - const row = page.getByRole("row").filter({ hasText: "cancel-drain" }).first(); + const row = page + .getByRole("row") + .filter({ hasText: "cancel-drain" }) + .filter({ hasText: "whisper-all" }) + .first(); await expect(row).toContainText("cancelled", { timeout: 20_000 }); await page.waitForTimeout(1_000); diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts @@ -53,8 +53,12 @@ test("list shows configured sites and edit updates branding", async ({ // site selector now, so a bare getByText would match two elements). await expect(page.getByRole("link", { name: /Alpha Site/ })).toBeVisible(); - await page.getByRole("link", { name: /Alpha Site/ }).click(); - await expect(page).toHaveURL(/\/sites\/alpha/); + // Drive the navigation with waitForURL alongside the click so a re-render + // between the click and a separate URL assertion can't drop it under load. + await Promise.all([ + page.waitForURL(/\/sites\/alpha/), + page.getByRole("link", { name: /Alpha Site/ }).click(), + ]); await page.getByLabel(/site title/i).fill("Alpha Renamed"); await page.getByRole("button", { name: /save site/i }).click(); await expect( diff --git a/editor/e2e/skip-live.spec.ts b/editor/e2e/skip-live.spec.ts @@ -100,11 +100,22 @@ test("skip-live on by default: live + upcoming skipped, VOD + normal download", /load-info-json:.*data\/normalvid001\/metadata\.info\.json/, ); - // Snapshot surfaces the skipped-live videos in their own bucket. + // Snapshot surfaces the skipped-live videos in their own bucket. The report + // is regenerated automatically after the download job finishes, via the + // global debounced scheduler, so poll until it lands rather than reading the + // file the instant the job completes. + await expect + .poll( + async () => { + const snap = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch( + () => null, + ); + return snap?.buckets.skippedByFilter ?? null; + }, + { timeout: 15_000, intervals: [200, 300, 500] }, + ) + .toEqual(expect.arrayContaining(["islive000001", "isupcoming01"])); const snapshot = await readJson<Snapshot>(`${ROOT}/snapshot.json`); - expect(snapshot.buckets.skippedByFilter ?? []).toEqual( - expect.arrayContaining(["islive000001", "isupcoming01"]), - ); expect(snapshot.buckets.skippedByFilter ?? []).not.toContain("waslive00001"); });