commit 92baa36653e06e563194774cfefb8ea02767ca5b parent 68a4413c6d5061d54cb824ad03b69ee13bd2b486 Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Fri, 28 Aug 2026 02:53:47 -0400 editor: the channel-work table and its census leave the /actionable route Seams only. /actionable renders from the shared module and its aria output is the same; nothing else about the page changes. The `SectionConfig` type and eight of its ten configs move to `app/components/channelWork/sections.tsx`. The type gains three fields that say who owns each section: `operation` (the operation page that renders it, or null for the two cleanup sections), `role` ("work" is what the runner will do, "attention" is what a human has to look at first) and `getRows`, so no page hand-pairs a config with a summary list again — that pairing was the same hand-exhaustiveness as the page's own "nothing pending" gate. `sectionsFor` returns work then attention in declaration order, and [] for an operation with no sections of its own. `channelWorkSections` is a function, not a constant, because `headerAction` is an element. `speakers` and `stale-reports` stay page-local: they die with the page. `Section` and `Row` become `ChannelWorkTable` — a server component, because `primaryAction` is a function and cannot cross to a client component. Its header comment records the aria contract four specs locate by, and the one rule the runner pages impose: never `role="status"` in here. Two branches go with `stale-reports`, which is not in the module: the "stale"/"missing" count cell and the guard that suppressed the row's own refresh button. `Row`'s private `isStale` was a copy of the loader's `isStaleOrMissing` and is now a call to it. The loader splits. `duplicates`, `duplicateOverrides`, `mediaScan` and `mediaScanOverrides` had exactly one reader — the review half of /actionable — and every dashboard render and every widget poll was paying the 6.7 MB duplicates parse to carry them. They become `ReviewSummary` / `loadReviewSummary` in `app/lib/review/loadReview.ts`, field comments intact. requestCache's note about who wants the report says /review now. Moves, imports repointed and nothing else edited: actionable/lib/loadActionable.ts -> app/lib/actionable/loadActionable.ts actionable/actions.ts -> app/lib/actionable/actions.ts actionable/components/InlineActionButton.tsx -> app/components/actions/InlineActionButton.tsx actionable/components/FixAllIncompleteButton.tsx -> app/components/channelWork/FixAllIncompleteButton.tsx The other five components stay under `actionable/components/` for now, each repointed at the moved actions module; they move in the commits that give them a home. `sections.test.ts` pins the ids, that every `getRows` returns a distinct field of the summary rather than a fresh array, and the four groupings. It assigns a global `React` first: the module has JSX, tsconfig says `jsx: "preserve"` for Next, and the unit runner's esbuild can only emit classic `React.createElement` — the hack lives in the test rather than as an import Next does not need. Nothing under transcripts/ was read or written for this commit. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Diffstat:
26 files changed, 1330 insertions(+), 1149 deletions(-)
diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts @@ -1,320 +0,0 @@ -"use server"; - -import { revalidatePath } from "next/cache"; -import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels"; -import { - excludedDownloadIdSet, - generateChannelSnapshot, -} from "yt-dlp-transcript-common/controller/channelSnapshot"; -import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; -import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand"; -import { drainStream } from "yt-dlp-transcript-common/jobs/drainStream"; -import { - detectDuplicateShorts, - readDuplicateReport, - updateDuplicateOverride, -} from "yt-dlp-transcript-common/controller/duplicateShorts"; -import { - scanCorruptMedia, - updateMediaScanOverride, -} from "yt-dlp-transcript-common/controller/scanCorruptMedia"; -import { mediaScanTotals } from "yt-dlp-transcript-common/lib/mediaScan"; -import { shareClusterFromCanonical } from "yt-dlp-transcript-common/controller/digestSharing"; -import { - clearIncompleteTranscriptsAction, - enableAutoRunners, - redownloadIncompleteBucketAction, -} from "../channels/[slug]/incompleteTranscriptActions"; -import { loadActionableSummary } from "./lib/loadActionable"; - -export type RefreshAllResult = { - queued: string[]; - skipped: { slug: string; reason: string }[]; -}; - -export type DuplicateScope = "shorts" | "all"; - -export type RunDuplicateDetectionResult = - | { ok: true; clusters: number; videosInClusters: number } - | { ok: false; error: string }; - -export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResult> { - const paths = getPaths(); - const channels = await listChannelConfigs(paths); - const active = new Set( - getRegistry() - .list() - .filter( - (j) => - j.kind === "refresh-report" && - (j.status === "queued" || j.status === "running") && - j.channelSlug, - ) - .map((j) => j.channelSlug as string), - ); - const queued: string[] = []; - const skipped: { slug: string; reason: string }[] = []; - const streams: ReadableStream<string>[] = []; - for (const c of channels) { - if (active.has(c.slug)) { - skipped.push({ slug: c.slug, reason: "already running" }); - continue; - } - const result = await runManagedFunction({ - kind: "refresh-report", - // Empty queueKey: bypass queue serialization. Snapshot regen is a - // local filesystem scan that never touches the platform, so there's - // no reason for it to wait behind sync/download work. See - // registry.ts:69-72 for the documented escape hatch. - queueKey: "", - paths, - channelSlug: c.slug, - fn: async (onLog) => { - onLog(`Regenerating report for ${c.slug}…`); - const snap = await generateChannelSnapshot(paths, c.slug); - const excluded = excludedDownloadIdSet(snap); - const awaitingTranscription = excluded.size - ? snap.buckets.downloadedNoTranscript.filter( - (id) => !excluded.has(id), - ).length - : snap.buckets.downloadedNoTranscript.length; - onLog( - `Done. ${snap.totals.videos} videos · ` + - `${snap.undownloadedIds.length} undownloaded · ` + - `${awaitingTranscription} awaiting transcription.`, - ); - // Deliberately no revalidatePath here — calling it from a - // background fn races with the in-flight re-render of /actionable - // that the action's own revalidatePath triggers. The action's - // single revalidate at the end picks up every fresh snapshot. - }, - }); - if (!result.ok) { - skipped.push({ slug: c.slug, reason: result.error }); - continue; - } - queued.push(c.slug); - streams.push(result.stream); - } - // Wait for all snapshots to finish writing before revalidating so the - // re-rendered /actionable reads fresh counts. With queueKey === "" the - // jobs all run in parallel, so this waits roughly the time of the - // slowest snapshot, not the sum. - await Promise.all(streams.map(drainStream)); - revalidatePath("/actionable"); - revalidatePath("/"); - return { queued, skipped }; -} - -// Runs the global cross-platform duplicate-shorts pass. `scope: "all"` removes -// the duration cutoff (a heavier one-off run that also surfaces clip-of-longer -// containment). Reads the cues + statsByPath written by build:index/build:stats, -// so a build must have run first for meaningful results. -export async function runDuplicateDetectionAction( - scope: DuplicateScope = "shorts", -): Promise<RunDuplicateDetectionResult> { - const paths = getPaths(); - let clusters = 0; - let videosInClusters = 0; - const result = await runManagedFunction({ - kind: "detect-duplicates", - // Empty queueKey: a local read-only scan, no reason to wait behind - // sync/download work (see refreshAllChannelSnapshotsAction). - queueKey: "", - paths, - fn: async (onLog, signal) => { - const report = await detectDuplicateShorts({ - paths, - thresholdSeconds: scope === "all" ? null : undefined, - onLog, - signal, - }); - clusters = report.totals.clusters; - videosInClusters = report.totals.videosInClusters; - }, - }); - if (!result.ok) return { ok: false, error: result.error }; - await drainStream(result.stream); - revalidatePath("/actionable"); - return { ok: true, clusters, videosInClusters }; -} - -// What a human can say about a cluster the detector could not decide. -// confirmed — "I looked; these really are the same video." Unblocks -// sharing AND publication for a title+duration suspect. -// not-duplicate — "They are not." Suppresses the cluster entirely. -// clear — undo, back to awaiting review. -export type DuplicateClusterDecision = "confirmed" | "not-duplicate" | "clear"; - -export type ReviewDuplicateClusterResult = - | { ok: true; shared: number; misaligned: number } - | { ok: false; error: string }; - -// Record a review decision for one cluster. -// -// `updateDuplicateOverride` has existed — with the `confirmed` flag, the -// read-modify-write, the atomic rename and the "an empty patch clears the -// decision" rule — since duplicate review was built, and until now **nothing in -// the repo called it**. `clusterMaySharePartial` fails closed, so every -// needsReview cluster shared nothing and there was no way for a human to change -// that. This is that missing caller. -// -// Confirming also attempts the share immediately, via the equally-uncalled -// `shareClusterFromCanonical`: the point of confirming is to let derived work -// flow, and making the operator wait for the canonical member's next sweep to -// find out whether it would have would make the button feel inert. It is a -// no-op when the canonical has no digest yet, which today is almost always. -export async function reviewDuplicateClusterAction( - clusterId: string, - decision: DuplicateClusterDecision, -): Promise<ReviewDuplicateClusterResult> { - const paths = getPaths(); - try { - const overrides = await updateDuplicateOverride( - paths, - clusterId, - decision === "confirmed" - ? { confirmed: true, notDuplicate: false } - : decision === "not-duplicate" - ? { notDuplicate: true, confirmed: false } - : { confirmed: false, notDuplicate: false }, - ); - - let shared = 0; - let misaligned = 0; - if (decision === "confirmed") { - const report = await readDuplicateReport(paths); - const cluster = report?.clusters.find((c) => c.clusterId === clusterId); - if (cluster) { - for (const outcome of await shareClusterFromCanonical(paths, cluster, { - overrides, - })) { - if (outcome.status === "shared") shared++; - else if (outcome.status === "misaligned") misaligned++; - } - } - } - - revalidatePath("/actionable"); - return { ok: true, shared, misaligned }; - } catch (e) { - return { ok: false, error: (e as Error)?.message ?? String(e) }; - } -} - -export type GlobalIncompleteResult = - | { ok: true; channels: number; affected: number } - | { ok: false; error: string }; - -// All channels with at least one truncated transcript right now. -async function affectedIncompleteSlugs(): Promise<string[]> { - const summary = await loadActionableSummary(getPaths()); - return summary.incompleteTranscripts.map((r) => r.channel.slug); -} - -// Clear & re-queue every truncated transcript across all channels, then enable -// the auto-runners once. Destructive — the caller confirms first. -export async function clearAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> { - const slugs = await affectedIncompleteSlugs(); - if (slugs.length === 0) { - return { ok: false, error: "No incomplete transcripts to clear." }; - } - let cleared = 0; - for (const slug of slugs) { - // Defer enabling the runners until the end so settings flip only once. - const r = await clearIncompleteTranscriptsAction(slug, undefined, { - enableRunners: false, - }); - cleared += r.succeeded; - } - await enableAutoRunners(); - revalidatePath("/actionable"); - return { ok: true, channels: slugs.length, affected: cleared }; -} - -// Queue one batch re-fix job per affected channel (fire-and-forget — the jobs -// keep running and surface in /jobs; cancel each stream so we don't hold them -// open). -export async function redownloadAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> { - const slugs = await affectedIncompleteSlugs(); - if (slugs.length === 0) { - return { ok: false, error: "No incomplete transcripts to fix." }; - } - let queued = 0; - for (const slug of slugs) { - const result = await redownloadIncompleteBucketAction(slug); - if (result.ok) { - void result.stream.cancel(); - queued++; - } - } - revalidatePath("/actionable"); - return { ok: true, channels: slugs.length, affected: queued }; -} - - -// --------------------------------------------------------------------------- -// Corrupt-media scan -// --------------------------------------------------------------------------- - -export type RunMediaScanResult = - | { ok: true; findings: number; filesScanned: number } - | { ok: false; error: string }; - -// Scan media already on disk. Reports; deletes nothing — deletion stays the -// separate, explicit per-file click that already exists on the video page. -// -// `deepProbe` turns on the tier-2 full decode for files whose duration looks -// wrong. Off by default because that tier is a real transcode per file: it has -// to be budgeted like the digest sweep, not like a stat() walk. -export async function runMediaScanAction( - opts: { channels?: string[]; deepProbe?: boolean } = {}, -): Promise<RunMediaScanResult> { - const paths = getPaths(); - let findings = 0; - let filesScanned = 0; - const perChannel = (opts.channels?.length ?? 0) > 0; - const result = await runManagedFunction({ - kind: perChannel ? "scan-media-channel" : "scan-media", - // Empty queueKey: local read-only disk work, no reason to wait behind - // sync/download work (see runDuplicateDetectionAction). - queueKey: "", - paths, - fn: async (onLog, signal) => { - const report = await scanCorruptMedia({ - paths, - channels: opts.channels, - deepProbe: opts.deepProbe, - onLog, - signal, - }); - findings = report.findings.length; - filesScanned = mediaScanTotals(report).filesScanned; - }, - }); - if (!result.ok) return { ok: false, error: result.error }; - await drainStream(result.stream); - revalidatePath("/actionable"); - return { ok: true, findings, filesScanned }; -} - -export type ReviewMediaFindingResult = - | { ok: true } - | { ok: false; error: string }; - -// "Looked at it; it is fine." Recorded in a sibling overrides file, because the -// report is regenerated wholesale by every scan and a decision written into it -// would be destroyed by the next run. -export async function reviewMediaFindingAction( - key: string, - reviewed: boolean, -): Promise<ReviewMediaFindingResult> { - try { - await updateMediaScanOverride(getPaths(), key, { reviewed }); - revalidatePath("/actionable"); - return { ok: true }; - } catch (e) { - return { ok: false, error: (e as Error).message }; - } -} diff --git a/editor/app/actionable/components/DuplicateClusterReview.tsx b/editor/app/actionable/components/DuplicateClusterReview.tsx @@ -5,7 +5,7 @@ import { reviewDuplicateClusterAction, type DuplicateClusterDecision, type ReviewDuplicateClusterResult, -} from "../actions"; +} from "../../lib/actionable/actions"; import type { DuplicateClusterOverride } from "yt-dlp-transcript-common/lib/duplicates"; type Status = diff --git a/editor/app/actionable/components/FixAllIncompleteButton.tsx b/editor/app/actionable/components/FixAllIncompleteButton.tsx @@ -1,101 +0,0 @@ -"use client"; - -import Link from "next/link"; -import { useState } from "react"; -import { - clearAllIncompleteTranscriptsAction, - redownloadAllIncompleteTranscriptsAction, - type GlobalIncompleteResult, -} from "../actions"; - -type Mode = "clear" | "redownload"; - -type Status = - | { kind: "idle" } - | { kind: "running"; mode: Mode } - | { kind: "done"; mode: Mode; result: GlobalIncompleteResult } - | { kind: "error"; message: string }; - -export function FixAllIncompleteButton() { - const [status, setStatus] = useState<Status>({ kind: "idle" }); - - async function run(mode: Mode) { - if ( - mode === "clear" && - !window.confirm( - "Delete the truncated audio + transcript for every flagged video across ALL channels and enable the auto-download/transcribe runners? The current partial transcripts are removed and re-fetched. This cannot be undone.", - ) - ) { - return; - } - setStatus({ kind: "running", mode }); - try { - const result = - mode === "clear" - ? await clearAllIncompleteTranscriptsAction() - : await redownloadAllIncompleteTranscriptsAction(); - setStatus({ kind: "done", mode, result }); - } catch (e) { - setStatus({ kind: "error", message: (e as Error).message }); - } - } - - const running = status.kind === "running"; - return ( - <div className="flex items-center gap-2 flex-wrap"> - <button - type="button" - onClick={() => run("redownload")} - disabled={running} - aria-label="re-download all incomplete transcripts" - className="px-2.5 py-1 rounded-md bg-primary text-primary-foreground text-xs font-medium hover:opacity-90 disabled:opacity-50 whitespace-nowrap" - > - {status.kind === "running" && status.mode === "redownload" - ? "Queuing…" - : "Re-fix all"} - </button> - <button - type="button" - onClick={() => run("clear")} - disabled={running} - aria-label="clear all incomplete transcripts" - className="px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted disabled:opacity-50 whitespace-nowrap" - > - {status.kind === "running" && status.mode === "clear" - ? "Clearing…" - : "Clear & re-queue all"} - </button> - {status.kind === "done" && ( - <span - aria-label="fix all incomplete transcripts result" - className="text-xs text-muted-foreground" - > - {status.result.ok ? ( - <> - {status.mode === "clear" ? "Cleared" : "Queued"}{" "} - {status.result.affected} across {status.result.channels} channel - {status.result.channels === 1 ? "" : "s"} ·{" "} - <Link - href="/jobs" - className="underline hover:text-foreground" - > - view jobs - </Link> - </> - ) : ( - status.result.error - )} - </span> - )} - {status.kind === "error" && ( - <span - role="alert" - aria-label="fix all incomplete transcripts error" - className="text-xs text-destructive" - > - {status.message} - </span> - )} - </div> - ); -} diff --git a/editor/app/actionable/components/MediaScanFindingRow.tsx b/editor/app/actionable/components/MediaScanFindingRow.tsx @@ -7,7 +7,7 @@ import { mediaScanKey, type MediaScanFinding, } from "yt-dlp-transcript-common/lib/mediaScan"; -import { reviewMediaFindingAction } from "../actions"; +import { reviewMediaFindingAction } from "../../lib/actionable/actions"; const VERDICT_LABEL: Record<MediaScanFinding["verdict"], string> = { ok: "ok", diff --git a/editor/app/actionable/components/RefreshAllReportsButton.tsx b/editor/app/actionable/components/RefreshAllReportsButton.tsx @@ -5,7 +5,7 @@ import { useState } from "react"; import { refreshAllChannelSnapshotsAction, type RefreshAllResult, -} from "../actions"; +} from "../../lib/actionable/actions"; type Status = | { kind: "idle" } diff --git a/editor/app/actionable/components/RunDuplicateDetectionButton.tsx b/editor/app/actionable/components/RunDuplicateDetectionButton.tsx @@ -5,7 +5,7 @@ import { runDuplicateDetectionAction, type DuplicateScope, type RunDuplicateDetectionResult, -} from "../actions"; +} from "../../lib/actionable/actions"; type Status = | { kind: "idle" } diff --git a/editor/app/actionable/components/RunMediaScanButton.tsx b/editor/app/actionable/components/RunMediaScanButton.tsx @@ -4,7 +4,7 @@ import { useState } from "react"; import { runMediaScanAction, type RunMediaScanResult, -} from "../actions"; +} from "../../lib/actionable/actions"; type Status = | { kind: "idle" } diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -1,295 +0,0 @@ -import type { Paths } from "yt-dlp-transcript-common/lib/paths"; -import { cache } from "react"; -import { - backfillLaneEntriesOf, - reachableOperationWork, -} from "yt-dlp-transcript-common/lib/operations"; -import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels"; -import { - getChannelBriefs, - getDuplicateReport, -} from "../../lib/requestCache"; -import { - digestWorkOf, - excludedDownloadIdSet, - type ChannelSnapshot, -} from "yt-dlp-transcript-common/controller/channelSnapshot"; -import { - readDuplicateOverrides, -} from "yt-dlp-transcript-common/controller/duplicateShorts"; -import type { - DuplicateOverrides, - DuplicateReport, -} from "yt-dlp-transcript-common/lib/duplicates"; -import type { - MediaScanOverrides, - MediaScanReport, -} from "yt-dlp-transcript-common/lib/mediaScan"; -import { - readMediaScanOverrides, - readMediaScanReport, -} from "yt-dlp-transcript-common/controller/scanCorruptMedia"; - -export type ActionableRow = { - channel: ChannelBrief; - // Always `channel.snapshot` — kept as a sibling field because every count - // helper and every consumer already reads `row.snapshot`. - snapshot: ChannelSnapshot | null; -}; - -export type ActionableSummary = { - rows: ActionableRow[]; - undownloaded: ActionableRow[]; - missingNeverFetched: ActionableRow[]; - untranscribed: ActionableRow[]; - incompleteTranscripts: ActionableRow[]; - shortAudio: ActionableRow[]; - cleanTranscribedAudio: ActionableRow[]; - cleanExtraFormats: ActionableRow[]; - staleOrMissing: ActionableRow[]; - digestWarnings: ActionableRow[]; - backfill: ActionableRow[]; - duplicates: DuplicateReport | null; - // The human decisions kept alongside the report — a cluster's canonical - // choice, "not a duplicate", and the `confirmed` flag that is the only thing - // letting a needsReview cluster share derived work. Loaded here because the - // review UI cannot show what has already been decided without it, and a - // review queue that forgets its own answers re-asks every question. - duplicateOverrides: DuplicateOverrides; - // The corrupt-media scan, and the "reviewed, this one's fine" decisions kept - // beside it. Null when the scan has never been run — which is NOT the same as - // "nothing is wrong", and the section says so rather than rendering an - // all-clear it has no evidence for. - mediaScan: MediaScanReport | null; - mediaScanOverrides: MediaScanOverrides; -}; - -export function isStaleOrMissing(row: ActionableRow): boolean { - if (!row.snapshot) return true; - const synced = row.channel.config.lastSyncedAt; - if (!synced) return false; - return new Date(synced).getTime() > new Date(row.snapshot.generatedAt).getTime(); -} - -// Counts that drive the actionable lists exclude IDs that the availability -// check has flagged as deleted / members-only / private — those videos -// can't be acted on, so they shouldn't inflate "needs attention" totals. -// `undownloadedIds` is already filtered at snapshot generation time, but we -// apply the filter again so a stale snapshot can't surface excluded IDs. -function countActionable( - snapshot: ChannelSnapshot | null | undefined, - ids: readonly string[] | undefined, -): number { - if (!snapshot || !ids) return 0; - const excluded = excludedDownloadIdSet(snapshot); - if (excluded.size === 0) return ids.length; - let n = 0; - for (const id of ids) if (!excluded.has(id)) n++; - return n; -} - -export function actionableUndownloadedCount(row: ActionableRow): number { - return countActionable(row.snapshot, row.snapshot?.undownloadedIds); -} - -// Videos the roster says we were told about, never downloaded, and that have -// since left the listing. Deliberately NOT run through countActionable: the -// availability exclusions are keyed on videos we have on disk, and these have no -// dir at all. Default 0 for snapshots written before the bucket existed. -export function actionableMissingNeverFetchedCount(row: ActionableRow): number { - return row.snapshot?.missingNeverFetched?.length ?? 0; -} - -export function actionableUntranscribedCount(row: ActionableRow): number { - return countActionable( - row.snapshot, - row.snapshot?.buckets.downloadedNoTranscript, - ); -} - -// Transcribed videos whose transcript is badly truncated (the audio download -// stopped early). Default 0 for snapshots written before the bucket existed. -export function actionableIncompleteTranscriptCount(row: ActionableRow): number { - return row.snapshot?.buckets.incompleteTranscript?.length ?? 0; -} - -// Downloads the duration guard flagged as truncated at the source (short audio -// kept on disk, not transcribed). Default 0 for snapshots predating the bucket. -export function actionableShortAudioCount(row: ActionableRow): number { - return row.snapshot?.buckets.shortAudio?.length ?? 0; -} - -// Cleanup buckets are filtered by "do not clean" at snapshot-generation time, -// so the length is the actionable count directly (default undefined → 0 for -// snapshots written before the bucket existed). -export function actionableCleanTranscribedCount(row: ActionableRow): number { - return row.snapshot?.buckets.transcribedWithAudio?.length ?? 0; -} - -export function actionableCleanExtraFormatsCount(row: ActionableRow): number { - return row.snapshot?.buckets.multipleAudioFormats?.length ?? 0; -} - -// The digest layer's two work lists. The first is "has no digest at the current -// identity" — missing, stale or part-done. `digestWarnings` is "the model -// produced something a human should look at", which includes the total failures -// that write no section and so are invisible to any count of files. -// -// Read through digestWorkOf: the registry's classification is the one the runner -// uses, and it excludes videos with no transcript (blocked) and videos whose -// cues.json is stale (deferred) — work the old `noDigest` bucket offered here -// and the runner then declined. -export function actionableNoDigestCount(row: ActionableRow): number { - return digestWorkOf(row.snapshot).reachable; -} - -export function actionableDigestWarningsCount(row: ActionableRow): number { - return row.snapshot?.buckets.digestWarnings?.length ?? 0; -} - -// The backfill lane's two numbers, and they are two FUNCTIONS on purpose so no -// caller can accidentally add them. -// -// `reachable` (missing + stale) is what the lane can do today and the only thing -// that decides whether a channel appears in the section at all. `missingInput` -// is the population that needs its media re-acquired first — measured at ~91x -// the reachable count corpus-wide, so filtering on it would put every channel in -// the list forever. That is not a hypothetical: it is the documented reason -// /api/widget/actionable refuses to filter on the digest work count. -// backfillLaneEntriesOf, not Object.values: the snapshot map is every catalog operation -// now, and digest is one of them. These two functions decide whether a channel -// appears in the BACKFILL section at all, so folding a ~75,000-video operation -// that runs on another queue into them would put every channel in the list -// forever — the same trap /api/widget/actionable documents for the digest work -// count, hit from the other direction. -export function actionableBackfillCount(row: ActionableRow): number { - return backfillLaneEntriesOf(row.snapshot?.backfill).reduce( - (n, e) => n + reachableOperationWork(e), - 0, - ); -} - -export function actionableBackfillMissingInputCount( - row: ActionableRow, -): number { - return backfillLaneEntriesOf(row.snapshot?.backfill).reduce( - (n, e) => n + e.missingInput, - 0, - ); -} - -// Estimated bytes each cleanup would reclaim (default 0 for snapshots written -// before cleanupBytes existed). -export function actionableCleanTranscribedBytes(row: ActionableRow): number { - return row.snapshot?.cleanupBytes?.transcribedWithAudio ?? 0; -} - -export function actionableCleanExtraFormatsBytes(row: ActionableRow): number { - return row.snapshot?.cleanupBytes?.multipleAudioFormats ?? 0; -} - -export async function loadActionableSummary( - paths: Paths, -): Promise<ActionableSummary> { - const [channels, duplicates, duplicateOverrides, mediaScan, mediaScanOverrides] = - await Promise.all([ - getChannelBriefs(paths), - getDuplicateReport(paths), - readDuplicateOverrides(paths), - readMediaScanReport(paths), - readMediaScanOverrides(paths), - ]); - // The brief already read the snapshot; this used to read each one a second - // time on top of a full corpus walk. - const rows: ActionableRow[] = channels.map((channel) => ({ - channel, - snapshot: channel.snapshot, - })); - - const undownloaded = rows - .filter((r) => actionableUndownloadedCount(r) > 0) - .sort( - (a, b) => actionableUndownloadedCount(b) - actionableUndownloadedCount(a), - ); - - const missingNeverFetched = rows - .filter((r) => actionableMissingNeverFetchedCount(r) > 0) - .sort( - (a, b) => - actionableMissingNeverFetchedCount(b) - - actionableMissingNeverFetchedCount(a), - ); - - const untranscribed = rows - .filter((r) => actionableUntranscribedCount(r) > 0) - .sort( - (a, b) => actionableUntranscribedCount(b) - actionableUntranscribedCount(a), - ); - - const incompleteTranscripts = rows - .filter((r) => actionableIncompleteTranscriptCount(r) > 0) - .sort( - (a, b) => - actionableIncompleteTranscriptCount(b) - - actionableIncompleteTranscriptCount(a), - ); - - const shortAudio = rows - .filter((r) => actionableShortAudioCount(r) > 0) - .sort((a, b) => actionableShortAudioCount(b) - actionableShortAudioCount(a)); - - const cleanTranscribedAudio = rows - .filter((r) => actionableCleanTranscribedCount(r) > 0) - .sort( - (a, b) => - actionableCleanTranscribedCount(b) - actionableCleanTranscribedCount(a), - ); - - const cleanExtraFormats = rows - .filter((r) => actionableCleanExtraFormatsCount(r) > 0) - .sort( - (a, b) => - actionableCleanExtraFormatsCount(b) - - actionableCleanExtraFormatsCount(a), - ); - - const digestWarnings = rows - .filter((r) => actionableDigestWarningsCount(r) > 0) - .sort( - (a, b) => - actionableDigestWarningsCount(b) - actionableDigestWarningsCount(a), - ); - - const backfill = rows - .filter((r) => actionableBackfillCount(r) > 0) - .sort((a, b) => actionableBackfillCount(b) - actionableBackfillCount(a)); - - const staleOrMissing = rows - .filter(isStaleOrMissing) - .sort((a, b) => a.channel.slug.localeCompare(b.channel.slug)); - - return { - rows, - undownloaded, - missingNeverFetched, - untranscribed, - incompleteTranscripts, - shortAudio, - cleanTranscribedAudio, - cleanExtraFormats, - staleOrMissing, - digestWarnings, - backfill, - duplicates, - duplicateOverrides, - mediaScan, - mediaScanOverrides, - }; -} - -// Per-request memoized. The dashboard renders this summary and several -// components derived from it in one pass; see ../../lib/requestCache for why a -// request is the only cache lifetime this data can safely have. -export const getActionableSummary = cache((paths: Paths) => - loadActionableSummary(paths), -); diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -1,30 +1,23 @@ import type { Metadata } from "next"; import Link from "next/link"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { backfillLaneOperations, operationsActionLabel, } from "yt-dlp-transcript-common/lib/operations"; import { - actionableCleanExtraFormatsBytes, - actionableCleanExtraFormatsCount, - actionableCleanTranscribedBytes, - actionableCleanTranscribedCount, actionableBackfillCount, actionableBackfillMissingInputCount, - actionableDigestWarningsCount, - actionableIncompleteTranscriptCount, - actionableMissingNeverFetchedCount, - actionableShortAudioCount, - actionableUndownloadedCount, - actionableUntranscribedCount, loadActionableSummary, - type ActionableRow, -} from "./lib/loadActionable"; -import { InlineActionButton } from "./components/InlineActionButton"; -import { FixAllIncompleteButton } from "./components/FixAllIncompleteButton"; +} from "../lib/actionable/loadActionable"; +import { loadReviewSummary } from "../lib/review/loadReview"; +import { + channelWorkSections, + type SectionConfig, +} from "../components/channelWork/sections"; +import { ChannelWorkTable } from "../components/channelWork/ChannelWorkTable"; +import { InlineActionButton } from "../components/actions/InlineActionButton"; import { RefreshAllReportsButton } from "./components/RefreshAllReportsButton"; import { RunDuplicateDetectionButton } from "./components/RunDuplicateDetectionButton"; import { RunMediaScanButton } from "./components/RunMediaScanButton"; @@ -47,27 +40,12 @@ export const dynamic = "force-dynamic"; export const metadata: Metadata = { title: "Actionable" }; -type SectionConfig = { - id: string; - title: string; - description: string; - countLabel: string; - emptyLabel: string; - getCount: (row: ActionableRow) => number; - // Optional extra column rendered after the count (used by the cleanup - // sections to show estimated reclaimable disk space). - extraColumn?: { - label: string; - getValue: (row: ActionableRow) => string; - }; - primaryAction: (row: ActionableRow) => React.ReactNode | null; - // Optional control rendered in the section header (e.g. an act-on-all button). - headerAction?: React.ReactNode; -}; - export default async function ActionablePage() { const paths = getPaths(); - const summary = await loadActionableSummary(paths); + const [summary, review] = await Promise.all([ + loadActionableSummary(paths), + loadReviewSummary(paths), + ]); // What the derived-data lane is actually called, given which operations are // enabled. "Backfill" is its queue key; on this install it stands for three // operations, and no button anyone presses should be named after a queue. @@ -89,236 +67,58 @@ export default async function ActionablePage() { // list has rows in it. summary.backfill.length === 0; - const sections: { config: SectionConfig; rows: ActionableRow[] }[] = [ - { - config: { - id: "undownloaded", - title: "Channels with undownloaded videos", - description: - "Playlist entries that have no audio/video on disk yet. Run “Download missing” to fetch them.", - countLabel: "undownloaded", - emptyLabel: "Nothing pending.", - getCount: actionableUndownloadedCount, - primaryAction: (r) => ( - <InlineActionButton - variant={{ kind: "downloadMissing", slug: r.channel.slug }} - /> - ), - }, - rows: summary.undownloaded, - }, - { - config: { - id: "missing-never-fetched", - title: "Channels with videos lost before they were ever downloaded", - description: - "Videos that were in the channel listing, were never fetched, and have since left it. Nothing of them exists on disk — only the URL the roster kept, which is enough to attempt a direct-link download. A video that merely went unlisted still downloads; a deleted one will not. Open the channel's Diagnostics stage to try them.", - countLabel: "never fetched", - emptyLabel: "None lost.", - getCount: actionableMissingNeverFetchedCount, - primaryAction: (r) => ( - <Link - href={`/channels/${r.channel.slug}`} - aria-label={`review never-fetched videos for ${r.channel.slug}`} - className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" - > - Review - </Link> - ), - }, - rows: summary.missingNeverFetched, - }, - { - config: { - id: "untranscribed", - title: "Channels with downloaded videos awaiting transcription", - description: - "Videos with audio on disk but no whisper or yt-vtt transcript yet. Run “Transcribe pending” to whisper them.", - countLabel: "awaiting transcript", - emptyLabel: "Nothing pending.", - getCount: actionableUntranscribedCount, - primaryAction: (r) => ( - <InlineActionButton - variant={{ - kind: "transcribeMissing", - slug: r.channel.slug, - audioFormat: r.channel.config.audioFormat, - }} - /> - ), - }, - rows: summary.untranscribed, - }, - { - config: { - id: "incomplete-transcripts", - title: "Channels with incomplete (truncated) transcripts", - description: - "Transcribed videos whose transcript covers only a small fraction of the runtime — the audio download stopped early. “Re-download & re-transcribe” queues a batch fix; “Clear & re-queue” deletes the truncated audio + transcript and hands them to the auto-runners.", - countLabel: "truncated", - emptyLabel: "None detected.", - getCount: actionableIncompleteTranscriptCount, - headerAction: <FixAllIncompleteButton />, - primaryAction: (r) => ( - <span className="inline-flex items-center justify-end gap-2 flex-wrap"> - <InlineActionButton - variant={{ kind: "redownloadIncomplete", slug: r.channel.slug }} - /> - <InlineActionButton - variant={{ kind: "clearIncomplete", slug: r.channel.slug }} - /> - <Link - href={`/channels/${r.channel.slug}?filter=incomplete_transcript`} - aria-label={`review incomplete transcripts for ${r.channel.slug}`} - className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" - > - Review - </Link> - </span> - ), - }, - rows: summary.incompleteTranscripts, - }, - { - config: { - id: "digest-warnings", - title: "Channels with digest passes that need a look", - description: - "Videos where the AI digest pass recorded warnings, or produced nothing usable at all. The second kind is the one worth opening: a total failure deliberately writes no section so the video retries, and “the model proposed nothing” and “the model proposed chapters and every one was rejected by a guard” look identical from outside but need different fixes.", - countLabel: "with warnings", - emptyLabel: "None recorded.", - getCount: actionableDigestWarningsCount, - primaryAction: (r) => ( - <Link - href={`/channels/${r.channel.slug}?filter=digest_warnings`} - aria-label={`review digest warnings for ${r.channel.slug}`} - className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" - > - Review - </Link> - ), - }, - rows: summary.digestWarnings, - }, - { - config: { - // "speakers" is the backfill lane's section: named for the operations - // it holds, as the channel page's speakers card and the /channels - // station are, not for the queue they share. - id: "speakers", - title: "Channels missing derived data the corpus predates", - description: - `Videos an enabled derived-data operation has nothing on disk for — no record, or one produced by a different engine/model/threshold than the current settings. Run \u201c${laneAction}\u201d to catch them up. The count is what the lane can do TODAY; the second column is the separate population whose source media has already been deleted, which needs the opt-in re-download to reach at all.`, - countLabel: "reachable", - emptyLabel: "Nothing pending.", - getCount: actionableBackfillCount, - // Same treatment "Est. reclaim" gets, and for a stronger reason: these - // two numbers differ by ~91x on the real corpus, so a single total would - // be dominated by work no button on this page can start. - extraColumn: { - label: "Needs media", - getValue: (r) => - actionableBackfillMissingInputCount(r).toLocaleString(), - }, - primaryAction: (r) => ( - <InlineActionButton - variant={{ kind: "backfillChannel", slug: r.channel.slug }} - // Named after the operations that are actually on, resolved here on - // the server. The button used to read "Backfill", which is the - // queue key three unrelated operations happen to share. - label={laneAction} - /> - ), - }, - rows: summary.backfill, + // The two configs that do NOT move to components/channelWork: they die with + // this page. `speakers` becomes the lane sections on /operations, and + // `stale-reports` becomes the Report column on /channels. + const speakersConfig: SectionConfig = { + // "speakers" is the backfill lane's section: named for the operations + // it holds, as the channel page's speakers card and the /channels + // station are, not for the queue they share. + id: "speakers", + operation: null, + role: "work", + getRows: (s) => s.backfill, + title: "Channels missing derived data the corpus predates", + description: + `Videos an enabled derived-data operation has nothing on disk for — no record, or one produced by a different engine/model/threshold than the current settings. Run “${laneAction}” to catch them up. The count is what the lane can do TODAY; the second column is the separate population whose source media has already been deleted, which needs the opt-in re-download to reach at all.`, + countLabel: "reachable", + emptyLabel: "Nothing pending.", + getCount: actionableBackfillCount, + // Same treatment "Est. reclaim" gets, and for a stronger reason: these + // two numbers differ by ~91x on the real corpus, so a single total would + // be dominated by work no button on this page can start. + extraColumn: { + label: "Needs media", + getValue: (r) => actionableBackfillMissingInputCount(r).toLocaleString(), }, - { - config: { - id: "short-audio", - title: "Channels with truncated downloads (short audio)", - description: - "Downloads that completed but whose audio is far shorter than the video — the source served a truncated stream, caught before transcription. The stub is kept (not transcribed). “Re-download (corrected format)” deletes it and re-fetches with the per-source default (Original for Odysee).", - countLabel: "truncated", - emptyLabel: "None detected.", - getCount: actionableShortAudioCount, - primaryAction: (r) => ( - <span className="inline-flex items-center justify-end gap-2 flex-wrap"> - <InlineActionButton - variant={{ kind: "redownloadShortAudio", slug: r.channel.slug }} - /> - <Link - href={`/channels/${r.channel.slug}?filter=short_audio`} - aria-label={`review short-audio downloads for ${r.channel.slug}`} - className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" - > - Review - </Link> - </span> - ), - }, - rows: summary.shortAudio, - }, - { - config: { - id: "clean-transcribed-audio", - title: "Channels with cleanable transcribed audio", - description: - "Videos that already have a whisper transcript but still keep their audio on disk. Run “Clean audio” to reclaim space. Videos marked “do not clean” are excluded.", - countLabel: "cleanable", - emptyLabel: "Nothing to clean.", - getCount: actionableCleanTranscribedCount, - extraColumn: { - label: "Est. reclaim", - getValue: (r) => `~${formatBytes(actionableCleanTranscribedBytes(r))}`, - }, - primaryAction: (r) => ( - <InlineActionButton - variant={{ kind: "cleanTranscribedAudio", slug: r.channel.slug }} - /> - ), - }, - rows: summary.cleanTranscribedAudio, - }, - { - config: { - id: "clean-extra-formats", - title: "Channels with extra audio formats", - description: - "Videos that have the channel's configured audio format plus other leftover audio files. Run “Clean extra formats” to keep only the target format. Videos marked “do not clean” are excluded.", - countLabel: "extra formats", - emptyLabel: "Nothing to clean.", - getCount: actionableCleanExtraFormatsCount, - extraColumn: { - label: "Est. reclaim", - getValue: (r) => - `~${formatBytes(actionableCleanExtraFormatsBytes(r))}`, - }, - primaryAction: (r) => ( - <InlineActionButton - variant={{ kind: "cleanExtraFormats", slug: r.channel.slug }} - /> - ), - }, - rows: summary.cleanExtraFormats, - }, - { - config: { - id: "stale-reports", - title: "Channels with stale or missing reports", - description: - "Reports are older than the most recent sync (or have never been generated). The counts above might be wrong until you refresh.", - countLabel: "report age", - emptyLabel: "All reports are current.", - getCount: () => 0, - primaryAction: (r) => ( - <InlineActionButton - variant={{ kind: "refreshReport", slug: r.channel.slug }} - /> - ), - }, - rows: summary.staleOrMissing, - }, - ]; + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "backfillChannel", slug: r.channel.slug }} + // Named after the operations that are actually on, resolved here on + // the server. The button used to read "Backfill", which is the + // queue key three unrelated operations happen to share. + label={laneAction} + /> + ), + }; + + const staleReportsConfig: SectionConfig = { + id: "stale-reports", + operation: null, + role: "attention", + getRows: (s) => s.staleOrMissing, + title: "Channels with stale or missing reports", + description: + "Reports are older than the most recent sync (or have never been generated). The counts above might be wrong until you refresh.", + countLabel: "report age", + emptyLabel: "All reports are current.", + getCount: () => 0, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "refreshReport", slug: r.channel.slug }} + /> + ), + }; return ( <div className="flex flex-col gap-6"> @@ -334,17 +134,22 @@ export default async function ActionablePage() { Everything is up to date. </p> ) : ( - sections.map(({ config, rows }) => ( - <Section key={config.id} config={config} rows={rows} /> - )) + <ChannelWorkTable + sections={[ + ...channelWorkSections(), + speakersConfig, + staleReportsConfig, + ]} + summary={summary} + /> )} <DuplicatesSection - report={summary.duplicates} - overrides={summary.duplicateOverrides} + report={review.duplicates} + overrides={review.duplicateOverrides} /> <MediaScanSection - report={summary.mediaScan} - overrides={summary.mediaScanOverrides} + report={review.mediaScan} + overrides={review.mediaScanOverrides} /> </div> ); @@ -549,148 +354,3 @@ function Badge({ children }: { children: React.ReactNode }) { </span> ); } - -function Section({ - config, - rows, -}: { - config: SectionConfig; - rows: ActionableRow[]; -}) { - return ( - <section - aria-label={config.id} - className="flex flex-col gap-2" - > - <div className="flex items-center justify-between flex-wrap gap-2"> - <h2 className="text-lg font-semibold">{config.title}</h2> - {config.headerAction} - </div> - <p className="text-sm text-muted-foreground">{config.description}</p> - {rows.length === 0 ? ( - <p - aria-label={`${config.id} empty`} - className="text-sm text-muted-foreground border border-dashed border-border rounded p-4" - > - {config.emptyLabel} - </p> - ) : ( - <div className="overflow-x-auto -mx-4 md:mx-0 md:overflow-visible"> - <table className="text-sm border-y md:border md:border-border border-border md:rounded-md md:overflow-hidden w-full"> - <thead className="bg-muted"> - <tr> - <th className="text-left font-medium px-3 py-2">Slug</th> - <th className="text-right font-medium px-3 py-2"> - {config.countLabel} - </th> - {config.extraColumn && ( - <th className="text-right font-medium px-3 py-2 whitespace-nowrap"> - {config.extraColumn.label} - </th> - )} - <th className="text-left font-medium px-3 py-2 whitespace-nowrap"> - Last report - </th> - <th className="text-left font-medium px-3 py-2 whitespace-nowrap"> - Last sync - </th> - <th className="text-right font-medium px-3 py-2">Actions</th> - </tr> - </thead> - <tbody> - {rows.map((row) => ( - <Row - key={row.channel.slug} - row={row} - count={config.getCount(row)} - extraValue={config.extraColumn?.getValue(row) ?? null} - primaryAction={config.primaryAction(row)} - sectionId={config.id} - /> - ))} - </tbody> - </table> - </div> - )} - </section> - ); -} - -function Row({ - row, - count, - extraValue, - primaryAction, - sectionId, -}: { - row: ActionableRow; - count: number; - extraValue: string | null; - primaryAction: React.ReactNode; - sectionId: string; -}) { - const { channel, snapshot } = row; - const lastSync = channel.config.lastSyncedAt; - const lastReport = snapshot?.generatedAt ?? null; - const isStale = - !snapshot || - (!!lastSync && - new Date(lastSync).getTime() > new Date(snapshot.generatedAt).getTime()); - return ( - <tr - aria-label={`${sectionId} row ${channel.slug}`} - className="border-t border-border" - > - <td className="px-3 py-2 font-mono"> - <Link - href={`/channels/${channel.slug}`} - className="underline hover:text-foreground" - > - {channel.slug} - </Link> - </td> - <td className="px-3 py-2 text-right tabular-nums"> - {sectionId === "stale-reports" ? ( - <span className="text-warning"> - {snapshot ? "stale" : "missing"} - </span> - ) : ( - count - )} - </td> - {extraValue !== null && ( - <td className="px-3 py-2 text-right tabular-nums whitespace-nowrap text-muted-foreground"> - {extraValue} - </td> - )} - <td - className={`px-3 py-2 text-xs whitespace-nowrap ${ - isStale - ? "text-warning" - : "text-muted-foreground" - }`} - > - {lastReport ? ( - <time dateTime={lastReport}> - {new Date(lastReport).toLocaleString()} - </time> - ) : ( - "never" - )} - </td> - <td className="px-3 py-2 text-xs text-muted-foreground whitespace-nowrap"> - {lastSync ? new Date(lastSync).toLocaleString() : "never"} - </td> - <td className="px-3 py-2 text-right"> - <span className="inline-flex items-center justify-end gap-3 flex-wrap"> - {primaryAction} - {sectionId !== "stale-reports" && ( - <InlineActionButton - variant={{ kind: "refreshReport", slug: channel.slug }} - /> - )} - </span> - </td> - </tr> - ); -} diff --git a/editor/app/api/widget/actionable/route.ts b/editor/app/api/widget/actionable/route.ts @@ -5,7 +5,7 @@ import { actionableUndownloadedCount, actionableUntranscribedCount, loadActionableSummary, -} from "../../../actionable/lib/loadActionable"; +} from "../../../lib/actionable/loadActionable"; export const dynamic = "force-dynamic"; diff --git a/editor/app/channels/[slug]/components/NoReportYet.tsx b/editor/app/channels/[slug]/components/NoReportYet.tsx @@ -1,5 +1,5 @@ import Link from "next/link"; -import { InlineActionButton } from "../../../actionable/components/InlineActionButton"; +import { InlineActionButton } from "../../../components/actions/InlineActionButton"; // Banner shown at the top of a channel page that has no generated report. // diff --git a/editor/app/channels/[slug]/components/flow/NextAction.tsx b/editor/app/channels/[slug]/components/flow/NextAction.tsx @@ -1,5 +1,5 @@ import type { ChannelFlow } from "../../lib/channelFlow"; -import { InlineActionButton } from "../../../../actionable/components/InlineActionButton"; +import { InlineActionButton } from "../../../../components/actions/InlineActionButton"; import type { StageId } from "../../lib/stageStatus"; // THE ONE PRIMARY ACTION, and the only brand-coloured thing on the page. diff --git a/editor/app/components/CommandPalette.tsx b/editor/app/components/CommandPalette.tsx @@ -27,7 +27,7 @@ import { syncAllChannelsAction } from "../channels/actions"; import { refreshAllChannelSnapshotsAction, runDuplicateDetectionAction, -} from "../actionable/actions"; +} from "../lib/actionable/actions"; import { retryAllFailedAction, drainAllAction } from "../jobs/actions"; import { pauseAllWorkersAction, diff --git a/editor/app/actionable/components/InlineActionButton.tsx b/editor/app/components/actions/InlineActionButton.tsx diff --git a/editor/app/components/channelWork/ChannelWorkTable.tsx b/editor/app/components/channelWork/ChannelWorkTable.tsx @@ -0,0 +1,178 @@ +import Link from "next/link"; +import { + isStaleOrMissing, + type ActionableRow, + type ActionableSummary, +} from "../../lib/actionable/loadActionable"; +import { InlineActionButton } from "../actions/InlineActionButton"; +import type { SectionConfig } from "./sections"; + +// A server component on purpose: `SectionConfig.primaryAction` is a function +// that returns an element, and a function cannot cross the server/client +// boundary as a prop. +// +// THE ARIA CONTRACT BELOW IS LOAD-BEARING. `channel-work.spec`, +// `cleanup-actionable.spec`, `incomplete-transcript.spec` and +// `download-format-guard.spec` all locate by these exact strings: +// <section aria-label="<id>">, <h2>{title}</h2>, +// <p aria-label="<id> empty">, <tr aria-label="<id> row <slug>">, +// columns Slug / countLabel / [extra] / Last report / Last sync / Actions, +// and a `refresh report <slug>` InlineActionButton on every row. +// +// NEVER add `role="status"` in here. This renders on the RUNNER operation +// pages, where `auto-subs-replace.spec.ts` takes `getByRole("status").first()` +// and the runner section reserves that role for its own console. +// +// On /cleanup the ChannelCleanupCard's own "Clean audio" control is a +// StreamActionLog whose button carries no aria-label, so the table's +// `clean audio <slug>` button is still the only match for that name; the log's +// `aria-label="Clean audio <slug> error"` differs from this table's only by +// case, and neither is asserted anywhere. +export function ChannelWorkTable({ + sections, + summary, +}: { + sections: SectionConfig[]; + summary: ActionableSummary; +}) { + return ( + <> + {sections.map((config) => ( + <Section key={config.id} config={config} rows={config.getRows(summary)} /> + ))} + </> + ); +} + +function Section({ + config, + rows, +}: { + config: SectionConfig; + rows: ActionableRow[]; +}) { + return ( + <section + aria-label={config.id} + className="flex flex-col gap-2" + > + <div className="flex items-center justify-between flex-wrap gap-2"> + <h2 className="text-lg font-semibold">{config.title}</h2> + {config.headerAction} + </div> + <p className="text-sm text-muted-foreground">{config.description}</p> + {rows.length === 0 ? ( + <p + aria-label={`${config.id} empty`} + className="text-sm text-muted-foreground border border-dashed border-border rounded p-4" + > + {config.emptyLabel} + </p> + ) : ( + <div className="overflow-x-auto -mx-4 md:mx-0 md:overflow-visible"> + <table className="text-sm border-y md:border md:border-border border-border md:rounded-md md:overflow-hidden w-full"> + <thead className="bg-muted"> + <tr> + <th className="text-left font-medium px-3 py-2">Slug</th> + <th className="text-right font-medium px-3 py-2"> + {config.countLabel} + </th> + {config.extraColumn && ( + <th className="text-right font-medium px-3 py-2 whitespace-nowrap"> + {config.extraColumn.label} + </th> + )} + <th className="text-left font-medium px-3 py-2 whitespace-nowrap"> + Last report + </th> + <th className="text-left font-medium px-3 py-2 whitespace-nowrap"> + Last sync + </th> + <th className="text-right font-medium px-3 py-2">Actions</th> + </tr> + </thead> + <tbody> + {rows.map((row) => ( + <Row + key={row.channel.slug} + row={row} + count={config.getCount(row)} + extraValue={config.extraColumn?.getValue(row) ?? null} + primaryAction={config.primaryAction(row)} + sectionId={config.id} + /> + ))} + </tbody> + </table> + </div> + )} + </section> + ); +} + +function Row({ + row, + count, + extraValue, + primaryAction, + sectionId, +}: { + row: ActionableRow; + count: number; + extraValue: string | null; + primaryAction: React.ReactNode; + sectionId: string; +}) { + const { channel, snapshot } = row; + const lastSync = channel.config.lastSyncedAt; + const lastReport = snapshot?.generatedAt ?? null; + // The loader's own definition, which this used to carry a copy of. + const isStale = isStaleOrMissing(row); + return ( + <tr + aria-label={`${sectionId} row ${channel.slug}`} + className="border-t border-border" + > + <td className="px-3 py-2 font-mono"> + <Link + href={`/channels/${channel.slug}`} + className="underline hover:text-foreground" + > + {channel.slug} + </Link> + </td> + <td className="px-3 py-2 text-right tabular-nums">{count}</td> + {extraValue !== null && ( + <td className="px-3 py-2 text-right tabular-nums whitespace-nowrap text-muted-foreground"> + {extraValue} + </td> + )} + <td + className={`px-3 py-2 text-xs whitespace-nowrap ${ + isStale + ? "text-warning" + : "text-muted-foreground" + }`} + > + {lastReport ? ( + <time dateTime={lastReport}> + {new Date(lastReport).toLocaleString()} + </time> + ) : ( + "never" + )} + </td> + <td className="px-3 py-2 text-xs text-muted-foreground whitespace-nowrap"> + {lastSync ? new Date(lastSync).toLocaleString() : "never"} + </td> + <td className="px-3 py-2 text-right"> + <span className="inline-flex items-center justify-end gap-3 flex-wrap"> + {primaryAction} + <InlineActionButton + variant={{ kind: "refreshReport", slug: channel.slug }} + /> + </span> + </td> + </tr> + ); +} diff --git a/editor/app/components/channelWork/FixAllIncompleteButton.tsx b/editor/app/components/channelWork/FixAllIncompleteButton.tsx @@ -0,0 +1,101 @@ +"use client"; + +import Link from "next/link"; +import { useState } from "react"; +import { + clearAllIncompleteTranscriptsAction, + redownloadAllIncompleteTranscriptsAction, + type GlobalIncompleteResult, +} from "../../lib/actionable/actions"; + +type Mode = "clear" | "redownload"; + +type Status = + | { kind: "idle" } + | { kind: "running"; mode: Mode } + | { kind: "done"; mode: Mode; result: GlobalIncompleteResult } + | { kind: "error"; message: string }; + +export function FixAllIncompleteButton() { + const [status, setStatus] = useState<Status>({ kind: "idle" }); + + async function run(mode: Mode) { + if ( + mode === "clear" && + !window.confirm( + "Delete the truncated audio + transcript for every flagged video across ALL channels and enable the auto-download/transcribe runners? The current partial transcripts are removed and re-fetched. This cannot be undone.", + ) + ) { + return; + } + setStatus({ kind: "running", mode }); + try { + const result = + mode === "clear" + ? await clearAllIncompleteTranscriptsAction() + : await redownloadAllIncompleteTranscriptsAction(); + setStatus({ kind: "done", mode, result }); + } catch (e) { + setStatus({ kind: "error", message: (e as Error).message }); + } + } + + const running = status.kind === "running"; + return ( + <div className="flex items-center gap-2 flex-wrap"> + <button + type="button" + onClick={() => run("redownload")} + disabled={running} + aria-label="re-download all incomplete transcripts" + className="px-2.5 py-1 rounded-md bg-primary text-primary-foreground text-xs font-medium hover:opacity-90 disabled:opacity-50 whitespace-nowrap" + > + {status.kind === "running" && status.mode === "redownload" + ? "Queuing…" + : "Re-fix all"} + </button> + <button + type="button" + onClick={() => run("clear")} + disabled={running} + aria-label="clear all incomplete transcripts" + className="px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted disabled:opacity-50 whitespace-nowrap" + > + {status.kind === "running" && status.mode === "clear" + ? "Clearing…" + : "Clear & re-queue all"} + </button> + {status.kind === "done" && ( + <span + aria-label="fix all incomplete transcripts result" + className="text-xs text-muted-foreground" + > + {status.result.ok ? ( + <> + {status.mode === "clear" ? "Cleared" : "Queued"}{" "} + {status.result.affected} across {status.result.channels} channel + {status.result.channels === 1 ? "" : "s"} ·{" "} + <Link + href="/jobs" + className="underline hover:text-foreground" + > + view jobs + </Link> + </> + ) : ( + status.result.error + )} + </span> + )} + {status.kind === "error" && ( + <span + role="alert" + aria-label="fix all incomplete transcripts error" + className="text-xs text-destructive" + > + {status.message} + </span> + )} + </div> + ); +} diff --git a/editor/app/components/channelWork/sections.test.ts b/editor/app/components/channelWork/sections.test.ts @@ -0,0 +1,88 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import React from "react"; +import type { ActionableRow, ActionableSummary } from "../../lib/actionable/loadActionable"; + +// The module under test contains JSX (`headerAction` is an element). This +// repo's tsconfig sets `jsx: "preserve"` for Next, which the unit runner's +// esbuild can only turn into classic `React.createElement` calls — so the +// module needs a global `React` when it is loaded outside Next. Next itself +// uses the automatic runtime and never reaches this line. Kept here rather +// than adding an import Next does not need to the module itself. +(globalThis as unknown as { React: unknown }).React = React; +const { channelWorkSections, sectionsFor } = await import("./sections"); + +// A summary whose every list is a DISTINCT array object, so "which list does +// this section draw" is answerable by identity. The pairing used to be done by +// hand at the render site, which is exactly the hand-exhaustiveness that let a +// section quietly render the wrong rows. +function distinctSummary(): ActionableSummary { + const fresh = () => [] as ActionableRow[]; + return { + rows: fresh(), + undownloaded: fresh(), + missingNeverFetched: fresh(), + untranscribed: fresh(), + incompleteTranscripts: fresh(), + shortAudio: fresh(), + cleanTranscribedAudio: fresh(), + cleanExtraFormats: fresh(), + staleOrMissing: fresh(), + digestWarnings: fresh(), + backfill: fresh(), + }; +} + +test("section ids are unique", () => { + const ids = channelWorkSections().map((s) => s.id); + assert.equal(new Set(ids).size, ids.length); +}); + +test("every section draws a distinct list of the summary", () => { + const summary = distinctSummary(); + const drawn = channelWorkSections().map((s) => s.getRows(summary)); + assert.equal(new Set(drawn).size, drawn.length); + for (const list of drawn) { + assert.ok( + Object.values(summary).includes(list), + "getRows must return a field of the summary, not a new array", + ); + } +}); + +test("sectionsFor groups the sections by the page that renders them", () => { + assert.deepEqual( + sectionsFor("download").map((s) => s.id), + ["undownloaded", "missing-never-fetched", "short-audio"], + ); + assert.deepEqual( + sectionsFor("transcription").map((s) => s.id), + ["untranscribed", "incomplete-transcripts"], + ); + assert.deepEqual( + sectionsFor("digest").map((s) => s.id), + ["digest-warnings"], + ); + assert.deepEqual( + sectionsFor(null).map((s) => s.id), + ["clean-transcribed-audio", "clean-extra-formats"], + ); +}); + +test("sectionsFor puts work before attention", () => { + for (const op of ["download", "transcription", "digest", null] as const) { + const roles = sectionsFor(op).map((s) => s.role); + assert.deepEqual(roles, [...roles].sort((a, b) => (a === b ? 0 : a === "work" ? -1 : 1))); + } +}); + +test("an operation with no sections of its own gets none", () => { + // Every runner/sweep page calls this with its own operation id; the ones + // with no work sections (diarization, attribution-*, transcode) must get an + // empty list rather than throwing, which is what makes the cast at the + // /operations/[id] call site safe. + assert.deepEqual( + sectionsFor("diarization" as never), + [], + ); +}); diff --git a/editor/app/components/channelWork/sections.tsx b/editor/app/components/channelWork/sections.tsx @@ -0,0 +1,250 @@ +import Link from "next/link"; +import { formatBytes } from "yt-dlp-transcript-common/lib/format"; +import { + actionableCleanExtraFormatsBytes, + actionableCleanExtraFormatsCount, + actionableCleanTranscribedBytes, + actionableCleanTranscribedCount, + actionableDigestWarningsCount, + actionableIncompleteTranscriptCount, + actionableMissingNeverFetchedCount, + actionableShortAudioCount, + actionableUndownloadedCount, + actionableUntranscribedCount, + type ActionableRow, + type ActionableSummary, +} from "../../lib/actionable/loadActionable"; +import { InlineActionButton } from "../actions/InlineActionButton"; +import { FixAllIncompleteButton } from "./FixAllIncompleteButton"; + +// The per-channel work sections that used to be a literal array inside +// /actionable/page.tsx. They live here because they belong to the OPERATION, +// not to a page: each one is now rendered by the operation page that runs it +// (and the two cleanup sections by /cleanup). The extension point moved with +// the sections; it did not disappear. +export type SectionConfig = { + id: string; + // The page that renders this section: an operation id for sections that are + // one operation's work, or null for the two cleanup sections (Storage, + // /cleanup's). + operation: "download" | "transcription" | "digest" | null; + // "work" is what the runner or sweep will do; "attention" is what a human + // must look at first — a truncated download is not one the runner can retry. + role: "work" | "attention"; + // Which summary list this section draws — declared here so no page + // hand-pairs a config with a list. The old page.tsx paired them by hand, the + // same hand-exhaustiveness as its "nothing pending" gate. + getRows: (summary: ActionableSummary) => ActionableRow[]; + title: string; + description: string; + countLabel: string; + emptyLabel: string; + getCount: (row: ActionableRow) => number; + // Optional extra column rendered after the count (used by the cleanup + // sections to show estimated reclaimable disk space). + extraColumn?: { + label: string; + getValue: (row: ActionableRow) => string; + }; + primaryAction: (row: ActionableRow) => React.ReactNode | null; + // Optional control rendered in the section header (e.g. an act-on-all button). + headerAction?: React.ReactNode; +}; + +// A function, not a constant: `headerAction` is a React element, and an element +// built once at module scope would be shared across every request. +export function channelWorkSections(): SectionConfig[] { + return [ + { + id: "undownloaded", + operation: "download", + role: "work", + getRows: (s) => s.undownloaded, + title: "Channels with undownloaded videos", + description: + "Playlist entries that have no audio/video on disk yet. Run “Download missing” to fetch them.", + countLabel: "undownloaded", + emptyLabel: "Nothing pending.", + getCount: actionableUndownloadedCount, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "downloadMissing", slug: r.channel.slug }} + /> + ), + }, + { + id: "missing-never-fetched", + operation: "download", + role: "attention", + getRows: (s) => s.missingNeverFetched, + title: "Channels with videos lost before they were ever downloaded", + description: + "Videos that were in the channel listing, were never fetched, and have since left it. Nothing of them exists on disk — only the URL the roster kept, which is enough to attempt a direct-link download. A video that merely went unlisted still downloads; a deleted one will not. Open the channel's Diagnostics stage to try them.", + countLabel: "never fetched", + emptyLabel: "None lost.", + getCount: actionableMissingNeverFetchedCount, + primaryAction: (r) => ( + <Link + href={`/channels/${r.channel.slug}`} + aria-label={`review never-fetched videos for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" + > + Review + </Link> + ), + }, + { + id: "untranscribed", + operation: "transcription", + role: "work", + getRows: (s) => s.untranscribed, + title: "Channels with downloaded videos awaiting transcription", + description: + "Videos with audio on disk but no whisper or yt-vtt transcript yet. Run “Transcribe pending” to whisper them.", + countLabel: "awaiting transcript", + emptyLabel: "Nothing pending.", + getCount: actionableUntranscribedCount, + primaryAction: (r) => ( + <InlineActionButton + variant={{ + kind: "transcribeMissing", + slug: r.channel.slug, + audioFormat: r.channel.config.audioFormat, + }} + /> + ), + }, + { + id: "incomplete-transcripts", + operation: "transcription", + role: "attention", + getRows: (s) => s.incompleteTranscripts, + title: "Channels with incomplete (truncated) transcripts", + description: + "Transcribed videos whose transcript covers only a small fraction of the runtime — the audio download stopped early. “Re-download & re-transcribe” queues a batch fix; “Clear & re-queue” deletes the truncated audio + transcript and hands them to the auto-runners.", + countLabel: "truncated", + emptyLabel: "None detected.", + getCount: actionableIncompleteTranscriptCount, + headerAction: <FixAllIncompleteButton />, + primaryAction: (r) => ( + <span className="inline-flex items-center justify-end gap-2 flex-wrap"> + <InlineActionButton + variant={{ kind: "redownloadIncomplete", slug: r.channel.slug }} + /> + <InlineActionButton + variant={{ kind: "clearIncomplete", slug: r.channel.slug }} + /> + <Link + href={`/channels/${r.channel.slug}?filter=incomplete_transcript`} + aria-label={`review incomplete transcripts for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" + > + Review + </Link> + </span> + ), + }, + { + id: "digest-warnings", + operation: "digest", + role: "attention", + getRows: (s) => s.digestWarnings, + title: "Channels with digest passes that need a look", + description: + "Videos where the AI digest pass recorded warnings, or produced nothing usable at all. The second kind is the one worth opening: a total failure deliberately writes no section so the video retries, and “the model proposed nothing” and “the model proposed chapters and every one was rejected by a guard” look identical from outside but need different fixes.", + countLabel: "with warnings", + emptyLabel: "None recorded.", + getCount: actionableDigestWarningsCount, + primaryAction: (r) => ( + <Link + href={`/channels/${r.channel.slug}?filter=digest_warnings`} + aria-label={`review digest warnings for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" + > + Review + </Link> + ), + }, + { + id: "short-audio", + operation: "download", + role: "attention", + getRows: (s) => s.shortAudio, + title: "Channels with truncated downloads (short audio)", + description: + "Downloads that completed but whose audio is far shorter than the video — the source served a truncated stream, caught before transcription. The stub is kept (not transcribed). “Re-download (corrected format)” deletes it and re-fetches with the per-source default (Original for Odysee).", + countLabel: "truncated", + emptyLabel: "None detected.", + getCount: actionableShortAudioCount, + primaryAction: (r) => ( + <span className="inline-flex items-center justify-end gap-2 flex-wrap"> + <InlineActionButton + variant={{ kind: "redownloadShortAudio", slug: r.channel.slug }} + /> + <Link + href={`/channels/${r.channel.slug}?filter=short_audio`} + aria-label={`review short-audio downloads for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap" + > + Review + </Link> + </span> + ), + }, + { + id: "clean-transcribed-audio", + operation: null, + role: "work", + getRows: (s) => s.cleanTranscribedAudio, + title: "Channels with cleanable transcribed audio", + description: + "Videos that already have a whisper transcript but still keep their audio on disk. Run “Clean audio” to reclaim space. Videos marked “do not clean” are excluded.", + countLabel: "cleanable", + emptyLabel: "Nothing to clean.", + getCount: actionableCleanTranscribedCount, + extraColumn: { + label: "Est. reclaim", + getValue: (r) => `~${formatBytes(actionableCleanTranscribedBytes(r))}`, + }, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "cleanTranscribedAudio", slug: r.channel.slug }} + /> + ), + }, + { + id: "clean-extra-formats", + operation: null, + role: "work", + getRows: (s) => s.cleanExtraFormats, + title: "Channels with extra audio formats", + description: + "Videos that have the channel's configured audio format plus other leftover audio files. Run “Clean extra formats” to keep only the target format. Videos marked “do not clean” are excluded.", + countLabel: "extra formats", + emptyLabel: "Nothing to clean.", + getCount: actionableCleanExtraFormatsCount, + extraColumn: { + label: "Est. reclaim", + getValue: (r) => `~${formatBytes(actionableCleanExtraFormatsBytes(r))}`, + }, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "cleanExtraFormats", slug: r.channel.slug }} + /> + ), + }, + ]; +} + +// Work first, then attention, declaration order within each. An unknown +// operation id returns [] — which is what makes the cast at the /operations +// call site safe. +export function sectionsFor( + operation: SectionConfig["operation"], +): SectionConfig[] { + const mine = channelWorkSections().filter((s) => s.operation === operation); + return [ + ...mine.filter((s) => s.role === "work"), + ...mine.filter((s) => s.role === "attention"), + ]; +} diff --git a/editor/app/components/dashboard/ChannelsTable.tsx b/editor/app/components/dashboard/ChannelsTable.tsx @@ -4,7 +4,7 @@ import Link from "next/link"; import { useState, useTransition } from "react"; import { syncAction } from "../../channels/[slug]/pipelineActions"; import { prioritizeChannelDownloadAction } from "../../operations/actions"; -import { InlineActionButton } from "../../actionable/components/InlineActionButton"; +import { InlineActionButton } from "../actions/InlineActionButton"; import { fmtTime } from "../../widget/lib/relativeTime"; import type { DashboardChannel } from "./types"; diff --git a/editor/app/components/dashboard/NeedsWorkPanel.tsx b/editor/app/components/dashboard/NeedsWorkPanel.tsx @@ -2,7 +2,7 @@ import Link from "next/link"; import type { WidgetActionablePayload } from "../../api/widget/actionable/route"; -import { InlineActionButton } from "../../actionable/components/InlineActionButton"; +import { InlineActionButton } from "../actions/InlineActionButton"; const LIMIT = 10; diff --git a/editor/app/lib/actionable/actions.ts b/editor/app/lib/actionable/actions.ts @@ -0,0 +1,320 @@ +"use server"; + +import { revalidatePath } from "next/cache"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels"; +import { + excludedDownloadIdSet, + generateChannelSnapshot, +} from "yt-dlp-transcript-common/controller/channelSnapshot"; +import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; +import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand"; +import { drainStream } from "yt-dlp-transcript-common/jobs/drainStream"; +import { + detectDuplicateShorts, + readDuplicateReport, + updateDuplicateOverride, +} from "yt-dlp-transcript-common/controller/duplicateShorts"; +import { + scanCorruptMedia, + updateMediaScanOverride, +} from "yt-dlp-transcript-common/controller/scanCorruptMedia"; +import { mediaScanTotals } from "yt-dlp-transcript-common/lib/mediaScan"; +import { shareClusterFromCanonical } from "yt-dlp-transcript-common/controller/digestSharing"; +import { + clearIncompleteTranscriptsAction, + enableAutoRunners, + redownloadIncompleteBucketAction, +} from "../../channels/[slug]/incompleteTranscriptActions"; +import { loadActionableSummary } from "./loadActionable"; + +export type RefreshAllResult = { + queued: string[]; + skipped: { slug: string; reason: string }[]; +}; + +export type DuplicateScope = "shorts" | "all"; + +export type RunDuplicateDetectionResult = + | { ok: true; clusters: number; videosInClusters: number } + | { ok: false; error: string }; + +export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResult> { + const paths = getPaths(); + const channels = await listChannelConfigs(paths); + const active = new Set( + getRegistry() + .list() + .filter( + (j) => + j.kind === "refresh-report" && + (j.status === "queued" || j.status === "running") && + j.channelSlug, + ) + .map((j) => j.channelSlug as string), + ); + const queued: string[] = []; + const skipped: { slug: string; reason: string }[] = []; + const streams: ReadableStream<string>[] = []; + for (const c of channels) { + if (active.has(c.slug)) { + skipped.push({ slug: c.slug, reason: "already running" }); + continue; + } + const result = await runManagedFunction({ + kind: "refresh-report", + // Empty queueKey: bypass queue serialization. Snapshot regen is a + // local filesystem scan that never touches the platform, so there's + // no reason for it to wait behind sync/download work. See + // registry.ts:69-72 for the documented escape hatch. + queueKey: "", + paths, + channelSlug: c.slug, + fn: async (onLog) => { + onLog(`Regenerating report for ${c.slug}…`); + const snap = await generateChannelSnapshot(paths, c.slug); + const excluded = excludedDownloadIdSet(snap); + const awaitingTranscription = excluded.size + ? snap.buckets.downloadedNoTranscript.filter( + (id) => !excluded.has(id), + ).length + : snap.buckets.downloadedNoTranscript.length; + onLog( + `Done. ${snap.totals.videos} videos · ` + + `${snap.undownloadedIds.length} undownloaded · ` + + `${awaitingTranscription} awaiting transcription.`, + ); + // Deliberately no revalidatePath here — calling it from a + // background fn races with the in-flight re-render of /actionable + // that the action's own revalidatePath triggers. The action's + // single revalidate at the end picks up every fresh snapshot. + }, + }); + if (!result.ok) { + skipped.push({ slug: c.slug, reason: result.error }); + continue; + } + queued.push(c.slug); + streams.push(result.stream); + } + // Wait for all snapshots to finish writing before revalidating so the + // re-rendered /actionable reads fresh counts. With queueKey === "" the + // jobs all run in parallel, so this waits roughly the time of the + // slowest snapshot, not the sum. + await Promise.all(streams.map(drainStream)); + revalidatePath("/actionable"); + revalidatePath("/"); + return { queued, skipped }; +} + +// Runs the global cross-platform duplicate-shorts pass. `scope: "all"` removes +// the duration cutoff (a heavier one-off run that also surfaces clip-of-longer +// containment). Reads the cues + statsByPath written by build:index/build:stats, +// so a build must have run first for meaningful results. +export async function runDuplicateDetectionAction( + scope: DuplicateScope = "shorts", +): Promise<RunDuplicateDetectionResult> { + const paths = getPaths(); + let clusters = 0; + let videosInClusters = 0; + const result = await runManagedFunction({ + kind: "detect-duplicates", + // Empty queueKey: a local read-only scan, no reason to wait behind + // sync/download work (see refreshAllChannelSnapshotsAction). + queueKey: "", + paths, + fn: async (onLog, signal) => { + const report = await detectDuplicateShorts({ + paths, + thresholdSeconds: scope === "all" ? null : undefined, + onLog, + signal, + }); + clusters = report.totals.clusters; + videosInClusters = report.totals.videosInClusters; + }, + }); + if (!result.ok) return { ok: false, error: result.error }; + await drainStream(result.stream); + revalidatePath("/actionable"); + return { ok: true, clusters, videosInClusters }; +} + +// What a human can say about a cluster the detector could not decide. +// confirmed — "I looked; these really are the same video." Unblocks +// sharing AND publication for a title+duration suspect. +// not-duplicate — "They are not." Suppresses the cluster entirely. +// clear — undo, back to awaiting review. +export type DuplicateClusterDecision = "confirmed" | "not-duplicate" | "clear"; + +export type ReviewDuplicateClusterResult = + | { ok: true; shared: number; misaligned: number } + | { ok: false; error: string }; + +// Record a review decision for one cluster. +// +// `updateDuplicateOverride` has existed — with the `confirmed` flag, the +// read-modify-write, the atomic rename and the "an empty patch clears the +// decision" rule — since duplicate review was built, and until now **nothing in +// the repo called it**. `clusterMaySharePartial` fails closed, so every +// needsReview cluster shared nothing and there was no way for a human to change +// that. This is that missing caller. +// +// Confirming also attempts the share immediately, via the equally-uncalled +// `shareClusterFromCanonical`: the point of confirming is to let derived work +// flow, and making the operator wait for the canonical member's next sweep to +// find out whether it would have would make the button feel inert. It is a +// no-op when the canonical has no digest yet, which today is almost always. +export async function reviewDuplicateClusterAction( + clusterId: string, + decision: DuplicateClusterDecision, +): Promise<ReviewDuplicateClusterResult> { + const paths = getPaths(); + try { + const overrides = await updateDuplicateOverride( + paths, + clusterId, + decision === "confirmed" + ? { confirmed: true, notDuplicate: false } + : decision === "not-duplicate" + ? { notDuplicate: true, confirmed: false } + : { confirmed: false, notDuplicate: false }, + ); + + let shared = 0; + let misaligned = 0; + if (decision === "confirmed") { + const report = await readDuplicateReport(paths); + const cluster = report?.clusters.find((c) => c.clusterId === clusterId); + if (cluster) { + for (const outcome of await shareClusterFromCanonical(paths, cluster, { + overrides, + })) { + if (outcome.status === "shared") shared++; + else if (outcome.status === "misaligned") misaligned++; + } + } + } + + revalidatePath("/actionable"); + return { ok: true, shared, misaligned }; + } catch (e) { + return { ok: false, error: (e as Error)?.message ?? String(e) }; + } +} + +export type GlobalIncompleteResult = + | { ok: true; channels: number; affected: number } + | { ok: false; error: string }; + +// All channels with at least one truncated transcript right now. +async function affectedIncompleteSlugs(): Promise<string[]> { + const summary = await loadActionableSummary(getPaths()); + return summary.incompleteTranscripts.map((r) => r.channel.slug); +} + +// Clear & re-queue every truncated transcript across all channels, then enable +// the auto-runners once. Destructive — the caller confirms first. +export async function clearAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> { + const slugs = await affectedIncompleteSlugs(); + if (slugs.length === 0) { + return { ok: false, error: "No incomplete transcripts to clear." }; + } + let cleared = 0; + for (const slug of slugs) { + // Defer enabling the runners until the end so settings flip only once. + const r = await clearIncompleteTranscriptsAction(slug, undefined, { + enableRunners: false, + }); + cleared += r.succeeded; + } + await enableAutoRunners(); + revalidatePath("/actionable"); + return { ok: true, channels: slugs.length, affected: cleared }; +} + +// Queue one batch re-fix job per affected channel (fire-and-forget — the jobs +// keep running and surface in /jobs; cancel each stream so we don't hold them +// open). +export async function redownloadAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> { + const slugs = await affectedIncompleteSlugs(); + if (slugs.length === 0) { + return { ok: false, error: "No incomplete transcripts to fix." }; + } + let queued = 0; + for (const slug of slugs) { + const result = await redownloadIncompleteBucketAction(slug); + if (result.ok) { + void result.stream.cancel(); + queued++; + } + } + revalidatePath("/actionable"); + return { ok: true, channels: slugs.length, affected: queued }; +} + + +// --------------------------------------------------------------------------- +// Corrupt-media scan +// --------------------------------------------------------------------------- + +export type RunMediaScanResult = + | { ok: true; findings: number; filesScanned: number } + | { ok: false; error: string }; + +// Scan media already on disk. Reports; deletes nothing — deletion stays the +// separate, explicit per-file click that already exists on the video page. +// +// `deepProbe` turns on the tier-2 full decode for files whose duration looks +// wrong. Off by default because that tier is a real transcode per file: it has +// to be budgeted like the digest sweep, not like a stat() walk. +export async function runMediaScanAction( + opts: { channels?: string[]; deepProbe?: boolean } = {}, +): Promise<RunMediaScanResult> { + const paths = getPaths(); + let findings = 0; + let filesScanned = 0; + const perChannel = (opts.channels?.length ?? 0) > 0; + const result = await runManagedFunction({ + kind: perChannel ? "scan-media-channel" : "scan-media", + // Empty queueKey: local read-only disk work, no reason to wait behind + // sync/download work (see runDuplicateDetectionAction). + queueKey: "", + paths, + fn: async (onLog, signal) => { + const report = await scanCorruptMedia({ + paths, + channels: opts.channels, + deepProbe: opts.deepProbe, + onLog, + signal, + }); + findings = report.findings.length; + filesScanned = mediaScanTotals(report).filesScanned; + }, + }); + if (!result.ok) return { ok: false, error: result.error }; + await drainStream(result.stream); + revalidatePath("/actionable"); + return { ok: true, findings, filesScanned }; +} + +export type ReviewMediaFindingResult = + | { ok: true } + | { ok: false; error: string }; + +// "Looked at it; it is fine." Recorded in a sibling overrides file, because the +// report is regenerated wholesale by every scan and a decision written into it +// would be destroyed by the next run. +export async function reviewMediaFindingAction( + key: string, + reviewed: boolean, +): Promise<ReviewMediaFindingResult> { + try { + await updateMediaScanOverride(getPaths(), key, { reviewed }); + revalidatePath("/actionable"); + return { ok: true }; + } catch (e) { + return { ok: false, error: (e as Error).message }; + } +} diff --git a/editor/app/lib/actionable/loadActionable.ts b/editor/app/lib/actionable/loadActionable.ts @@ -0,0 +1,253 @@ +import type { Paths } from "yt-dlp-transcript-common/lib/paths"; +import { cache } from "react"; +import { + backfillLaneEntriesOf, + reachableOperationWork, +} from "yt-dlp-transcript-common/lib/operations"; +import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels"; +import { getChannelBriefs } from "../requestCache"; +import { + digestWorkOf, + excludedDownloadIdSet, + type ChannelSnapshot, +} from "yt-dlp-transcript-common/controller/channelSnapshot"; + +export type ActionableRow = { + channel: ChannelBrief; + // Always `channel.snapshot` — kept as a sibling field because every count + // helper and every consumer already reads `row.snapshot`. + snapshot: ChannelSnapshot | null; +}; + +export type ActionableSummary = { + rows: ActionableRow[]; + undownloaded: ActionableRow[]; + missingNeverFetched: ActionableRow[]; + untranscribed: ActionableRow[]; + incompleteTranscripts: ActionableRow[]; + shortAudio: ActionableRow[]; + cleanTranscribedAudio: ActionableRow[]; + cleanExtraFormats: ActionableRow[]; + staleOrMissing: ActionableRow[]; + digestWarnings: ActionableRow[]; + backfill: ActionableRow[]; +}; + +export function isStaleOrMissing(row: ActionableRow): boolean { + if (!row.snapshot) return true; + const synced = row.channel.config.lastSyncedAt; + if (!synced) return false; + return new Date(synced).getTime() > new Date(row.snapshot.generatedAt).getTime(); +} + +// Counts that drive the actionable lists exclude IDs that the availability +// check has flagged as deleted / members-only / private — those videos +// can't be acted on, so they shouldn't inflate "needs attention" totals. +// `undownloadedIds` is already filtered at snapshot generation time, but we +// apply the filter again so a stale snapshot can't surface excluded IDs. +function countActionable( + snapshot: ChannelSnapshot | null | undefined, + ids: readonly string[] | undefined, +): number { + if (!snapshot || !ids) return 0; + const excluded = excludedDownloadIdSet(snapshot); + if (excluded.size === 0) return ids.length; + let n = 0; + for (const id of ids) if (!excluded.has(id)) n++; + return n; +} + +export function actionableUndownloadedCount(row: ActionableRow): number { + return countActionable(row.snapshot, row.snapshot?.undownloadedIds); +} + +// Videos the roster says we were told about, never downloaded, and that have +// since left the listing. Deliberately NOT run through countActionable: the +// availability exclusions are keyed on videos we have on disk, and these have no +// dir at all. Default 0 for snapshots written before the bucket existed. +export function actionableMissingNeverFetchedCount(row: ActionableRow): number { + return row.snapshot?.missingNeverFetched?.length ?? 0; +} + +export function actionableUntranscribedCount(row: ActionableRow): number { + return countActionable( + row.snapshot, + row.snapshot?.buckets.downloadedNoTranscript, + ); +} + +// Transcribed videos whose transcript is badly truncated (the audio download +// stopped early). Default 0 for snapshots written before the bucket existed. +export function actionableIncompleteTranscriptCount(row: ActionableRow): number { + return row.snapshot?.buckets.incompleteTranscript?.length ?? 0; +} + +// Downloads the duration guard flagged as truncated at the source (short audio +// kept on disk, not transcribed). Default 0 for snapshots predating the bucket. +export function actionableShortAudioCount(row: ActionableRow): number { + return row.snapshot?.buckets.shortAudio?.length ?? 0; +} + +// Cleanup buckets are filtered by "do not clean" at snapshot-generation time, +// so the length is the actionable count directly (default undefined → 0 for +// snapshots written before the bucket existed). +export function actionableCleanTranscribedCount(row: ActionableRow): number { + return row.snapshot?.buckets.transcribedWithAudio?.length ?? 0; +} + +export function actionableCleanExtraFormatsCount(row: ActionableRow): number { + return row.snapshot?.buckets.multipleAudioFormats?.length ?? 0; +} + +// The digest layer's two work lists. The first is "has no digest at the current +// identity" — missing, stale or part-done. `digestWarnings` is "the model +// produced something a human should look at", which includes the total failures +// that write no section and so are invisible to any count of files. +// +// Read through digestWorkOf: the registry's classification is the one the runner +// uses, and it excludes videos with no transcript (blocked) and videos whose +// cues.json is stale (deferred) — work the old `noDigest` bucket offered here +// and the runner then declined. +export function actionableNoDigestCount(row: ActionableRow): number { + return digestWorkOf(row.snapshot).reachable; +} + +export function actionableDigestWarningsCount(row: ActionableRow): number { + return row.snapshot?.buckets.digestWarnings?.length ?? 0; +} + +// The backfill lane's two numbers, and they are two FUNCTIONS on purpose so no +// caller can accidentally add them. +// +// `reachable` (missing + stale) is what the lane can do today and the only thing +// that decides whether a channel appears in the section at all. `missingInput` +// is the population that needs its media re-acquired first — measured at ~91x +// the reachable count corpus-wide, so filtering on it would put every channel in +// the list forever. That is not a hypothetical: it is the documented reason +// /api/widget/actionable refuses to filter on the digest work count. +// backfillLaneEntriesOf, not Object.values: the snapshot map is every catalog operation +// now, and digest is one of them. These two functions decide whether a channel +// appears in the BACKFILL section at all, so folding a ~75,000-video operation +// that runs on another queue into them would put every channel in the list +// forever — the same trap /api/widget/actionable documents for the digest work +// count, hit from the other direction. +export function actionableBackfillCount(row: ActionableRow): number { + return backfillLaneEntriesOf(row.snapshot?.backfill).reduce( + (n, e) => n + reachableOperationWork(e), + 0, + ); +} + +export function actionableBackfillMissingInputCount( + row: ActionableRow, +): number { + return backfillLaneEntriesOf(row.snapshot?.backfill).reduce( + (n, e) => n + e.missingInput, + 0, + ); +} + +// Estimated bytes each cleanup would reclaim (default 0 for snapshots written +// before cleanupBytes existed). +export function actionableCleanTranscribedBytes(row: ActionableRow): number { + return row.snapshot?.cleanupBytes?.transcribedWithAudio ?? 0; +} + +export function actionableCleanExtraFormatsBytes(row: ActionableRow): number { + return row.snapshot?.cleanupBytes?.multipleAudioFormats ?? 0; +} + +export async function loadActionableSummary( + paths: Paths, +): Promise<ActionableSummary> { + const channels = await getChannelBriefs(paths); + // The brief already read the snapshot; this used to read each one a second + // time on top of a full corpus walk. + const rows: ActionableRow[] = channels.map((channel) => ({ + channel, + snapshot: channel.snapshot, + })); + + const undownloaded = rows + .filter((r) => actionableUndownloadedCount(r) > 0) + .sort( + (a, b) => actionableUndownloadedCount(b) - actionableUndownloadedCount(a), + ); + + const missingNeverFetched = rows + .filter((r) => actionableMissingNeverFetchedCount(r) > 0) + .sort( + (a, b) => + actionableMissingNeverFetchedCount(b) - + actionableMissingNeverFetchedCount(a), + ); + + const untranscribed = rows + .filter((r) => actionableUntranscribedCount(r) > 0) + .sort( + (a, b) => actionableUntranscribedCount(b) - actionableUntranscribedCount(a), + ); + + const incompleteTranscripts = rows + .filter((r) => actionableIncompleteTranscriptCount(r) > 0) + .sort( + (a, b) => + actionableIncompleteTranscriptCount(b) - + actionableIncompleteTranscriptCount(a), + ); + + const shortAudio = rows + .filter((r) => actionableShortAudioCount(r) > 0) + .sort((a, b) => actionableShortAudioCount(b) - actionableShortAudioCount(a)); + + const cleanTranscribedAudio = rows + .filter((r) => actionableCleanTranscribedCount(r) > 0) + .sort( + (a, b) => + actionableCleanTranscribedCount(b) - actionableCleanTranscribedCount(a), + ); + + const cleanExtraFormats = rows + .filter((r) => actionableCleanExtraFormatsCount(r) > 0) + .sort( + (a, b) => + actionableCleanExtraFormatsCount(b) - + actionableCleanExtraFormatsCount(a), + ); + + const digestWarnings = rows + .filter((r) => actionableDigestWarningsCount(r) > 0) + .sort( + (a, b) => + actionableDigestWarningsCount(b) - actionableDigestWarningsCount(a), + ); + + const backfill = rows + .filter((r) => actionableBackfillCount(r) > 0) + .sort((a, b) => actionableBackfillCount(b) - actionableBackfillCount(a)); + + const staleOrMissing = rows + .filter(isStaleOrMissing) + .sort((a, b) => a.channel.slug.localeCompare(b.channel.slug)); + + return { + rows, + undownloaded, + missingNeverFetched, + untranscribed, + incompleteTranscripts, + shortAudio, + cleanTranscribedAudio, + cleanExtraFormats, + staleOrMissing, + digestWarnings, + backfill, + }; +} + +// Per-request memoized. The dashboard renders this summary and several +// components derived from it in one pass; see ../requestCache for why a +// request is the only cache lifetime this data can safely have. +export const getActionableSummary = cache((paths: Paths) => + loadActionableSummary(paths), +); diff --git a/editor/app/lib/requestCache.ts b/editor/app/lib/requestCache.ts @@ -26,15 +26,15 @@ import { getSettings } from "yt-dlp-transcript-common/lib/settings"; // call. Pass it straight through; don't spread or rebuild it. // // (loadActionableSummary gets the same treatment, but its cached wrapper lives -// beside it in ../actionable/lib/loadActionable — this module deliberately +// beside it in ./actionable/loadActionable — this module deliberately // imports nothing from app/ so those loaders can import IT.) export const getChannelBriefs = cache((paths: Paths) => listChannelBriefs(paths), ); -// The duplicates report is a 6.7 MB JSON parse. It measures at ~65 ms, so this -// is tidiness rather than a headline win — but /actionable and the dashboard -// both want it in one render. +// The duplicates report is a 6.7 MB JSON parse. It measures at ~65 ms. Only +// /review reads it now — the dashboard and the widget poll used to pay for it +// through loadActionableSummary and never showed a cluster. export const getDuplicateReport = cache((paths: Paths) => readDuplicateReport(paths), ); diff --git a/editor/app/lib/review/loadReview.ts b/editor/app/lib/review/loadReview.ts @@ -0,0 +1,47 @@ +import type { Paths } from "yt-dlp-transcript-common/lib/paths"; +import { readDuplicateOverrides } from "yt-dlp-transcript-common/controller/duplicateShorts"; +import type { + DuplicateOverrides, + DuplicateReport, +} from "yt-dlp-transcript-common/lib/duplicates"; +import type { + MediaScanOverrides, + MediaScanReport, +} from "yt-dlp-transcript-common/lib/mediaScan"; +import { + readMediaScanOverrides, + readMediaScanReport, +} from "yt-dlp-transcript-common/controller/scanCorruptMedia"; +import { getDuplicateReport } from "../requestCache"; + +// The half of the old actionable summary that is a HUMAN judgement rather than +// a lane's work: findings someone decides about, with the decisions kept beside +// them. Split out of loadActionable because nothing but the review page ever +// read these four — and the dashboard and the widget poll were paying the +// 6.7 MB duplicates parse on every render for a list they never showed. +export type ReviewSummary = { + duplicates: DuplicateReport | null; + // The human decisions kept alongside the report — a cluster's canonical + // choice, "not a duplicate", and the `confirmed` flag that is the only thing + // letting a needsReview cluster share derived work. Loaded here because the + // review UI cannot show what has already been decided without it, and a + // review queue that forgets its own answers re-asks every question. + duplicateOverrides: DuplicateOverrides; + // The corrupt-media scan, and the "reviewed, this one's fine" decisions kept + // beside it. Null when the scan has never been run — which is NOT the same as + // "nothing is wrong", and the section says so rather than rendering an + // all-clear it has no evidence for. + mediaScan: MediaScanReport | null; + mediaScanOverrides: MediaScanOverrides; +}; + +export async function loadReviewSummary(paths: Paths): Promise<ReviewSummary> { + const [duplicates, duplicateOverrides, mediaScan, mediaScanOverrides] = + await Promise.all([ + getDuplicateReport(paths), + readDuplicateOverrides(paths), + readMediaScanReport(paths), + readMediaScanOverrides(paths), + ]); + return { duplicates, duplicateOverrides, mediaScan, mediaScanOverrides }; +} diff --git a/editor/app/page.tsx b/editor/app/page.tsx @@ -12,7 +12,7 @@ import { actionableUndownloadedCount, actionableUntranscribedCount, getActionableSummary, -} from "./actionable/lib/loadActionable"; +} from "./lib/actionable/loadActionable"; import { resolveActiveSite } from "./lib/activeSite"; import { buildActiveJobsPayload } from "./jobs/active/buildActiveJobs"; import { buildWorkersPayload } from "./workers/buildWorkers"; @@ -20,7 +20,7 @@ import { buildWidgetSyncPayload } from "./api/widget/sync/route"; import { DashboardCockpit } from "./components/dashboard/DashboardCockpit"; import type { DashboardChannel } from "./components/dashboard/types"; import type { WidgetActionablePayload } from "./api/widget/actionable/route"; -import type { ActionableRow } from "./actionable/lib/loadActionable"; +import type { ActionableRow } from "./lib/actionable/loadActionable"; export const dynamic = "force-dynamic"; diff --git a/editor/app/widget/components/MonitorWidget.tsx b/editor/app/widget/components/MonitorWidget.tsx @@ -15,7 +15,7 @@ import type { } from "yt-dlp-transcript-common/jobs/registry"; import { jobKindLabel } from "../../jobs/jobKindLabels"; import type { WorkersPayload, WorkerView } from "../../workers/components/WorkersView"; -import { InlineActionButton } from "../../actionable/components/InlineActionButton"; +import { InlineActionButton } from "../../components/actions/InlineActionButton"; import type { WidgetActionablePayload } from "../../api/widget/actionable/route"; import type { CleanablePayload,