Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d32e6b2e726ab50d56b8c3b6620f16460fc9ea3d
parent fd2120dd6e8324b9bd6210c7414511005f328c90
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 29 Jun 2026 01:47:42 -0400

Detect & flag truncated transcripts in the editor

When an audio download silently stops early (yt-dlp exits "ok",
download-outcome.json records success), whisper transcribes only the few
minutes that landed — e.g. a 2h22m video ends with a ~7-minute transcript
and nothing warns you. Detect this via transcript coverage (last cue end ÷
duration) and surface it for re-download throughout the editor.

- common/lib/transcriptCoverage.ts: pure helper + named thresholds
  (>=10min, <50% coverage, livestreams/empty excluded), the single source
  of truth. readTranscriptCoverage() reads each video's transcript.cues.json
  so the existing corpus is flagged with no migration.
- channelSnapshot: new incompleteTranscript bucket.
- channel video list: "Incomplete transcript" filter chip + amber
  transcribed-dot glyph; composes with the Transcribed chip.
- video page: warning banner ("covers 6:52 of 2:22:21 (4.8%)…") with a
  one-click Re-download & re-transcribe button. The fix action deletes the
  truncated audio first, then re-downloads and re-transcribes — re-running
  whisper alone would just reproduce the short transcript.
- /actionable: aggregates affected channels in a new section.
- e2e: incomplete-transcript.spec.ts covers flagged/not-flagged fixtures
  (truncated, full, short, livestream) through the list, panel, actionable.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelSnapshot.ts | 30+++++++++++++++++++++++++++++-
Mcommon/controller/normalizeTranscript.ts | 20++++++++++++++++++++
Acommon/lib/transcriptCoverage.ts | 54++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/actionable/lib/loadActionable.ts | 16++++++++++++++++
Meditor/app/actionable/page.tsx | 23+++++++++++++++++++++++
Meditor/app/channels/[slug]/components/VideoListPane.tsx | 18++++++++++++++----
Meditor/app/channels/[slug]/lib/stageStatus.ts | 1+
Meditor/app/channels/[slug]/lib/videoRows.ts | 8++++++++
Meditor/app/channels/[slug]/lib/videoRowsServer.ts | 2++
Meditor/app/channels/[slug]/page.tsx | 27+++++++++++++++++++++------
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 101++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 14++++++++++++++
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 61+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/incomplete-transcript.spec.ts | 128+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
15 files changed, 490 insertions(+), 14 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -28,6 +28,8 @@ import { loadFailedTranscriptions } from "./failedTranscriptions"; import { readChannelConfig } from "./channels"; import { computeKeptVideoIds } from "./keptVideos"; import { loadMaybeMissing } from "./quickAvailabilityCheck"; +import { readTranscriptCoverage } from "./normalizeTranscript"; +import { isIncompleteTranscript } from "../lib/transcriptCoverage"; export type AvailabilitySnapshot = { byStatus: Record<Availability, string[]>; @@ -80,6 +82,13 @@ export type ChannelSnapshot = { // declined as currently-live/upcoming and will be retried on a later sync. // Optional: older snapshots lack it; readers must default to []. skippedByFilter: string[]; + // Videos whose transcript covers only a small fraction of the video's + // duration — the audio download silently truncated (yt-dlp exited "ok") so + // whisper transcribed just the first few minutes. The detection threshold + // lives in ../lib/transcriptCoverage (isIncompleteTranscript). Surfaced so + // the user can re-download/re-transcribe. Optional: older snapshots lack it; + // readers must default to []. + incompleteTranscript: string[]; }; undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; @@ -274,6 +283,12 @@ export async function generateChannelSnapshot( const effectiveAvailability = await resolveEffectiveAvailability(dir); const doNotClean = await isDoNotClean(dir); const outcome = await loadDownloadOutcome(dir); + // Only transcribed (non-untranscribable) dirs can have a truncated + // transcript; skip the cues.json read for everything else. + const coverage = + isVideoTranscribed(files) && !files.isUntranscribable + ? await readTranscriptCoverage(dir) + : null; return { id, files, @@ -283,6 +298,7 @@ export async function generateChannelSnapshot( effectiveAvailability, doNotClean, outcome, + coverage, }; }), ), @@ -337,15 +353,26 @@ export async function generateChannelSnapshot( const corruptSource: string[] = []; const nonStandardVtt: string[] = []; const skippedByFilter: string[] = []; + const incompleteTranscript: string[] = []; let transcribedWithAudioBytes = 0; let multipleAudioFormatsBytes = 0; let foreignAudioBytes = 0; let transcribed = 0; let downloaded = 0; - for (const { id, files, audioSizes, outcome } of perVideo) { + for (const { id, files, audioSizes, outcome, coverage } of perVideo) { if (isVideoTranscribed(files)) transcribed++; if (isVideoDownloaded(files)) downloaded++; if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); + // A transcribed video whose cues stop far short of its duration — the audio + // download truncated silently. Threshold lives in transcriptCoverage. + if ( + coverage && + isIncompleteTranscript(coverage.cov, { + isLivestream: coverage.isLivestream, + }) + ) { + incompleteTranscript.push(id); + } // A video the filter declined (e.g. live/upcoming) that hasn't since been // downloaded. Once it lands an artifact it drops out of this bucket. if ( @@ -505,6 +532,7 @@ export async function generateChannelSnapshot( corruptSource: corruptSource.sort(), nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), + incompleteTranscript: incompleteTranscript.sort(), }, undownloadedIds, excludedFromDownload, diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts @@ -14,6 +14,10 @@ import type { TranscriptOutputFormat } from "../lib/transcriptionApps"; import { summarize, type RawMetadata } from "../lib/transcripts-server"; import type { TranscriptDetail } from "../lib/transcripts"; import { + transcriptCoverage, + type TranscriptCoverage, +} from "../lib/transcriptCoverage"; +import { CUES_JSON_FILENAME, META_FILENAME, VTT_FILENAME, @@ -165,6 +169,22 @@ export async function readNormalizedTranscript( } } +// Read transcript coverage (last cue end vs. duration) for a video dir from its +// transcript.cues.json. Returns null when there's no normalized transcript. +// Used by the snapshot builder and the editor pages to flag truncated downloads +// — the shared math lives in ../lib/transcriptCoverage. +export async function readTranscriptCoverage( + videoDir: string, +): Promise<{ cov: TranscriptCoverage; isLivestream: boolean } | null> { + const cuesPath = path.join(videoDir, CUES_JSON_FILENAME); + const t = await readNormalizedTranscript(cuesPath); + if (!t) return null; + return { + cov: transcriptCoverage(t.cues, t.duration), + isLivestream: Boolean(t.isLivestream), + }; +} + // Helper: given a video dir, decide whether transcript.cues.json (if present) // is at least as new as metadata.info.json and the raw transcript file. Used // by buildIndex to know whether it can trust cues.json without re-parsing. diff --git a/common/lib/transcriptCoverage.ts b/common/lib/transcriptCoverage.ts @@ -0,0 +1,54 @@ +// Detect badly truncated transcripts: when the audio download silently stopped +// early (yt-dlp exits 0, download-outcome "ok"), whisper transcribes only the +// few minutes that landed, so the transcript covers a tiny fraction of the real +// runtime. We catch that by comparing the last cue's end time against the +// recorded video duration. +// +// This module is pure (no fs) and is the single source of truth for the +// threshold — the snapshot builder, the editor pages, and buildStats all call +// it so the definition never drifts. + +import type { Cue } from "./vtt"; + +// A flagged video must be this long; short videos with a few cues are normal. +export const INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC = 600; // 10 min +// Flag when the transcript covers less than this fraction of the duration. +export const INCOMPLETE_TRANSCRIPT_MAX_COVERAGE = 0.5; // < 50% covered + +export type TranscriptCoverage = { + lastCueEnd: number; // max cue.end in seconds; 0 when there are no cues + duration: number; // recorded video duration in seconds + // lastCueEnd / duration. null when duration is unknown/zero (can't judge). + // May exceed 1 when a cue end runs slightly past the recorded duration. + coverage: number | null; +}; + +export function transcriptCoverage( + cues: ReadonlyArray<Pick<Cue, "end">> | undefined, + duration: number, +): TranscriptCoverage { + let lastCueEnd = 0; + if (cues) { + for (const c of cues) { + const end = c.end ?? 0; + if (end > lastCueEnd) lastCueEnd = end; + } + } + const coverage = duration > 0 ? lastCueEnd / duration : null; + return { lastCueEnd, duration, coverage }; +} + +export function isIncompleteTranscript( + cov: TranscriptCoverage, + opts?: { isLivestream?: boolean }, +): boolean { + // Livestream durations are unreliable (the recorded length often doesn't match + // the captured stream), so don't judge coverage for them. + if (opts?.isLivestream) return false; + if (cov.duration < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC) return false; + // No cues at all (lastCueEnd === 0) is an empty/untranscribable marker, not a + // truncation — leave it to the untranscribable path rather than flag it here. + if (cov.lastCueEnd <= 0) return false; + if (cov.coverage === null) return false; + return cov.coverage < INCOMPLETE_TRANSCRIPT_MAX_COVERAGE; +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Truncated transcripts are now detected and flagged for re-download.** When an audio download silently stops early (yt-dlp exits `ok`, `download-outcome.json` records success), whisper transcribes only the few minutes that landed — so a 2h22m video ends up with a ~7-minute transcript and nothing warns you. A new coverage check (last cue end ÷ video duration) flags any non-livestream video ≥10min whose transcript covers <50% of its runtime. The single source of truth is `common/lib/transcriptCoverage.ts` (`transcriptCoverage` + `isIncompleteTranscript`, with named thresholds), read from each video's `transcript.cues.json` so the existing corpus is flagged with no migration. Surfaced everywhere: a new **`incompleteTranscript`** channel-snapshot bucket → an **"Incomplete transcript"** filter chip and an **amber transcribed-dot** in the per-channel video list; a warning banner on the video page ("Transcript covers 6:52 of 2:22:21 (4.8%)…") with a one-click **Re-download & re-transcribe** button; and an **"Channels with incomplete (truncated) transcripts"** section on `/actionable`. The fix action (`redownloadIncompleteTranscriptAction`) deletes the truncated audio first, then re-downloads and re-transcribes — re-running whisper alone would just reproduce the short transcript. See `common/controller/channelSnapshot.ts`, `editor/app/channels/[slug]/{lib/videoRows.ts,lib/videoRowsServer.ts,lib/stageStatus.ts,components/VideoListPane.tsx,videos/[id]/{components/VideoPanel.tsx,videoActions.ts,page.tsx},page.tsx}`, and `editor/app/actionable/{lib/loadActionable.ts,page.tsx}`. - **Auto-queue rules with no bucket now draw from *all* of a runner's buckets, and auto-download can resume partial downloads.** A policy-tree rule left at the **"all buckets (default)"** setting (previously just labeled *default*) now draws from the **union** of every bucket that runner kind tracks — deduped, in priority order — instead of only the single primary bucket. This fixes channels (e.g. an Odysee channel mid-download) that quietly stopped being auto-downloaded once their remaining work drifted entirely into **partially-downloaded** videos: those have a `.part` file but no completed audio, so they live in the `partialDownloads` bucket and were **absent from `undownloadedIds`** — the only bucket auto-download used to load. The download runner now loads `partialDownloads` alongside `undownloadedIds` (partials first, so in-progress downloads resume via `downloadOneManaged` before fresh ones start), and exposes `partialDownloads` as a selectable bucket in the policy editor so you can dedicate a high-priority rule to resuming partials. The per-kind bucket lists are consolidated behind a single `bucketsForKind` source of truth shared by the runner, the per-rule pending-count helper, and the editor's bucket picker (so they can't drift). Note: a *bucketless* auto-transcribe rule now also drains `failedListed` after `downloadedNoTranscript` (it already loaded both); platform rate-limit backoff is unchanged and remains an independent reason a throttled platform may pause. See `common/jobs/autoQueuePolicy.ts` (`buildPendingByLeaf` + `bucketsForKind` + unit tests), `common/controller/autoRunner.ts`, and `editor/app/auto-queue/{page.tsx,components/PolicyTreeEditor.tsx}`. - **Hub homepage redesigned into a cross-site landing; the homepage page-creator is removed.** The hub's home page is now a single mobile-first cross-site landing (headline KPIs and one stacked activity chart with Metric [Transcribed/Downloaded] · Breakdown [By site/By channel] · Bucket [Week/Month/Cumulative] · Range [90d/12mo/All] · Display [Share/Counts] controls, plus a metric-aware site-links grid with sparklines and a "#1 this month" badge), built from a small `homepage-summary.json` pre-computed by `compose-homepage`. The separate `/stats` dashboard route folds into it. Consequently the hub's **Markdown-pages subsystem is dropped**: **Manage → Homepage** now edits only branding (the Pages list, New-page, and the page editor are gone), and the homepage config no longer carries a `nav`. The page server actions (`saveHomepagePageAction`/`deleteHomepagePageAction`), `editor/app/homepage/pages/*`, `PageEditor.tsx`, `common/lib/{homepagePages,homepageConstants}.ts`, and `paths.homepagePagesDir` are removed. See `editor/app/homepage/{page.tsx,actions.ts}`, `common/bin/compose-homepage.ts`, `common/lib/{homepageSummary,homepageChart}.ts`, and the `homepage/` package. (Re-addable later if needed.) - **One-click Retry for failed jobs (plus "Retry all failed").** A failed job that carries a replay descriptor (any bookmarkable kind — sync, download-missing, transcribe-all, retry-bucket, …) now shows a **Retry** button on the Jobs history table, and the page header gains a **Retry all failed** button whenever at least one such job is listed. Retry re-runs the job from its stored spec exactly like a bookmark re-run (so bucket jobs re-derive from the channel's *current* state), and the re-run **jumps ahead of other queued work** (it's promoted to the front of its queue, reusing the new reorder machinery) so a fix-and-retry runs next rather than at the back of the line. The spec is resolved from the live registry or, for an evicted/archived job, from its on-disk `<id>.meta.json` sidecar — so even a failure the 100-job cap has dropped is still retryable. Kinds with no replay descriptor (e.g. `import-one`) intentionally offer no Retry. See `editor/app/jobs/actions.ts` (`retryJobAction` / `retryAllFailedAction`), the new `RetryJobButton` / `RetryAllFailedButton`, and `editor/e2e/jobs-retry.spec.ts`. diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -20,6 +20,7 @@ export type ActionableSummary = { rows: ActionableRow[]; undownloaded: ActionableRow[]; untranscribed: ActionableRow[]; + incompleteTranscripts: ActionableRow[]; cleanTranscribedAudio: ActionableRow[]; cleanExtraFormats: ActionableRow[]; staleOrMissing: ActionableRow[]; @@ -61,6 +62,12 @@ export function actionableUntranscribedCount(row: ActionableRow): number { ); } +// Transcribed videos whose transcript is badly truncated (the audio download +// stopped early). Default 0 for snapshots written before the bucket existed. +export function actionableIncompleteTranscriptCount(row: ActionableRow): number { + return row.snapshot?.buckets.incompleteTranscript?.length ?? 0; +} + // Cleanup buckets are filtered by "do not clean" at snapshot-generation time, // so the length is the actionable count directly (default undefined → 0 for // snapshots written before the bucket existed). @@ -108,6 +115,14 @@ export async function loadActionableSummary( (a, b) => actionableUntranscribedCount(b) - actionableUntranscribedCount(a), ); + const incompleteTranscripts = rows + .filter((r) => actionableIncompleteTranscriptCount(r) > 0) + .sort( + (a, b) => + actionableIncompleteTranscriptCount(b) - + actionableIncompleteTranscriptCount(a), + ); + const cleanTranscribedAudio = rows .filter((r) => actionableCleanTranscribedCount(r) > 0) .sort( @@ -131,6 +146,7 @@ export async function loadActionableSummary( rows, undownloaded, untranscribed, + incompleteTranscripts, cleanTranscribedAudio, cleanExtraFormats, staleOrMissing, diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -7,6 +7,7 @@ import { actionableCleanExtraFormatsCount, actionableCleanTranscribedBytes, actionableCleanTranscribedCount, + actionableIncompleteTranscriptCount, actionableUndownloadedCount, actionableUntranscribedCount, loadActionableSummary, @@ -46,6 +47,7 @@ export default async function ActionablePage() { const nothingPending = summary.undownloaded.length === 0 && summary.untranscribed.length === 0 && + summary.incompleteTranscripts.length === 0 && summary.cleanTranscribedAudio.length === 0 && summary.cleanExtraFormats.length === 0 && summary.staleOrMissing.length === 0; @@ -91,6 +93,27 @@ export default async function ActionablePage() { }, { config: { + id: "incomplete-transcripts", + title: "Channels with incomplete (truncated) transcripts", + description: + "Transcribed videos whose transcript covers only a small fraction of the runtime — the audio download stopped early. Open the channel (filtered) to re-download & re-transcribe the affected videos.", + countLabel: "truncated", + emptyLabel: "None detected.", + getCount: actionableIncompleteTranscriptCount, + primaryAction: (r) => ( + <Link + href={`/channels/${r.channel.slug}?filter=incomplete_transcript`} + aria-label={`review incomplete transcripts for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-zinc-300 dark:border-zinc-700 text-xs font-medium hover:bg-zinc-100 dark:hover:bg-zinc-800 whitespace-nowrap" + > + Review + </Link> + ), + }, + rows: summary.incompleteTranscripts, + }, + { + config: { id: "clean-transcribed-audio", title: "Channels with cleanable transcribed audio", description: diff --git a/editor/app/channels/[slug]/components/VideoListPane.tsx b/editor/app/channels/[slug]/components/VideoListPane.tsx @@ -58,6 +58,7 @@ const FILTER_OPTIONS: { value: VideoFilter; label: string }[] = [ { value: "no_audio", label: "No audio" }, { value: "downloaded_no_transcript", label: "No transcript" }, { value: "partial", label: "Partial" }, + { value: "incomplete_transcript", label: "Incomplete transcript" }, { value: "untranscribable", label: "Untranscribable" }, { value: "running", label: "Running" }, { value: "transcribed", label: "Transcribed" }, @@ -594,9 +595,11 @@ function StatusGlyphs({ row }: { row: VideoRow }) { : "bg-zinc-300 dark:bg-zinc-700"; const trColor = row.untranscribable ? "bg-zinc-400 dark:bg-zinc-600" - : row.transcribed - ? "bg-emerald-500" - : "bg-zinc-300 dark:bg-zinc-700"; + : row.incompleteTranscript + ? "bg-amber-500" + : row.transcribed + ? "bg-emerald-500" + : "bg-zinc-300 dark:bg-zinc-700"; const failureColor = row.failedTranscription || row.failedTranscoding ? "bg-red-500" : null; return ( @@ -609,7 +612,14 @@ function StatusGlyphs({ row }: { row: VideoRow }) { className={`w-2 h-2 rounded-full ${dlColor}`} /> <span title="transcoded" className={`w-2 h-2 rounded-full ${tcColor}`} /> - <span title="transcribed" className={`w-2 h-2 rounded-full ${trColor}`} /> + <span + title={ + row.incompleteTranscript + ? "transcript truncated (covers a fraction of the video — re-download)" + : "transcribed" + } + className={`w-2 h-2 rounded-full ${trColor}`} + /> {failureColor && ( <span title="failure recorded" diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -27,6 +27,7 @@ export function normalizeBuckets( corruptSource: raw?.corruptSource ?? [], nonStandardVtt: raw?.nonStandardVtt ?? [], skippedByFilter: raw?.skippedByFilter ?? [], + incompleteTranscript: raw?.incompleteTranscript ?? [], }; } diff --git a/editor/app/channels/[slug]/lib/videoRows.ts b/editor/app/channels/[slug]/lib/videoRows.ts @@ -29,6 +29,10 @@ export type VideoRow = { wrongFormatAudio: boolean; // Members-only / deleted / private — listed in the channel but unactionable. excluded: boolean; + // Transcribed, but the transcript covers only a small fraction of the video's + // duration — the audio download truncated silently. Independent flag (not a + // `status`) so it composes with `transcribed`. See transcriptCoverage. + incompleteTranscript: boolean; running: boolean; status: VideoRowStatus; }; @@ -47,6 +51,7 @@ export type VideoFilter = | "downloaded_no_transcript" | "transcribed" | "partial" + | "incomplete_transcript" | "untranscribable" | "running"; @@ -57,6 +62,7 @@ const VIDEO_FILTERS: readonly VideoFilter[] = [ "downloaded_no_transcript", "transcribed", "partial", + "incomplete_transcript", "untranscribable", "running", ]; @@ -85,6 +91,8 @@ function matchesFilter(r: VideoRow, filter: VideoFilter): boolean { return r.transcribed; case "partial": return r.partial; + case "incomplete_transcript": + return r.incompleteTranscript; case "untranscribable": return r.untranscribable; case "running": diff --git a/editor/app/channels/[slug]/lib/videoRowsServer.ts b/editor/app/channels/[slug]/lib/videoRowsServer.ts @@ -40,6 +40,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { const multipleAudioFormats = new Set(buckets.multipleAudioFormats); const untranscribable = new Set(buckets.untranscribable); const partial = new Set(buckets.partialDownloads); + const incompleteTranscript = new Set(buckets.incompleteTranscript); const corruptSourceSet = new Set(buckets.corruptSource); const failedTranscription = new Set(input.failedTranscriptionIds); const failedTranscoding = new Set(input.failedTranscodingIds); @@ -96,6 +97,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { wrongFormatAudio: untranscoded.has(id) || multipleAudioFormats.has(id), excluded, + incompleteTranscript: incompleteTranscript.has(id), running: runningIds.has(id), status, }); diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -26,6 +26,8 @@ import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcom import { loadAvailability } from "yt-dlp-transcript-common/lib/availability-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { resolvePrimaryVtt } from "yt-dlp-transcript-common/lib/videoStatus"; +import { readTranscriptCoverage } from "yt-dlp-transcript-common/controller/normalizeTranscript"; +import { isIncompleteTranscript } from "yt-dlp-transcript-common/lib/transcriptCoverage"; import { platformQueueKey, queueKeyForUrl, @@ -352,13 +354,25 @@ export default async function ChannelDetailPage({ let selectedPanel: ReactNode = null; let selectedTitle: string | null = null; if (selectedVideoId) { - const [dirData, title, outcome, availabilityRecord] = await Promise.all([ - loadVideoDir(channelDataDir, selectedVideoId), - loadVideoTitle(channelDataDir, selectedVideoId), - loadDownloadOutcome(path.join(channelDataDir, selectedVideoId)), - loadAvailability(path.join(channelDataDir, selectedVideoId)), - ]); + const [dirData, title, outcome, availabilityRecord, cov] = + await Promise.all([ + loadVideoDir(channelDataDir, selectedVideoId), + loadVideoTitle(channelDataDir, selectedVideoId), + loadDownloadOutcome(path.join(channelDataDir, selectedVideoId)), + loadAvailability(path.join(channelDataDir, selectedVideoId)), + readTranscriptCoverage(path.join(channelDataDir, selectedVideoId)), + ]); selectedTitle = title; + const coverage = cov + ? { + lastCueEnd: cov.cov.lastCueEnd, + duration: cov.cov.duration, + coverage: cov.cov.coverage, + incomplete: isIncompleteTranscript(cov.cov, { + isLivestream: cov.isLivestream, + }), + } + : null; const prevRow = selectedIndex > 0 ? orderedRows[selectedIndex - 1] : undefined; const nextRow = @@ -377,6 +391,7 @@ export default async function ChannelDetailPage({ downloadOutcome={outcome} availabilityHistory={availabilityRecord?.history ?? []} channelAudioFormat={config.audioFormat} + coverage={coverage} prevHref={prevRow ? buildVideoHref(prevRow.id) : undefined} nextHref={nextRow ? buildVideoHref(nextRow.id) : undefined} position={ diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -11,7 +11,7 @@ import { import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome"; import type { SavedVideoPointer } from "yt-dlp-transcript-common/lib/savedVideo"; import type { AvailabilityHistoryEntry } from "yt-dlp-transcript-common/lib/availability"; -import { formatBytes } from "yt-dlp-transcript-common/lib/format"; +import { formatBytes, formatDuration } from "yt-dlp-transcript-common/lib/format"; import { QueueControl } from "../../../../../components/QueueControl"; import { cancelJobAction } from "../../../../../jobs/actions"; import { PipelineStageCard } from "../../../components/PipelineStageCard"; @@ -23,6 +23,7 @@ import { downloadVideoPipelineAction, markVideoUntranscribableAction, setPrimaryTranscriptAction, + redownloadIncompleteTranscriptAction, redownloadToArchiveAction, toggleDoNotCleanAction, transcodeAudioAction, @@ -62,6 +63,16 @@ type Props = { // This video's saved-video pointer when its source container is persisted to // the store, else null. Drives the Source-video persistence card (Phase 5). savedVideo?: SavedVideoPointer | null; + // Transcript coverage vs. video duration. When `incomplete` the transcript + // covers only a fraction of the runtime (truncated audio download) — surfaced + // as a warning banner with a re-download/re-transcribe action. Absent when + // there's no transcript or no duration to judge against. + coverage?: { + lastCueEnd: number; + duration: number; + coverage: number | null; + incomplete: boolean; + } | null; prevHref?: string; nextHref?: string; position?: { index: number; total: number }; @@ -139,6 +150,7 @@ export function VideoPanel({ channelAudioFormat, doNotClean = false, savedVideo = null, + coverage = null, prevHref, nextHref, position, @@ -156,6 +168,7 @@ export function VideoPanel({ const activeTranscript = hasTranscriptJson ? WHISPER_FILENAME : primaryVtt; const noAudio = audioFiles.length === 0; const downloadFailed = downloadOutcome?.status === "failed"; + const incompleteTranscript = coverage?.incomplete ?? false; const downloadSummary = noAudio ? handling === "youtube" @@ -191,6 +204,15 @@ export function VideoPanel({ downloadFailed={downloadFailed} /> {downloadOutcome && <DownloadOutcomeBadge outcome={downloadOutcome} />} + {incompleteTranscript && coverage && ( + <IncompleteTranscriptBanner + slug={slug} + videoId={videoId} + coverage={coverage} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> + )} <PipelineStageCard id="availability-history" title="Availability history" @@ -218,7 +240,15 @@ export function VideoPanel({ title={noAudio ? "Download audio" : "Redownload audio"} summary={downloadSummary} defaultOpen={true} - tone={noAudio ? "attention" : downloadFailed ? "danger" : "neutral"} + tone={ + noAudio + ? "attention" + : downloadFailed + ? "danger" + : incompleteTranscript + ? "attention" + : "neutral" + } > <RedownloadSection slug={slug} @@ -258,7 +288,15 @@ export function VideoPanel({ title="Transcribe" summary={transcribeSummary} defaultOpen={true} - tone={hasTranscript ? "ok" : noAudio ? "neutral" : "attention"} + tone={ + incompleteTranscript + ? "attention" + : hasTranscript + ? "ok" + : noAudio + ? "neutral" + : "attention" + } > <div className="flex flex-col gap-4"> {audioFiles.map((f) => ( @@ -1238,6 +1276,63 @@ function DownloadOutcomeBadge({ ); } +function IncompleteTranscriptBanner({ + slug, + videoId, + coverage, + defaultQueueKey, + existingQueues, +}: { + slug: string; + videoId: string; + coverage: { lastCueEnd: number; duration: number; coverage: number | null }; + defaultQueueKey: string; + existingQueues: string[]; +}) { + const [queueKey, setQueueKey] = useState(defaultQueueKey); + const pct = + coverage.coverage === null + ? null + : Math.round(coverage.coverage * 1000) / 10; + const actionLabel = `Re-download & re-transcribe ${videoId}`; + return ( + <div + role="alert" + aria-label="incomplete transcript" + className="flex flex-col gap-3 rounded border px-3 py-2 text-sm border-amber-300 bg-amber-50 text-amber-900 dark:border-amber-900 dark:bg-amber-950 dark:text-amber-200" + > + <div className="flex flex-col gap-1"> + <span className="font-medium">Transcript looks truncated</span> + <span> + The transcript covers only {formatDuration(coverage.lastCueEnd)} of{" "} + {formatDuration(coverage.duration)} + {pct !== null ? ` (${pct}%)` : ""}. The audio download likely stopped + early. Re-download the audio and re-transcribe to fix it — this deletes + the truncated audio first so the full file is fetched. + </span> + </div> + <StreamActionLog + trigger={() => + redownloadIncompleteTranscriptAction(slug, videoId, queueKey) + } + cancelAction={cancelJobAction} + buttonLabel="Re-download & re-transcribe" + runningLabel="Re-downloading…" + label={actionLabel} + extraControls={ + <QueueControl + value={queueKey} + onChange={setQueueKey} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel={actionLabel} + /> + } + /> + </div> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -11,6 +11,8 @@ import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { resolvePrimaryVtt } from "yt-dlp-transcript-common/lib/videoStatus"; +import { readTranscriptCoverage } from "yt-dlp-transcript-common/controller/normalizeTranscript"; +import { isIncompleteTranscript } from "yt-dlp-transcript-common/lib/transcriptCoverage"; import { platformQueueKey, queueKeyForUrl, @@ -103,6 +105,17 @@ export default async function VideoDetailPage({ const availabilityHistory = availabilityRecord?.history ?? []; const doNotClean = await isDoNotClean(videoDir); const savedVideo = await loadSavedVideo(videoDir); + const cov = await readTranscriptCoverage(videoDir); + const coverage = cov + ? { + lastCueEnd: cov.cov.lastCueEnd, + duration: cov.cov.duration, + coverage: cov.cov.coverage, + incomplete: isIncompleteTranscript(cov.cov, { + isLivestream: cov.isLivestream, + }), + } + : null; const registry = getRegistry(); const existingQueues = registry.activeQueueNames(); @@ -182,6 +195,7 @@ export default async function VideoDetailPage({ channelAudioFormat={config.audioFormat} doNotClean={doNotClean} savedVideo={savedVideo} + coverage={coverage} /> </div> ); diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -305,6 +305,67 @@ export async function whisperVideoAction( }); } +// Fix a truncated transcript: the audio download silently stopped early, so the +// audio on disk is itself truncated and re-running whisper on it would reproduce +// the short transcript. Delete the truncated audio first, then re-download the +// full audio and re-transcribe in one job. The new transcript overwrites the +// old transcript.json (transcribeWithWorker), and normalize regenerates cues. +export async function redownloadIncompleteTranscriptAction( + slug: string, + videoId: string, + queueKey?: string, +): Promise<StreamActionResult> { + const r = await loadConfigOrError(slug); + if (!r.ok) return r; + const paths = getPaths(); + const videoDir = videoDirOf(slug, videoId); + return runManagedFunction({ + kind: "whisper-video", + queueKey: videoQueueKey(r.config, queueKey), + paths, + channelSlug: slug, + videoId, + fn: async (onLog, signal, _setProgress, ctx) => { + const audioFormat = r.config.audioFormat ?? "mp3"; + const url = await findVideoSourceUrl(paths, slug, videoId, r.config); + if (!url) { + throw new Error( + "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", + ); + } + // Remove the truncated audio so the download below re-fetches the full + // file rather than seeing it as already present. + const entries = await readdir(videoDir).catch(() => [] as string[]); + for (const name of entries.filter(isRealAudioFile)) { + await rm(path.join(videoDir, name), { force: true }); + onLog(`Removed truncated audio ${name}.`); + } + onLog(`Re-downloading audio for ${videoId}…`); + await runYtdlp({ + channelSlug: slug, + mode: "download-one-audio", + channelConfig: r.config, + paths, + onLog, + signal, + singleVideoUrl: url, + audioFormatOverride: audioFormat, + }); + await transcribeWithWorker({ + paths, + videoDir, + videoId, + audioFilename: `audio.${audioFormat}`, + tracker: makeTaskTracker(ctx, onLog), + onLog, + signal, + }); + revalidatePath(`/channels/${slug}/videos/${videoId}`); + revalidatePath(`/channels/${slug}`); + }, + }); +} + function safeJoinUnderDir( baseDir: string, filename: string, diff --git a/editor/e2e/incomplete-transcript.spec.ts b/editor/e2e/incomplete-transcript.spec.ts @@ -0,0 +1,128 @@ +// A transcript that covers only a small fraction of the video's duration means +// the audio download silently truncated (yt-dlp exited "ok" but only a few +// minutes landed, so whisper transcribed only those). The editor flags any +// non-livestream video >=10min whose transcript covers <50% of its runtime: +// an "Incomplete transcript" list filter + amber glyph, a warning banner on the +// video page with a re-download button, and an /actionable section. Detection +// reads each video's transcript.cues.json (duration + cues). See +// common/lib/transcriptCoverage.ts. + +import { mkdir, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { resetData, resolvePath } from "./helpers"; + +const CHANNEL = "test-transcribe"; +const DATA = `test-transcripts/channels/${CHANNEL}/data`; + +// Non-empty whisper transcript so the video reads as transcribed (and isn't +// flagged untranscribable for being empty). +const WHISPER = JSON.stringify({ transcription: [{ text: "hello" }] }); + +// Write a transcribed video with a transcript.cues.json of a given duration and +// last-cue end. version:1 is required for readNormalizedTranscript to parse it. +async function writeVideo( + id: string, + opts: { + duration: number; + lastCueEnd: number; + isLivestream?: boolean; + empty?: boolean; + }, +) { + const dir = resolvePath(`${DATA}/${id}`); + await mkdir(dir, { recursive: true }); + await writeFile( + `${dir}/transcript.json`, + opts.empty ? JSON.stringify({ transcription: [] }) : WHISPER, + ); + await writeFile( + `${dir}/audio.m4a`, + // Tiny placeholder; the truncated-audio detail doesn't matter to the test. + "fake-audio", + ); + await writeFile( + `${dir}/transcript.cues.json`, + JSON.stringify({ + version: 1, + source: "whisper", + duration: opts.duration, + isLivestream: Boolean(opts.isLivestream), + cues: opts.empty + ? [] + : [{ start: 0, end: opts.lastCueEnd, text: "hello" }], + }), + ); +} + +async function seed() { + await resetData("one-transcribe-channel"); + // Flagged: 2h22m video, transcript stops at ~6:52 → 4.8% coverage. + await writeVideo("vidTrunc", { duration: 8541, lastCueEnd: 412 }); + // Not flagged: full coverage of a >=10min video. + await writeVideo("vidFull", { duration: 1200, lastCueEnd: 1180 }); + // Not flagged: low coverage but under the 10min minimum. + await writeVideo("vidShort", { duration: 300, lastCueEnd: 10 }); + // Not flagged: low coverage but a livestream (duration unreliable). + await writeVideo("vidLive", { + duration: 8541, + lastCueEnd: 100, + isLivestream: true, + }); +} + +test("incomplete-transcript filter, glyph, panel banner, and actionable", async ({ + page, +}) => { + await seed(); + + // --- Channel list: the chip filters to exactly the flagged video --- + await page.goto(`/channels/${CHANNEL}`); + const list = page.getByLabel("videos", { exact: true }); + await expect(list.getByLabel("open vidTrunc")).toBeVisible(); + + await page + .getByRole("button", { name: "Incomplete transcript", exact: true }) + .click(); + await expect(list.getByLabel("open vidTrunc")).toBeVisible(); + await expect(list.getByLabel("open vidFull")).toBeHidden(); + await expect(list.getByLabel("open vidShort")).toBeHidden(); + await expect(list.getByLabel("open vidLive")).toBeHidden(); + // URL carries the filter. + await expect + .poll(() => new URL(page.url()).searchParams.get("filter")) + .toBe("incomplete_transcript"); + + // Composes with Transcribed (the flagged video is still transcribed). + await page.getByRole("button", { name: "Transcribed", exact: true }).click(); + await expect(list.getByLabel("open vidTrunc")).toBeVisible(); + await expect(list.getByLabel(/^open vid/)).toHaveCount(1); + + // The amber "truncated" glyph is shown for the flagged row. + await expect(page.getByTitle(/transcript truncated/)).toBeVisible(); + + // --- Video page: warning banner + re-download button on the flagged video --- + await page.goto(`/channels/${CHANNEL}/videos/vidTrunc`); + const banner = page.getByLabel("incomplete transcript"); + await expect(banner).toBeVisible(); + // Coverage line: covers 6:52 of 2:22:21 (4.8%). + await expect(banner).toContainText("2:22:21"); + await expect(banner).toContainText("4.8%"); + await expect( + banner.getByRole("button", { name: /Re-download & re-transcribe/ }), + ).toBeVisible(); + + // No banner on a fully-covered video. + await page.goto(`/channels/${CHANNEL}/videos/vidFull`); + await expect(page.getByLabel("incomplete transcript")).toBeHidden(); + + // --- Actionable view lists the channel under incomplete transcripts --- + await page.goto(`/actionable`); + const section = page.getByRole("region", { + name: "incomplete-transcripts", + exact: true, + }); + await expect(section).toBeVisible(); + await expect( + section.getByLabel(`incomplete-transcripts row ${CHANNEL}`), + ).toBeVisible(); +});