import "server-only"; import type { Dirent } from "node:fs"; import { readdir } from "node:fs/promises"; import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; import type { JobRecord } from "yt-dlp-transcript-common/jobs/registry"; import { normalizeBuckets } from "yt-dlp-transcript-common/views/pipeline/stageStatus"; import type { VideoRow, VideoRowStatus } from "./videoRows"; export async function readDataDirVideoIds( channelDataDir: string, ): Promise { let entries: Dirent[]; try { entries = await readdir(channelDataDir, { withFileTypes: true }); } catch { return []; } return entries .filter((e) => e.isDirectory() && !e.name.startsWith(".")) .map((e) => e.name); } export type ComputeRowsInput = { channelDataDirIds: string[]; snapshot: ChannelSnapshot; failedTranscriptionIds: string[]; runningJobs: JobRecord[]; excludedIds: Set; // id -> title, from readChannelVideoTitles. Optional so a caller that does // not show titles need not pay for them. titles?: ReadonlyMap; }; export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { const buckets = normalizeBuckets(input.snapshot.buckets); const undownloaded = new Set(input.snapshot.undownloadedIds ?? []); const onDisk = new Set(input.channelDataDirIds); const noTranscript = new Set(buckets.noTranscript); const downloadedNoTranscript = new Set(buckets.downloadedNoTranscript); const wrongFormatAudio = new Set(buckets.wrongFormatAudio); const multipleAudioFormats = new Set(buckets.multipleAudioFormats); const untranscribable = new Set(buckets.untranscribable); const partial = new Set(buckets.partialDownloads); const incompleteTranscript = new Set(buckets.incompleteTranscript); const shortAudioSet = new Set(buckets.shortAudio ?? []); const digestWarningsSet = new Set(buckets.digestWarnings ?? []); const corruptSourceSet = new Set(buckets.corruptSource); const corruptFullSourceSet = new Set(buckets.corruptFullSource); const failedTranscription = new Set(input.failedTranscriptionIds); const runningIds = new Set(); for (const j of input.runningJobs) { if (j.videoId && (j.status === "queued" || j.status === "running")) { runningIds.add(j.videoId); } } const allIds = new Set([...onDisk, ...undownloaded]); const rows: VideoRow[] = []; for (const id of allIds) { const isOnDisk = onDisk.has(id); const isUndownloaded = undownloaded.has(id) && !isOnDisk; const isUntranscribable = untranscribable.has(id); const isPartial = partial.has(id); const isCorruptSource = corruptSourceSet.has(id); const isCorruptFullSource = corruptFullSourceSet.has(id); // A corrupt-source video is a download/source problem, not a failed // transcription — don't let a stale failed-transcriptions entry (pruned on // the next transcribe pass) mark it failed here. Same for a kept // corrupt-full-source (it has no transcript and isn't a transcription error). const isFailedT = failedTranscription.has(id) && !isCorruptSource && !isCorruptFullSource; const inNoTranscript = noTranscript.has(id); const inDownloadedNoTranscript = downloadedNoTranscript.has(id); const downloaded = isOnDisk && !inNoTranscript; // A kept corrupt-full-source has a raw container on disk but no usable // transcript and is excluded from the no-transcript buckets, so guard the // "transcribed" fallback explicitly or it would read as done. const transcribed = isOnDisk && !inNoTranscript && !inDownloadedNoTranscript && !isCorruptFullSource; const excluded = input.excludedIds.has(id); let status: VideoRowStatus; if (isFailedT) status = "failed"; else if (isCorruptSource) status = "corrupt_source"; else if (isCorruptFullSource) status = "corrupt_full_source"; else if (isPartial) status = "partial_download"; else if (isUndownloaded) status = "not_downloaded"; else if (isUntranscribable) status = "untranscribable"; else if (inNoTranscript) status = "no_audio"; else if (inDownloadedNoTranscript) status = "downloaded_no_transcript"; else status = "transcribed"; const title = input.titles?.get(id)?.title; rows.push({ id, ...(title ? { title } : {}), downloaded, transcribed, untranscribable: isUntranscribable, partial: isPartial, corruptSource: isCorruptSource, corruptFullSource: isCorruptFullSource, failedTranscription: isFailedT, wrongFormatAudio: wrongFormatAudio.has(id) || multipleAudioFormats.has(id), excluded, incompleteTranscript: incompleteTranscript.has(id), shortAudio: shortAudioSet.has(id), digestWarnings: digestWarningsSet.has(id), running: runningIds.has(id), status, }); } rows.sort((a, b) => a.id.localeCompare(b.id)); return rows; }