Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 6d4a06d76f4382ad3d56d372b284a84cb94f98fe
parent 3cacaa177064da3e59457c391a5f63469f70537c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 22 May 2026 10:05:20 -0400

exclude unavailable videos from actionable/undownloaded count

Diffstat:
Mcommon/controller/channelSnapshot.ts | 7+++++++
Mcommon/lib/availability-server.ts | 52+++++++++++++++++++++++++++++++++++++++-------------
Meditor/CHANGELOG.md | 2++
Meditor/app/actionable/actions.ts | 13+++++++++++--
Meditor/app/actionable/lib/loadActionable.ts | 41+++++++++++++++++++++++++++++++++--------
Meditor/app/actionable/page.tsx | 6++++--
Meditor/app/channels/[slug]/lib/stageStatus.ts | 20++++++++------------
Meditor/app/channels/[slug]/page.tsx | 17++++++++++-------
8 files changed, 114 insertions(+), 44 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -72,6 +72,13 @@ export function normalizeExcludedFromDownload( }; } +export function excludedDownloadIdSet( + snapshot: Pick<ChannelSnapshot, "excludedFromDownload"> | null | undefined, +): Set<string> { + const e = normalizeExcludedFromDownload(snapshot?.excludedFromDownload); + return new Set<string>([...e.membersOnly, ...e.deleted, ...e.private]); +} + function emptyAvailability(): AvailabilitySnapshot { const byStatus = {} as Record<Availability, string[]>; for (const v of AVAILABILITY_VALUES) byStatus[v] = []; diff --git a/common/lib/availability-server.ts b/common/lib/availability-server.ts @@ -44,17 +44,41 @@ export async function writeAvailability( await rename(tmp, file); } -// Resolve a video's availability class, preferring an explicit availability.json -// (the authoritative record written by the availability-check or metadata-backfill -// passes) and falling back to the last attempt's availabilityClass in -// download-outcome.json. This second source matters for videos that have -// never successfully downloaded — their availability never reaches -// availability.json otherwise. +// Resolve a video's availability class by reconciling two sources: +// - availability.json (written by the explicit availability-check pass) +// - download-outcome.json's last attempt's availabilityClass (recorded +// whenever a download attempt finishes, including the failure cases +// where yt-dlp reports the video is removed/private/members-only) +// Both can be authoritative; the older one can also be stale. We use +// whichever was written more recently. The outcome source matters not just +// for videos that have never reached the availability check, but also for +// videos whose status changed after the last check — e.g. a public video +// that the uploader has since deleted will fail a download attempt with +// `availabilityClass: "deleted"` long before the next availability sweep. export async function resolveEffectiveAvailability( videoDir: string, ): Promise<Availability | null> { - const explicit = await loadAvailability(videoDir); + const [explicit, outcome] = await Promise.all([ + loadAvailability(videoDir), + loadDownloadOutcomeAvailability(videoDir), + ]); + + if (explicit && outcome) { + if ( + outcome.finishedAt && + new Date(outcome.finishedAt) > new Date(explicit.checkedAt) + ) { + return outcome.availability; + } + return explicit.availability; + } if (explicit) return explicit.availability; + return outcome?.availability ?? null; +} + +async function loadDownloadOutcomeAvailability( + videoDir: string, +): Promise<{ availability: Availability; finishedAt: string | null } | null> { try { const raw = await readFile( path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME), @@ -62,12 +86,14 @@ export async function resolveEffectiveAvailability( ); const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>; const attempts = parsed.attempts; - if (Array.isArray(attempts) && attempts.length > 0) { - const last = attempts[attempts.length - 1]; - if (last?.availabilityClass) return last.availabilityClass; - } + if (!Array.isArray(attempts) || attempts.length === 0) return null; + const last = attempts[attempts.length - 1]; + if (!last?.availabilityClass) return null; + return { + availability: last.availabilityClass, + finishedAt: typeof parsed.finishedAt === "string" ? parsed.finishedAt : null, + }; } catch { - // No outcome file or unreadable → unknown + return null; } - return null; } diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -24,6 +24,8 @@ - Sidebar nav badges are no longer redundant: **Jobs** continues to count running + queued, while **Active** now counts only running jobs. ### Fixed +- A failed download attempt that classifies a video as `deleted` / `private` / `members_only` (e.g. yt-dlp reports "removed by the uploader") is now respected even when an older `availability.json` says the video was once public. `resolveEffectiveAvailability` previously preferred `availability.json` unconditionally, so a video that went public → deleted between availability sweeps stayed classified as public until the next sweep, leaving it counted in undownloaded / awaiting-transcription lists indefinitely. The resolver now picks whichever source — `availability.json` or `download-outcome.json`'s last attempt — was written more recently. +- Actionable lists no longer count videos that the availability check has flagged as `deleted`, `members_only`, or `private`. The "Channels with downloaded videos awaiting transcription" row count on `/actionable`, the dashboard "Needs attention" summary, the channel page's Transcribe stage count + disclosure list, and the Download stage's "dirs missing transcript & audio" list all now exclude unactionable videos. `undownloadedIds` was already filtered at snapshot generation, but the transcript buckets weren't — a deleted video with audio on disk used to inflate "awaiting transcription" indefinitely. Per-video row status in the channel video list is unchanged; excluded videos still display their true on-disk state (so the existing "needs action" filter can hide them via the `excluded` flag). - Channel page: clicking a pipeline stage in the side rail (or a mobile stage badge) no longer jumps up into the selected video's viewer panel. The per-video stage cards were emitting the same DOM ids as the channel-level pipeline sections, so the browser scrolled to the first match. The per-video cards now use a `video-stage-` prefix. - Channel page sticky/anchor offsets bumped to clear the sticky `StatusHeader`. The left video list and the pipeline stage rail previously pinned too high (`top: 1rem` and `top: 8rem` respectively) and sat partially behind the header. Stage anchors used a fixed `scroll-margin-top: 8rem` that wasn't enough on narrow widths where the header's mobile badge row wraps. The video-list and stage-rail sticky tops are now `top: 7rem` (lg+), and stage sections use a responsive `scroll-margin-top` (11rem on mobile, 7rem on lg+). - Clicking a mobile stage badge (or a stage rail item) now auto-expands the collapsible "Channel pipeline & settings" wrapper if the user had collapsed it. Previously the hash navigation succeeded but the target section was inside a closed `<details>` and not visible. The target is also re-scrolled into view after the expansion so it lands at the correct position. diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts @@ -3,7 +3,10 @@ import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { listChannels } from "yt-dlp-transcript-common/controller/channels"; -import { generateChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; +import { + excludedDownloadIdSet, + generateChannelSnapshot, +} from "yt-dlp-transcript-common/controller/channelSnapshot"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand"; @@ -58,10 +61,16 @@ export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResu fn: async (onLog) => { onLog(`Regenerating report for ${c.slug}…`); const snap = await generateChannelSnapshot(paths, c.slug); + const excluded = excludedDownloadIdSet(snap); + const awaitingTranscription = excluded.size + ? snap.buckets.downloadedNoTranscript.filter( + (id) => !excluded.has(id), + ).length + : snap.buckets.downloadedNoTranscript.length; onLog( `Done. ${snap.totals.videos} videos · ` + `${snap.undownloadedIds.length} undownloaded · ` + - `${snap.buckets.downloadedNoTranscript.length} awaiting transcription.`, + `${awaitingTranscription} awaiting transcription.`, ); // Deliberately no revalidatePath here — calling it from a // background fn races with the in-flight re-render of /actionable diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -4,6 +4,7 @@ import { type ChannelStat, } from "yt-dlp-transcript-common/controller/channels"; import { + excludedDownloadIdSet, readChannelSnapshot, type ChannelSnapshot, } from "yt-dlp-transcript-common/controller/channelSnapshot"; @@ -27,6 +28,34 @@ export function isStaleOrMissing(row: ActionableRow): boolean { return new Date(synced).getTime() > new Date(row.snapshot.generatedAt).getTime(); } +// Counts that drive the actionable lists exclude IDs that the availability +// check has flagged as deleted / members-only / private — those videos +// can't be acted on, so they shouldn't inflate "needs attention" totals. +// `undownloadedIds` is already filtered at snapshot generation time, but we +// apply the filter again so a stale snapshot can't surface excluded IDs. +function countActionable( + snapshot: ChannelSnapshot | null | undefined, + ids: readonly string[] | undefined, +): number { + if (!snapshot || !ids) return 0; + const excluded = excludedDownloadIdSet(snapshot); + if (excluded.size === 0) return ids.length; + let n = 0; + for (const id of ids) if (!excluded.has(id)) n++; + return n; +} + +export function actionableUndownloadedCount(row: ActionableRow): number { + return countActionable(row.snapshot, row.snapshot?.undownloadedIds); +} + +export function actionableUntranscribedCount(row: ActionableRow): number { + return countActionable( + row.snapshot, + row.snapshot?.buckets.downloadedNoTranscript, + ); +} + export async function loadActionableSummary( paths: Paths, ): Promise<ActionableSummary> { @@ -39,19 +68,15 @@ export async function loadActionableSummary( ); const undownloaded = rows - .filter((r) => (r.snapshot?.undownloadedIds.length ?? 0) > 0) + .filter((r) => actionableUndownloadedCount(r) > 0) .sort( - (a, b) => - (b.snapshot?.undownloadedIds.length ?? 0) - - (a.snapshot?.undownloadedIds.length ?? 0), + (a, b) => actionableUndownloadedCount(b) - actionableUndownloadedCount(a), ); const untranscribed = rows - .filter((r) => (r.snapshot?.buckets.downloadedNoTranscript.length ?? 0) > 0) + .filter((r) => actionableUntranscribedCount(r) > 0) .sort( - (a, b) => - (b.snapshot?.buckets.downloadedNoTranscript.length ?? 0) - - (a.snapshot?.buckets.downloadedNoTranscript.length ?? 0), + (a, b) => actionableUntranscribedCount(b) - actionableUntranscribedCount(a), ); const staleOrMissing = rows diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -2,6 +2,8 @@ import type { Metadata } from "next"; import Link from "next/link"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { + actionableUndownloadedCount, + actionableUntranscribedCount, loadActionableSummary, type ActionableRow, } from "./lib/loadActionable"; @@ -39,7 +41,7 @@ export default async function ActionablePage() { "Playlist entries that have no audio/video on disk yet. Run “Download missing” to fetch them.", countLabel: "undownloaded", emptyLabel: "Nothing pending.", - getCount: (r) => r.snapshot?.undownloadedIds.length ?? 0, + getCount: actionableUndownloadedCount, primaryAction: (r) => ( <InlineActionButton variant={{ kind: "downloadMissing", slug: r.channel.slug }} @@ -56,7 +58,7 @@ export default async function ActionablePage() { "Videos with audio on disk but no whisper or yt-vtt transcript yet. Run “Transcribe pending” to whisper them.", countLabel: "awaiting transcript", emptyLabel: "Nothing pending.", - getCount: (r) => r.snapshot?.buckets.downloadedNoTranscript.length ?? 0, + getCount: actionableUntranscribedCount, primaryAction: (r) => ( <InlineActionButton variant={{ diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -1,6 +1,6 @@ import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import { - normalizeExcludedFromDownload, + excludedDownloadIdSet, type ChannelSnapshot, } from "yt-dlp-transcript-common/controller/channelSnapshot"; import type { JobRecord } from "yt-dlp-transcript-common/jobs/registry"; @@ -100,17 +100,13 @@ export function computeStageStatuses( const buckets = normalizeBuckets(snapshot.buckets); const undownloadedIds = snapshot.undownloadedIds ?? []; - const excludedFromDownload = normalizeExcludedFromDownload( - snapshot.excludedFromDownload, - ); - const excludedDownloadIds = new Set<string>([ - ...excludedFromDownload.membersOnly, - ...excludedFromDownload.deleted, - ...excludedFromDownload.private, - ]); + const excludedDownloadIds = excludedDownloadIdSet(snapshot); const actionableNoTranscript = buckets.noTranscript.filter( (id) => !excludedDownloadIds.has(id), ); + const actionableDownloadedNoTranscript = buckets.downloadedNoTranscript.filter( + (id) => !excludedDownloadIds.has(id), + ); const runningByStage = new Set<StageId>(); for (const job of runningJobs) { @@ -128,7 +124,7 @@ export function computeStageStatuses( buckets.partialDownloads.length; const transcodePending = transcodeApplies ? buckets.untranscoded.length : 0; const transcodeFailed = transcodeApplies ? failedTranscodingIds.length : 0; - const transcribePending = buckets.downloadedNoTranscript.length; + const transcribePending = actionableDownloadedNoTranscript.length; const transcribeFailed = failedVideoIds.length; const cleanupPending = buckets.multipleAudioFormats.length; // `untranscribable` is deliberately excluded: those videos are an intentional @@ -254,10 +250,10 @@ export function computeStageStatuses( const transcribeRunning = runningByStage.has("transcribe"); const transcribeParts: string[] = []; - if (buckets.downloadedNoTranscript.length > 0) { + if (actionableDownloadedNoTranscript.length > 0) { transcribeParts.push( pluralize( - buckets.downloadedNoTranscript.length, + actionableDownloadedNoTranscript.length, "video awaiting whisper", "videos awaiting whisper", ), diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -6,6 +6,7 @@ import type { Metadata } from "next"; import { notFound } from "next/navigation"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; import { + excludedDownloadIdSet, generateChannelSnapshot, normalizeAvailability, normalizeExcludedFromDownload, @@ -172,14 +173,16 @@ export default async function ChannelDetailPage({ // Hide failed-transcription entries whose video can no longer be acted on // (members_only / deleted / private). Same rationale as the snapshot's // bucket filter; this path is fresh-loaded so it needs its own pass. - const excludedDownloadIds = new Set<string>([ - ...excludedFromDownload.membersOnly, - ...excludedFromDownload.deleted, - ...excludedFromDownload.private, - ]); + const excludedDownloadIds = excludedDownloadIdSet(snapshot); const failedVideoIds = rawFailedVideoIds.filter( (id) => !excludedDownloadIds.has(id), ); + const actionableNoTranscriptIds = buckets.noTranscript.filter( + (id) => !excludedDownloadIds.has(id), + ); + const actionableDownloadedNoTranscriptIds = buckets.downloadedNoTranscript.filter( + (id) => !excludedDownloadIds.has(id), + ); const availability = normalizeAvailability(snapshot.availability); const stages = computeStageStatuses({ snapshot, @@ -233,7 +236,7 @@ export default async function ChannelDetailPage({ existingQueues={existingQueues} undownloadedIds={undownloadedIds} excludedFromDownload={excludedFromDownload} - noTranscriptIds={buckets.noTranscript} + noTranscriptIds={actionableNoTranscriptIds} partialDownloadIds={buckets.partialDownloads} missingShard={downloadMissingShard} /> @@ -243,7 +246,7 @@ export default async function ChannelDetailPage({ slug={slug} existingQueues={existingQueues} failedVideoIds={failedVideoIds} - downloadedNoTranscriptIds={buckets.downloadedNoTranscript} + downloadedNoTranscriptIds={actionableDownloadedNoTranscriptIds} defaultConcurrency={paths.parallelTranscribeLimit} defaultQueueKey={TRANSCRIPTION_QUEUE} missingShard={transcribeMissingShard}