Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 54df7aca3f424b2d6be0c03acc357d8d6d077631
parent 5e68c64bb6ebcc503e6bb54c071bbc0ec1fdb4a5
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 31 Aug 2026 12:52:56 -0400

snapshot: the untranscoded bucket is wrong-format audio

The bucket's name presupposed a transcode: it is the list of videos with audio
on disk, none of it in the channel's audioFormat — the orphan half of what
"Remove wrong-format audio" deletes. That is the sweep's population, not a
to-do list for an operation that no longer exists, so it is called
wrongFormatAudio.

The writer and both readers move in one commit because the rename crosses the
package boundary. No regen tooling: there is no snapshot schema version and
every field is optional by design, so an old snapshot keeps a stray
`untranscoded` key and reads the new one as [] until its next refresh — and
production has 0 ids in the bucket across all 68 channels, so nothing rendered
moves. An offline rewriter would only race snapshotScheduler.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelSnapshot.ts | 22+++++++++++++++-------
Meditor/app/channels/[slug]/lib/stageStatus.ts | 2+-
Meditor/app/channels/[slug]/lib/videoRowsServer.ts | 4++--
Meditor/app/channels/[slug]/page.tsx | 2+-
4 files changed, 19 insertions(+), 11 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -107,7 +107,14 @@ export type ChannelSnapshot = { buckets: { noTranscript: string[]; downloadedNoTranscript: string[]; - untranscoded: string[]; + // Has audio on disk, none of it in the channel's audioFormat — the orphan + // half of what "Remove wrong-format audio" deletes (the other half is + // multipleAudioFormats: target present plus extras). Was `untranscoded` + // until 2026-08-30, when the transcode operation that read it as a to-do + // list was removed; the population itself is still the sweep's. Old + // snapshots keep a stray `untranscoded` key and read this one as [] until + // their next regen — no schema version, per-field optionality, as always. + wrongFormatAudio: string[]; multipleAudioFormats: string[]; // Dirs with a whisper transcript AND audio still on disk — the audio is // redundant and can be cleaned. Mirrors what cleanAudioFromTranscribed @@ -762,7 +769,7 @@ export async function generateChannelSnapshot( const noTranscript: string[] = []; const downloadedNoTranscript: string[] = []; - const untranscoded: string[] = []; + const wrongFormatAudio: string[] = []; const multipleAudioFormats: string[] = []; const transcribedWithAudio: string[] = []; const untranscribable: string[] = []; @@ -870,9 +877,10 @@ export async function generateChannelSnapshot( // A download that completed but stayed malformed after one re-download. The // raw container (e.g. audio.mp4) is kept on disk for inspection. It IS an // artifact (so it won't be re-queued for download), but it is NOT usable - // audio — short-circuit so it doesn't land in untranscoded / - // downloadedNoTranscript (which would re-transcode/transcribe corrupt audio) - // or the wrong-format cleanup estimate (which would delete the kept file). + // audio — short-circuit so it doesn't land in wrongFormatAudio / + // downloadedNoTranscript (which would transcribe corrupt audio, or offer + // the kept file to the wrong-format sweep) or the wrong-format cleanup + // estimate (which would delete it). if (outcome?.status === "corrupt-full-source") { corruptFullSource.push(id); continue; @@ -920,7 +928,7 @@ export async function generateChannelSnapshot( files.audioFiles.length > 0 && !files.audioFiles.includes(targetAudioFile) ) { - untranscoded.push(id); + wrongFormatAudio.push(id); } // Reclaim estimate for the wrong-format sweep: every non-target audio file // in a cleanable (not do-not-clean) dir. Covers orphans (no target) and the @@ -1142,7 +1150,7 @@ export async function generateChannelSnapshot( buckets: { noTranscript: noTranscript.sort(), downloadedNoTranscript: downloadedNoTranscript.sort(), - untranscoded: untranscoded.sort(), + wrongFormatAudio: wrongFormatAudio.sort(), multipleAudioFormats: multipleAudioFormats.sort(), transcribedWithAudio: transcribedWithAudio.sort(), untranscribable: untranscribable.sort(), diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -21,7 +21,7 @@ export function normalizeBuckets( return { noTranscript: raw?.noTranscript ?? [], downloadedNoTranscript: raw?.downloadedNoTranscript ?? [], - untranscoded: raw?.untranscoded ?? [], + wrongFormatAudio: raw?.wrongFormatAudio ?? [], multipleAudioFormats: raw?.multipleAudioFormats ?? [], transcribedWithAudio: raw?.transcribedWithAudio ?? [], untranscribable: raw?.untranscribable ?? [], diff --git a/editor/app/channels/[slug]/lib/videoRowsServer.ts b/editor/app/channels/[slug]/lib/videoRowsServer.ts @@ -34,7 +34,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { const onDisk = new Set(input.channelDataDirIds); const noTranscript = new Set(buckets.noTranscript); const downloadedNoTranscript = new Set(buckets.downloadedNoTranscript); - const untranscoded = new Set(buckets.untranscoded); + const wrongFormatAudio = new Set(buckets.wrongFormatAudio); const multipleAudioFormats = new Set(buckets.multipleAudioFormats); const untranscribable = new Set(buckets.untranscribable); const partial = new Set(buckets.partialDownloads); @@ -102,7 +102,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { corruptFullSource: isCorruptFullSource, failedTranscription: isFailedT, wrongFormatAudio: - untranscoded.has(id) || multipleAudioFormats.has(id), + wrongFormatAudio.has(id) || multipleAudioFormats.has(id), excluded, incompleteTranscript: incompleteTranscript.has(id), shortAudio: shortAudioSet.has(id), diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -455,7 +455,7 @@ export default async function ChannelDetailPage({ supersededAutoSubsIds={buckets.supersededAutoSubs} foreignAudioIds={[ ...new Set([ - ...buckets.untranscoded, + ...buckets.wrongFormatAudio, ...buckets.multipleAudioFormats, ]), ].sort()}