commit 19e70e1131b587d222e8467ca34780d06683aa3a
parent 088ccc391c38445f6ad69b91f54ef653284b1e55
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 28 Jun 2026 22:49:45 -0400
Auto-queue: bucketless rules union all buckets; auto-download resumes partials
A policy-tree rule with no bucket selected ("all buckets (default)") now draws
from the union of every bucket its runner kind tracks, deduped and in priority
order, instead of only the single primary bucket. This fixes channels that
quietly stopped being auto-downloaded once their remaining work drifted entirely
into partially-downloaded videos: those carry a .part file but no completed
audio, so they live in partialDownloads and were absent from undownloadedIds —
the only bucket the download runner used to load.
The download runner now loads partialDownloads alongside undownloadedIds
(partials first, so in-progress downloads resume via downloadOneManaged before
fresh ones start) and exposes partialDownloads as a selectable bucket in the
policy editor. Per-kind bucket lists are consolidated behind a single
bucketsForKind() shared by the runner, the pending-count helper, and the editor
picker so they can't drift.
Verified against the live cornbreadman snapshot + real download policy: its
pending count rises 11 -> 18, surfacing 7 previously-skipped partial-only videos.
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Diffstat:
6 files changed, 121 insertions(+), 42 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -17,8 +17,7 @@ import {
type WorkPick,
buildPendingByLeaf,
selectNextWork,
- DEFAULT_TRANSCRIBE_BUCKET,
- DEFAULT_DOWNLOAD_BUCKET,
+ bucketsForKind,
} from "../jobs/autoQueuePolicy";
import {
type AutoQueueKind,
@@ -150,8 +149,6 @@ async function listChannelMeta(paths: Paths): Promise<ChannelMeta[]> {
return out;
}
-const BUCKET_NAMES = ["downloadedNoTranscript", "failedListed"] as const;
-
// Read each channel's snapshot and project the buckets this runner kind cares
// about into ChannelWork, plus a videoId -> owning channel map (a platform/all
// leaf spans channels, so the pick needs the owner to locate the video dir).
@@ -166,15 +163,14 @@ async function buildChannelWork(
const snap = await readChannelSnapshot(paths, slug);
if (!snap) continue;
const buckets: Record<string, string[]> = {};
- if (kind === "transcription") {
- for (const name of BUCKET_NAMES) {
- const ids = snap.buckets?.[name] ?? [];
- buckets[name] = ids;
- for (const id of ids) if (!owner.has(id)) owner.set(id, slug);
- }
- } else {
- const ids = snap.undownloadedIds ?? [];
- buckets[DEFAULT_DOWNLOAD_BUCKET] = ids;
+ for (const name of bucketsForKind(kind)) {
+ // `undownloadedIds` lives at the snapshot top level; every other bucket
+ // is under snap.buckets.
+ const ids =
+ name === "undownloadedIds"
+ ? (snap.undownloadedIds ?? [])
+ : ((snap.buckets as Record<string, string[]>)?.[name] ?? []);
+ buckets[name] = ids;
for (const id of ids) if (!owner.has(id)) owner.set(id, slug);
}
channels.push({ slug, platform, buckets });
@@ -190,11 +186,9 @@ export async function computeLeafPendingCounts(
paths: Paths = getPaths(),
): Promise<Record<string, number>> {
const policy = getSettings().autoQueue[kind];
- const defaultBucket =
- kind === "transcription" ? DEFAULT_TRANSCRIBE_BUCKET : DEFAULT_DOWNLOAD_BUCKET;
const meta = await listChannelMeta(paths);
const { channels } = await buildChannelWork(paths, kind, meta);
- const pending = buildPendingByLeaf(policy.root, channels, defaultBucket);
+ const pending = buildPendingByLeaf(policy.root, channels, bucketsForKind(kind));
const counts: Record<string, number> = {};
for (const [leafId, ids] of Object.entries(pending)) counts[leafId] = ids.length;
return counts;
@@ -233,8 +227,7 @@ async function runLoop(
signal: AbortSignal,
ctx: JobRunContext,
): Promise<void> {
- const defaultBucket =
- kind === "transcription" ? DEFAULT_TRANSCRIBE_BUCKET : DEFAULT_DOWNLOAD_BUCKET;
+ const defaultBuckets = bucketsForKind(kind);
const tracker = makeTaskTracker(ctx, onLog);
const state = await readAutoQueueState(paths);
const kindState = state[kind];
@@ -345,7 +338,7 @@ async function runLoop(
metaCache.map((m) => [m.slug, m.platform ?? "unknown"]),
);
const { channels, owner } = await buildChannelWork(paths, kind, metaCache);
- const pending = buildPendingByLeaf(policy.root, channels, defaultBucket);
+ const pending = buildPendingByLeaf(policy.root, channels, defaultBuckets);
const exclude = new Set<string>([...live.inFlight.keys(), ...completed]);
removeIds(pending, exclude);
// Download only: drop videos whose platform already has an in-flight
diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts
@@ -5,6 +5,7 @@ import {
type AutoQueueRuntime,
type ChannelWork,
buildPendingByLeaf,
+ bucketsForKind,
defaultAutoQueue,
emptyAutoQueueRuntime,
flattenLeaves,
@@ -219,7 +220,7 @@ test("buildPendingByLeaf: channel + platform + all matchers", () => {
{ id: "rest", match: { type: "all" } },
],
};
- const pending = buildPendingByLeaf(root, CHANNELS, "downloadedNoTranscript");
+ const pending = buildPendingByLeaf(root, CHANNELS, ["downloadedNoTranscript"]);
assert.deepEqual(pending.corn, ["c1", "c2"]);
assert.deepEqual(pending.tw, ["h1"]); // hasanabi is twitch
assert.deepEqual(pending.rest, []); // both channels already claimed
@@ -234,7 +235,7 @@ test("buildPendingByLeaf: first-match-wins prevents double-claiming a video", ()
{ id: "corn", match: { type: "channel", value: "cornbreadman" } },
],
};
- const pending = buildPendingByLeaf(root, CHANNELS, "downloadedNoTranscript");
+ const pending = buildPendingByLeaf(root, CHANNELS, ["downloadedNoTranscript"]);
assert.deepEqual(pending.all1.sort(), ["c1", "c2", "h1"]);
assert.deepEqual(pending.corn, []); // already claimed by the catch-all above it
});
@@ -251,11 +252,78 @@ test("buildPendingByLeaf: a bucket leaf draws from its named bucket", () => {
{ id: "corn", match: { type: "channel", value: "cornbreadman" } },
],
};
- const pending = buildPendingByLeaf(root, CHANNELS, "downloadedNoTranscript");
+ const pending = buildPendingByLeaf(root, CHANNELS, ["downloadedNoTranscript"]);
assert.deepEqual(pending.retry, ["cf1"]);
assert.deepEqual(pending.corn, ["c1", "c2"]);
});
+test("buildPendingByLeaf: a bucketless leaf unions all default buckets, deduped", () => {
+ const root: AutoQueueGroup = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "corn", match: { type: "channel", value: "cornbreadman" } }],
+ };
+ // cornbreadman has c1,c2 in downloadedNoTranscript and cf1 in failedListed; a
+ // bucketless leaf draws from both.
+ const pending = buildPendingByLeaf(root, CHANNELS, [
+ "downloadedNoTranscript",
+ "failedListed",
+ ]);
+ assert.deepEqual(pending.corn, ["c1", "c2", "cf1"]);
+});
+
+test("buildPendingByLeaf: union follows default-bucket list order", () => {
+ const channels: ChannelWork[] = [
+ {
+ slug: "cornbreadman",
+ platform: "odysee",
+ // a4 is in BOTH buckets — it must be taken once, under the earlier bucket.
+ buckets: { partialDownloads: ["p1", "a4"], undownloadedIds: ["a4", "u1"] },
+ },
+ ];
+ const root: AutoQueueGroup = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "corn", match: { type: "channel", value: "cornbreadman" } }],
+ };
+ // Partials first (DOWNLOAD_BUCKETS order), then the fresh ids, deduped.
+ const pending = buildPendingByLeaf(root, channels, [
+ "partialDownloads",
+ "undownloadedIds",
+ ]);
+ assert.deepEqual(pending.corn, ["p1", "a4", "u1"]);
+});
+
+test("buildPendingByLeaf: partial-only channel is still picked (regression)", () => {
+ // The cornbreadman bug: remaining work drifted entirely into partialDownloads,
+ // leaving undownloadedIds empty — the old single-bucket default skipped it.
+ const channels: ChannelWork[] = [
+ {
+ slug: "cornbreadman",
+ platform: "odysee",
+ buckets: { undownloadedIds: [], partialDownloads: ["p1", "p2"] },
+ },
+ ];
+ const root: AutoQueueGroup = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "corn", match: { type: "channel", value: "cornbreadman" } }],
+ };
+ const pending = buildPendingByLeaf(root, channels, bucketsForKind("download"));
+ assert.deepEqual(pending.corn, ["p1", "p2"]);
+});
+
+test("bucketsForKind: per-kind ordered bucket lists", () => {
+ assert.deepEqual(
+ [...bucketsForKind("transcription")],
+ ["downloadedNoTranscript", "failedListed"],
+ );
+ assert.deepEqual(
+ [...bucketsForKind("download")],
+ ["partialDownloads", "undownloadedIds"],
+ );
+});
+
test("flattenLeaves: pre-order priority order", () => {
const root: AutoQueueGroup = {
id: "root",
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -76,10 +76,20 @@ export type AutoQueueSettings = {
download: AutoQueuePolicy;
};
-// Default snapshot bucket each runner kind draws from when a leaf doesn't
-// specify one.
-export const DEFAULT_TRANSCRIBE_BUCKET = "downloadedNoTranscript";
-export const DEFAULT_DOWNLOAD_BUCKET = "undownloadedIds";
+// Buckets each runner kind draws from, in priority order. A leaf with no
+// explicit bucket draws from the whole list (union, deduped); the list order is
+// its internal priority. Single source of truth for the runner, the pending-
+// count helper, and the editor's bucket picker.
+export const TRANSCRIBE_BUCKETS = ["downloadedNoTranscript", "failedListed"] as const;
+export const DOWNLOAD_BUCKETS = ["partialDownloads", "undownloadedIds"] as const;
+
+// Local kind type — do NOT import AutoQueueKind from autoQueueState.ts, which
+// already imports from this module (the reverse edge would be a cycle).
+export function bucketsForKind(
+ kind: "transcription" | "download",
+): readonly string[] {
+ return kind === "transcription" ? TRANSCRIBE_BUCKETS : DOWNLOAD_BUCKETS;
+}
export const AUTO_QUEUE_MAX_WORKERS_MAX = 64;
@@ -135,26 +145,32 @@ function matchesChannel(match: AutoQueueMatch, ch: ChannelWork): boolean {
// pre-order) leaf whose match covers the video's channel and whose bucket
// contains it. First-match-wins prevents the same video being claimed — and thus
// double-processed — by two overlapping leaves (e.g. a channel leaf and an `all`
-// catch-all). Returns leafId -> available video ids (in their snapshot order).
+// catch-all). A leaf with an explicit `match.bucket` draws from only that bucket;
+// a leaf with none draws from the union of `defaultBuckets` (the kind's full
+// list), in list order. The `claimed` Set dedups across AND within leaves, so a
+// video present in two buckets is taken once, under the earlier (higher-priority)
+// bucket. Returns leafId -> available video ids (in priority order).
export function buildPendingByLeaf(
root: AutoQueueNode,
channels: ReadonlyArray<ChannelWork>,
- defaultBucket: string,
+ defaultBuckets: ReadonlyArray<string>,
): Record<string, string[]> {
const leaves = flattenLeaves(root);
const pending: Record<string, string[]> = {};
for (const leaf of leaves) pending[leaf.id] = [];
const claimed = new Set<string>();
for (const leaf of leaves) {
- const bucket = leaf.match.bucket ?? defaultBucket;
- for (const ch of channels) {
- if (!matchesChannel(leaf.match, ch)) continue;
- const ids = ch.buckets[bucket];
- if (!ids) continue;
- for (const id of ids) {
- if (claimed.has(id)) continue;
- claimed.add(id);
- pending[leaf.id].push(id);
+ const bucketNames = leaf.match.bucket ? [leaf.match.bucket] : defaultBuckets;
+ for (const bucket of bucketNames) {
+ for (const ch of channels) {
+ if (!matchesChannel(leaf.match, ch)) continue;
+ const ids = ch.buckets[bucket];
+ if (!ids) continue;
+ for (const id of ids) {
+ if (claimed.has(id)) continue;
+ claimed.add(id);
+ pending[leaf.id].push(id);
+ }
}
}
}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Auto-queue rules with no bucket now draw from *all* of a runner's buckets, and auto-download can resume partial downloads.** A policy-tree rule left at the **"all buckets (default)"** setting (previously just labeled *default*) now draws from the **union** of every bucket that runner kind tracks — deduped, in priority order — instead of only the single primary bucket. This fixes channels (e.g. an Odysee channel mid-download) that quietly stopped being auto-downloaded once their remaining work drifted entirely into **partially-downloaded** videos: those have a `.part` file but no completed audio, so they live in the `partialDownloads` bucket and were **absent from `undownloadedIds`** — the only bucket auto-download used to load. The download runner now loads `partialDownloads` alongside `undownloadedIds` (partials first, so in-progress downloads resume via `downloadOneManaged` before fresh ones start), and exposes `partialDownloads` as a selectable bucket in the policy editor so you can dedicate a high-priority rule to resuming partials. The per-kind bucket lists are consolidated behind a single `bucketsForKind` source of truth shared by the runner, the per-rule pending-count helper, and the editor's bucket picker (so they can't drift). Note: a *bucketless* auto-transcribe rule now also drains `failedListed` after `downloadedNoTranscript` (it already loaded both); platform rate-limit backoff is unchanged and remains an independent reason a throttled platform may pause. See `common/jobs/autoQueuePolicy.ts` (`buildPendingByLeaf` + `bucketsForKind` + unit tests), `common/controller/autoRunner.ts`, and `editor/app/auto-queue/{page.tsx,components/PolicyTreeEditor.tsx}`.
- **Hub homepage redesigned into a cross-site landing; the homepage page-creator is removed.** The hub's home page is now a single mobile-first cross-site landing (headline KPIs and one stacked activity chart with Metric [Transcribed/Downloaded] · Breakdown [By site/By channel] · Bucket [Week/Month/Cumulative] · Range [90d/12mo/All] · Display [Share/Counts] controls, plus a metric-aware site-links grid with sparklines and a "#1 this month" badge), built from a small `homepage-summary.json` pre-computed by `compose-homepage`. The separate `/stats` dashboard route folds into it. Consequently the hub's **Markdown-pages subsystem is dropped**: **Manage → Homepage** now edits only branding (the Pages list, New-page, and the page editor are gone), and the homepage config no longer carries a `nav`. The page server actions (`saveHomepagePageAction`/`deleteHomepagePageAction`), `editor/app/homepage/pages/*`, `PageEditor.tsx`, `common/lib/{homepagePages,homepageConstants}.ts`, and `paths.homepagePagesDir` are removed. See `editor/app/homepage/{page.tsx,actions.ts}`, `common/bin/compose-homepage.ts`, `common/lib/{homepageSummary,homepageChart}.ts`, and the `homepage/` package. (Re-addable later if needed.)
- **One-click Retry for failed jobs (plus "Retry all failed").** A failed job that carries a replay descriptor (any bookmarkable kind — sync, download-missing, transcribe-all, retry-bucket, …) now shows a **Retry** button on the Jobs history table, and the page header gains a **Retry all failed** button whenever at least one such job is listed. Retry re-runs the job from its stored spec exactly like a bookmark re-run (so bucket jobs re-derive from the channel's *current* state), and the re-run **jumps ahead of other queued work** (it's promoted to the front of its queue, reusing the new reorder machinery) so a fix-and-retry runs next rather than at the back of the line. The spec is resolved from the live registry or, for an evicted/archived job, from its on-disk `<id>.meta.json` sidecar — so even a failure the 100-job cap has dropped is still retryable. Kinds with no replay descriptor (e.g. `import-one`) intentionally offer no Retry. See `editor/app/jobs/actions.ts` (`retryJobAction` / `retryAllFailedAction`), the new `RetryJobButton` / `RetryAllFailedButton`, and `editor/e2e/jobs-retry.spec.ts`.
- **Reorder and promote queued jobs from Active Jobs.** A queued job's row now carries **Promote / ↑ / ↓** controls (mirroring the auto-queue policy editor's move buttons) to change its order within its queue — Promote sends it to the front so it runs next, ↑/↓ nudge it one slot. Only actionable moves render (the first-queued job shows no up/promote, the last no down), and a running job is never displaced. See `editor/app/jobs/components/ReorderJobButtons.tsx`, the `reorderJobAction` / `promoteJobAction` server actions, and `editor/e2e/jobs-reorder.spec.ts`.
diff --git a/editor/app/auto-queue/components/PolicyTreeEditor.tsx b/editor/app/auto-queue/components/PolicyTreeEditor.tsx
@@ -481,7 +481,7 @@ function LeafControls(props: EditorProps & { leaf: AutoQueueLeaf }) {
}
className={inputClass}
>
- <option value="">default</option>
+ <option value="">all buckets (default)</option>
{buckets.map((b) => (
<option key={b} value={b}>
{b}
diff --git a/editor/app/auto-queue/page.tsx b/editor/app/auto-queue/page.tsx
@@ -3,6 +3,7 @@ import Link from "next/link";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { listChannels } from "yt-dlp-transcript-common/controller/channels";
import { PLATFORM_VALUES } from "yt-dlp-transcript-common/lib/platform";
+import { bucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueuePolicy";
import { buildAutoQueueStatusPayload } from "./status";
import { AutoQueueView } from "./components/AutoQueueView";
@@ -10,11 +11,11 @@ export const dynamic = "force-dynamic";
export const metadata: Metadata = { title: "Auto-queue" };
-// Buckets each runner can draw from. Transcription: fresh downloaded-no-
-// transcript plus retry-the-failed. Download: the channel's undownloaded ids.
+// Buckets each runner can draw from — derived from the policy engine's single
+// source of truth (bucketsForKind) so the picker can't drift from the runner.
const BUCKETS_BY_KIND = {
- transcription: ["downloadedNoTranscript", "failedListed"],
- download: ["undownloadedIds"],
+ transcription: [...bucketsForKind("transcription")],
+ download: [...bucketsForKind("download")],
};
export default async function AutoQueuePage() {