commit 6e001f2b97a0e6476041c190f812435bb9141928
parent aa295e2cf4372e30b05ad00779c5cd81b6b92ccf
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 21 Aug 2026 23:29:39 -0400
sweeps: order the CHANNEL plan by recency, behind a reach setting
Ordering videos inside a channel does nothing for a video uploaded
this morning whose channel is fortieth in a heaviest-first plan. Adds
AutoQueueReach ("channel" | "corpus") plus digest.recencyReach and
backfill.reach; "corpus" additionally orders the channels.
Digest gets its dates FREE: buildDigestSweepPlan's row loop already
holds `stat`, and uploadDate is already on it, so newest/oldestPending
cost one comparison and zero extra I/O. Only rows that will actually
be GENERATED count — a video a duplicate cluster will share costs the
queue nothing and must not pull its channel up the plan.
Backfill has no dates on a snapshot, so it pays one buildRecencyKeys
call per pass over the union of every reachable id. Affordable because
passes are hours apart and recencyIndex memoizes process-wide, and it
is skipped entirely under "listed" — the ~78,000-string id collection
is not even built.
Both directions read their OWN field: "oldest" reads oldestPending,
not newestPending, or it would mean "the channel whose freshest video
is least fresh". The two-level rule is one function (planOrder.ts):
"listed" consults recency not at all, an unknown date is "" (last
under newest, first under oldest, exactly makeRecencyComparator's rule
for a video), and weight stays the tiebreak.
Resolved per PASS, not at launch, so switching mid-sweep takes effect
on the next plan rather than after re-arming a multi-week run.
Measured on the live corpus (66 channels): digest re-plans in ~1.5 s,
newest-first heads on 2026-08-20 work, oldest-first on 2008-10-04.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
8 files changed, 347 insertions(+), 5 deletions(-)
diff --git a/common/controller/backfillSweep.ts b/common/controller/backfillSweep.ts
@@ -41,6 +41,9 @@ import {
} from "../lib/backfillKinds";
import { listChannelStatsFromDisk, readChannelSnapshot } from "./channels";
import { countBackfillWork, runBackfillBatch } from "./backfillBatch";
+import type { AutoQueueOrder } from "../jobs/autoQueuePolicy";
+import { buildRecencyKeys } from "./recencyIndex";
+import { orderPlanByRecency } from "./planOrder";
export const BACKFILL_SWEEP_KIND = "backfill-sweep";
export const BACKFILL_CHANNEL_KIND = "backfill-channel";
@@ -95,6 +98,12 @@ export type BackfillPlanEntry = {
channelSlug: string;
reachable: number;
missingInput: number;
+ // Newest / oldest upload date (YYYYMMDD) among this channel's REACHABLE ids,
+ // present only when the plan was asked to order by recency. "" is "no dated
+ // reachable video": last under "newest", first under "oldest", the same rule
+ // makeRecencyComparator applies to a single video.
+ newestPending: string;
+ oldestPending: string;
};
// What is left, per channel, ordered heaviest-first by REACHABLE work.
@@ -125,6 +134,10 @@ export async function buildBackfillSweepPlan(opts: {
paths: Paths;
kindIds?: string[];
channelSlugs?: string[];
+ // Order the CHANNEL list by each channel's freshest (or oldest) reachable
+ // video instead of heaviest-first. Absent/"listed" leaves today's order
+ // untouched and costs nothing — no dates are read at all.
+ order?: AutoQueueOrder;
onLog?: (msg: string) => void;
}): Promise<BackfillPlanEntry[]> {
const wanted = new Set(opts.channelSlugs ?? []);
@@ -138,7 +151,13 @@ export async function buildBackfillSweepPlan(opts: {
const kinds = resolveBackfillKinds(getSettings(), opts.kindIds).map(
(k) => k.id,
);
+ const order = opts.order ?? "listed";
const plan: BackfillPlanEntry[] = [];
+ // Reachable ids per channel, collected only when they will be used. The
+ // snapshot's `ids` IS the reachable set (see BackfillSnapshotEntry), so this
+ // is a read of something already in hand — but on this corpus it is ~78,000
+ // strings, so it is not built when nothing will sort by it.
+ const idsBySlug = order === "listed" ? null : new Map<string, string[]>();
let walked = 0;
for (const ch of channels) {
const snapshot = await readChannelSnapshot(opts.paths, ch.slug);
@@ -146,10 +165,19 @@ export async function buildBackfillSweepPlan(opts: {
? countFromSnapshot(snapshot.backfill, kinds)
: ((walked++,
await countBackfillWork(opts.paths, ch.slug, opts.kindIds)));
+ if (idsBySlug && snapshot?.backfill) {
+ const ids = new Set<string>();
+ for (const id of kinds) {
+ for (const videoId of snapshot.backfill[id]?.ids ?? []) ids.add(videoId);
+ }
+ idsBySlug.set(ch.slug, [...ids]);
+ }
plan.push({
channelSlug: ch.slug,
reachable: counts.reachable,
missingInput: counts.missingInput,
+ newestPending: "",
+ oldestPending: "",
});
}
if (walked > 0) {
@@ -157,8 +185,54 @@ export async function buildBackfillSweepPlan(opts: {
`${walked} channel(s) had no snapshot yet and were counted by walking their videos.`,
);
}
- return plan.sort(
- (a, b) => b.reachable - a.reachable || a.channelSlug.localeCompare(b.channelSlug),
+
+ // ONE recency build for the whole plan. The digest plan gets its dates free
+ // (build:stats already put uploadDate on the row it was reading anyway); a
+ // backfill snapshot carries ids and no dates, so this is the one place that
+ // pays. It is affordable because a sweep pass is HOURS apart, and because
+ // recencyIndex memoizes a video's date process-wide once read.
+ if (idsBySlug) {
+ const candidateIds = new Set<string>();
+ const owner = new Map<string, string>();
+ for (const [slug, ids] of idsBySlug) {
+ for (const id of ids) {
+ candidateIds.add(id);
+ if (!owner.has(id)) owner.set(id, slug);
+ }
+ }
+ try {
+ const keys = await buildRecencyKeys({
+ paths: opts.paths,
+ meta: channels.map((c) => ({ slug: c.slug })),
+ candidateIds,
+ owner,
+ // Every reachable id is a directory on disk, so there is nothing for
+ // playlist interpolation to estimate and the playlist reads would be
+ // pure cost.
+ interpolate: false,
+ });
+ for (const entry of plan) {
+ for (const id of idsBySlug.get(entry.channelSlug) ?? []) {
+ const key = keys.get(id)?.key ?? "";
+ if (!key) continue;
+ if (key > entry.newestPending) entry.newestPending = key;
+ if (!entry.oldestPending || key < entry.oldestPending) {
+ entry.oldestPending = key;
+ }
+ }
+ }
+ } catch {
+ // Dates are an ordering nicety; failing to read them must not stop a
+ // sweep. Every key stays "", which sorts the plan exactly as it does
+ // today via the weight tiebreak below.
+ }
+ }
+
+ return orderPlanByRecency(
+ plan,
+ order,
+ (a, b) =>
+ b.reachable - a.reachable || a.channelSlug.localeCompare(b.channelSlug),
);
}
@@ -277,10 +351,19 @@ async function runSweepLoop(
}
pass++;
+ // Resolved per pass, not at launch: switching to newest-first mid-sweep
+ // takes effect on the next plan rather than only after stopping and
+ // re-arming. Only "corpus" reaches the CHANNEL order — under "channel" the
+ // plan stays heaviest-first and each per-channel batch applies the order
+ // itself from settings.backfill.order.
+ const backfillNow = getSettings().backfill;
+ const planOrder =
+ backfillNow.reach === "corpus" ? backfillNow.order : "listed";
const plan = await buildBackfillSweepPlan({
paths,
kindIds,
channelSlugs,
+ order: planOrder,
onLog,
});
const work = plan.filter((c) => c.reachable > 0);
@@ -296,7 +379,14 @@ async function runSweepLoop(
// BOTH NUMBERS, never their sum. See lib/backfillKinds.ts's header.
onLog(
`Pass ${pass}: ${work.length} channel(s), ${reachable.toLocaleString()} video(s) ` +
- `reachable now, ${missingInput.toLocaleString()} needing their media re-acquired.`,
+ `reachable now, ${missingInput.toLocaleString()} needing their media re-acquired` +
+ (planOrder === "listed"
+ ? ", heaviest channel first."
+ : `, ${planOrder} channel first (${work[0].channelSlug} @ ${
+ planOrder === "newest"
+ ? work[0].newestPending || "undated"
+ : work[0].oldestPending || "undated"
+ }).`),
);
let didWork = false;
diff --git a/common/controller/digestPlan.ts b/common/controller/digestPlan.ts
@@ -29,6 +29,8 @@
// GPU time and no transcript reads to compute.
import { open } from "lmdb";
+import type { AutoQueueOrder } from "../jobs/autoQueuePolicy";
+import { orderPlanByRecency } from "./planOrder";
import path from "node:path";
import type { Paths } from "../lib/paths";
import type { VideoStat } from "../lib/stats";
@@ -168,6 +170,18 @@ export type DigestChannelPlan = {
// because it is the one way the chunk total can be wrong, and it must not be
// invisible.
chunksEstimated: number;
+ // The newest and oldest upload dates (YYYYMMDD) among the videos this channel
+ // will actually GENERATE — not among everything remaining, because a video a
+ // duplicate cluster will share rather than generate costs the queue nothing
+ // and must not pull its channel up the plan.
+ //
+ // "" means no dated pending video: either the channel has nothing to generate
+ // or build:stats has no date for what it has. It sorts LAST under "newest"
+ // and FIRST under "oldest" — the same rule makeRecencyComparator applies to
+ // an individual video, so the two levels cannot disagree, and a corpus with
+ // no dates at all reproduces the historical plan exactly.
+ newestPending: string;
+ oldestPending: string;
};
export type DigestSweepPlan = {
@@ -206,6 +220,11 @@ export type BuildDigestSweepPlanOptions = {
// Restrict to these channels (the orchestrator uses it to re-price one).
channelSlugs?: string[];
clusterPlan?: DigestClusterPlan;
+ // Order the CHANNEL list by each channel's freshest (or oldest) pending
+ // video instead of heaviest-first. Absent/"listed" leaves today's
+ // chunks-descending order untouched, which is what the census wants — this is
+ // a dispatch decision, not a costing one.
+ order?: AutoQueueOrder;
onLog?: (msg: string) => void;
};
@@ -463,6 +482,8 @@ export async function buildDigestSweepPlan(
generateChunks: 0,
sharedChunks: 0,
chunksEstimated: 0,
+ newestPending: "",
+ oldestPending: "",
};
const resolved = checkFreshness
@@ -501,6 +522,17 @@ export async function buildDigestSweepPlan(
);
addTo(entry.remaining[planRole], stat.duration, chunks);
if (estimated) entry.chunksEstimated++;
+ // Free: `stat` is already in hand and `uploadDate` is already on it, so
+ // cross-channel ordering costs this comparison and no extra I/O. Only
+ // rows that will actually be generated count — see newestPending.
+ if (roleMustGenerate(planRole) && stat.uploadDate) {
+ if (stat.uploadDate > entry.newestPending) {
+ entry.newestPending = stat.uploadDate;
+ }
+ if (!entry.oldestPending || stat.uploadDate < entry.oldestPending) {
+ entry.oldestPending = stat.uploadDate;
+ }
+ }
}
for (const role of DIGEST_PLAN_ROLES) {
@@ -520,7 +552,14 @@ export async function buildDigestSweepPlan(
// ordering by audio would put the cheapest-per-hour work first while claiming to
// be heaviest-first. (It is still longest-first in practice, so early throughput
// will look better than the corpus average; see the census table above.)
- channels.sort(
+ //
+ // With `order` set, recency leads and the chunk order becomes the tiebreak —
+ // so two channels whose freshest pending video landed the same day are still
+ // visited heaviest-first, and a corpus where nothing is dated (every key "")
+ // sorts exactly as it does today.
+ orderPlanByRecency(
+ channels,
+ opts.order ?? "listed",
(a, b) =>
b.generateChunks - a.generateChunks ||
b.generateSeconds - a.generateSeconds ||
diff --git a/common/controller/digestSweep.ts b/common/controller/digestSweep.ts
@@ -201,10 +201,21 @@ async function runSweepLoop(
}
pass++;
+ // Reach, resolved per pass rather than at launch: an operator who switches
+ // to newest-first mid-sweep should see it take effect on the next plan, not
+ // only after stopping and re-arming a multi-week run.
+ //
+ // Only "corpus" reaches the CHANNEL order; "channel" leaves the plan
+ // heaviest-first and lets each per-channel batch apply the order itself
+ // (it reads settings.digest.recencyOrder on its own).
+ const digestNow = getSettings().digest;
+ const planOrder =
+ digestNow.recencyReach === "corpus" ? digestNow.recencyOrder : "listed";
const plan: DigestSweepPlan = await buildDigestSweepPlan({
paths,
lane,
channelSlugs,
+ order: planOrder,
onLog,
});
const work = plan.channels.filter((c) => c.generateSeconds > 0);
@@ -220,7 +231,14 @@ async function runSweepLoop(
`${audioHours(plan.generateSeconds).toFixed(0)} audio-hours to generate ` +
`(~${sweepDays(plan.generateChunks).toFixed(1)} days at the measured rate, ` +
`${chunksPerAudioHour(plan.generateChunks, plan.generateSeconds).toFixed(2)} chunks/audio-h), ` +
- `${audioHours(plan.sharedSeconds).toFixed(0)} audio-hours covered by cluster sharing.`,
+ `${audioHours(plan.sharedSeconds).toFixed(0)} audio-hours covered by cluster sharing` +
+ (planOrder === "listed"
+ ? ", heaviest channel first."
+ : `, ${planOrder} channel first (${work[0].channelSlug} @ ${
+ planOrder === "newest"
+ ? work[0].newestPending
+ : work[0].oldestPending
+ }).`),
);
let didWork = false;
diff --git a/common/controller/planOrder.test.ts b/common/controller/planOrder.test.ts
@@ -0,0 +1,101 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { orderPlanByRecency } from "./planOrder";
+
+// Run with: node_modules/.bin/tsx --test common/controller/planOrder.test.ts
+
+type Entry = {
+ slug: string;
+ weight: number;
+ newestPending: string;
+ oldestPending: string;
+};
+
+// Weight order and date order disagree on purpose: `heavy` is the biggest
+// channel but its work is old, `fresh` is the smallest but holds today's upload.
+// Any comparator that quietly ignores one of the two axes shows up here.
+const PLAN: Entry[] = [
+ { slug: "heavy", weight: 9000, newestPending: "20240101", oldestPending: "20170101" },
+ { slug: "fresh", weight: 12, newestPending: "20260820", oldestPending: "20260101" },
+ { slug: "mid", weight: 400, newestPending: "20250601", oldestPending: "20200101" },
+];
+
+const byWeight = (a: Entry, b: Entry): number =>
+ b.weight - a.weight || a.slug.localeCompare(b.slug);
+
+const slugs = (entries: Entry[]): string[] => entries.map((e) => e.slug);
+
+test('order "listed" reproduces the weight order exactly', () => {
+ // The compatibility claim for cross-channel ordering. An armed sweep on a
+ // 78,000-video corpus must visit channels in exactly the order it does today
+ // unless someone asked otherwise.
+ const before = [...PLAN].sort(byWeight);
+ const after = orderPlanByRecency([...PLAN], "listed", byWeight);
+ assert.deepEqual(slugs(before), ["heavy", "mid", "fresh"]);
+ assert.deepEqual(after, before);
+});
+
+test("newest visits the channel holding the freshest pending video first", () => {
+ const out = orderPlanByRecency([...PLAN], "newest", byWeight);
+ assert.deepEqual(slugs(out), ["fresh", "mid", "heavy"]);
+});
+
+test("oldest reads the OLDEST date, not the newest one", () => {
+ // heavy's oldest is 2017 and fresh's is 2026, so oldest-first inverts the
+ // newest-first list here. Reading newestPending for both directions would
+ // give the same answer by accident on this fixture only if the two orders
+ // happened to agree — they do not, which is the point of the fixture.
+ const out = orderPlanByRecency([...PLAN], "oldest", byWeight);
+ assert.deepEqual(slugs(out), ["heavy", "mid", "fresh"]);
+});
+
+test("an all-undated plan is byte-identical to today, in both directions", () => {
+ // The degradation case, and the one that has to be right: a cold stats cache
+ // or an index mid-rebuild gives every channel "". Every recency comparison
+ // then ties and the whole plan falls through to weight — today's order —
+ // rather than shuffling into an arbitrary one.
+ const undated = PLAN.map((e) => ({
+ ...e,
+ newestPending: "",
+ oldestPending: "",
+ }));
+ const today = [...undated].sort(byWeight);
+ for (const order of ["newest", "oldest"] as const) {
+ assert.deepEqual(
+ orderPlanByRecency([...undated], order, byWeight),
+ today,
+ `${order} over an undated plan must reproduce the weight order`,
+ );
+ }
+});
+
+test("an undated channel sorts last under newest and first under oldest", () => {
+ // The same rule makeRecencyComparator applies to an individual video, so the
+ // two levels of the feature cannot disagree about what "no date" means.
+ const withUndated: Entry[] = [
+ ...PLAN,
+ { slug: "undated", weight: 5, newestPending: "", oldestPending: "" },
+ ];
+ assert.equal(
+ slugs(orderPlanByRecency([...withUndated], "newest", byWeight)).at(-1),
+ "undated",
+ );
+ assert.equal(
+ slugs(orderPlanByRecency([...withUndated], "oldest", byWeight))[0],
+ "undated",
+ );
+});
+
+test("weight is the tiebreak, never discarded", () => {
+ // Two channels whose freshest pending video landed the same day are still
+ // visited heaviest-first — the ordering the sweep was built on, and still the
+ // right answer once recency has nothing left to say.
+ const sameDay: Entry[] = [
+ { slug: "light", weight: 1, newestPending: "20260820", oldestPending: "20260820" },
+ { slug: "heavy", weight: 900, newestPending: "20260820", oldestPending: "20260820" },
+ ];
+ assert.deepEqual(slugs(orderPlanByRecency(sameDay, "newest", byWeight)), [
+ "heavy",
+ "light",
+ ]);
+});
diff --git a/common/controller/planOrder.ts b/common/controller/planOrder.ts
@@ -0,0 +1,52 @@
+import type { AutoQueueOrder } from "../jobs/autoQueuePolicy";
+
+// A plan entry that can be ordered by recency: one channel's worth of pending
+// work, with the newest and oldest upload dates in it.
+//
+// Both dates, not one. Reading `newestPending` for both directions would make
+// "oldest first" mean "the channel whose freshest video is least fresh", which
+// is a different and much less useful question than "the channel holding the
+// oldest un-done work".
+export type RecencyPlanEntry = {
+ // YYYYMMDD, or "" for a channel with no dated pending work.
+ newestPending: string;
+ oldestPending: string;
+};
+
+// Sort a sweep's CHANNEL list by recency, in place, falling back to the plan's
+// own weight order.
+//
+// This is the cross-channel half of "newest first". The within-channel half is
+// batchRecency; both exist so that the same setting means the same thing at
+// both levels, and this function is where the three rules that make that true
+// live:
+//
+// 1. `order: "listed"` sorts by weight ALONE — byte-for-byte today's plan.
+// Not "sorts by a recency key that happens to tie": the historical order
+// is reproduced by not consulting recency at all.
+// 2. An unknown date is "", which sorts LAST under "newest" and FIRST under
+// "oldest" — exactly what makeRecencyComparator does to an undatable
+// video. So a corpus where nothing is dated (a cold stats cache, an index
+// mid-rebuild) falls entirely through to the weight order and reproduces
+// today, rather than shuffling into an arbitrary one.
+// 3. Weight is the TIEBREAK, never discarded. Two channels whose freshest
+// pending video landed the same day are still visited heaviest-first,
+// which is the ordering the sweep was built on and still the right answer
+// once recency has nothing left to say.
+export function orderPlanByRecency<T extends RecencyPlanEntry>(
+ entries: T[],
+ order: AutoQueueOrder,
+ byWeight: (a: T, b: T) => number,
+): T[] {
+ if (order === "listed") return entries.sort(byWeight);
+ return entries.sort((a, b) => {
+ const ka = order === "newest" ? a.newestPending : a.oldestPending;
+ const kb = order === "newest" ? b.newestPending : b.oldestPending;
+ if (ka !== kb) {
+ // Keys are YYYYMMDD, so lexicographic order IS chronological order.
+ // "newest" therefore sorts DESCENDING: a smaller (older) key comes later.
+ return order === "newest" ? (ka < kb ? 1 : -1) : ka < kb ? -1 : 1;
+ }
+ return byWeight(a, b);
+ });
+}
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -44,6 +44,33 @@ export const AUTO_QUEUE_ORDERS: ReadonlyArray<AutoQueueOrder> = [
"oldest",
];
+// How far an order REACHES.
+//
+// "channel" — sort each channel's own candidates (today, and the default). A
+// corpus-wide sweep still visits channels heaviest-first, so a video
+// uploaded this morning waits for its channel's turn.
+// "corpus" — additionally order the CHANNELS by their freshest (or oldest)
+// pending video, so the channel holding the newest work goes first.
+//
+// It is a separate axis from AutoQueueOrder rather than two more enum members
+// because it is meaningless without one: reach only says how widely an order
+// applies, and "listed" has no order to apply. The console disables the control
+// while "listed" is selected for exactly that reason.
+//
+// This deliberately does NOT interleave individual videos across channels —
+// that would break one-job-per-channel, and 77,000 single-video jobs would evict
+// the registry's 100 records.
+export type AutoQueueReach = "channel" | "corpus";
+
+export const AUTO_QUEUE_REACHES: ReadonlyArray<AutoQueueReach> = [
+ "channel",
+ "corpus",
+];
+
+export function sanitizeAutoQueueReach(value: unknown): AutoQueueReach {
+ return value === "corpus" ? "corpus" : "channel";
+}
+
// Coerce a stored/raw value to a legal order. Anything unrecognised — including
// a missing field on a settings file written before the field existed — means
// "listed", i.e. today's behaviour. One sanitizer, because the same enum is now
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -21,10 +21,12 @@ import {
} from "./workers";
import {
type AutoQueueOrder,
+ type AutoQueueReach,
type AutoQueueSettings,
defaultAutoQueue,
sanitizeAutoQueue,
sanitizeAutoQueueOrder,
+ sanitizeAutoQueueReach,
} from "../jobs/autoQueuePolicy";
import {
DEFAULT_DIARIZATION_ENGINE,
@@ -317,6 +319,9 @@ export type BackfillSettings = {
// policies' `order`, so an operator learns the control once. Default
// "listed", i.e. today's behaviour.
order: AutoQueueOrder;
+ // How far `order` reaches (see AutoQueueReach). Same two values and the same
+ // meaning as digest.recencyReach.
+ reach: AutoQueueReach;
// Re-acquire media for videos whose input is GONE (audio deleted after
// transcription). OFF by default and deliberately so: measured on this corpus,
// 836 videos still have media and ~76,270 would need a re-download — 91x the
@@ -520,6 +525,11 @@ export type DigestSettings = {
// video within a day. One enum carrying both would make them look mutually
// exclusive, which they are not.
recencyOrder: AutoQueueOrder;
+ // How far `recencyOrder` reaches (see AutoQueueReach). "channel" (the
+ // default) orders each channel's own candidates and leaves the sweep visiting
+ // channels heaviest-first; "corpus" additionally orders the CHANNELS by their
+ // freshest pending video, so the channel holding the newest work goes first.
+ recencyReach: AutoQueueReach;
// A free-text label for a non-default prompt shape, folded into the recorded
// provenance by digestPromptVariant(). Setting it invalidates every digest
// generated under a different label, which is exactly what makes a bake-off
@@ -885,6 +895,7 @@ export function defaultDigest(): DigestSettings {
timestampMode: DEFAULT_DIGEST_TIMESTAMP_MODE,
// "listed" — today's behaviour exactly. Ordering is opt-in.
recencyOrder: "listed",
+ recencyReach: "channel",
promptVariant: "",
};
}
@@ -977,6 +988,7 @@ export function sanitizeDigest(value: unknown): DigestSettings {
? r.timestampMode
: d.timestampMode,
recencyOrder: sanitizeAutoQueueOrder(r.recencyOrder),
+ recencyReach: sanitizeAutoQueueReach(r.recencyReach),
// Trimmed and length-capped: it goes into provenance on every record, and a
// runaway value would bloat 119k sidecars.
promptVariant:
@@ -1090,6 +1102,7 @@ export function defaultBackfill(): BackfillSettings {
sweepChannels: [],
// "listed" — today's behaviour exactly. Ordering is opt-in.
order: "listed",
+ reach: "channel",
// See BackfillSettings.allowRedownload — this one holds disk.
allowRedownload: false,
};
@@ -1116,6 +1129,7 @@ export function sanitizeBackfill(value: unknown): BackfillSettings {
sweepKinds: slugs(r.sweepKinds),
sweepChannels: slugs(r.sweepChannels),
order: sanitizeAutoQueueOrder(r.order),
+ reach: sanitizeAutoQueueReach(r.reach),
allowRedownload: r.allowRedownload === true,
};
}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -6,6 +6,7 @@
- **The GPU is usable from the container, including on AMD.** Alongside the default CPU whisper.cpp image there is a **Vulkan** target running parakeet.cpp — one build that covers AMD (RADV), Intel and NVIDIA, needing nothing on the host but a render node at `/dev/dri` and no vendor container toolkit — and a CUDA target for NVIDIA whisper. Measured on an RX 6600 XT, through the app's own overlapping-window wrapper: **3.4 s against 36.3 s** for the same 33-second clip pinned to the CPU. That gap is also the thing to watch for, because the failure here is silent — a Vulkan container with no `/dev/dri` does not error, it transcribes correctly on the CPU about ten times slower. The entrypoint prints which one it got on every boot, and the image ships `vulkaninfo` so you can ask directly.
- **The editor can be started without resuming whatever it was in the middle of.** Booting arms the sync heartbeat, both auto-queue runners and the digest and backfill sweeps — right for a host install, where a restart interrupts work you own, and wrong the first time a container is pointed at a corpus somebody else configured: its stored policies may say *sweep*, and a corpus-wide digest sweep is GPU-**weeks** that would start seconds after `docker compose up`. `ARCHILYZER_IDLE_BOOT=1` starts the server with all of that stopped, and you can start any of it from the UI afterwards. The shutdown reaper and the persisted-pause restore stay armed either way, because both only ever *stop* work.
- **The digest and backfill lanes can do the newest uploads first too.** *Newest first* only ever existed on two of the four pipelines. The digest batch ordered a channel by **duration** — shortest first, which is the right default and turns a backlog into visible coverage fastest — and the backfill batch used whatever order a `readdir` happened to return, which is neither of the two orders anyone would choose. So on a channel with eleven thousand undigested videos, a video uploaded this morning sat behind every one of them, and there was no setting that changed that. Both lanes now take the same three-value **Order** the auto-queue runners already use: *Listed order* (exactly today's behaviour, and still the default), *Newest first*, *Oldest first*. It is opt-in, and with it left alone every list is byte-for-byte the list it was before — that is asserted, not assumed. Digest **composes** the two rules rather than choosing between them: videos are ordered by date first and by duration inside a date, so *newest first* reads as "newest day first, shortest video within that day" and the duration rule you already had is still doing its job. Measured on the live corpus, ordering a channel's 11,329 pending digests costs about a tenth of a second.
+- **"Newest first" can now mean newest in the whole archive, not just newest in a channel.** Ordering videos inside a channel does nothing for the video uploaded this morning if its channel is fortieth in the queue — and a corpus-wide sweep visits channels heaviest-first, which on this archive means the largest channel gets eight days of attention before the second one is looked at. So Order is joined by **Reach**: *Within each channel* (today's behaviour, and the default) or *Across all channels*, which additionally orders the **channels** by the freshest — or oldest — piece of work each is holding. It is one control, not two mechanisms: an order without a reach is meaningless, so Reach is disabled while Order is left at *Listed*. The two levels share one rule for a video whose date is unknown — last under newest, first under oldest — which is what makes an archive with no dates at all come out in exactly the order it comes out in today, and heaviest-first survives as the tiebreak, so two channels holding work from the same day are still visited biggest-first. Measured on the live corpus: sixty-six channels re-ordered in about 1.5 seconds for digest, and the freshest channel under *Across all channels* is holding work from today.
- **Newest-first could quietly hand a video another channel's upload date.** The date lookup's first and cheapest layer scans the transcript index, which is keyed by the id the *platform* reports; every video the auto-queue actually asks about is named by its **folder**, and on this archive those two disagree for about one video in seven — a Rumble folder is named for the URL, while its recorded id is the embed id. Matched across the whole corpus, one channel's folder name could collide with a different channel's recorded id and inherit its date, which is a *wrong* answer rather than a missing one, and wrong dates are exactly what an ordering setting cannot survive. The lookup is now scoped to the channel the video belongs to, so a folder name that is not an id in its own channel falls through to reading the date out of the folder it actually names. Separately, the first pass over a very large backlog can only date so many videos at once; it now says so once in the log, because the remainder sorting to the back and being picked up on the next pass is the design, not a fault.
- **The auto-queue can be told to do the newest uploads first, and it now genuinely does.** Both runners always took the first video off a rule's pile, and that pile's order came straight from the channel snapshot, which sorts most buckets **alphabetically by video id** — arbitrary for YouTube ids, and oldest-first for the date-prefixed folder names some sites use. So when a channel uploaded today, nothing made that video jump the nine-thousand-video backlog in front of it; the only reason auto-download roughly worked was that its one bucket happens to be left in playlist order. There is now an **Order** setting per runner — *Listed order* (what you have today, and still the default), *Newest first*, *Oldest first*. It sorts the videos **inside** each rule, across every channel and bucket that rule claims; the rule list still decides which rule goes first, because that is what the rule list is for. For a straight newest-first archive, use one catch-all rule. Worth knowing before you switch it on: under *Newest first* the retry and partial-download buckets lose their head start, so a half-finished download can end up waiting behind fresh work. The page says so next to the setting.
- **Working out how recent 79,000 videos are turned out to be nearly free, once we stopped guessing where the dates were.** The obvious source — reading each video's metadata file — is about six and a half minutes and 41 GB of reading, on every scheduling decision, which is a non-starter. The transcript index already holds a date per video in a form that can be scanned without decoding anything: **78,583 videos in well under a second**. That covers the corpus, but it turned out **not** to cover the videos auto-transcribe actually queues, because the index only holds videos that already *have* a transcript and auto-transcribe's whole job is the ones that don't — of the 870 videos genuinely pending here, it knew the date of **122**. The gap is closed by reading the last 8 KB of each remaining video's metadata file, where the upload date happens to sit: **868 of the 870, at a fifth of a millisecond each**, and remembered afterwards so it is paid once rather than every few seconds. Videos not downloaded yet have no date anywhere on disk at all, so auto-download estimates one from the video's position in the channel's newest-first listing; those show with a `≈`, and a brand-new upload with nothing dated above it goes to the front, which is the entire point. Anything still undatable sorts to the back rather than disappearing, and a missing or busy index degrades to the old ordering instead of stopping the runner.