commit 980624e9c2e7496bb19956e891ce6d092ae46ae4
parent 685a84c8e0e13f334d705a82f21d446e8933e204
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 21 Aug 2026 23:16:54 -0400
recency: scope the index lookup to the owning channel
Layer 1 scans the transcript index, which is keyed by summary.id — the
metadata id. Every candidate id that reaches buildRecencyKeys is a
DIRECTORY NAME: snapshot.backfill[op].ids, the snapshot buckets and a
readdir of data/ all name directories. They diverge for ~14.5% of this
corpus (the-quartering-rumble/data/v1007ay has metadata id vxe1ae),
and a flat corpus-wide map turns that divergence into a wrong date
rather than a missing one — channel A's directory name can collide
with channel B's metadata id and inherit B's date.
Scope the lookup through datesBySlug when `owner` is supplied (every
scheduling path supplies it). A dir name that is not a metadata id in
its own channel then falls to layer 2, which reads the date out of the
directory it names and is correct by construction.
Also log once when a build hits TAIL_READS_PER_BUILD: the first
corpus-wide pass will, and it self-corrects on the next refresh, so it
should read as expected rather than as a bug.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
3 files changed, 123 insertions(+), 2 deletions(-)
diff --git a/common/controller/recencyIndex.test.ts b/common/controller/recencyIndex.test.ts
@@ -241,3 +241,82 @@ test("makeRecencyComparator: an unknown id sorts oldest, never crashes", () => {
["known", "ghost"],
);
});
+
+// --- Layer 1: the index scan, and the two key spaces it straddles ------------
+
+// Write a byChannel sub-DB the way buildIndex does: key [slug, YYYYMMDD, id],
+// constant value, msgpack. That id is the METADATA id, which is not always the
+// directory name — see the layer-1 comment in recencyIndex.ts.
+async function writeIndex(
+ lmdbPath: string,
+ rows: ReadonlyArray<[string, string, string]>,
+): Promise<void> {
+ const { open } = await import("lmdb");
+ const root = open({ path: lmdbPath, maxDbs: 14 });
+ const byChannel = root.openDB<number, [string, string, string]>({
+ name: "byChannel",
+ encoding: "msgpack",
+ });
+ for (const row of rows) await byChannel.put(row, 1);
+ await byChannel.flushed;
+ await root.close();
+}
+
+test("layer 1 is scoped to the owning channel, not the whole corpus", async () => {
+ // The bug this pins: candidate ids are DIRECTORY NAMES, the index is keyed by
+ // METADATA ids, and on this corpus they diverge for ~14.5% of videos (a
+ // Rumble dir is the URL slug, its metadata id is the embed id). A flat
+ // corpus-wide map lets channel A's directory name collide with channel B's
+ // metadata id and inherit B's date — a wrong answer, not a missing one.
+ const dir = await mkdtemp(path.join(tmpdir(), "recency-scope-"));
+ try {
+ clearRecencyCache();
+ const lmdbPath = path.join(dir, "index.mdb");
+ const channelsDir = path.join(dir, "channels");
+ // "v1007ay" is a metadata id in channel `other`, and a DIRECTORY name in
+ // channel `rumble` whose metadata id is `vxe1ae`.
+ await writeIndex(lmdbPath, [
+ ["other", "20200101", "v1007ay"],
+ ["rumble", "20260812", "vxe1ae"],
+ ]);
+ await writeMeta(channelsDir, "rumble", "v1007ay", "20260812");
+ const keys = await buildRecencyKeys({
+ paths: { lmdbPath, channelsDir } as Paths,
+ meta: [{ slug: "other" }, { slug: "rumble" }],
+ candidateIds: new Set(["v1007ay"]),
+ owner: new Map([["v1007ay", "rumble"]]),
+ interpolate: false,
+ fresh: true,
+ });
+ // Not 20200101 — that is the other channel's video. Layer 1 misses, and
+ // layer 2 reads the truth out of the directory the id actually names.
+ assert.deepEqual(keys.get("v1007ay"), {
+ key: "20260812",
+ estimated: false,
+ });
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("layer 1 still serves an id that IS its channel's metadata id", async () => {
+ // The 85.5% case must not regress: same-space ids are answered by the scan,
+ // with no tail read at all (there is no metadata.info.json to read here).
+ const dir = await mkdtemp(path.join(tmpdir(), "recency-scope-hit-"));
+ try {
+ clearRecencyCache();
+ const lmdbPath = path.join(dir, "index.mdb");
+ await writeIndex(lmdbPath, [["ch", "20260812", "vid"]]);
+ const keys = await buildRecencyKeys({
+ paths: { lmdbPath, channelsDir: path.join(dir, "channels") } as Paths,
+ meta: [{ slug: "ch" }],
+ candidateIds: new Set(["vid"]),
+ owner: new Map([["vid", "ch"]]),
+ interpolate: false,
+ fresh: true,
+ });
+ assert.deepEqual(keys.get("vid"), { key: "20260812", estimated: false });
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
diff --git a/common/controller/recencyIndex.ts b/common/controller/recencyIndex.ts
@@ -161,6 +161,13 @@ const TAIL_MEMO_CAP = 200_000;
const TAIL_READS_PER_BUILD = 8000;
const TAIL_READ_CONCURRENCY = 32;
+// Hitting the cap is EXPECTED once, not a fault: the first pass that orders a
+// corpus-wide operation (digest's 77,508 pending, say) has tens of thousands of
+// undated ids and the memo is cold. It self-corrects — each refresh keys another
+// 8,000 and the rest sort oldest meanwhile — so the line says so explicitly,
+// and says it once, or a 3-second poll would print it for an hour.
+let tailCapLogged = false;
+
async function readTailUploadDate(file: string): Promise<string | null> {
let fh;
try {
@@ -200,7 +207,16 @@ async function datesFromMetadata(
}
if (!owner.has(id)) continue;
todo.push(id);
- if (todo.length >= TAIL_READS_PER_BUILD) break;
+ if (todo.length >= TAIL_READS_PER_BUILD) {
+ if (!tailCapLogged) {
+ tailCapLogged = true;
+ console.log(
+ `[recency] dating ${TAIL_READS_PER_BUILD} of ${wanted.size} undated candidates this pass; ` +
+ `the remainder sort oldest until a later refresh keys them (expected on a first corpus-wide pass)`,
+ );
+ }
+ break;
+ }
}
if (todo.length === 0) return;
const dates = await mapConcurrent(todo, TAIL_READ_CONCURRENCY, (id) =>
@@ -343,6 +359,7 @@ export function clearRecencyCache(): void {
cache = null;
cacheKey = "";
tailMemo.clear();
+ tailCapLogged = false;
}
async function loadSources(
@@ -416,9 +433,33 @@ export async function buildRecencyKeys({
const sources = await loadSources(paths, slugs, interpolate, fresh);
// Layer 1: the index scan.
+ //
+ // TWO KEY SPACES MEET HERE, and they are not the same space. The index is
+ // keyed by the METADATA id (buildIndex writes `summary.id`, i.e. the id the
+ // platform reports), while every candidate id reaching this function is a
+ // DIRECTORY NAME — snapshot.backfill[op].ids, the snapshot buckets and a
+ // readdir of data/ all name directories. They agree for ~85.5% of this
+ // corpus and diverge for the rest: `the-quartering-rumble/data/v1007ay` has
+ // metadata id `vxe1ae`, because Rumble's dir is the URL slug and its
+ // metadata id is the embed id.
+ //
+ // A flat corpus-wide `dates` map turns that divergence into a WRONG DATE
+ // rather than a missing one: channel A's directory name can collide with
+ // channel B's metadata id and silently inherit B's upload date. So when the
+ // caller supplies `owner` — every scheduling path does — the lookup is
+ // scoped to the owning channel's own map, which contains that channel's ids
+ // only. A dir name that is not a metadata id in its own channel then falls
+ // through to layer 2, which reads the date out of the directory it names and
+ // is therefore correct by construction.
+ //
+ // Without `owner` there is nothing to scope by, so the flat map stands; that
+ // path is tests and the offline sanity script, not a runner.
const missing = new Set<string>();
for (const id of candidateIds) {
- const date = sources.dates.get(id);
+ const slug = owner?.get(id);
+ const date = slug
+ ? sources.datesBySlug.get(slug)?.get(id)
+ : sources.dates.get(id);
if (date) out.set(id, { key: date, estimated: false });
else missing.add(id);
}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -5,6 +5,7 @@
- **Nothing the container runs is reachable from outside the machine until you say so.** The editor has no authentication of any kind, shells out to yt-dlp, and deletes media — so **no application container publishes a port at all**. Caddy is the single front door, and every one of the four ports it publishes binds `127.0.0.1` by default, the public sites included; opening one is a deliberate edit of a single line in `.env`. Because a bind address is exactly the sort of thing that gets changed in a hurry, there is also a rail: if a private app is bound off-loopback with nothing checking credentials, **the containers refuse to start** — both the app and Caddy, which is the process that actually opens the ports — and print the four ways to fix it. `basic_auth` is built into Caddy so a password needs nothing installed; Tinyauth and Authelia attach through `forward_auth` as documented drop-in overlays; `ARCHILYZER_AUTH_MODE=none` is the one explicit escape hatch for people who already have their own front door.
- **The GPU is usable from the container, including on AMD.** Alongside the default CPU whisper.cpp image there is a **Vulkan** target running parakeet.cpp — one build that covers AMD (RADV), Intel and NVIDIA, needing nothing on the host but a render node at `/dev/dri` and no vendor container toolkit — and a CUDA target for NVIDIA whisper. Measured on an RX 6600 XT, through the app's own overlapping-window wrapper: **3.4 s against 36.3 s** for the same 33-second clip pinned to the CPU. That gap is also the thing to watch for, because the failure here is silent — a Vulkan container with no `/dev/dri` does not error, it transcribes correctly on the CPU about ten times slower. The entrypoint prints which one it got on every boot, and the image ships `vulkaninfo` so you can ask directly.
- **The editor can be started without resuming whatever it was in the middle of.** Booting arms the sync heartbeat, both auto-queue runners and the digest and backfill sweeps — right for a host install, where a restart interrupts work you own, and wrong the first time a container is pointed at a corpus somebody else configured: its stored policies may say *sweep*, and a corpus-wide digest sweep is GPU-**weeks** that would start seconds after `docker compose up`. `ARCHILYZER_IDLE_BOOT=1` starts the server with all of that stopped, and you can start any of it from the UI afterwards. The shutdown reaper and the persisted-pause restore stay armed either way, because both only ever *stop* work.
+- **Newest-first could quietly hand a video another channel's upload date.** The date lookup's first and cheapest layer scans the transcript index, which is keyed by the id the *platform* reports; every video the auto-queue actually asks about is named by its **folder**, and on this archive those two disagree for about one video in seven — a Rumble folder is named for the URL, while its recorded id is the embed id. Matched across the whole corpus, one channel's folder name could collide with a different channel's recorded id and inherit its date, which is a *wrong* answer rather than a missing one, and wrong dates are exactly what an ordering setting cannot survive. The lookup is now scoped to the channel the video belongs to, so a folder name that is not an id in its own channel falls through to reading the date out of the folder it actually names. Separately, the first pass over a very large backlog can only date so many videos at once; it now says so once in the log, because the remainder sorting to the back and being picked up on the next pass is the design, not a fault.
- **The auto-queue can be told to do the newest uploads first, and it now genuinely does.** Both runners always took the first video off a rule's pile, and that pile's order came straight from the channel snapshot, which sorts most buckets **alphabetically by video id** — arbitrary for YouTube ids, and oldest-first for the date-prefixed folder names some sites use. So when a channel uploaded today, nothing made that video jump the nine-thousand-video backlog in front of it; the only reason auto-download roughly worked was that its one bucket happens to be left in playlist order. There is now an **Order** setting per runner — *Listed order* (what you have today, and still the default), *Newest first*, *Oldest first*. It sorts the videos **inside** each rule, across every channel and bucket that rule claims; the rule list still decides which rule goes first, because that is what the rule list is for. For a straight newest-first archive, use one catch-all rule. Worth knowing before you switch it on: under *Newest first* the retry and partial-download buckets lose their head start, so a half-finished download can end up waiting behind fresh work. The page says so next to the setting.
- **Working out how recent 79,000 videos are turned out to be nearly free, once we stopped guessing where the dates were.** The obvious source — reading each video's metadata file — is about six and a half minutes and 41 GB of reading, on every scheduling decision, which is a non-starter. The transcript index already holds a date per video in a form that can be scanned without decoding anything: **78,583 videos in well under a second**. That covers the corpus, but it turned out **not** to cover the videos auto-transcribe actually queues, because the index only holds videos that already *have* a transcript and auto-transcribe's whole job is the ones that don't — of the 870 videos genuinely pending here, it knew the date of **122**. The gap is closed by reading the last 8 KB of each remaining video's metadata file, where the upload date happens to sit: **868 of the 870, at a fifth of a millisecond each**, and remembered afterwards so it is paid once rather than every few seconds. Videos not downloaded yet have no date anywhere on disk at all, so auto-download estimates one from the video's position in the channel's newest-first listing; those show with a `≈`, and a brand-new upload with nothing dated above it goes to the front, which is the entire point. Anything still undatable sorts to the back rather than disappearing, and a missing or busy index degrades to the old ordering instead of stopping the runner.
- **The auto-queue page now answers "what is it doing, and why not?".** It was two identical panels of pending counts and a pick log — and the payload it was already receiving contained the answers to both questions, thrown away on arrival. A **dispatch deck** across the top gives both runners at a glance, so you never scroll to find out about the other one. Each lane then reads top to bottom as the questions you actually arrive with. **Next up** names the exact video the policy would hand out next, with its upload date, the rule that claimed it, and *why that rule* — "rule 1 has no pending work" — which is also the fastest way to see that an ordering change did what you asked. **In flight** lists what is running right now with how long each has been going, which the page received and rendered as a bare count. The policy tree became a **claim ladder**: one rung per rule, numbered by its real priority, saying in plain language what it matches, with a hairline down its left edge whose fill shows how much of the runner that rung is currently holding. The pending count on each rung opens to show the actual videos, in the exact order they will be handed out. This folds three previously separate sections — the tree, "Pending per rule" and "Recent picks", which you had to cross-reference by eye — into one object. The lane header gains how long the runner has been up and a link to its job log, both of which were on the wire and discarded.