Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c9c0207c83bfa6294f9f559c95ae8acfe93c76fa
parent fa2f521343c6c9a2b4a94e7b9871d66926ea8c4b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  3 Oct 2026 13:20:25 -0400

mcp: a search date bound is normalised, or refused — never silently dropped

date_from/date_to accepted only YYYYMMDD; any other spelling became a footer
warning and the search ran unbounded. normalizeDateArg (mcp/src/dateArg.ts)
accepts YYYYMMDD, YYYY-MM-DD (also / or .) and ISO timestamps, and returns
null for anything else, which parseSearchArgs turns into an error result:
nothing is searched.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Meditor/CHANGELOG.md | 1+
Amcp/src/dateArg.test.ts | 27+++++++++++++++++++++++++++
Amcp/src/dateArg.ts | 29+++++++++++++++++++++++++++++
Mmcp/src/server.ts | 43+++++++++++++++++++++++++------------------
4 files changed, 82 insertions(+), 18 deletions(-)

diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched. - **`pnpm ops transcribe-bucket` transcribes a channel's downloaded-but-untranscribed videos**, as the channel page's **Transcribe N downloaded** button does, on the transcription queue. `"ids"` runs only some of them; each must be in the bucket, and a stray id is refused by name. `retry-bucket` is not the way to do this: it retries downloads, and counts a video whose audio is on disk as complete. - **`pnpm ops retry-bucket` can run part of a bucket.** Its body takes `"ids"`, a list of video ids, and runs only those, as ticking them on the bucket's card does. Every id must be in the named bucket: one that is not is refused with a 400 naming it, and nothing runs. A job started with `ids` is not replayable, like a checkbox selection in the UI. - **An X post fetch keeps what it has read when it is cancelled, times out or fails part-way, and the next fetch picks up where it stopped.** The gallery-dl fetcher used to receive an account's posts all at once when gallery-dl finished, so a long fetch that X's rate limit held past the 30-minute limit — or one you cancelled — ended with nothing saved. gallery-dl now hands over each post as it reads it: the fetch saves posts every 200 posts or every minute along with gallery-dl's own resume point, and the next fetch of the channel continues from that point instead of starting again. A fetch of new posts stops once it reaches 100 already-archived posts in a row instead of reading the whole timeline, and a fetch of an account's history may run for up to 3 hours (new-post fetches keep the 30-minute limit). gallery-dl's rate-limit waits now appear in the job's log as they happen. diff --git a/mcp/src/dateArg.test.ts b/mcp/src/dateArg.test.ts @@ -0,0 +1,27 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { normalizeDateArg } from "./dateArg"; + +test("normalizeDateArg: YYYYMMDD passes through", () => { + assert.equal(normalizeDateArg("20241026"), "20241026"); +}); + +test("normalizeDateArg: dashed, slashed, dotted and ISO timestamps normalise", () => { + assert.equal(normalizeDateArg("2024-10-26"), "20241026"); + assert.equal(normalizeDateArg(" 2024/10/26 "), "20241026"); + assert.equal(normalizeDateArg("2024.10.26"), "20241026"); + assert.equal(normalizeDateArg("2024-10-26T00:00:00Z"), "20241026"); + assert.equal(normalizeDateArg("2024-10-26T12:30:00+02:00"), "20241026"); +}); + +test("normalizeDateArg: absent or blank is no bound", () => { + assert.equal(normalizeDateArg(undefined), undefined); + assert.equal(normalizeDateArg(null), undefined); + assert.equal(normalizeDateArg(" "), undefined); +}); + +test("normalizeDateArg: anything else is refused, never read as no bound", () => { + for (const bad of ["2024", "202410", "2024-1-5", "Oct 26 2024", "26/10/2024", "20241332", "20240230", 20241026, "yesterday"]) { + assert.equal(normalizeDateArg(bad), null, String(bad)); + } +}); diff --git a/mcp/src/dateArg.ts b/mcp/src/dateArg.ts @@ -0,0 +1,29 @@ +// An upload-date bound as a tool caller may spell it, normalised to the index's +// YYYYMMDD. +// +// THE INDEX COMPARES YYYYMMDD LEXICOGRAPHICALLY, so a date in any other shape +// cannot be used as given. It used to be dropped with a footer warning, and the +// search ran with NO bound — a complete-set count over the whole corpus that +// read exactly like the bounded count the caller asked for. Now the common +// spellings are accepted, and anything else is refused (`null`) so the caller +// gets an error and nothing is searched. +// +// Returns the YYYYMMDD string, `undefined` for an absent or blank value, or +// `null` for a value that is not a date. +export function normalizeDateArg(raw: unknown): string | undefined | null { + if (raw === undefined || raw === null) return undefined; + if (typeof raw !== "string") return null; + const v = raw.trim(); + if (!v) return undefined; + const m = + /^(\d{4})(\d{2})(\d{2})$/.exec(v) ?? + /^(\d{4})[-/.](\d{2})[-/.](\d{2})(?:[T ][\d:.]+(?:Z|[+-]\d{2}:?\d{2})?)?$/.exec(v); + if (!m) return null; + const [, y, mo, d] = m; + const month = Number(mo); + const day = Number(d); + if (month < 1 || month > 12 || day < 1) return null; + const last = new Date(Date.UTC(Number(y), month, 0)).getUTCDate(); + if (day > last) return null; + return `${y}${mo}${d}`; +} diff --git a/mcp/src/server.ts b/mcp/src/server.ts @@ -1,3 +1,4 @@ +import { normalizeDateArg } from "./dateArg"; import { Server, type Prompt, @@ -245,11 +246,11 @@ const SEARCH_FILTER_ARGS = { }, date_from: { type: "string", - description: "Keep only videos uploaded on or after this date (YYYYMMDD).", + description: "Keep only videos uploaded on or after this date (YYYYMMDD or YYYY-MM-DD; anything else is an error).", }, date_to: { type: "string", - description: "Keep only videos uploaded on or before this date (YYYYMMDD).", + description: "Keep only videos uploaded on or before this date (YYYYMMDD or YYYY-MM-DD; anything else is an error).", }, media_type: { type: "string", @@ -1426,6 +1427,7 @@ async function handleSearch( const includeSnippets = args.include_snippets !== false; const base = linkStyleOf(args) === "base"; const parsed = parseSearchArgs(args); + if (parsed.error) return errorText(parsed.error); const result = await searchTranscripts(source, { query, filters: parsed.filters, @@ -1604,6 +1606,7 @@ async function handleEnumerateMatches( const batchSize = batchSizeArg >= 1 ? batchSizeArg : 8; const parsed = parseSearchArgs(args); + if (parsed.error) return errorText(parsed.error); const result = await searchTranscripts(source, { query, filters: parsed.filters, @@ -1710,6 +1713,9 @@ type ParsedSearchArgs = { scopes?: LayerScope[]; collapseDuplicates: boolean; warnings: string[]; + // Set when an argument must stop the call rather than be dropped (a date + // bound that is not a date). Callers return it as an error and search nothing. + error?: string; }; const SCOPE_TOKENS: ReadonlyArray<LayerScope> = [ @@ -1753,20 +1759,20 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs { warnings.push(`unknown age ignored: ${String(args.age)}`); } - const dateFrom = typeof args.date_from === "string" ? args.date_from.trim() : ""; - const dateTo = typeof args.date_to === "string" ? args.date_to.trim() : ""; - for (const [name, v] of [ - ["date_from", dateFrom], - ["date_to", dateTo], - ] as const) { - // The comparison is lexicographic on YYYYMMDD, so a differently-shaped date - // wouldn't error — it would quietly filter wrongly. Say so instead. - if (v && !/^\d{8}$/.test(v)) { - warnings.push(`${name}="${v}" is not YYYYMMDD — it was ignored`); - } - } - const from = /^\d{8}$/.test(dateFrom) ? dateFrom : undefined; - const to = /^\d{8}$/.test(dateTo) ? dateTo : undefined; + // A date bound the caller cannot have meant is an ERROR, not a warning: an + // ignored bound widens the search to the whole corpus and the count reads as + // the bounded one (see dateArg.ts). The common spellings are normalised. + const from = normalizeDateArg(args.date_from); + const to = normalizeDateArg(args.date_to); + const badDates = [ + ["date_from", args.date_from, from], + ["date_to", args.date_to, to], + ] + .filter(([, , v]) => v === null) + .map(([name, raw]) => `${name}=${JSON.stringify(raw)}`); + const error = badDates.length + ? `not a date: ${badDates.join(", ")} — use YYYYMMDD or YYYY-MM-DD. Nothing was searched.` + : undefined; // Curated tags. A malformed id is REPORTED, never silently dropped: dropping // the only tag in the list would widen the search back to the whole corpus @@ -1794,8 +1800,8 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs { states !== undefined || mediaType !== undefined || age !== undefined || - from !== undefined || - to !== undefined || + Boolean(from) || + Boolean(to) || tags !== undefined; const filters: SearchFilters | null = anyFilter @@ -1835,6 +1841,7 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs { scopes, collapseDuplicates: args.collapse_duplicates !== false, warnings, + ...(error ? { error } : {}), }; }