Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a0e73ebe37e6f0f743a387712455d066e758163f
parent 763578d6327c28883f45c596af328f39bc846ef2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed,  7 Oct 2026 13:13:11 -0400

Merge x-older-from (older X walk can start at a date; Kiwi Farms player uploads are video media)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 2+-
Mcommon/bin/fetch-posts.ts | 9++++++---
Mcommon/controller/fetchOlderPosts.test.ts | 55+++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/fetchPosts.ts | 28+++++++++++++++++++++++-----
Mcommon/social/xenforoParse.test.ts | 18++++++++++++++++++
Mcommon/social/xenforoParse.ts | 14++++++++++++++
Meditor/app/api/ops/fetch-posts/route.ts | 8+++++---
Meditor/app/channels/[slug]/socialActions.ts | 11+++++++++--
Meditor/app/jobs/jobReplayRegistry.ts | 1+
Mscripts/archilyzer-ops.mjs | 4+++-
10 files changed, 135 insertions(+), 15 deletions(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -394,7 +394,7 @@ export const COMMANDS: Command[] = [ script(["duplicates"], "duplicate-shorts.ts", "[--threshold N] [--all-durations] [--blocking title|duration|both] [--near F] [--tolerance N] … on-demand duplicate detection (after index + stats)", 8192), script(["posts", "fetch"], "fetch-posts.ts", - "--slug <channel> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post; --pages: a forum thread's latest N pages)"), + "--slug <channel> [--full | --older [--floor YYYY-MM-DD] [--from YYYY-MM-DD] [--force]] [--limit N] [--pages N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post; --pages: a forum thread's latest N pages)"), { path: ["posts", "import-html"], usage: diff --git a/common/bin/fetch-posts.ts b/common/bin/fetch-posts.ts @@ -2,7 +2,7 @@ // Fetch social posts for one channel into its on-disk posts corpus. // // pnpm --filter yt-dlp-transcript-common exec tsx bin/fetch-posts.ts \ -// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N] +// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--from YYYY-MM-DD] [--force]] [--limit N] [--pages N] // // --pages caps how many pages one run reads, for a source read page by page (a // forum thread: its latest N pages, newest first; the next run continues where @@ -14,7 +14,9 @@ // // --older walks the account's history backwards from the oldest archived post // (X: search windows), below what the timeline reaches; --floor is the date it -// stops at. Its position is saved apart from the timeline's, so a later run +// stops at; --from starts it afresh at a date, replacing its saved position +// (and a "complete") — for a gap above one surviving old post. Its position +// is saved apart from the timeline's, so a later run // continues it and a normal fetch is unaffected. An account that shows no // posts (nothing archived, and the last timeline fetch read none) is refused // a search walk unless --force. @@ -28,7 +30,7 @@ const flags = parseFlags(process.argv.slice(2)); const slug = flags.slug; if (!slug) { console.error( - "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N]", + "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--from YYYY-MM-DD] [--force]] [--limit N] [--pages N]", ); process.exit(2); } @@ -52,6 +54,7 @@ fetchPosts({ full: flags.full === "true", older: flags.older === "true", floor: flags.floor, + from: flags.from, force: flags.force === "true", limit, pages, diff --git a/common/controller/fetchOlderPosts.test.ts b/common/controller/fetchOlderPosts.test.ts @@ -592,3 +592,58 @@ test("a drain already set when a window ends stops before the pause, never waiti assert.equal((await argvLines()).length - before, 1); assert.equal((await readPostFetchState(channelRoot))?.older?.since, "2020-09-11"); }); + +test("from starts a completed walk afresh above a sparse old post, and covers the gap", async () => { + const slug = "older-from"; + const channelRoot = await makeChannel(slug); + // Only the OLDEST post survives in the archive (a timeline that returned + // one 2019 post), and the walk below it is already complete. + const oldest = POST_DATES[POST_DATES.length - 1]; + const post = normalizeXTweet( + { tweet_id: idOf(oldest), date: oldest, content: "old", author: { name: HANDLE }, user: { name: HANDLE } }, + slug, + ); + assert.ok(post); + await writePosts(channelRoot, [post]); + await writePostFetchState(channelRoot, { + lastFetchedAt: "2026-01-01T00:00:00.000Z", + cursor: "1/TIMELINE-RESUME", + older: { since: "2019-08-20", until: "2019-11-21", emptyWindows: 4, complete: true, completeReason: "done" }, + }); + + // Without a start date there is nothing to walk. + const idle = await fetchPosts({ paths: getPaths(), slug, settings: LOGIN, older: true, olderWindowPauseMs: 0 }); + assert.equal(idle.written, 0); + + const before = (await argvLines()).length; + const log: string[] = []; + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + from: "2021-03-10", + olderWindowPauseMs: 0, + onLog: (l) => log.push(l), + }); + assert.equal(result.ok, true, log.join("\n")); + assert.equal(result.written, 4); + assert.equal((await readSeenPostIds(channelRoot)).size, 5); + const queries = (await argvLines()).slice(before).map(queryOf); + assert.equal(queries[0], `from:${HANDLE} since:2020-12-11 until:2021-03-11 include:nativeretweets`); + assert.ok(log.some((l) => /starting afresh from 2021-03-10/.test(l)), log.join("\n")); + const state = await readPostFetchState(channelRoot); + assert.equal(state?.older?.complete, true); + assert.equal(state?.cursor, "1/TIMELINE-RESUME"); +}); + +test("from is refused without older, and when it is not a date", async () => { + const slug = "older-from-refusals"; + await makeChannel(slug); + const a = await fetchPosts({ paths: getPaths(), slug, settings: LOGIN, from: "2021-01-01" }); + assert.equal(a.ok, false); + assert.match(a.error ?? "", /only to an older-posts fetch/); + const b = await fetchPosts({ paths: getPaths(), slug, settings: LOGIN, older: true, from: "2021-13-01" }); + assert.equal(b.ok, false); + assert.match(b.error ?? "", /not a date/); +}); diff --git a/common/controller/fetchPosts.ts b/common/controller/fetchPosts.ts @@ -78,6 +78,13 @@ export type FetchPostsOptions = { // later runs of the walk. Default: none (it stops after a run of empty // windows, or at the account's creation date). floor?: string; + // Start the older walk afresh, backwards from this date (YYYY-MM-DD), in + // place of its saved position — and of a "complete" it reached. The walk + // otherwise starts below the OLDEST archived post, so one surviving old post + // (a timeline that returns a handful of 2018 posts beside its 3,200 recent + // ones) puts years of history below a gap no window ever covers. Only with + // `older`. + from?: string; // The older walk's pause between two windows. Default: the fetcher's own. olderWindowPauseMs?: number; // Walk older posts even on an account that shows none (see @@ -96,6 +103,8 @@ export type FetchPostsOptions = { // The refusal for both walks at once, shared with the server action so the // button, the ops route and this controller say the same thing. +export const OLDER_FROM_REFUSAL = "A start date (\"from\") applies only to an older-posts fetch."; + export const FULL_AND_OLDER_REFUSAL = "A full re-fetch and an older-posts fetch are different walks — run one at a time."; @@ -149,6 +158,14 @@ export async function fetchPosts( if (opts.full && opts.older) { return { ok: false, written: 0, skipped: 0, complete: false, error: FULL_AND_OLDER_REFUSAL }; } + for (const [name, day] of [["from", opts.from]] as const) { + if (day !== undefined && !isUtcDay(day)) { + return { ok: false, written: 0, skipped: 0, complete: false, error: `"${day}" is not a date (YYYY-MM-DD) for "${name}"` }; + } + } + if (opts.from !== undefined && !opts.older) { + return { ok: false, written: 0, skipped: 0, complete: false, error: OLDER_FROM_REFUSAL }; + } if (opts.floor !== undefined && !isUtcDay(opts.floor)) { return { ok: false, @@ -399,7 +416,8 @@ async function fetchOlderPosts(ctx: { const { slug } = opts; const log = (line: string) => opts.onLog?.(line); const fetchOlder = fetcher.fetchOlder!; - const prior = priorState?.older; + // A start date replaces the saved walk, complete or not. + const prior = opts.from === undefined ? priorState?.older : undefined; if (prior?.complete) { log( @@ -418,8 +436,8 @@ async function fetchOlderPosts(ctx: { } if (empty) log("[warn] The account shows no posts; walking anyway, as forced."); - const floor = opts.floor ?? prior?.floor; - let accountCreatedAt = prior?.accountCreatedAt; + const floor = opts.floor ?? priorState?.older?.floor; + let accountCreatedAt = priorState?.older?.accountCreatedAt; const startedAt = new Date().toISOString(); let position: OlderBackfillPosition; if (prior) { @@ -430,7 +448,7 @@ async function fetchOlderPosts(ctx: { ...(prior.maxId ? { maxId: prior.maxId } : {}), }; } else { - const oldest = await oldestPostCreatedAt(channelRoot); + const oldest = opts.from ?? (await oldestPostCreatedAt(channelRoot)); position = firstOlderWindow({ oldestArchivedAt: oldest ?? undefined, floor: olderFloorDay(floor, accountCreatedAt), @@ -439,7 +457,7 @@ async function fetchOlderPosts(ctx: { log( `Fetching older posts for ${slug} via ${fetcher.label} (@${handle}): ` + - (prior ? "resuming at " : "starting at ") + + (prior ? "resuming at " : opts.from ? `starting afresh from ${opts.from} at ` : "starting at ") + `${position.since} – ${position.until}` + (floor ? `, down to ${floor}` : "") + `; ${ctx.seenIds.size} already archived.`, diff --git a/common/social/xenforoParse.test.ts b/common/social/xenforoParse.test.ts @@ -243,3 +243,21 @@ test("a message with no id or no date is not a post", () => { .replace(/datetime="[^"]*"/, ""); assert.equal(parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts.length, 0); }); + +test("a Kiwi Farms player upload is video media named by its file, not a bare duration", () => { + const player = + `<div class="ephyra-player ephyra-player--video" id="media-9660049" data-engine="videojs"` + + ` data-manifest="/ephyra-stream/9619135/master.m3u8?attachment_id=9660049"` + + ` data-source-fallback="//uploads.kiwifarms.st/data/video/9619/9619135-10fa.mp4?hash=lZ6"` + + ` data-duration="209.207" data-duration-label="3:29" data-attachment-id="9660049"` + + ` data-media-type="video" data-filename="Teapot Intro [aLfKTn4x7q8].mp4">` + + `<button type="button" class="ephyra-lazy-poster"><img src="/ephyra-stream/9619135/poster.avif" alt=""></button>` + + `<span class="ephyra-duration">3:29</span></div>`; + const html = threadPage({ page: 1, last: 1, posts: [post(700, 7, `Before it goes:${player}Watch it.`)] }); + const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts; + assert.deepEqual(p.media, [ + { kind: "video", url: "https://uploads.kiwifarms.st/data/video/9619/9619135-10fa.mp4?hash=lZ6", name: "Teapot Intro [aLfKTn4x7q8].mp4" }, + ]); + assert.ok(p.text.includes("[video: Teapot Intro [aLfKTn4x7q8].mp4 (3:29)]"), p.text); + assert.ok(!/^3:29$/m.test(p.text), "the duration label is not a line of its own"); +}); diff --git a/common/social/xenforoParse.ts b/common/social/xenforoParse.ts @@ -410,6 +410,20 @@ function render(node: HtmlNode, ctx: BodyCtx, out: string[]): void { return; } + // Kiwi Farms' own player ("ephyra"): a div carrying the upload as data + // attributes, its <video> built by script. The file is the fallback source; + // without this branch only the player's duration label ("1:11") reached the + // text, and a forum-hosted clip — often the only copy — was not media. + if ((el.attrs["data-media-type"] === "video" || el.attrs["data-media-type"] === "audio") && + el.attrs["data-source-fallback"]) { + const url = resolveForumUrl(el.attrs["data-source-fallback"], ctx.origin); + const name = el.attrs["data-filename"]; + const label = el.attrs["data-duration-label"]; + if (url) addMedia(ctx, { kind: "video", url, ...(name ? { name } : {}) }); + out.push(SOFT, `[${el.attrs["data-media-type"]}${name ? `: ${name}` : ""}${label ? ` (${label})` : ""}]`, SOFT); + return; + } + if (el.tag === "video" || el.tag === "audio") { const src = el.attrs.src ?? diff --git a/editor/app/api/ops/fetch-posts/route.ts b/editor/app/api/ops/fetch-posts/route.ts @@ -10,14 +10,15 @@ import { export const dynamic = "force-dynamic"; -// POST { slug, full?, older?, floor?, force?, limit?, pages?, queueKey? } -> { ok: true, jobId } +// POST { slug, full?, older?, floor?, from?, force?, limit?, pages?, queueKey? } -> { ok: true, jobId } // // `pages` caps how many pages a run reads, for a channel read page by page (a // forum thread: its latest N pages). // // A social channel's "Fetch posts" button, over HTTP — and its "Re-fetch full // history" (`full`) and "Fetch older posts" (`older`, with an optional `floor` -// date, YYYY-MM-DD, that the walk stops at). The job runs on the platform's +// date, YYYY-MM-DD, that the walk stops at, and an optional `from` date it +// starts afresh at, replacing its saved position). The job runs on the platform's // queue, as the buttons' does. // // Every refusal is the action's own sentence: a channel that is not a social @@ -28,7 +29,7 @@ export const dynamic = "force-dynamic"; export async function POST(request: Request) { return ops( request, - ["slug", "full", "older", "floor", "force", "limit", "pages", "queueKey"], + ["slug", "full", "older", "floor", "from", "force", "limit", "pages", "queueKey"], async (body) => { const slug = reqSlug(body, "slug"); return jobResponse( @@ -41,6 +42,7 @@ export async function POST(request: Request) { optString(body, "floor"), optBool(body, "force"), optPositiveInt(body, "pages"), + optString(body, "from"), ), ); }, diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts @@ -23,7 +23,7 @@ import { isSocialChannel } from "yt-dlp-transcript-common/lib/channelConfig"; import { emptyAccountOlderProblem, fetchPosts, - FULL_AND_OLDER_REFUSAL, + FULL_AND_OLDER_REFUSAL, OLDER_FROM_REFUSAL, olderPostsProblem, } from "yt-dlp-transcript-common/controller/fetchPosts"; import { isUtcDay } from "yt-dlp-transcript-common/social/olderBackfill"; @@ -166,6 +166,8 @@ export async function fetchPostsAction( force?: boolean, // A page walker's cap (a forum thread: its latest N pages this run). pages?: number, + // The older walk's fresh start date (FetchPostsOptions.from). + from?: string, ): Promise<StreamActionResult> { if (full && older) return { ok: false, error: FULL_AND_OLDER_REFUSAL }; if (pages !== undefined && !(Number.isInteger(pages) && pages > 0)) { @@ -174,6 +176,10 @@ export async function fetchPostsAction( if (floor !== undefined && !older) { return { ok: false, error: "A floor date applies only to an older-posts fetch." }; } + if (from !== undefined && !older) return { ok: false, error: OLDER_FROM_REFUSAL }; + if (from !== undefined && !isUtcDay(from)) { + return { ok: false, error: `"${from}" is not a date (YYYY-MM-DD).` }; + } if (force && !older) { return { ok: false, error: "\"force\" applies only to an older-posts fetch." }; } @@ -215,7 +221,7 @@ export async function fetchPostsAction( spec: { kind: "fetch-posts", slug, - params: { queueKey, full, limit, older, floor, force, ...(pages ? { pages } : {}) }, + params: { queueKey, full, limit, older, floor, force, ...(pages ? { pages } : {}), ...(from ? { from } : {}) }, }, fn: async (onLog, signal, _progress, ctx) => { const result = await fetchPosts({ @@ -225,6 +231,7 @@ export async function fetchPostsAction( full, older, floor, + from, force, limit, pages, diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -298,6 +298,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { str(p.floor), bool(p.force), num(p.pages), + str(p.from), ); }, // The ids are the spec's own (a capture is OF specific posts, unlike a diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -403,7 +403,9 @@ export function usage() { 'fetch-posts fetches a social channel\'s new posts: {"slug"}. "full": true', ' re-walks the whole timeline; "older": true walks back from the oldest', " archived post through search (X; needs a login), saving its place for", - ' the next run, down to "floor": "YYYY-MM-DD" when given. "limit": N caps', + ' the next run, down to "floor": "YYYY-MM-DD" when given; "from":', + ' "YYYY-MM-DD" starts the walk afresh there, replacing its saved place', + ' (and a "complete") — for a gap above one surviving old post. "limit": N caps', ' the posts one run reads; "pages": N caps the pages (a forum thread: its', ' latest N pages). "full" and "older" together are refused. An', ' older walk over an account that shows no posts (nothing archived, and',