Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit ec7018490b752bfb2696e9c93e5dff560cb97644
parent b56921eda36d3ea632d9c2c446c9677745492d1f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri,  9 Oct 2026 19:43:39 -0400

social: an X fetch's limit caps posts (--post-range), not files — --range never stopped a metadata-only walk

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/social/xGalleryDlFetcher.test.ts | 2+-
Mcommon/social/xGalleryDlFetcher.ts | 7+++++--
Mcommon/social/xNormalize.test.ts | 6++++--
Meditor/CHANGELOG.md | 1+
4 files changed, 11 insertions(+), 5 deletions(-)

diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts @@ -296,7 +296,7 @@ test("search argv: the timeline's flags and login, latest-first max_id paging, n limit: 50, }); assert.equal(flagValue(argv, "--cookies"), JAR); - assert.equal(flagValue(argv, "--range"), "1-50"); + assert.equal(flagValue(argv, "--post-range"), "1-50"); for (const opt of [ "output.jsonl=true", "extractor.twitter.text-tweets=true", diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts @@ -134,7 +134,7 @@ export function buildGalleryDlArgs( export const X_SLEEP_REQUEST = "4.0-10.0"; // The flags every gallery-dl read shares: metadata only, streamed, text tweets -// with their retweets and replies, the login, and the --range cap. +// with their retweets and replies, the login, and the --post-range cap. function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] { const args = [ // One JSON object per item on stdout — we want metadata, never files. @@ -166,8 +166,11 @@ function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] { ...galleryDlPacingArgs(), ...galleryDlLoginArgs(opts), ]; + // A cap on POSTS. `--range` counts files, and with downloads off most tweets + // yield none, so it never stopped the walk: a 400-post fetch of a new + // channel read 3,803. `--post-range` counts the tweets themselves. if (opts.limit && opts.limit > 0) { - args.push("--range", `1-${Math.floor(opts.limit)}`); + args.push("--post-range", `1-${Math.floor(opts.limit)}`); } return args; } diff --git a/common/social/xNormalize.test.ts b/common/social/xNormalize.test.ts @@ -280,7 +280,9 @@ test("gallery-dl argv enables text-tweets and downloads nothing", () => { args.slice(args.indexOf("--cookies-from-browser"), args.indexOf("--cookies-from-browser") + 2), ["--cookies-from-browser", "firefox"], ); - assert.match(joined, /--range 1-50/); + assert.match(joined, /--post-range 1-50/); + // `--range` counts FILES: with downloads off it never stopped the walk. + assert.ok(!args.includes("--range"), "a limit caps posts, not files"); // Normalized to the timeline sub-extractor — a bare profile URL yields no // tweets at all (see timelineUrlFor). assert.equal(args[args.length - 1], "https://x.com/someaccount/timeline"); @@ -289,7 +291,7 @@ test("gallery-dl argv enables text-tweets and downloads nothing", () => { test("gallery-dl argv omits cookies when none are resolved", () => { const args = buildGalleryDlArgs({ accountUrl: "https://x.com/a" }); assert.ok(!args.includes("--cookies-from-browser")); - assert.ok(!args.join(" ").includes("--range")); + assert.ok(!args.join(" ").includes("range 1-")); }); test("parses JSON-lines, whole-array and tuple dump forms", () => { diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **An X fetch with a `limit` stops at that many posts.** "Fetch posts" with `limit` (`pnpm ops fetch-posts {"limit": 400}`) on a gallery-dl X channel read the whole history instead — a new channel walked 3,803 posts under the rate limit and held the platform queue for hours — because the cap counted media files, which a metadata-only read has almost none of. It now caps the posts themselves. - **A home seeder of last resort, behind a VPN.** `archilyzer seed` seeds the playable torrents of the sites named in the new `settings.seeder` (`sites`, `trackers`, `maxUploadKiBps`, `maxConnections`, `pollSeconds`, `standbyAfterSeconds`, `bindInterface`; SETTINGS.md) to desktop clients over TCP and to browsers over WebRTC — but each torrent only while no other seeder has it: other seeders seen on every poll for `standbyAfterSeconds` puts that torrent on standby (it stops announcing and closes its peers, keeping the data), and it comes back at once when a leecher is waiting with no other source, or after the same window with no other seeder. Every change is logged with its reason. No DHT, no local discovery, no UPnP. `archilyzer tracker` is a self-hosted HTTP + WebSocket tracker that tracks only those torrents. `docker-compose.seeder.yml` (profile `seeder`) runs both inside a WireGuard container's network namespace (gluetun, its firewall always on), so a tunnel that is down means no network, never the home connection; the WireGuard config is yours (`SEEDER_WG_CONF`, required, mounted read-only). `archilyzer doctor` compares the seeder's egress address with the host's and fails when they are the same; it says "seeder not configured" until `seeder.sites` names a site. - **Saved videos can be made browser-playable, with a torrent each.** `pnpm ops prepare-playable` (`POST /api/ops/prepare-playable`) and `archilyzer media playable <slug>` remux each of a channel's saved containers — without re-encoding (`-c copy`) — into an mp4 with its index in front, or a webm when it already is one (VP9/AV1 with Opus), drop subtitles, metadata and chapters, and make one single-file torrent of the copy: named `<id>.<ext>`, no web seed, no comment, no "created by", 256 KiB–1 MiB pieces. They go to `playable/<slug>/<id>/` beside the saved-video store, listed in `playable/<slug>/playable.json` with each infohash. `"trackers"` is the announce list written into each torrent (none by default); it is not part of the infohash, so the same torrent can be announced elsewhere later. A video already prepared from the same source (by sha256) is skipped, so a re-run is a no-op; a codec a browser cannot play without re-encoding (HEVC, MPEG-4 Part 2) is listed and left alone. - **A channel's videos can get their media from a local archive.** `pnpm ops attach-media` (`POST /api/ops/attach-media`) and `archilyzer media attach <slug> <source>` take `{"slug", "source"}` — an absolute path to a directory, a `.zip` (read in place: a stored entry is copied straight out of it, with no temp dir) or a `.7z` — and put each held video's file into the saved-video store as its source container, so clip windows and report clips can be cut from it with nothing fetched. The id is the folder's trailing `(<id>)`, else the file's yt-dlp suffix; `"items": [{"id", "path"}]` names exact files and `"match"` narrows the folders. The pointer records where the file came from: `origin: {kind: "local-archive", archive, entry, sha256, attachedAt}`. A video that already has a saved container is left alone unless `"replace": true`; `"createRecords": true` writes a record for a video the channel does not hold (from the folder's yt-dlp `.info.json`, else its `description.txt` and name); `[LOST]` folders are listed and never attached. `"dryRun": true` lists what each folder is, and the held videos the archive has no media for, and writes nothing. The archive is never written to.