commit 66e5133c912d8dfbe11b5545092dec8ed0557358
parent 9dc1bd697f8948aa1dfcf6e798f62880240a1a24
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 2 Oct 2026 17:54:59 -0400
common: an X post from gallery-dl carries its reply and repost flags
gallery-dl flattens a tweet's references into reply_id / reply_to /
retweet_id (0 when unset); the normalizer read only in_reply_to_* and
retweeted_status, so every fetched post was isReply=false, isRepost=false.
A flat retweet is credited to `user` (the timeline owner), the original's
author goes to repostOf. quote_id is the reverse link and is not read.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 135 insertions(+), 9 deletions(-)
diff --git a/common/social/xNormalize.test.ts b/common/social/xNormalize.test.ts
@@ -142,6 +142,108 @@ test("marks a retweet and points repostOf at the original", () => {
assert.equal(post!.repostOf?.author, "original");
});
+// ─── gallery-dl's flat reference fields (regression) ───
+//
+// gallery-dl 1.32.9 does not pass X's in_reply_to_* / retweeted_status through:
+// it emits `reply_id`, `retweet_id`, `quote_id` (0 when unset) and `reply_to`.
+// Reading only X's names marked every archived post as neither a reply nor a
+// repost, and credited each retweet to the account that was retweeted.
+
+test("a gallery-dl reply is a reply, with its parent", () => {
+ const post = normalizeXTweet(
+ {
+ ...GALLERY_DL_RECORD,
+ content: "@otheraccount agreed",
+ reply_id: "1799999999999999990",
+ reply_to: "otheraccount",
+ retweet_id: 0,
+ quote_id: 0,
+ },
+ "xchan",
+ )!;
+ assert.equal(post.isReply, true);
+ assert.equal(post.replyTo?.id, "1799999999999999990");
+ assert.equal(post.replyTo?.author, "otheraccount");
+ assert.equal(post.replyTo?.url, "https://x.com/otheraccount/status/1799999999999999990");
+ assert.equal(post.isRepost, false);
+});
+
+test("gallery-dl's zeroed references mean none", () => {
+ const post = normalizeXTweet(
+ { ...GALLERY_DL_RECORD, reply_id: 0, retweet_id: 0, quote_id: 0, conversation_id: 0 },
+ "xchan",
+ )!;
+ assert.equal(post.isReply, false);
+ assert.equal(post.isRepost, false);
+ assert.equal(post.replyTo, undefined);
+ assert.equal(post.repostOf, undefined);
+ assert.equal(post.quoted, undefined);
+});
+
+test("a gallery-dl retweet is the timeline owner's repost of the original", () => {
+ // Shape from real output: `author` is the retweeted account, `user` the
+ // timeline's owner, and gallery-dl prefixes the body with "RT @author:".
+ const post = normalizeXTweet(
+ {
+ tweet_id: "2058535364466192493",
+ retweet_id: "2058530000000000000",
+ reply_id: 0,
+ quote_id: 0,
+ conversation_id: "2058530000000000000",
+ date: "2026-05-24 13:07:31",
+ content: "RT @originalacct: the original words",
+ author: { name: "originalacct", nick: "Original Account" },
+ user: { name: "timelineowner", nick: "Timeline Owner" },
+ favorite_count: 0,
+ retweet_count: 1246,
+ },
+ "xchan",
+ )!;
+ assert.equal(post.isRepost, true);
+ assert.equal(post.id, "2058535364466192493");
+ assert.equal(post.author, "timelineowner");
+ assert.equal(post.authorName, "Timeline Owner");
+ assert.equal(post.url, "https://x.com/timelineowner/status/2058535364466192493");
+ assert.equal(post.repostOf?.id, "2058530000000000000");
+ assert.equal(post.repostOf?.author, "originalacct");
+ assert.equal(post.text, "RT @originalacct: the original words");
+ assert.deepEqual(parsePost(JSON.parse(JSON.stringify(post))), post);
+});
+
+test("gallery-dl's quote_id names the QUOTING tweet, so it is never read as the quoted one", () => {
+ const post = normalizeXTweet(
+ { ...GALLERY_DL_RECORD, quote_id: "1799999999999999995", quote_by: "quoter" },
+ "xchan",
+ )!;
+ assert.equal(post.quoted, undefined);
+});
+
+test("X's quoted_status_id_str gives the quoted tweet", () => {
+ const post = normalizeXTweet(
+ { ...RAW_GRAPHQL_RECORD, quoted_status_id_str: "1700000000000000001" },
+ "xchan",
+ )!;
+ assert.equal(post.quoted?.id, "1700000000000000001");
+ assert.equal(post.quoted?.url, "https://x.com/i/status/1700000000000000001");
+});
+
+test("gallery-dl's unquoted 64-bit reply_id and retweet_id survive parsing", () => {
+ const reply =
+ '{"tweet_id": 2085320225776427457, "reply_id": 2085116299223425392, "retweet_id": 0, "quote_id": 0, "reply_to": "someone", "date": "2026-08-06 11:00:59", "content": "@someone yes", "author": {"name": "NASA"}, "user": {"name": "NASA"}}';
+ const [r] = parseGalleryDlOutput(reply);
+ const rp = normalizeXTweet(r, "xchan")!;
+ assert.equal(rp.isReply, true);
+ assert.equal(rp.replyTo?.id, "2085116299223425392");
+
+ const retweet =
+ '{"tweet_id": 2085320225776427458, "retweet_id": 2085116299223425393, "reply_id": 0, "quote_id": 0, "date": "2026-08-06 11:00:59", "content": "RT @someone: hi", "author": {"name": "someone"}, "user": {"name": "NASA"}}';
+ const [t] = parseGalleryDlOutput(retweet);
+ const rt = normalizeXTweet(t, "xchan")!;
+ assert.equal(rt.isRepost, true);
+ assert.equal(rt.author, "NASA");
+ assert.equal(rt.repostOf?.id, "2085116299223425393");
+});
+
test("rejects records with no id or no timestamp", () => {
assert.equal(normalizeXTweet({}, "c"), null);
assert.equal(normalizeXTweet({ tweet_id: "1" }, "c"), null, "no date -> rejected");
diff --git a/common/social/xNormalize.ts b/common/social/xNormalize.ts
@@ -40,6 +40,15 @@ function obj(v: unknown): Record<string, unknown> | undefined {
: undefined;
}
+// A reference to another tweet, as a string id, or undefined when absent.
+// gallery-dl writes an unset reference as 0; a 64-bit id arrives as a string
+// (`quoteBigIntegers`), a short one as a number.
+function refId(v: unknown): string | undefined {
+ if (typeof v === "string") return /^\d+$/.test(v) && !/^0+$/.test(v) ? v : undefined;
+ if (typeof v === "number" && Number.isSafeInteger(v) && v > 0) return String(v);
+ return undefined;
+}
+
// First defined value among a record's alias keys.
function pick(rec: XTweetRaw, keys: string[]): unknown {
for (const k of keys) {
@@ -147,27 +156,41 @@ export function normalizeXTweet(
const createdAt = xCreatedAt(rec);
if (!createdAt) return null;
- const { handle, name } = xAuthor(rec);
const text = xText(rec);
+ // gallery-dl flattens a tweet's references into `reply_id`, `retweet_id` and
+ // `quote_id` (plus `reply_to`, the replied-to handle), each 0 when unset.
const replyToId =
- str(pick(rec, ["in_reply_to_status_id_str", "in_reply_to_tweet_id"])) ??
- (num(pick(rec, ["in_reply_to_status_id"])) !== undefined
- ? String(num(pick(rec, ["in_reply_to_status_id"])))
- : undefined);
+ refId(pick(rec, ["in_reply_to_status_id_str", "in_reply_to_tweet_id", "in_reply_to_status_id"])) ??
+ refId(rec.reply_id);
const replyToHandle = str(
- pick(rec, ["in_reply_to_screen_name", "in_reply_to_user"]),
+ pick(rec, ["in_reply_to_screen_name", "in_reply_to_user", "reply_to"]),
);
+ // gallery-dl's `quote_id` is the reverse link: it is set on a quoted tweet
+ // (yielded only with its `quoted` option) and names the tweet QUOTING it, so
+ // it is never read as this tweet's quoted id.
const quotedRec = obj(rec.quoted) ?? obj(rec.quoted_status);
- const quotedId = quotedRec ? xIdOf(quotedRec) : str(rec.quote_id);
+ const quotedId = quotedRec ? xIdOf(quotedRec) : refId(rec.quoted_status_id_str);
const quotedHandle = quotedRec ? xAuthor(quotedRec).handle : undefined;
+ // A gallery-dl retweet is one flat record: `tweet_id` is the retweet itself,
+ // `retweet_id` the original, `author` the ORIGINAL's author and `user` the
+ // timeline's owner — who is the one who posted the retweet.
const retweetRec = obj(rec.retweeted_status) ?? obj(rec.retweet);
- const retweetId = retweetRec ? xIdOf(retweetRec) : undefined;
- const retweetHandle = retweetRec ? xAuthor(retweetRec).handle : undefined;
+ const flatRetweetId = retweetRec ? undefined : refId(rec.retweet_id);
+ const retweetId = retweetRec ? xIdOf(retweetRec) : flatRetweetId;
+ const retweetHandle = retweetRec
+ ? xAuthor(retweetRec).handle
+ : flatRetweetId
+ ? xAuthor(rec).handle || undefined
+ : undefined;
const isRepost = Boolean(retweetId) || rec.retweeted === true;
+ const owner = obj(rec.user);
+ const { handle, name } =
+ flatRetweetId && owner ? xAuthor({ author: owner }) : xAuthor(rec);
+
// X's conversation_id IS the thread root, which is exactly our threadId.
const threadId =
str(pick(rec, ["conversation_id", "conversation_id_str"])) ?? id;
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -2,6 +2,7 @@
## [Unreleased]
- **An X post fetch keeps what it has read when it is cancelled, times out or fails part-way, and the next fetch picks up where it stopped.** The gallery-dl fetcher used to receive an account's posts all at once when gallery-dl finished, so a long fetch that X's rate limit held past the 30-minute limit — or one you cancelled — ended with nothing saved. gallery-dl now hands over each post as it reads it: the fetch saves posts every 200 posts or every minute along with gallery-dl's own resume point, and the next fetch of the channel continues from that point instead of starting again. A fetch of new posts stops once it reaches 100 already-archived posts in a row instead of reading the whole timeline, and a fetch of an account's history may run for up to 3 hours (new-post fetches keep the 30-minute limit). gallery-dl's rate-limit waits now appear in the job's log as they happen.
+- **An X post fetched by gallery-dl says whether it is a reply or a repost.** gallery-dl writes a tweet's references as `reply_id`, `reply_to` and `retweet_id` (0 when unset), and the normalizer read only X's own `in_reply_to_*` and `retweeted_status` fields, so every fetched post was stored as an original post. A reply now carries the replied-to post and handle; a repost carries the original's id and author and is credited to the account that reposted it. gallery-dl's `quote_id` names the tweet quoting this one, not the one quoted, and is no longer read as a quote. Posts already fetched keep their old flags until the account is fetched again.
- **umtool can keep each report's render folder on a media drive.** With `UMTOOL_MEDIA_DIR` set, in umtool's environment (restart umtool after setting it), to a directory inside that drive, a report project's `out/` (its fetched windows, segments and finished video) is a link to the same path under that directory: a project's first build makes it there, and `umtool storage move-out <project>` (or `--all`) moves an existing one, copying it, checking the copy and only then leaving the link; `--dry-run` says how much would move, and `umtool storage move-back` brings one home. The manifest, its revisions, notes and sources stay where they are, and nothing in umtool reads a project differently. When the drive is not mounted, a build or source check refuses and says so instead of starting a new folder on the main disk; umtool never creates the media directory itself. `umtool storage` lists where each project's `out/` is. With `UMTOOL_MEDIA_DIR` unset nothing changes.
- **umtool can move a report's cut clips and share batches to the media drive, one project at a time.** The **Move deliverables** button in a report's deliver panel, or `umtool storage deliverables <project> --to media` (`--to local` to bring them back, `--dry-run` to see what would move), moves the project's `clips/` folder and every `share-…` batch folder to the same path under `UMTOOL_MEDIA_DIR`, checks each copy, leaves a link in its place, and then records the choice in the report's `video.manifest.json` as `"storage": { "deliverables": "media" }`. From then on a cut or a new batch is made on the media drive, and everything that opens `clips/<id>.mp4` keeps working through the link. The panel shows where the deliverables are, and the button is disabled, with the reason beside it, while a cut or a batch is running, while the drive is not mounted, or (to the media drive) when `UMTOOL_MEDIA_DIR` is not set. When the drive is not mounted, a cut or a batch refuses and says so instead of starting again on the main disk, and `umtool check` reports it as blocking, for the render folder too. Nothing moves until you ask: without the switch, a report's deliverables stay in the project.
- **umtool's cache moves to `~/.cache/archilyzer/umtool`** (`$XDG_CACHE_HOME/archilyzer/umtool` when that is set, or `UMTOOL_CACHE_DIR`). It was inside the song project's data folder, so it followed that folder onto whatever drive it was on. Run `umtool index` once after updating to rebuild the project index in its new place; umtool works without it, only slower, and the rest of the cache is remade as it is needed. `umtool doctor` now also shows the reports, media and cache folders, and the old cache folder while it is still there; it can be deleted.