commit b3b47e28964866e4e3189b8354a997740a7cd9a1
parent 7aff8253f112e67eaa9aed9ef7d72759c46bc5a8
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 09:04:31 -0400
wayback: the video page and a citation say a record is an archived copy
The video page shows "Archived copy (Wayback Machine, <capture date>) of <original>" from
`wayback.json`. Report composition loads it for a cited record: the citation links the
original (at the cited second where its platform takes one), labelled "Original (may be
gone)", and "Wayback Machine copy, <date>", the capture page, which plays. README section
and changelogs.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 135 insertions(+), 4 deletions(-)
diff --git a/README.md b/README.md
@@ -212,6 +212,29 @@ nothing asked while BitChute is in a rate-limit cooldown, which a 429 starts; an
on disk never fetched again (`common/ytdlp/platformArgs.mjs`,
`common/controller/bitchuteImport.ts`).
+### Wayback Machine captures
+
+**Import video** takes a Wayback Machine capture
+(`https://web.archive.org/web/<timestamp>/<original>`, with or without a replay
+modifier such as `id_`) and yt-dlp downloads it: an archived YouTube page through its
+Wayback extractor, a raw media file as the file. The record is named by what the capture
+is OF: an archived YouTube page by its YouTube id, a JW Player file
+(`…/videos/<id>-<rendition>.mp4` on `cdn.jwplayer.com`, `content.jwplatform.com`,
+`videos-fms.jwpsrv.com`) by its media id. Every download of a capture writes
+`wayback.json` beside its metadata: the original URL (an archived YouTube page's watch
+URL), the capture's timestamp, the capture as a page that plays and as its raw bytes.
+
+The video page says "Archived copy (Wayback Machine, <capture date>) of <original>". A
+citation of one links the original, marked as the original and as possibly gone, and the
+Wayback copy; its moment link is the capture, which plays.
+
+`pnpm archilyzer wayback refresh <channel> [--titles <file>] [--dry-run]` brings records
+imported before this up to it, offline: `wayback.json`, the dir renamed to its id through
+the snapshot's own reconcile pass (the roster entry moves with it), and with `--titles` (a
+JSON file of `id → {title, upload_date}`) the title and date of a raw file that has none.
+A record a running job holds is skipped and named. It prints old → new; a second run
+changes nothing.
+
### Requirements
Always needed, to install and run the apps:
diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts
@@ -140,11 +140,13 @@ export type RecordView = {
originalUrl?: string;
// What `originalUrl` is, when "Original" would not say: "archive.org" for an
// archive.org record, "YouTube" for an archive.org mirror of a YouTube
- // upload (whose originalUrl is the upload at the cited second).
+ // upload (whose originalUrl is the upload at the cited second), "Original
+ // (may be gone)" for a Wayback Machine copy (lib/wayback.ts).
originalLabel?: string;
// Where a reader can fetch the recording itself to check it, derived from
// the record's provenance (lib/archiveOrg.ts archiveOrgCitationLinks): the
- // archive.org page and the item's torrent. Absent for a record with none.
+ // archive.org page and the item's torrent; a Wayback copy's capture page
+ // (lib/wayback.ts waybackCitationLinks). Absent for a record with none.
downloads?: { label: string; url: string }[];
// The record in this site's corpus (`/?v=<channel>/<id>&t=<s>`): a FULL site
// only — a cited site has no corpus to open.
diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts
@@ -78,6 +78,8 @@ import { WHISPER_FILENAME, isEnglishVtt, resolvePrimaryVtt } from "../lib/videoS
import { platformMomentUrl } from "../lib/momentUrl";
import { archiveOrgCitationLinks, type ArchiveOrgProvenance } from "../lib/archiveOrg";
import { loadArchiveOrgProvenance } from "../lib/archiveOrg-server";
+import { WAYBACK_PROVENANCE_FILENAME, waybackCitationLinks, type WaybackProvenance } from "../lib/wayback";
+import { loadWaybackProvenance } from "../lib/wayback-server";
import type { Platform } from "../lib/platform";
import { readAllPosts } from "../lib/posts-server";
import type { Post } from "../lib/posts";
@@ -236,6 +238,9 @@ type CitedRecord = {
// An archive.org record's provenance (its torrent, a mirror's original);
// null for every other record.
archiveOrg: ArchiveOrgProvenance | null;
+ // A Wayback Machine capture's provenance (lib/wayback.ts): what the record
+ // is an archived copy of. Null for every other record.
+ wayback: WaybackProvenance | null;
};
// The English VTT tracks of a video dir, `en-orig` first, then the order
@@ -316,7 +321,8 @@ export async function readCitedRecord(
}
if (tracks.length === 0 && cues.length > 0) tracks.push({ name: "cues", cues });
const archiveOrg = summary.platform === "archiveorg" ? await loadArchiveOrgProvenance(dir) : null;
- return { summary, cues, tracks, archiveOrg };
+ const wayback = entries.includes(WAYBACK_PROVENANCE_FILENAME) ? await loadWaybackProvenance(dir) : null;
+ return { summary, cues, tracks, archiveOrg, wayback };
}
const isoDay = (uploadDate: string | undefined): string | undefined =>
@@ -629,9 +635,28 @@ export async function resolveSiteReports(opts: ResolveSiteReportsOptions): Promi
corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }),
});
}
- const { summary, archiveOrg } = (await recordOf(c.channel, c.id))!;
+ const { summary, archiveOrg, wayback } = (await recordOf(c.channel, c.id))!;
const audioOnly = isAudioOnlyPlatform(config?.platform);
const seconds = Math.max(0, Math.floor(c.start));
+ if (wayback) {
+ // An archived copy (Wayback Machine): the original, named as the
+ // original and as possibly gone, then the copy, which plays.
+ const links = waybackCitationLinks(wayback, {
+ originalMomentUrl: platformMomentUrl(wayback.originalUrl, null, c.start),
+ });
+ return defined({
+ channel: c.channel,
+ channelTitle: config?.name ?? (summary.channel || undefined),
+ id: c.id,
+ title: summary.title,
+ date: isoDay(summary.uploadDate),
+ platform: summary.platform,
+ originalUrl: links.original.url,
+ originalLabel: links.original.label,
+ downloads: [links.copy],
+ corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}),
+ });
+ }
if (summary.platform === "archiveorg") {
// archive.org: the original (YouTube at the second, for a mirror; else
// the archive.org page) plus the downloads a reader can check it from.
diff --git a/common/publish/composeReportsWayback.test.ts b/common/publish/composeReportsWayback.test.ts
@@ -0,0 +1,54 @@
+// A cited Wayback Machine copy carries its provenance into compose, and the
+// links a citation of it shows are derived from that (lib/wayback.ts
+// waybackCitationLinks): the original at the second, named as the original and
+// as possibly gone, then the Wayback copy. Every id here is invented.
+//
+// Run with: node_modules/.bin/tsx --test publish/composeReportsWayback.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { readCitedRecord } from "./composeReports";
+import { buildWaybackProvenance, waybackCitationLinks } from "../lib/wayback";
+import { platformMomentUrl } from "../lib/momentUrl";
+
+const YT = "Xyz987abc65";
+const CAPTURE = `https://web.archive.org/web/20190807060504/https://www.youtube.com/watch?v=${YT}`;
+
+test("readCitedRecord loads a Wayback copy's provenance; others get null", async () => {
+ const channels = mkdtempSync(path.join(tmpdir(), "compose-wayback-"));
+ const dir = path.join(channels, "demo-wayback", "data", YT);
+ mkdirSync(dir, { recursive: true });
+ writeFileSync(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({
+ id: YT,
+ extractor_key: "YoutubeWebArchive",
+ title: "An archived upload",
+ upload_date: "20090102",
+ webpage_url: CAPTURE,
+ }),
+ );
+ const prov = buildWaybackProvenance(CAPTURE)!;
+ writeFileSync(path.join(dir, "wayback.json"), JSON.stringify(prov));
+
+ const rec = await readCitedRecord(channels, "demo-wayback", YT);
+ assert.ok(rec);
+ assert.equal(rec.summary.platform, "youtube");
+ assert.equal(rec.summary.id, YT);
+ assert.deepEqual(rec.wayback, prov);
+ // The record's own moment link is the capture, which plays.
+ assert.equal(platformMomentUrl(rec.summary.webpageUrl, "youtube", 42), CAPTURE);
+ const links = waybackCitationLinks(rec.wayback!, {
+ originalMomentUrl: platformMomentUrl(rec.wayback!.originalUrl, null, 42),
+ });
+ assert.deepEqual(links.original, { label: "Original (may be gone)", url: `https://www.youtube.com/watch?v=${YT}&t=42s` });
+ assert.deepEqual(links.copy, { label: "Wayback Machine copy, 2019-08-07", url: CAPTURE });
+
+ const plain = path.join(channels, "demo-wayback", "data", "Plain12345a");
+ mkdirSync(plain, { recursive: true });
+ writeFileSync(path.join(plain, "metadata.info.json"), JSON.stringify({ id: "Plain12345a", extractor_key: "Youtube" }));
+ assert.equal((await readCitedRecord(channels, "demo-wayback", "Plain12345a"))?.wayback, null);
+});
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,8 @@
# Changelog
## [Unreleased]
+- **A Wayback Machine capture is a copy, and says of what.** A capture URL (`web.archive.org/web/<timestamp>[id_|im_|…]/<original>`) names its record by what it is a capture of: an archived YouTube page by its YouTube id (no longer `watch`), a JW Player file by its media id (no longer `<id>-<rendition>.mp4`). Every download of a capture writes `wayback.json` (the original URL, the capture's timestamp, the capture page and its raw bytes); the video page says "Archived copy (Wayback Machine, <date>) of <original>"; a citation links the original, marked as possibly gone, and the Wayback copy, and its moment link is the capture, which plays (a capture URL never takes a time param). An existing record whose page is a capture is renamed to its id by the next snapshot.
+- **`archilyzer wayback refresh <slug> [--titles <file>] [--dry-run]`** brings a channel's Wayback copies up to that offline: `wayback.json`, the dir renamed through the snapshot's own reconcile pass with its roster entry moved, and with `--titles` (`id → {title, upload_date}`) the title and date of a raw file that has none, recorded in the metadata history as `wayback-provenance`. A record a live job holds is skipped and named; a second run changes nothing.
- **A cited moment at the very end of a recording prepares.** Prepare evidence media cuts a clip whose padding runs past the recording's end at the end (the recording's duration from its metadata), where it found no media for the padded span; a span that starts past the end is still refused. report-to-video keeps its strict rule.
- **Exporting a changed report records a new revision of it.** `reports export` (and **Export reports** on a site's Reports tab, and the end of a prepare) commits a revision to the report's own git history, `sites/<site>/reports/<id>/history-git/`, whenever its `report.json` changed since the last one: the `report.json`, its Markdown export and the checksums of every export file, with a message of `Revision N` and a summary of the change. A re-export of an unchanged report records nothing. The commits carry the site's name and a `noreply@<site>.invalid` address with dates in UTC, never your git name, email or time zone. The Reports tab shows each report's revision, its commit and the last change under **Exports**, and the site's next build publishes the history. Add `history-git/` to the corpus repository's `.gitignore`.
- **archive.org files come over BitTorrent when possible, else straight from archive.org — never through yt-dlp.** The chosen file of an archive.org import is fetched from the item's own torrent (`<identifier>_archive.torrent`, which lists archive.org as a web seed, so other peers take load off archive.org) with aria2c, only that file of the item, and seeded afterwards for 10 minutes or to a ratio of 1, whichever comes first; the log shows "torrent: <file> (n of m pieces, peers p, web seed yes)" and "seeding 10 min…". With no aria2c, a torrent that does not carry the file, or no progress for 5 minutes, it is downloaded directly from `archive.org/download/…` instead (resumable, backing off on 429/503), and the log says "fell back to direct download: <reason>". Every file is checked against archive.org's sha1/md5: a mismatch is downloaded once more directly, a second one fails the record. The record is written from the item's metadata: `metadata.info.json` with the file's page, the canonical id, the duration ffprobe measures and archive.org's playable copies of the file, the `archiveorg.json` provenance (a mirror's original title, date and uploader), and `audio.<fmt>` — an audio file already in the channel's format is used as is, anything else goes through the app's audio extraction, a video kept in the saved-video store when the channel keeps sources. An .avi/.mpeg/.flac/.wav original is fetched as archive.org's mp4 or mp3 of it. aria2c runs in its own process group: cancelling the job stops it and everything it started, and it stops itself if the editor exits. New settings block `archiveOrg` (`torrent`, `seedMinutes`, `seedRatio`, `stallMinutes`, `maxPeers`, `maxDownloadKiBps`, `maxUploadKiBps`), `ARIA2C_BIN`, an aria2c row in `archilyzer doctor`, and `aria2` in the runtime Docker images.
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -13,6 +13,8 @@ import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/exclu
import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server";
import { loadArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg-server";
import type { ArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg";
+import { loadWaybackProvenance } from "yt-dlp-transcript-common/lib/wayback-server";
+import { waybackCaptureDate, type WaybackProvenance } from "yt-dlp-transcript-common/lib/wayback";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
readVideoMetadataForDisplay,
@@ -139,6 +141,8 @@ export default async function VideoDetailPage({
// Where an archive.org record came from (lib/archiveOrg-server.ts): the
// item and its torrent, and a mirror's original. Absent everywhere else.
const archiveOrg = await loadArchiveOrgProvenance(videoDir);
+ // What a Wayback Machine copy is a copy of (lib/wayback-server.ts).
+ const wayback = await loadWaybackProvenance(videoDir);
// The windows another tool asked this editor to fetch. One readdir of
// data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere,
// because that would be a walk of every video dir to draw one number.
@@ -191,6 +195,7 @@ export default async function VideoDetailPage({
excludedFromTruncatedCheck,
savedVideo,
archiveOrg,
+ wayback,
clipWindows,
vttProvenance,
coverage,
@@ -219,6 +224,7 @@ export default async function VideoDetailPage({
excludedFromTruncatedCheck,
savedVideo,
archiveOrg,
+ wayback,
clipWindows,
vttProvenance,
coverage,
@@ -285,6 +291,7 @@ export default async function VideoDetailPage({
)}
</div>
{archiveOrg && <ArchiveOrgProvenanceLine prov={archiveOrg} />}
+ {wayback && <WaybackProvenanceLine prov={wayback} />}
{meta.description && (
<details className="text-sm">
<summary className="cursor-pointer text-muted-foreground hover:text-foreground">
@@ -385,6 +392,23 @@ function ArchiveOrgProvenanceLine({ prov }: { prov: ArchiveOrgProvenance }) {
);
}
+// "Archived copy (Wayback Machine, <capture date>) of <original>".
+function WaybackProvenanceLine({ prov }: { prov: WaybackProvenance }) {
+ const link = "underline hover:text-foreground";
+ return (
+ <div aria-label="Wayback provenance" className="text-sm text-muted-foreground">
+ Archived copy (
+ <a href={prov.waybackUrl} target="_blank" rel="noreferrer" className={link}>
+ Wayback Machine, {waybackCaptureDate(prov.captureTs)}
+ </a>
+ ) of{" "}
+ <a href={prov.originalUrl} target="_blank" rel="noreferrer" className={`${link} break-all`}>
+ {prov.originalUrl}
+ </a>
+ </div>
+ );
+}
+
function formatUploadDate(s: string): string {
// yt-dlp emits YYYYMMDD. Render as YYYY-MM-DD; pass through anything else.
if (/^\d{8}$/.test(s)) {
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A citation of a Wayback Machine copy links its original and the copy.** A cited record downloaded from a Wayback capture shows "Original (may be gone)", the original at the cited second where its platform takes one, and "Wayback Machine copy, <capture date>", the capture page, which plays. Its moment link is the capture: a capture URL never takes a time param.
- **A report shows its revision, and every edit to it can be checked.** A report's date line ends with "revision N", linking to its history ("edited since revision N" when the report has changed since). The history page, `/reports/<id>/history/`, lists every revision, newest first: its number, date (UTC), commit hash and the sha256 of its `report.json`, what changed (claims added or removed, verdicts changed, claims edited, citations added or removed, quotes edited, title, series or subtitle changed), and each changed claim's title, text, verdict and findings with the words removed struck through and the words added marked. The same data is in `history.json` beside the page. Each report's history is its own git repository, published for cloning: `git clone <site>/reports/<id>/history/repo`. Each commit names the site as its author, with its date in UTC. The footer of the report's HTML, PDF and Markdown downloads begins with the revision number, and its sha256 can be looked up on the history page. Needs `reports export` and a rebuild and deploy of each site with reports.
- **A report can be saved whole: as one HTML page, a PDF, Markdown, or an evidence pack.** A report page's download line reads HTML · PDF · Markdown · Evidence pack · Citations JSON · CSV, each listed only when the site publishes it. The HTML is one file that opens with no network: the report with its verdicts, the document's sentences and the post screenshots inside it, numbered citations, and a reference list giving each quote's speaker, date, record, the original at its time and the moment page on the site. The PDF is that page printed. The Markdown is the same report as plain text with numbered references. The evidence pack is a zip of the page with its clips, stills and screenshots beside it, so the clips play offline. Each ends with a line naming the report's revision, its date and the start of its checksum. Needs `reports export` (or prepare) and a rebuild and deploy of each site with reports.
- **A report's claim can carry a flag, its header names the document under review, and a site with one report names it in the browser tab.** `report.json` claim `flag` (one line, at most 60 characters) shows as a small pill in the accent colour beside the claim's verdict, e.g. "No source given". On a report-only site with one report, the home page's tab title is the report's, as on the report's own page. A report's page header is its name — with a `series`, the series on one line in the accent colour and the title on the line below; without one, the title — then one small line of dates and the revision ("2026-10-04 · updated 2026-10-05 · revision 1"; "updated" only when it differs), then a card for the document under review: its title linking to the document, "<author> · <publisher> · <date>", and its archive links folded away, on a left rail in the document's colour (a source's `accent`, `"#rrggbb"`; without one, the border colour), then the subtitle. The page names no byline or site of its own. A claim that cites the document's own sentence shows that sentence — its still, else its words — on the same rail, with no link up to the card and no paraphrase beside it; a sentence of another document links "from <title>" to that document's box; a titled claim with no such sentence shows its text under the title, plain. A report's citation can say where its evidence came from (`origin`: `"subject"`, the document under review gave it; `"added"`, the report's author found it). A claim lists what the report added first, each card marked with the Archilyzer mark under its number ("Not in the article", or "Not in the source", is the mark's tooltip and what a screen reader says), then evidence of unknown origin, then what the document gave itself folded under "In the article (n)"; the reference list marks an added citation with the mark too, and a claim's flag pill wears the same mark. A fact-check's page reads in three tiers, each opened by a hairline with one, two or three dots: the quick take (the tally, the summary, and links to what the check found, every claim and the downloads); **What the check found**, every ruled claim grouped by verdict (contradicted, not found, partly, untestable, corroborated), one line each linking to the claim, with its `gist` (a new optional claim field, one line, at most 240 characters) and its flag; and **Every claim, with its evidence**, which opens with **How it was checked** (`method`, a new optional report field in markdown). A report of kind `sweep` has the first and last tiers only. Needs a rebuild and deploy of the site.