Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit dc0fc16597f90dfcc40cd0dd58e33e480c3739a5
parent b63208b86560533079814136b1c5195159d13d43
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 18:51:57 -0400

reports: a cited moment's clip stops at the recording's end when its padding runs past it

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/lib/evidenceClip-server.test.ts | 8++++++++
Mcommon/lib/evidenceClip-server.ts | 11+++++++++++
Mcommon/publish/reportMedia.ts | 13++++++++++++-
Meditor/CHANGELOG.md | 1+
4 files changed, 32 insertions(+), 1 deletion(-)

diff --git a/common/lib/evidenceClip-server.test.ts b/common/lib/evidenceClip-server.test.ts @@ -238,3 +238,11 @@ test("size: a clip over the limit is refused, with its size, and nothing is kept assert.ok(!r.ok && (r.bytes ?? 0) > 1000); assert.deepEqual(await readdir(cacheDir), []); }); + +test("clampSpanToDuration: padding past a recording's end stops at the end; nothing else moves", async () => { + const { clampSpanToDuration } = await import("./evidenceClip-server"); + assert.deepEqual(clampSpanToDuration({ from: 189.02, to: 203.64 }, 201.48), { from: 189.02, to: 201.48 }); + assert.deepEqual(clampSpanToDuration({ from: 10, to: 20 }, 201.48), { from: 10, to: 20 }); + assert.deepEqual(clampSpanToDuration({ from: 202, to: 205 }, 201.48), { from: 202, to: 205 }, "a span starting past the end is not invented"); + assert.deepEqual(clampSpanToDuration({ from: 189, to: 203 }, null), { from: 189, to: 203 }); +}); diff --git a/common/lib/evidenceClip-server.ts b/common/lib/evidenceClip-server.ts @@ -80,6 +80,17 @@ export const EVIDENCE_EXT: Record<EvidenceKind, string> = { video: ".mp4", audio // The seconds a clip covers, in the record's clock. export type EvidenceSpan = { from: number; to: number }; +// A span whose padding runs past the recording's end, cut at the end: the +// file holds every second the citation names, and a clip simply stops where the +// recording does. Only the end moves, and only for a span that starts inside +// the recording; with no known duration the span is as given. (report-to-video +// keeps the strict rule — its timeline needs every padded second.) +export function clampSpanToDuration(span: EvidenceSpan, duration: number | null | undefined): EvidenceSpan { + if (!Number.isFinite(duration) || !duration || duration <= 0) return span; + if (span.to <= duration || span.from >= duration) return span; + return { from: span.from, to: Number(duration.toFixed(3)) }; +} + // The span a citation's clip covers: its moment's (rounded) start and end, // widened by its pad, the start clamped at 0. Rounded to the millisecond so a // float's last digits never change a hash. diff --git a/common/publish/reportMedia.ts b/common/publish/reportMedia.ts @@ -56,6 +56,7 @@ import type { CitationPad } from "../lib/citations/schema"; import { parseReport } from "../lib/report/validate"; import type { Report } from "../lib/report/schema"; import { + clampSpanToDuration, evidenceSpan, isAudioOnlyPlatform, prepareEvidenceClip, @@ -258,7 +259,10 @@ export async function prepareReportMedia(opts: PrepareReportMediaOptions): Promi continue; } if (m.moment.kind !== "span") continue; - const span = evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad }); + const span = clampSpanToDuration( + evidenceSpan({ start: m.moment.start, end: m.moment.end, pad: m.pad }), + await recordDuration(paths.channelsDir, slug, m.moment.id), + ); const r = await prepareEvidenceClip({ channelsDir: paths.channelsDir, slug, @@ -368,3 +372,10 @@ export async function readReportMediaIndex(paths: Paths, siteId: string): Promis if (!v || v.format !== REPORT_MEDIA_FORMAT || v.version !== REPORT_MEDIA_VERSION) return null; return v as ReportMediaIndex; } + +// A record's duration (metadata.info.json), or null when it is not recorded. +async function recordDuration(channelsDir: string, slug: string, id: string): Promise<number | null> { + const read = await readJsonFile(path.join(channelsDir, slug, "data", id, "metadata.info.json")); + const d = read.ok ? Number((read.value as { duration?: unknown } | null)?.duration) : NaN; + return Number.isFinite(d) && d > 0 ? d : null; +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **A cited moment at the very end of a recording prepares.** Prepare evidence media cuts a clip whose padding runs past the recording's end at the end (the recording's duration from its metadata), where it found no media for the padded span; a span that starts past the end is still refused. report-to-video keeps its strict rule. - **Exporting a changed report records a new revision of it.** `reports export` (and **Export reports** on a site's Reports tab, and the end of a prepare) commits a revision to the report's own git history, `sites/<site>/reports/<id>/history-git/`, whenever its `report.json` changed since the last one: the `report.json`, its Markdown export and the checksums of every export file, with a message of `Revision N` and a summary of the change. A re-export of an unchanged report records nothing. The commits carry the site's name and a `noreply@<site>.invalid` address with dates in UTC, never your git name, email or time zone. The Reports tab shows each report's revision, its commit and the last change under **Exports**, and the site's next build publishes the history. Add `history-git/` to the corpus repository's `.gitignore`. - **archive.org files come over BitTorrent when possible, else straight from archive.org — never through yt-dlp.** The chosen file of an archive.org import is fetched from the item's own torrent (`<identifier>_archive.torrent`, which lists archive.org as a web seed, so other peers take load off archive.org) with aria2c, only that file of the item, and seeded afterwards for 10 minutes or to a ratio of 1, whichever comes first; the log shows "torrent: <file> (n of m pieces, peers p, web seed yes)" and "seeding 10 min…". With no aria2c, a torrent that does not carry the file, or no progress for 5 minutes, it is downloaded directly from `archive.org/download/…` instead (resumable, backing off on 429/503), and the log says "fell back to direct download: <reason>". Every file is checked against archive.org's sha1/md5: a mismatch is downloaded once more directly, a second one fails the record. The record is written from the item's metadata: `metadata.info.json` with the file's page, the canonical id, the duration ffprobe measures and archive.org's playable copies of the file, the `archiveorg.json` provenance (a mirror's original title, date and uploader), and `audio.<fmt>` — an audio file already in the channel's format is used as is, anything else goes through the app's audio extraction, a video kept in the saved-video store when the channel keeps sources. An .avi/.mpeg/.flac/.wav original is fetched as archive.org's mp4 or mp3 of it. aria2c runs in its own process group: cancelling the job stops it and everything it started, and it stops itself if the editor exits. New settings block `archiveOrg` (`torrent`, `seedMinutes`, `seedRatio`, `stallMinutes`, `maxPeers`, `maxDownloadKiBps`, `maxUploadKiBps`), `ARIA2C_BIN`, an aria2c row in `archilyzer doctor`, and `aria2` in the runtime Docker images. - **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated; a page saved from one of a forum's mirror domains (kiwifarms.net for a kiwifarms.st channel) is a page of the same thread. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines.