Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit f1beea234e970ccf4f858435efd7bfce6cfbf5a5
parent d401d7d896448d0c33f9522fd87db5e0622f8719
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri,  2 Oct 2026 10:05:35 -0400

report-to-video: a post's QR links its page on the archive by default -- postQrUrl in deck.mjs (siteUrl, else siteOrigin + siteChannel + the post's id, else its own url), posts.links "archive" | "original", optional siteChannel/siteUrl/postId on a post; post-links.mjs finds the archive channel that keeps each post (corpus.json -> posts manifests' slugToPage, handle-matching channels first, cues.mjs's JSON cache, one fresh re-read on a miss) just before the build writes its schedule, in memory only, a note per post, the original on any miss or failure; umtool's preview links through the same resolver; the e2e post fixtures pin siteChannel

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mumtool/components/projects/OnscreenSection.tsx | 1+
Mumtool/e2e/fixtures/make-fixture.mjs | 17++++++++++-------
Mumtool/lib/report/onscreen.mjs | 29++++++++++++++++++++++++-----
Mumtool/lib/report/onscreen.test.mjs | 28++++++++++++++++++++++++++++
Mumtool/report-to-video/build-video.mjs | 23+++++++++++++++++++++++
Mumtool/report-to-video/cues.mjs | 50+++++++++++++++++++++++++++++++++++++++-----------
Mumtool/report-to-video/deck.mjs | 92+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Mumtool/report-to-video/package.json | 1+
Aumtool/report-to-video/post-links.mjs | 158+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/report-to-video/post-links.test.mjs | 230+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
10 files changed, 599 insertions(+), 30 deletions(-)

diff --git a/umtool/components/projects/OnscreenSection.tsx b/umtool/components/projects/OnscreenSection.tsx @@ -559,6 +559,7 @@ const GROUPS: { name: string; fields: Field[] }[] = [ { key: "posts.position", label: "side", kind: "select", options: ["top-right", "top-left"], hint: "the side the column hangs from: the frame's edge when making room, else the footage's" }, { key: "posts.width", label: "width", kind: "int", hint: "px, 320–900" }, { key: "posts.qrSize", label: "qr", kind: "num", hint: "px, 80–200, at most half the card" }, + { key: "posts.links", label: "qr links", kind: "select", options: ["archive", "original"], hint: "archive: each post's QR opens its page on the archive the manifest names (which links on to the original), when the build finds the post there; original: the bsky.app / x.com link itself" }, { key: "posts.maxLines", label: "lines", kind: "int", hint: "2–14: longer posts end in an ellipsis" }, { key: "posts.inset", label: "inset", kind: "num", hint: "px, 0–80 from the edges" }, { key: "posts.shift", label: "make room", kind: "bool", hint: "the footage moves aside while a clip's posts are up; off leaves it in its box under the column" }, diff --git a/umtool/e2e/fixtures/make-fixture.mjs b/umtool/e2e/fixtures/make-fixture.mjs @@ -1518,7 +1518,9 @@ const ONSCREEN_BUILD = writeProject( // p-early Aug 1 older than every clip -> c01 ("first") // p-mid Sep 5 after c01's Sep 3 -> c01 ("date") // p-late Sep 12 after c02's Sep 10 -> c02 ("date") -// Never built, so the posts' timing is the estimate's. +// Never built, so the posts' timing is the estimate's. Every post pins its +// archive channel (`siteChannel`), so its QR is the archive's page for it and +// neither the preview nor a build asks archive.example where the post is kept. const ONSCREEN_POSTS = writeProject( "onscreen-posts-fixture", { @@ -1531,17 +1533,17 @@ const ONSCREEN_POSTS = writeProject( { id: "p-early", platform: "bluesky", author: "Fixture Author", handle: "fixture.example", date: "2024-08-01T12:00:00.000Z", text: "Older than every clip in the cut.", - url: "https://bsky.app/profile/fixture.example/post/early", + url: "https://bsky.app/profile/fixture.example/post/early", siteChannel: "fixture-bsky", }, { id: "p-mid", platform: "bluesky", author: "Fixture Author", handle: "fixture.example", date: "2024-09-05T09:30:00.000Z", text: "Two days after the first clip.\nA second line.", - url: "https://bsky.app/profile/fixture.example/post/mid", + url: "https://bsky.app/profile/fixture.example/post/mid", siteChannel: "fixture-bsky", }, { id: "p-late", platform: "x", author: "Fixture Author", handle: "fixture", date: "2024-09-12T18:00:00.000Z", text: "Two days after the second clip.", - url: "https://x.com/fixture/status/1", + url: "https://x.com/fixture/status/1", siteChannel: "fixture-x", }, ], }, @@ -1551,7 +1553,8 @@ const ONSCREEN_POSTS = writeProject( // onscreen-posts.spec.ts -- every segment framed into the feed's box, one feed // sequence for the whole cut from the stub renderer, no hold -- then refused // a --chrome-only once the layout says popup. Clips only, as the build -// fixture above: a card needs Pango. +// fixture above: a card needs Pango. Its posts pin `siteChannel` too: the +// build looks nothing up on the network. const ONSCREEN_FEED = (() => { const m = deckManifest("onscreen-feed-fixture", "The On-screen Feed Fixture", [ { type: "clip", id: "c01", video: "vid1", start: 3.0, end: 6.0, cite: 3, section: 0, lock: true, quote: "and because", date: "2024-09-03" }, @@ -1564,12 +1567,12 @@ const ONSCREEN_FEED = (() => { { id: "f-one", platform: "bluesky", author: "Fixture Author", handle: "fixture.example", date: "2024-09-05T09:30:00.000Z", text: "Rides on the first clip, in from its start.", - url: "https://bsky.app/profile/fixture.example/post/one", + url: "https://bsky.app/profile/fixture.example/post/one", siteChannel: "fixture-bsky", }, { id: "f-two", platform: "bluesky", author: "Fixture Author", handle: "fixture.example", date: "2024-09-12T18:00:00.000Z", text: "Rides on the second clip.", - url: "https://bsky.app/profile/fixture.example/post/two", + url: "https://bsky.app/profile/fixture.example/post/two", siteChannel: "fixture-bsky", }, ], }); diff --git a/umtool/lib/report/onscreen.mjs b/umtool/lib/report/onscreen.mjs @@ -19,6 +19,7 @@ import { readFile, rm } from "node:fs/promises"; import path from "node:path"; import { composeChrome } from "umtool-report-to-video/compose-chrome"; import { selectVariant } from "umtool-report-to-video/build-video"; +import { createPostChannelResolver, resolvePostLinks } from "umtool-report-to-video/post-links"; import { attachPosts, clipDay, @@ -183,7 +184,7 @@ export function previewSchedule({ variantManifest, built, draft, metas, postsDra }); const total = round(built.total + shift); const placed = deck.posts.show - ? postSchedule({ posts, entries: patchedEntries, metas, segments, D: built.transition, total, render }) + ? postSchedule({ posts, entries: patchedEntries, metas, segments, D: built.transition, total, render, provenance }) : []; // The layout as it is NOW: a feed (no holds, no moves) only with posts to draw. const feed = placed.length > 0 && deck.posts.layout === "feed"; @@ -327,16 +328,34 @@ async function readBuiltSchedule(dir, variant) { * @param {string} variant * @param {Map<string, { title?: string, subtitle?: string } | null>} draft */ -export async function scheduleForPreview(project, manifest, variant, draft = new Map(), postsDraft = {}) { - const variantManifest = selectVariant(manifest, variant); - const entries = variantManifest.timeline ?? []; - const [built, metas] = await Promise.all([ +export async function scheduleForPreview(project, manifest, variant, draft = new Map(), postsDraft = {}, { resolver = previewPostResolver() } = {}) { + const selected = selectVariant(manifest, variant); + const entries = selected.timeline ?? []; + const [built, metas, linked] = await Promise.all([ readBuiltSchedule(project.dir, variant), deckMetas(project.dir, manifest, entries), + resolvePostLinks({ posts: selected.posts ?? [], provenance: selected.provenance ?? {}, render: selected.render ?? {}, resolver }), ]); + // The posts with the archive channel each is kept in, found as the build + // finds it (post-links.mjs, the same cache on disk), so the preview's QRs + // are the build's. Never written to the manifest. + const variantManifest = linked.posts === selected.posts ? selected : { ...selected, posts: linked.posts }; return { variantManifest, metas, schedule: previewSchedule({ variantManifest, built, draft, metas, postsDraft }) }; } +/** + * The preview's post-link resolver: one for the server's life, so its memory + * cache spans requests. A post missing from the cached archive is looked for + * in a fresh copy at most every ten minutes, and a request waits at most eight + * seconds on an archive that does not answer (then the QR links the original, + * as the build's would). + */ +let previewResolver = null; +function previewPostResolver() { + previewResolver ??= createPostChannelResolver({ timeoutMs: 8000, refreshAfterMs: 10 * 60 * 1000 }); + return previewResolver; +} + // compose-chrome writes one directory per cut. Two requests composing into it // at once would interleave their writes, so they queue per directory. /** @type {Map<string, Promise<unknown>>} */ diff --git a/umtool/lib/report/onscreen.test.mjs b/umtool/lib/report/onscreen.test.mjs @@ -17,6 +17,7 @@ import { normalizePostsDraft, postRows, previewSchedule, + scheduleForPreview, scheduleMatches, segmentBoxes, stillTimeOf, @@ -384,3 +385,30 @@ test("segmentBoxes: each segment's recorded framing box, else the deck's; an odd await rm(dir, { recursive: true, force: true }); } }); + +// ---- post links --------------------------------------------------------------- + +test("a post that names its archive channel previews with the archive QR, built or estimated", () => { + const m = withPosts([{ ...POSTS[1], siteChannel: "b-x" }]); + for (const b of [built(), null]) { + const s = previewSchedule({ variantManifest: m, built: b, draft: new Map(), metas }); + assert.equal(s.posts[0].qrUrl, "https://example.test/?v=b-x%2F2&vm=post"); + assert.equal(s.posts[0].url, "https://x.com/b/status/2"); + } +}); + +test("scheduleForPreview: the posts are linked by the resolver the build uses, in memory only", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "onscreen-links-")); + try { + const manifest = withPosts([{ ...POSTS[1] }]); + const asked = []; + const resolver = { find: async (origin, post) => (asked.push([origin, post.id]), "b-x") }; + const { variantManifest, schedule } = await scheduleForPreview({ dir }, manifest, "sourced", new Map(), {}, { resolver }); + assert.deepEqual(asked, [["https://example.test", "p2"]]); + assert.equal(variantManifest.posts[0].siteChannel, "b-x"); + assert.equal(manifest.posts[0].siteChannel, undefined, "the manifest is not changed"); + assert.equal(schedule.posts[0].qrUrl, "https://example.test/?v=b-x%2F2&vm=post"); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); diff --git a/umtool/report-to-video/build-video.mjs b/umtool/report-to-video/build-video.mjs @@ -77,6 +77,7 @@ import { cardWidth, contentWidth, reservedFooterHeight, } from "./render-cards.mjs"; import { createCueSource, siteOriginFromManifest } from "./cues.mjs"; +import { createPostChannelResolver, resolvePostLinks } from "./post-links.mjs"; import { ensureWriteDir } from "../lib/report/storage.mjs"; // The deck (`render.chrome`): its geometry, validation and schedule are pure // and live in deck.mjs. This file only frames segments into its box and writes @@ -3360,6 +3361,25 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly // beside the pictures it cites -- not to the cwd the build was started from. const manifestDir = path.dirname(path.resolve(manifestPath)); + // Where each post's QR goes: its page on the archive (`posts.links` + // "archive", the default), which needs the archive channel that keeps it. + // Found once, just before the first schedule is written (a full build, + // --chrome-only, --chrome-preview: after every refusal, so nothing is asked + // of the network for a run that stops), and set on this run's copy of the + // posts -- never written back to the manifest. A post the archive does not + // have, or an archive that does not answer, links the original: a note, + // not a failure. + let postsLinked = false; + const linkPosts = async () => { + if (postsLinked) return; + postsLinked = true; + if (!Array.isArray(manifest.posts) || !manifest.posts.length) return; + const resolver = opts.postResolver ?? createPostChannelResolver({ log: (m) => EMIT("log", { message: m }) }); + const linked = await resolvePostLinks({ posts: manifest.posts, provenance, render, resolver }); + manifest.posts = linked.posts; + for (const message of linked.notes) EMIT("note", { message }); + }; + // The manifest already records which archive it was built against, so a clone // with no corpus needs no extra configuration to read cue windows. CUES = createCueSource({ @@ -3496,6 +3516,7 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly return { out: finalPath, failures: [] }; } + // Re-run the rail over a cached concat instead of rebuilding the timeline. // The rail is the part that gets iterated on; the 40-minute concat is not. if (opts.railOnly) { @@ -3566,6 +3587,7 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly EMIT("card", { id: e.id, i: entries.indexOf(e), n: entries.length }); await buildTeaserSegment(e, { manifestPath, render, outDir, variant, transition: D }); } + await linkPosts(); const schedule = await writeChromeSchedule({ manifest, entries, segments: segs, D, outDir }); EMIT("chrome", { phase: "schedule", total: schedule.total, segments: schedule.segments.length }); // The holds, the moves, the mutes and the end fade, joined on their @@ -3694,6 +3716,7 @@ export async function buildVideo({ manifestPath, opts = {}, out, only, fetchOnly // laid in the concat itself (or, for hard cuts, in one pass after it). let schedule = null; if (deck) { + await linkPosts(); schedule = await writeChromeSchedule({ manifest, entries, segments, D, outDir }); EMIT("chrome", { phase: "schedule", total: schedule.total, segments: schedule.segments.length }); } diff --git a/umtool/report-to-video/cues.mjs b/umtool/report-to-video/cues.mjs @@ -63,7 +63,7 @@ const REPO_ROOT = path.resolve(/* turbopackIgnore: true */ path.dirname(/* turbo export const DEFAULT_CHANNELS_DIR = process.env.CHANNELS_DIR ?? path.join(/* turbopackIgnore: true */ REPO_ROOT, "transcripts", "channels"); -const DEFAULT_CACHE_DIR = +export const DEFAULT_CACHE_DIR = process.env.REPORT_CACHE_DIR ?? path.join(os.homedir(), ".cache", "archilyzer-report-to-video"); @@ -116,23 +116,34 @@ export class CueLookupError extends Error { } } -export function createCueSource({ - channelsDir = DEFAULT_CHANNELS_DIR, - siteOrigin = null, +/** + * A JSON GET with an in-memory and an on-disk cache, keyed by URL: the cache + * the cue walk reads the archive through, and the one the post-link resolver + * (post-links.mjs) reads it through too: the same files on disk, so an + * archive's corpus.json fetched for one is on disk for the other. + * + * `getJson(url, { refresh: true })` goes back to the network for a URL -- + * unless this cache already fetched it from the network less than + * `refreshAfterMs` ago. The default, Infinity, makes that "at most once per + * process": a build re-reads a stale cached document once, never twice. A + * long-lived server passes a window instead. + */ +export function createJsonCache({ cacheDir = DEFAULT_CACHE_DIR, fetchImpl = globalThis.fetch, log = () => {}, - // "auto" | "local" | "http" — see the note on divergence above. - prefer = "auto", - // Opt-in, because it is expensive: see resolveSiteId below. - resolveSiteIds = false, + refreshAfterMs = Infinity, } = {}) { const mem = new Map(); + /** URL -> when this cache last fetched it from the network (ms). */ + const fetchedAt = new Map(); - async function getJson(url) { - if (mem.has(url)) return mem.get(url); + async function getJson(url, { refresh = false } = {}) { + const recent = fetchedAt.has(url) && Date.now() - fetchedAt.get(url) < refreshAfterMs; + const useCache = !refresh || recent; + if (useCache && mem.has(url)) return mem.get(url); const disk = cacheDir ? path.join(cacheDir, cacheKey(url)) : null; - if (disk) { + if (useCache && disk) { try { const cached = JSON.parse(await readFile(disk, "utf8")); mem.set(url, cached); @@ -146,6 +157,7 @@ export function createCueSource({ if (!res.ok) throw new Error(`GET ${url} -> ${res.status}`); const json = await res.json(); mem.set(url, json); + fetchedAt.set(url, Date.now()); if (disk) { try { await mkdir(path.dirname(disk), { recursive: true }); @@ -157,6 +169,22 @@ export function createCueSource({ return json; } + return { getJson }; +} + +export function createCueSource({ + channelsDir = DEFAULT_CHANNELS_DIR, + siteOrigin = null, + cacheDir = DEFAULT_CACHE_DIR, + fetchImpl = globalThis.fetch, + log = () => {}, + // "auto" | "local" | "http" — see the note on divergence above. + prefer = "auto", + // Opt-in, because it is expensive: see resolveSiteId below. + resolveSiteIds = false, +} = {}) { + const { getJson } = createJsonCache({ cacheDir, fetchImpl, log }); + async function channelEntry(origin, channelSlug) { const corpus = await getJson(`${origin}/corpus.json`); const found = (corpus.channels ?? []).find((c) => c.slug === channelSlug); diff --git a/umtool/report-to-video/deck.mjs b/umtool/report-to-video/deck.mjs @@ -45,9 +45,13 @@ export const DECK_DEFAULTS = Object.freeze({ // `layout` "popup" draws them as above; "feed" keeps them in a column of // their own beside the footage for the whole cut (feedGeometry), each one // ticking in as its clip starts -- no hold, no move. + // `links` is where a post's QR goes: "archive" (the post's page on the + // archive the manifest names, `provenance.siteOrigin`, which survives the + // post or the platform going away, and links on to the original) or + // "original" (the bsky.app / x.com link itself). posts: Object.freeze({ show: true, layout: "popup", seconds: 4, hold: 2.5, position: "top-right", width: 600, qrSize: 120, maxLines: 7, - inset: 24, shift: Object.freeze({ scale: 0.86, seconds: 0.6 }), + inset: 24, shift: Object.freeze({ scale: 0.86, seconds: 0.6 }), links: "archive", }), }); @@ -190,10 +194,13 @@ export function validateChrome(chrome, render = {}) { errors.push(`${w}.qr.size ${size} does not fit a ${h}px deck (at most ${h - 20})`); } }); - sub("posts", ["show", "layout", "seconds", "hold", "position", "width", "qrSize", "maxLines", "inset", "shift"], (p) => { + sub("posts", ["show", "layout", "seconds", "hold", "position", "width", "qrSize", "maxLines", "inset", "shift", "links"], (p) => { if (p.layout !== undefined && !POST_LAYOUTS.includes(p.layout)) { errors.push(`${w}.posts.layout must be ${POST_LAYOUTS.map((l) => `"${l}"`).join(" or ")}`); } + if (p.links !== undefined && !POST_LINKS.includes(p.links)) { + errors.push(`${w}.posts.links must be ${POST_LINKS.map((l) => `"${l}"`).join(" or ")}`); + } if (p.hold !== undefined && !numIn(p.hold, 0, 10)) errors.push(`${w}.posts.hold must be from 0 to 10 seconds`); if (p.shift !== undefined && p.shift !== false) { if (!isObj(p.shift)) errors.push(`${w}.posts.shift must be false or { scale, seconds }`); @@ -525,7 +532,7 @@ export function deckSchedule({ const multiChannel = isMultiChannel(entries, provenance); const round = (v) => Math.round(v * 1000) / 1000; const segs = entries.map((e, i) => ({ id: e.id, start: starts[i], duration: full[i] })); - const placed = deck.posts.show ? postSchedule({ posts, entries, metas, segments: segs, D, total, render }) : []; + const placed = deck.posts.show ? postSchedule({ posts, entries, metas, segments: segs, D, total, render, provenance }) : []; // The feed is a layout only when it has posts to draw: a feed with none is // the deck alone, and writes the schedule a deck without posts always did. const feed = placed.length > 0 && deck.posts.layout === "feed"; @@ -671,7 +678,9 @@ export function pipSegments(schedule) { export const POST_PLATFORMS = Object.freeze(["bluesky", "x"]); /** How posts are drawn (`posts.layout`): cards at the end of a clip, or a column for the whole cut. */ export const POST_LAYOUTS = Object.freeze(["popup", "feed"]); -const POST_KEYS = ["id", "platform", "author", "handle", "date", "text", "url", "attachTo", "hide"]; +/** Where a post's QR links (`posts.links`): its page on the archive, or the platform's own link. */ +export const POST_LINKS = Object.freeze(["archive", "original"]); +const POST_KEYS = ["id", "platform", "author", "handle", "date", "text", "url", "attachTo", "hide", "siteChannel", "siteUrl", "postId"]; /** * Every reason the manifest's `posts` cannot be built, as sentences. `timeline` @@ -707,6 +716,18 @@ export function validatePosts(posts, timeline = [], render = null) { errors.push(`${w}.attachTo ${JSON.stringify(p.attachTo)} is not a clip in the timeline`); } if (p.hide !== undefined && typeof p.hide !== "boolean") errors.push(`${w}.hide must be true or false`); + // Where the archive keeps the post: its own channel's slug (a post channel + // is not the video channel -- `piratesoftware-bsky`, not `piratesoftware`), + // a page to link instead, or the post's id when its url does not carry one. + if (p.siteChannel !== undefined && (typeof p.siteChannel !== "string" || !SLUG_RE.test(p.siteChannel))) { + errors.push(`${w}.siteChannel must be the archive's channel slug (letters, digits, dots, dashes, underscores)`); + } + if (p.siteUrl !== undefined && (typeof p.siteUrl !== "string" || !/^https?:\/\/\S+$/.test(p.siteUrl))) { + errors.push(`${w}.siteUrl must be an http(s) link`); + } + if (p.postId !== undefined && (typeof p.postId !== "string" || !POST_ID_RE.test(p.postId))) { + errors.push(`${w}.postId must be the post's id on its platform (letters, digits, dashes, underscores)`); + } }); // Posts that will be drawn need a column that fits the footage box. if (render && deckOn(render) && resolveDeck(render).posts.show && posts.some((p) => isObj(p) && !p.hide)) { @@ -799,11 +820,15 @@ export function attachPosts({ posts = [], entries = [], metas = [] }) { * short for that shares what it has after its incoming dissolve evenly. They * all leave together over `out`. * + * Each one's `qrUrl` is postQrUrl's: its archive page under `posts.links` + * "archive" when the post names its archive channel, else its own `url`. + * * @returns {Array<{ id, segment, slot, of, appear, out: [number, number], date, text, * author, handle, platform, url, qrUrl }>} */ -export function postSchedule({ posts = [], entries, metas = [], segments, D, total, render }) { +export function postSchedule({ posts = [], entries, metas = [], segments, D, total, render, provenance = {} }) { const settings = resolveDeck(render).posts; + const postFields = (p) => postFieldsOf(p, provenance, settings.links); const feed = settings.layout === "feed"; const byId = new Map(posts.map((p) => [p.id, p])); const groups = new Map(); @@ -848,7 +873,7 @@ export function postSchedule({ posts = [], entries, metas = [], segments, D, tot } /** What a placed post carries for the page that draws it. */ -function postFields(p) { +function postFieldsOf(p, provenance, links) { return { date: p.date, text: p.text, @@ -856,10 +881,63 @@ function postFields(p) { handle: p.handle ?? "", platform: p.platform, url: p.url, - qrUrl: p.url, + qrUrl: postQrUrl(p, provenance, { links }), }; } +const SLUG_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$/; +const POST_ID_RE = /^[A-Za-z0-9_-]{1,128}$/; + +/** + * The post's id on its platform -- what the archive keys it by (a post + * channel's `slugToPage`): an explicit `postId`, else read off its url -- a + * Bluesky `/post/<rkey>`, an X (or twitter.com) `/status/<id>`. Null when + * neither says. + * + * @returns {string|null} + */ +export function postNativeId(post) { + if (typeof post?.postId === "string" && POST_ID_RE.test(post.postId)) return post.postId; + let u; + try { + u = new URL(String(post?.url ?? "")); + } catch { + return null; + } + const m = /\/post\/([A-Za-z0-9]+)\/?$/.exec(u.pathname) ?? /\/status(?:es)?\/(\d+)(?:\/|$)/.exec(u.pathname); + return m ? m[1] : null; +} + +/** + * A post's page on an archive: the site's post modal for `<channel>/<id>`, + * which shows the post and links on to the original. + */ +export function postArchiveUrl(siteOrigin, siteChannel, nativeId) { + const origin = String(siteOrigin).replace(/\/+$/, ""); + return `${origin}/?v=${encodeURIComponent(`${siteChannel}/${nativeId}`)}&vm=post`; +} + +/** + * Where a post's QR (and any link to it the cut draws) goes. + * + * `links: "original"` (the deck's `posts.links`) is the post's own `url`, + * always. Under "archive" (the default): the post's explicit `siteUrl`; else, + * with an archive to link (`provenance.siteOrigin`), the channel it is kept in + * there (`siteChannel` -- pinned in the manifest, or found by the build's + * resolver, post-links.mjs) and its id, the post's archive page; else its own + * `url`, as before. + * + * @returns {string} + */ +export function postQrUrl(post, provenance = {}, { links = DECK_DEFAULTS.posts.links } = {}) { + if (links === "original") return post.url; + if (typeof post.siteUrl === "string" && post.siteUrl) return post.siteUrl; + const origin = typeof provenance?.siteOrigin === "string" ? provenance.siteOrigin.trim() : ""; + const id = postNativeId(post); + if (origin && post.siteChannel && id) return postArchiveUrl(origin, post.siteChannel, id); + return post.url; +} + /** * The windows the posts region is rendered for: one per clip that carries * posts, from its first post's appearance to the end of its leave. Frames are diff --git a/umtool/report-to-video/package.json b/umtool/report-to-video/package.json @@ -26,6 +26,7 @@ "./ledger-totals": "./ledger-totals.mjs", "./mute": "./mute.mjs", "./package.json": "./package.json", + "./post-links": "./post-links.mjs", "./render-cards": "./render-cards.mjs", "./resolve-windows": "./resolve-windows.mjs", "./verify-build": "./verify-build.mjs" diff --git a/umtool/report-to-video/post-links.mjs b/umtool/report-to-video/post-links.mjs @@ -0,0 +1,158 @@ +// Which archive channel keeps each post, so its QR can link the post's page +// on the archive rather than bsky.app / x.com. +// +// A post channel is a channel of its own on the archive, with its own slug +// (`piratesoftware-bsky`, not the video channel `piratesoftware`), and a manifest +// rarely says which: it carries the platform's link. So the build finds it, by +// the walk `/corpus.json` publishes under `postScheme`: +// +// 1. GET <origin>/corpus.json -> channels[] with `manifests.posts` +// 2. GET each one's posts manifest -> { slugToPage: { <postId>: N } } +// 3. the first channel whose slugToPage has the post's id keeps it +// +// Channels whose slug or name carries the post's handle are asked first, so a +// post that several channels hold (a mirror, a repost) links the account that +// posted it. Nothing here writes the manifest: a found channel is set on the +// build's in-memory copy of the post, and deck.mjs `postQrUrl` makes the link. +// A post the manifest pins (`siteChannel` or `siteUrl`) is not looked up. +// +// Failure is never the build's: a post not in the archive, or an archive that +// does not answer, links the original, and says so in a note. +// +// The fetches go through cues.mjs's JSON cache -- the same files on disk the +// cue walk reads -- with one refresh: a post missing from a CACHED corpus or +// manifest is looked for again in a fresh copy (once per process for a build, +// once per `refreshAfterMs` for a server), so a post published since the cache +// was filled is found. +import { createJsonCache } from "./cues.mjs"; +import { postNativeId, resolveDeck } from "./deck.mjs"; + +/** How long one archive request may take before the post links its original. */ +export const POST_LINK_TIMEOUT_MS = 15000; + +const origin = (provenance) => { + const o = typeof provenance?.siteOrigin === "string" ? provenance.siteOrigin.trim() : ""; + return o ? o.replace(/\/+$/, "") : null; +}; + +const squash = (v) => String(v ?? "").toLowerCase().replace(/^@/, ""); + +/** Does this corpus channel look like the account that posted? Its slug or name carries the handle. */ +export function channelMatchesHandle(channel, handle) { + const h = squash(handle); + if (!h) return false; + const first = h.split(".")[0]; + const hay = [squash(channel.slug), squash(channel.name)]; + return hay.some((s) => s.includes(h) || (first.length >= 3 && s.includes(first))); +} + +/** + * The resolver. `getJson` (url, { refresh }) => document is injectable for + * tests; by default a cues.mjs JSON cache over `fetchImpl` with a timeout. + * + * @returns {{ find(origin: string, post: Record<string, any>): Promise<string|null> }} + * `find` resolves the slug of the channel that keeps the post, or null when + * none does; it rejects when the archive cannot be read. + */ +export function createPostChannelResolver({ + getJson = null, + cacheDir, + fetchImpl = globalThis.fetch, + log = () => {}, + timeoutMs = POST_LINK_TIMEOUT_MS, + refreshAfterMs = Infinity, +} = {}) { + const timed = (url) => fetchImpl(url, { signal: AbortSignal.timeout(timeoutMs) }); + const get = getJson ?? createJsonCache({ ...(cacheDir !== undefined ? { cacheDir } : {}), fetchImpl: timed, log, refreshAfterMs }).getJson; + // One failed read of an archive is remembered for the resolver's life, so a + // cut of twenty posts against an archive that does not answer waits once. + const failed = new Map(); + + async function read(url, refresh) { + if (failed.has(url) && !refresh) throw failed.get(url); + try { + return await get(url, { refresh }); + } catch (e) { + const err = e instanceof Error ? e : new Error(String(e)); + failed.set(url, err); + throw err; + } + } + + async function lookIn(base, id, handle, refresh) { + const corpus = await read(`${base}/corpus.json`, refresh); + const channels = (corpus?.channels ?? []).filter((c) => typeof c?.manifests?.posts === "string" && c.postCount !== 0); + const ordered = [ + ...channels.filter((c) => channelMatchesHandle(c, handle)), + ...channels.filter((c) => !channelMatchesHandle(c, handle)), + ]; + for (const c of ordered) { + let manifest; + try { + manifest = await read(c.manifests.posts, refresh); + } catch { + // One channel's manifest missing is that channel not answering, not the archive. + continue; + } + if (manifest?.slugToPage && Object.hasOwn(manifest.slugToPage, id)) return c.slug; + } + return null; + } + + return { + async find(base, post) { + const id = postNativeId(post); + if (!id) return null; + const at = String(base).replace(/\/+$/, ""); + return (await lookIn(at, id, post.handle, false)) ?? (await lookIn(at, id, post.handle, true)); + }, + }; +} + +/** + * The posts with their archive channel found, and a note per post saying + * where its QR goes. Only under `posts.links: "archive"` with an archive named + * (`provenance.siteOrigin`) and posts to draw; otherwise the posts come back + * as they were, with no notes. + * + * Never throws for the archive: a post it cannot place keeps its own link. + * + * @param {{ posts?: Array<Record<string, any>>, provenance?: Record<string, any>, + * render?: Record<string, any>, resolver: ReturnType<typeof createPostChannelResolver> }} args + * @returns {Promise<{ posts: Array<Record<string, any>>, notes: string[] }>} + */ +export async function resolvePostLinks({ posts = [], provenance = {}, render = {}, resolver }) { + const settings = resolveDeck(render).posts; + const base = origin(provenance); + if (!Array.isArray(posts) || !posts.length || !settings.show || settings.links !== "archive") { + return { posts, notes: [] }; + } + if (!base) { + return { posts, notes: ["posts: no provenance.siteOrigin names an archive -- every post's QR links the original"] }; + } + const notes = []; + const out = []; + for (const p of posts) { + if (p.siteUrl) { notes.push(`${p.id}: QR links ${p.siteUrl} (siteUrl)`); out.push(p); continue; } + if (p.siteChannel) { notes.push(`${p.id}: archive link via ${p.siteChannel} (pinned)`); out.push(p); continue; } + if (!postNativeId(p)) { + notes.push(`${p.id}: no post id in its url -- QR links the original`); + out.push(p); + continue; + } + try { + const slug = await resolver.find(base, p); + if (slug) { + notes.push(`${p.id}: archive link via ${slug}`); + out.push({ ...p, siteChannel: slug }); + } else { + notes.push(`${p.id}: not in the archive at ${base} -- QR links the original`); + out.push(p); + } + } catch (e) { + notes.push(`${p.id}: the archive at ${base} did not answer (${e instanceof Error ? e.message : String(e)}) -- QR links the original`); + out.push(p); + } + } + return { posts: out, notes }; +} diff --git a/umtool/report-to-video/post-links.test.mjs b/umtool/report-to-video/post-links.test.mjs @@ -0,0 +1,230 @@ +// Where a post's QR goes: deck.mjs's pure postQrUrl, the deck's `posts.links` +// setting and the post keys it reads, and post-links.mjs's resolver (with a +// stub archive -- nothing here touches the network). +// +// Run with: pnpm test:scripts +import assert from "node:assert/strict"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import test from "node:test"; + +import { createJsonCache } from "./cues.mjs"; +import { + DECK_DEFAULTS, + estimateSchedule, + postArchiveUrl, + postNativeId, + postQrUrl, + validateChrome, + validatePosts, +} from "./deck.mjs"; +import { channelMatchesHandle, createPostChannelResolver, resolvePostLinks } from "./post-links.mjs"; + +const ORIGIN = "https://jasolyzer.pages.dev"; +const PROV = { siteOrigin: ORIGIN, channelSlug: "piratesoftware" }; +const BSKY = { + id: "bs-3l6uxj6esfr2z", platform: "bluesky", handle: "piratesoftware.live", date: "2024-10-19T17:01:17.640Z", + text: "words", url: "https://bsky.app/profile/piratesoftware.live/post/3l6uxj6esfr2z", +}; +const X = { + id: "x-1", platform: "x", handle: "PirateSoftware", date: "2024-10-20", text: "words", + url: "https://x.com/PirateSoftware/status/1847712345678901234", +}; + +// ---- the pure half (deck.mjs) ------------------------------------------------ + +test("postNativeId: a bsky rkey, an X status id, an explicit postId, else null", () => { + assert.equal(postNativeId(BSKY), "3l6uxj6esfr2z"); + assert.equal(postNativeId(X), "1847712345678901234"); + assert.equal(postNativeId({ url: "https://twitter.com/a/status/42/photo/1" }), "42"); + assert.equal(postNativeId({ ...X, postId: "999" }), "999"); + assert.equal(postNativeId({ url: "https://bsky.app/p/1" }), null); + assert.equal(postNativeId({ url: "not a url" }), null); +}); + +test("postArchiveUrl: the site's post modal for <channel>/<id>", () => { + assert.equal( + postArchiveUrl(`${ORIGIN}/`, "piratesoftware-bsky", "3l6uxj6esfr2z"), + "https://jasolyzer.pages.dev/?v=piratesoftware-bsky%2F3l6uxj6esfr2z&vm=post", + ); +}); + +test("postQrUrl: the archive page when the post names its channel, else its own url", () => { + // bsky, with its archive channel: the archive page. + assert.equal( + postQrUrl({ ...BSKY, siteChannel: "piratesoftware-bsky" }, PROV), + "https://jasolyzer.pages.dev/?v=piratesoftware-bsky%2F3l6uxj6esfr2z&vm=post", + ); + // X, the same. + assert.equal( + postQrUrl({ ...X, siteChannel: "piratesoftware-x" }, PROV), + "https://jasolyzer.pages.dev/?v=piratesoftware-x%2F1847712345678901234&vm=post", + ); + // An explicit siteUrl wins over everything but links "original". + assert.equal(postQrUrl({ ...BSKY, siteChannel: "c", siteUrl: "https://elsewhere.example/p" }, PROV), "https://elsewhere.example/p"); + assert.equal(postQrUrl({ ...BSKY, siteUrl: "https://elsewhere.example/p" }, {}), "https://elsewhere.example/p"); + // links "original": today's behaviour, whatever the post names. + assert.equal(postQrUrl({ ...BSKY, siteChannel: "piratesoftware-bsky", siteUrl: "https://e/p" }, PROV, { links: "original" }), BSKY.url); + // No siteOrigin, no channel, or no id: the post's own url. + assert.equal(postQrUrl({ ...BSKY, siteChannel: "piratesoftware-bsky" }, {}), BSKY.url); + assert.equal(postQrUrl(BSKY, PROV), BSKY.url); + assert.equal(postQrUrl({ ...BSKY, url: "https://bsky.app/p/1", siteChannel: "c" }, PROV), "https://bsky.app/p/1"); + // A postId supplies the id a url does not carry. + assert.equal( + postQrUrl({ ...BSKY, url: "https://bsky.app/p/1", siteChannel: "c", postId: "abc" }, PROV), + "https://jasolyzer.pages.dev/?v=c%2Fabc&vm=post", + ); +}); + +test("the schedule's posts carry postQrUrl's link; links \"original\" keeps the post's own", () => { + const render = { fps: 30, transition: 0.5, chrome: { engine: "hyperframes", layout: "deck", deck: { posts: { layout: "feed" } } } }; + const timeline = [{ type: "clip", id: "c01", video: "v1", start: 0, end: 10, date: "2024-10-01" }]; + const posts = [{ ...BSKY, siteChannel: "piratesoftware-bsky" }, X]; + const s = estimateSchedule({ render, provenance: PROV, timeline, posts }); + const by = Object.fromEntries(s.posts.map((p) => [p.id, p])); + assert.equal(by[BSKY.id].qrUrl, "https://jasolyzer.pages.dev/?v=piratesoftware-bsky%2F3l6uxj6esfr2z&vm=post"); + assert.equal(by[BSKY.id].url, BSKY.url, "the post's own link stays beside it"); + assert.equal(by[X.id].qrUrl, X.url, "a post with no archive channel links its original"); + const orig = { ...render, chrome: { ...render.chrome, deck: { posts: { layout: "feed", links: "original" } } } }; + const o = estimateSchedule({ render: orig, provenance: PROV, timeline, posts }); + assert.ok(o.posts.every((p) => p.qrUrl === p.url)); +}); + +test("posts.links: archive by default; archive or original, nothing else", () => { + assert.equal(DECK_DEFAULTS.posts.links, "archive"); + assert.deepEqual(validateChrome({ engine: "hyperframes", layout: "deck", deck: { posts: { links: "original" } } }, { width: 1920, height: 1080 }), []); + assert.deepEqual(validateChrome({ engine: "hyperframes", layout: "deck", deck: { posts: { links: "archive" } } }, { width: 1920, height: 1080 }), []); + const bad = validateChrome({ engine: "hyperframes", layout: "deck", deck: { posts: { links: "platform" } } }, { width: 1920, height: 1080 }); + assert.ok(bad.some((e) => /posts\.links must be "archive" or "original"/.test(e)), bad.join("; ")); +}); + +test("validatePosts: siteChannel, siteUrl and postId are optional and checked", () => { + assert.deepEqual(validatePosts([{ ...BSKY, siteChannel: "piratesoftware-bsky", siteUrl: "https://a.example/p", postId: "3l6uxj6esfr2z" }]), []); + const errors = validatePosts([{ ...BSKY, siteChannel: "has space", siteUrl: "ftp://a/p", postId: "a/b" }]); + assert.equal(errors.length, 3, errors.join("; ")); + assert.ok(errors.some((e) => /siteChannel/.test(e))); + assert.ok(errors.some((e) => /siteUrl must be an http\(s\) link/.test(e))); + assert.ok(errors.some((e) => /postId/.test(e))); + assert.ok(validatePosts([{ ...BSKY, siteChannel: 7 }]).some((e) => /siteChannel/.test(e))); +}); + +// ---- the resolver (post-links.mjs) ------------------------------------------- + +/** A stub archive: corpus.json plus each channel's posts manifest, as getJson answers them. */ +function archive(channels, { failing = new Set() } = {}) { + const docs = new Map(); + docs.set(`${ORIGIN}/corpus.json`, { + channels: channels.map((c) => ({ + slug: c.slug, name: c.name ?? c.slug, videoCount: 0, + ...(c.ids ? { postCount: c.ids.length, manifests: { posts: `${ORIGIN}/posts/${c.slug}/manifest.json` } } : { manifests: {} }), + })), + }); + for (const c of channels) { + if (c.ids) docs.set(`${ORIGIN}/posts/${c.slug}/manifest.json`, { slugToPage: Object.fromEntries(c.ids.map((id) => [id, 0])) }); + } + const asked = []; + const getJson = async (url, { refresh = false } = {}) => { + asked.push({ url, refresh }); + if (failing.has(url)) throw new Error(`GET ${url} -> 503`); + if (!docs.has(url)) throw new Error(`GET ${url} -> 404`); + return docs.get(url); + }; + return { getJson, asked, docs }; +} + +const DECK = { chrome: { engine: "hyperframes", layout: "deck", deck: {} } }; + +test("resolver: finds the channel that keeps the post, and the QR is its archive page", async () => { + const a = archive([ + { slug: "piratesoftware" }, + { slug: "piratesoftware-bsky", name: "piratesoftware.live (BlueSky)", ids: ["3l6uxj6esfr2z", "other"] }, + ]); + const resolver = createPostChannelResolver({ getJson: a.getJson }); + const r = await resolvePostLinks({ posts: [BSKY], provenance: PROV, render: DECK, resolver }); + assert.equal(r.posts[0].siteChannel, "piratesoftware-bsky"); + assert.deepEqual(r.notes, [`${BSKY.id}: archive link via piratesoftware-bsky`]); + assert.equal(postQrUrl(r.posts[0], PROV), "https://jasolyzer.pages.dev/?v=piratesoftware-bsky%2F3l6uxj6esfr2z&vm=post"); + assert.equal(BSKY.siteChannel, undefined, "the manifest's post is not changed"); +}); + +test("resolver: a post the archive does not have links its original, with a note", async () => { + const a = archive([{ slug: "piratesoftware-bsky", ids: ["nope"] }]); + const resolver = createPostChannelResolver({ getJson: a.getJson }); + const r = await resolvePostLinks({ posts: [BSKY], provenance: PROV, render: DECK, resolver }); + assert.equal(r.posts[0].siteChannel, undefined); + assert.deepEqual(r.notes, [`${BSKY.id}: not in the archive at ${ORIGIN} -- QR links the original`]); + // A miss is looked for again in a fresh copy, once. + assert.deepEqual(a.asked.map((x) => x.refresh), [false, false, true, true]); +}); + +test("resolver: an archive that does not answer links the original, says why, and is asked once", async () => { + const a = archive([], { failing: new Set([`${ORIGIN}/corpus.json`]) }); + const resolver = createPostChannelResolver({ getJson: a.getJson }); + const r = await resolvePostLinks({ posts: [BSKY, { ...X }], provenance: PROV, render: DECK, resolver }); + assert.ok(r.posts.every((p) => p.siteChannel === undefined)); + assert.equal(r.notes.length, 2); + assert.match(r.notes[0], /did not answer \(GET .*corpus\.json -> 503\) -- QR links the original/); + // The failure is remembered: the second post asks nothing new without refresh. + assert.equal(a.asked.filter((x) => !x.refresh).length, 1); +}); + +test("resolver: several channels hold the id -- the one whose name carries the handle wins", async () => { + const a = archive([ + { slug: "mirror-bsky", name: "a mirror (BlueSky)", ids: ["3l6uxj6esfr2z"] }, + { slug: "piratesoftware-bsky", name: "piratesoftware.live (BlueSky)", ids: ["3l6uxj6esfr2z"] }, + ]); + const resolver = createPostChannelResolver({ getJson: a.getJson }); + assert.equal(await resolver.find(ORIGIN, BSKY), "piratesoftware-bsky"); + // With no handle to go by, corpus order. + assert.equal(await resolver.find(ORIGIN, { ...BSKY, handle: "" }), "mirror-bsky"); + assert.equal(channelMatchesHandle({ slug: "piratesoftware-x", name: "PirateSoftware (X)" }, "PirateSoftware"), true); + assert.equal(channelMatchesHandle({ slug: "other", name: "Other" }, "piratesoftware.live"), false); +}); + +test("resolvePostLinks: pinned posts, links \"original\", no siteOrigin, no id", async () => { + const a = archive([{ slug: "piratesoftware-bsky", ids: ["3l6uxj6esfr2z"] }]); + const resolver = createPostChannelResolver({ getJson: a.getJson }); + // Pinned: not looked up. + const pinned = await resolvePostLinks({ + posts: [{ ...BSKY, siteChannel: "pinned-bsky" }, { ...X, siteUrl: "https://e.example/x" }], provenance: PROV, render: DECK, resolver, + }); + assert.deepEqual(pinned.notes, [`${BSKY.id}: archive link via pinned-bsky (pinned)`, `${X.id}: QR links https://e.example/x (siteUrl)`]); + assert.equal(a.asked.length, 0); + // links "original": nothing to find, nothing said. + const orig = await resolvePostLinks({ + posts: [BSKY], provenance: PROV, render: { chrome: { ...DECK.chrome, deck: { posts: { links: "original" } } } }, resolver, + }); + assert.deepEqual(orig, { posts: [BSKY], notes: [] }); + // No archive named: one note for the lot. + const none = await resolvePostLinks({ posts: [BSKY], provenance: {}, render: DECK, resolver }); + assert.equal(none.posts[0], BSKY); + assert.deepEqual(none.notes, ["posts: no provenance.siteOrigin names an archive -- every post's QR links the original"]); + // No id in the url. + const noId = await resolvePostLinks({ posts: [{ ...BSKY, url: "https://bsky.app/p/1" }], provenance: PROV, render: DECK, resolver }); + assert.deepEqual(noId.notes, [`${BSKY.id}: no post id in its url -- QR links the original`]); + assert.equal(a.asked.length, 0); +}); + +test("createJsonCache: a refresh refetches once per process, then reads what it fetched", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "post-links-cache-")); + try { + let n = 0; + const fetchImpl = async () => ({ ok: true, status: 200, json: async () => ({ n: ++n }) }); + const url = "https://archive.example/corpus.json"; + // A first process fills the disk cache. + assert.deepEqual(await createJsonCache({ cacheDir: dir, fetchImpl }).getJson(url), { n: 1 }); + // A second reads it from disk; a refresh goes to the network once, and only once. + const c = createJsonCache({ cacheDir: dir, fetchImpl }); + assert.deepEqual(await c.getJson(url), { n: 1 }); + assert.deepEqual(await c.getJson(url, { refresh: true }), { n: 2 }); + assert.deepEqual(await c.getJson(url, { refresh: true }), { n: 2 }); + assert.deepEqual(await c.getJson(url), { n: 2 }); + // A window lets a long-lived reader refresh again once it has passed. + const w = createJsonCache({ cacheDir: dir, fetchImpl, refreshAfterMs: 0 }); + assert.deepEqual(await w.getJson(url, { refresh: true }), { n: 3 }); + assert.deepEqual(await w.getJson(url, { refresh: true }), { n: 4 }); + } finally { + await rm(dir, { recursive: true, force: true }); + } +});