commit adecfb2b8992a406b1c8ce650bd75afb70ff06b2 parent 0580ebcf23bb5785601785a447db246361b6e61c Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Mon, 5 Oct 2026 15:14:05 -0400 sources: import from archive.org in the editor; play the file; citations link archive.org and the torrent - import-video: an archive.org URL is resolved first (multi-file items refused with the way to choose files), canonicalized, run on platform:archiveorg with archive.org's args, and never re-fetched when already on disk - pnpm ops import-archive-org {slug, item, files | match, dryRun?}: one drainable job (kind import-archive-org) importing the files one at a time - channel form: archive.org in both platform lists; duplicates label - player: archive.org records play their file in a native <video> (FilePlayer, summary mediaUrl) — seeks and highlights cues; link fallback - video page: "Archived on archive.org: <item> · torrent", and for a mirror "Originally on YouTube: <url> (uploaded <date>)" - citations: RecordView.originalLabel + downloads, derived at compose from the provenance — "YouTube · archive.org · torrent" for a mirror, "archive.org · torrent" otherwise — on cards, moment pages and the MCP report text Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Diffstat:
16 files changed, 471 insertions(+), 12 deletions(-)
diff --git a/common/components/FilePlayer.tsx b/common/components/FilePlayer.tsx @@ -0,0 +1,94 @@ +"use client"; + +import { + forwardRef, + useEffect, + useImperativeHandle, + useRef, +} from "react"; + +export type FilePlayerHandle = { + seekTo: (seconds: number, unit?: "seconds") => void; +}; + +type Props = { + url: string; + playing?: boolean; + onReady?: () => void; + onPlay?: () => void; + onPause?: () => void; + onProgress?: (state: { playedSeconds: number }) => void; + onError?: () => void; +}; + +// A plain media file in a native <video> — archive.org records, whose own +// embed takes no start time we can rely on but whose files are served over +// ranged HTTP, so the browser seeks them itself. The same imperative `seekTo` +// handle and ready/play/pause/progress/error events as KickPlayer, so +// PlayerProvider drives it the same way (full seeking + cue highlighting). An +// audio file plays in the same element. +const FilePlayer = forwardRef<FilePlayerHandle, Props>(function FilePlayer( + { url, playing, onReady, onPlay, onPause, onProgress, onError }, + ref, +) { + const videoRef = useRef<HTMLVideoElement | null>(null); + const cbRef = useRef({ onReady, onPlay, onPause, onProgress, onError }); + cbRef.current = { onReady, onPlay, onPause, onProgress, onError }; + + useImperativeHandle( + ref, + () => ({ + seekTo(seconds: number) { + const video = videoRef.current; + if (!video) return; + video.currentTime = Math.max(0, seconds); + void video.play().catch(() => {}); + }, + }), + [], + ); + + useEffect(() => { + const video = videoRef.current; + if (!video) return; + const emitReady = () => cbRef.current.onReady?.(); + const emitPlay = () => cbRef.current.onPlay?.(); + const emitPause = () => cbRef.current.onPause?.(); + const emitError = () => cbRef.current.onError?.(); + const emitProgress = () => + cbRef.current.onProgress?.({ playedSeconds: video.currentTime }); + video.addEventListener("loadedmetadata", emitReady, { once: true }); + video.addEventListener("play", emitPlay); + video.addEventListener("pause", emitPause); + video.addEventListener("error", emitError); + video.addEventListener("timeupdate", emitProgress); + video.src = url; + return () => { + video.removeEventListener("loadedmetadata", emitReady); + video.removeEventListener("play", emitPlay); + video.removeEventListener("pause", emitPause); + video.removeEventListener("error", emitError); + video.removeEventListener("timeupdate", emitProgress); + }; + }, [url]); + + // Autoplay may be refused without a gesture; the native controls start it. + useEffect(() => { + const video = videoRef.current; + if (!video) return; + if (playing) void video.play().catch(() => {}); + else video.pause(); + }, [playing]); + + return ( + <video + ref={videoRef} + controls + playsInline + preload="metadata" + style={{ width: "100%", height: "100%", background: "black" }} + /> + ); +}); + +export default FilePlayer; diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx @@ -30,6 +30,7 @@ import type { RumblePlayerHandle } from "./RumblePlayer"; import type { OdyseePlayerHandle } from "./OdyseePlayer"; import type { TwitchPlayerHandle } from "./TwitchPlayer"; import type { KickPlayerHandle } from "./KickPlayer"; +import type { FilePlayerHandle } from "./FilePlayer"; const ReactPlayer = dynamic(() => import("react-player/youtube"), { ssr: false, @@ -54,12 +55,19 @@ const KickPlayer = dynamic(() => import("./KickPlayer"), { ssr: false, }); +// archive.org records play their file in a native <video> (FilePlayer): the +// embed takes no start time we can rely on, the file seeks. +const FilePlayer = dynamic(() => import("./FilePlayer"), { + ssr: false, +}); + type PlayerHandle = | Pick<ReactPlayerType, "seekTo"> | RumblePlayerHandle | OdyseePlayerHandle | TwitchPlayerHandle - | KickPlayerHandle; + | KickPlayerHandle + | FilePlayerHandle; export type Cue = { start: number; end: number; text: string }; @@ -77,6 +85,7 @@ export type TranscriptData = { platform: Platform; webpageUrl: string; hlsUrl?: string; + mediaUrl?: string; cues?: Cue[]; }; @@ -100,6 +109,7 @@ type Detail = { platform: Platform; webpageUrl: string; hlsUrl?: string; + mediaUrl?: string; cues?: Cue[]; }; @@ -387,6 +397,8 @@ export function PlayerProvider({ // Set when the Kick HLS manifest fails to load (usually an expired VOD), so // the player swaps to the expiry/link fallback. Reset per active video below. const [kickError, setKickError] = useState(false); + // The same for an archive.org file that will not load: swap to the link. + const [fileError, setFileError] = useState(false); const playerRef = useRef<PlayerHandle | null>(null); const pendingSeekRef = useRef<number | null>(null); const readyForSlugRef = useRef<string | null>(null); @@ -431,6 +443,7 @@ export function PlayerProvider({ platform: detail.platform, webpageUrl: detail.webpageUrl, hlsUrl: detail.hlsUrl, + mediaUrl: detail.mediaUrl, cues: detail.cues, }; }, [detail, detailMatches]); @@ -511,7 +524,9 @@ export function PlayerProvider({ const copyShareUrl = useCallback(async (): Promise<boolean> => { if (!data) return false; const t = - data.platform === "youtube" || data.platform === "kick" + data.platform === "youtube" || + data.platform === "kick" || + (data.platform === "archiveorg" && data.mediaUrl) ? currentTime : (urlTime ?? 0); const secs = Math.max(0, Math.floor(t)); @@ -636,6 +651,7 @@ export function PlayerProvider({ dispatch({ type: "DIGEST_RESET" }); digestInFlightForRef.current = null; setKickError(false); + setFileError(false); if (!urlSlug) return; // Posts are not transcripts — never run the video fetch for one. // A post slug always arrives together with a slug change, so this closure's @@ -664,6 +680,7 @@ export function PlayerProvider({ platform: full.platform, webpageUrl: full.webpageUrl, hlsUrl: full.hlsUrl, + mediaUrl: full.mediaUrl, cues: full.cues, }, }); @@ -1020,6 +1037,22 @@ export function PlayerProvider({ } onError={() => setKickError(true)} /> + ) : data.platform === "archiveorg" && data.mediaUrl && !fileError ? ( + <FilePlayer + key={data.id} + ref={(p: FilePlayerHandle | null) => { + playerRef.current = p; + }} + url={data.mediaUrl} + playing={playing} + onReady={handleReady} + onPlay={() => setPlaying(true)} + onPause={() => setPlaying(false)} + onProgress={(s: { playedSeconds: number }) => + setCurrentTime(s.playedSeconds) + } + onError={() => setFileError(true)} + /> ) : data.platform === "kick" ? ( // Kick VOD with no playable manifest (expired or pre-feature // archive): show the expiry notice + a link to the source. diff --git a/common/components/citations/CitationCard.tsx b/common/components/citations/CitationCard.tsx @@ -137,7 +137,7 @@ function SpanLinks({ c }: { c: SpanCitationView }) { <Play className="size-3 shrink-0" aria-hidden /> Play {spanLabel(c.start, c.end)} </a> - {c.record.originalUrl && <ExternalA href={c.record.originalUrl}>Original</ExternalA>} + <RecordLinks r={c.record} /> {c.record.corpusUrl && ( <a href={c.record.corpusUrl} className={linkClass}> Transcript @@ -147,6 +147,21 @@ function SpanLinks({ c }: { c: SpanCitationView }) { ); } +// The record's original, then where it can be downloaded to check it +// ("YouTube · archive.org · torrent" for an archive.org mirror). +function RecordLinks({ r }: { r: SpanCitationView["record"] }) { + return ( + <> + {r.originalUrl && <ExternalA href={r.originalUrl}>{r.originalLabel ?? "Original"}</ExternalA>} + {r.downloads?.map((d) => ( + <ExternalA key={d.url} href={d.url}> + {d.label} + </ExternalA> + ))} + </> + ); +} + function PostLinks({ c }: { c: PostCitationView }) { return ( <> diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -281,6 +281,18 @@ const JOB_KINDS: Record<string, JobKindMeta> = { queueKeyStrategy: "custom", needsMedia: true, }, + // CHOSEN FILES OF ONE archive.org ITEM, imported one at a time on + // archive.org's own queue (controller/archiveOrgImport.ts). Drainable: a + // drain lets the file in flight finish and starts no more; re-running the + // same command resumes, skipping what landed. + "import-archive-org": { + kind: "import-archive-org", + label: "Import from archive.org", + drainable: true, + replayable: false, + queueKeyStrategy: "platform", + needsMedia: true, + }, // ONE WINDOW of a video's source media, fetched into data/<id>/clips/ for a // tool that asked for it by name (umtool's clip bench). Platform-queued like // every other fetch so it takes its turn behind the channel's own downloads; diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts @@ -118,6 +118,14 @@ export type RecordView = { // The record on its own platform at the cited time (lib/momentUrl.ts // platformMomentUrl), or the post's own URL. originalUrl?: string; + // What `originalUrl` is, when "Original" would not say: "archive.org" for an + // archive.org record, "YouTube" for an archive.org mirror of a YouTube + // upload (whose originalUrl is the upload at the cited second). + originalLabel?: string; + // Where a reader can fetch the recording itself to check it, derived from + // the record's provenance (lib/archiveOrg.ts archiveOrgCitationLinks): the + // archive.org page and the item's torrent. Absent for a record with none. + downloads?: { label: string; url: string }[]; // The record in this site's corpus (`/?v=<channel>/<id>&t=<s>`): a FULL site // only — a cited site has no corpus to open. corpusUrl?: string; diff --git a/common/publish/composeReports.ts b/common/publish/composeReports.ts @@ -63,6 +63,8 @@ import { parseVtt, type Cue } from "../lib/vtt"; import { parseTranscriptJson } from "../lib/whisper"; import { WHISPER_FILENAME, isEnglishVtt, resolvePrimaryVtt } from "../lib/videoStatus"; import { platformMomentUrl } from "../lib/momentUrl"; +import { archiveOrgCitationLinks, type ArchiveOrgProvenance } from "../lib/archiveOrg"; +import { loadArchiveOrgProvenance } from "../lib/archiveOrg-server"; import type { Platform } from "../lib/platform"; import { readAllPosts } from "../lib/posts-server"; import type { Post } from "../lib/posts"; @@ -209,6 +211,9 @@ type CitedRecord = { // first). A served `en` track can be a rewrite of what was said, so a // quote is checked against each and the best match is the one shown. tracks: { name: string; cues: Cue[] }[]; + // An archive.org record's provenance (its torrent, a mirror's original); + // null for every other record. + archiveOrg: ArchiveOrgProvenance | null; }; // The English VTT tracks of a video dir, `en-orig` first, then the order @@ -287,7 +292,8 @@ export async function readCitedRecord( if (v.length > 0) tracks.push({ name, cues: v }); } if (tracks.length === 0 && cues.length > 0) tracks.push({ name: "cues", cues }); - return { summary, cues, tracks }; + const archiveOrg = summary.platform === "archiveorg" ? await loadArchiveOrgProvenance(dir) : null; + return { summary, cues, tracks, archiveOrg }; } const isoDay = (uploadDate: string | undefined): string | undefined => @@ -588,9 +594,26 @@ export async function composeReports(opts: ComposeReportsOptions): Promise<Compo corpusUrl: cited ? undefined : corpusLink(`${c.channel}/${c.id}`, { vm: "post" }), }); } - const { summary } = (await recordOf(c.channel, c.id))!; + const { summary, archiveOrg } = (await recordOf(c.channel, c.id))!; const audioOnly = isAudioOnlyPlatform(config?.platform); const seconds = Math.max(0, Math.floor(c.start)); + if (summary.platform === "archiveorg") { + // archive.org: the original (YouTube at the second, for a mirror; else + // the archive.org page) plus the downloads a reader can check it from. + const links = archiveOrgCitationLinks({ webpageUrl: summary.webpageUrl, provenance: archiveOrg, seconds: c.start }); + return defined({ + channel: c.channel, + channelTitle: config?.name ?? (summary.channel || undefined), + id: c.id, + title: summary.title, + date: isoDay(summary.uploadDate), + platform: summary.platform, + originalUrl: links.original?.url ?? (summary.webpageUrl || undefined), + originalLabel: links.original?.label, + downloads: links.downloads.length > 0 ? links.downloads : undefined, + corpusUrl: cited ? undefined : corpusLink(summary.slug ?? `${c.channel}/${summary.id}`, seconds > 0 ? { t: String(seconds) } : {}), + }); + } return defined({ channel: c.channel, channelTitle: config?.name ?? (summary.channel || undefined), diff --git a/common/publish/composeReportsArchiveOrg.test.ts b/common/publish/composeReportsArchiveOrg.test.ts @@ -0,0 +1,60 @@ +// A cited archive.org record carries its provenance into compose, and the +// links a citation of it shows are derived from that (lib/archiveOrg.ts +// archiveOrgCitationLinks): YouTube at the second for a mirror, the +// archive.org page and the torrent as downloads. Every name here is invented. +// +// Run with: node_modules/.bin/tsx --test publish/composeReportsArchiveOrg.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { readCitedRecord } from "./composeReports"; +import { archiveOrgCitationLinks, buildArchiveOrgProvenance } from "../lib/archiveOrg"; +import { archiveOrgDetailsUrl, archiveOrgVideoId } from "../lib/archiveOrgId"; + +const ITEM = "example-item"; +const FILE = "First Upload-AbC123xyz_9.mp4"; + +test("readCitedRecord loads an archive.org record's provenance; others get null", async () => { + const channels = mkdtempSync(path.join(tmpdir(), "compose-archiveorg-")); + const id = archiveOrgVideoId({ identifier: ITEM, file: FILE }); + const dir = path.join(channels, "demo-archive", "data", id); + mkdirSync(dir, { recursive: true }); + writeFileSync( + path.join(dir, "metadata.info.json"), + JSON.stringify({ + id: `${ITEM}/${FILE}`, + extractor_key: "ArchiveOrg", + title: "First Upload (original)", + upload_date: "20230405", + webpage_url: archiveOrgDetailsUrl({ identifier: ITEM, file: FILE }), + }), + ); + const prov = buildArchiveOrgProvenance({ + ref: { identifier: ITEM, file: FILE }, + item: { + metadata: { identifier: ITEM, title: "Example Archive" }, + files: [{ name: FILE, source: "original" }, { name: `${ITEM}_archive.torrent`, source: "metadata" }], + }, + infoJson: { id: "AbC123xyz_9", extractor_key: "Youtube", upload_date: "20230405" }, + fetchedAt: "2026-01-01T00:00:00.000Z", + }); + writeFileSync(path.join(dir, "archiveorg.json"), JSON.stringify(prov)); + + const rec = await readCitedRecord(channels, "demo-archive", id); + assert.ok(rec); + assert.equal(rec.summary.platform, "archiveorg"); + assert.equal(rec.summary.id, id); + assert.deepEqual(rec.archiveOrg, prov); + const links = archiveOrgCitationLinks({ webpageUrl: rec.summary.webpageUrl, provenance: rec.archiveOrg, seconds: 12 }); + assert.equal(links.original?.label, "YouTube"); + assert.equal(links.original?.url, "https://www.youtube.com/watch?v=AbC123xyz_9&t=12s"); + assert.deepEqual(links.downloads.map((d) => d.label), ["archive.org", "torrent"]); + + const yt = path.join(channels, "demo-archive", "data", "AbC123xyz_9"); + mkdirSync(yt, { recursive: true }); + writeFileSync(path.join(yt, "metadata.info.json"), JSON.stringify({ id: "AbC123xyz_9", extractor_key: "Youtube" })); + assert.equal((await readCitedRecord(channels, "demo-archive", "AbC123xyz_9"))?.archiveOrg, null); +}); diff --git a/editor/app/api/ops/import-archive-org/route.ts b/editor/app/api/ops/import-archive-org/route.ts @@ -0,0 +1,44 @@ +import { importArchiveOrgAction } from "../../../channels/[slug]/pipelineActions"; +import { + OpsInputError, + jobResponse, + ops, + optBool, + optString, + reqSlug, + reqString, + type OpsBody, +} from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, item, files?: string[], match?: string, dryRun? } +// -> { ok: true, jobId } +// +// Import chosen media files of ONE archive.org item into an existing channel, +// as one job on archive.org's own queue: one file at a time, a jittered pause +// between them, files already downloaded skipped, a rate limit or three +// failures in a row ending it (controller/archiveOrgImport.ts). Exactly one of +// `files` (exact paths in the item, untrimmed) and `match` (a case-insensitive +// regex over them). `dryRun` logs what would be fetched and fetches nothing. +function optFiles(body: OpsBody): string[] | undefined { + const v = body.files; + if (v === undefined) return undefined; + if (!Array.isArray(v) || v.length === 0 || v.some((s) => typeof s !== "string" || !s)) { + throw new OpsInputError('"files" must be a non-empty array of file names'); + } + return v as string[]; +} + +export async function POST(request: Request) { + return ops(request, ["slug", "item", "files", "match", "dryRun"], async (body) => + jobResponse( + await importArchiveOrgAction(reqSlug(body, "slug"), { + item: reqString(body, "item"), + files: optFiles(body), + match: optString(body, "match"), + dryRun: optBool(body, "dryRun"), + }), + ), + ); +} diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -12,7 +12,14 @@ import { downloadQueueKey, resolveQueueKey, } from "yt-dlp-transcript-common/lib/queueKeys"; -import { detectPlatform } from "yt-dlp-transcript-common/lib/platform"; +import { + detectPlatform, + platformQueueKey, +} from "yt-dlp-transcript-common/lib/platform"; +import { + resolveArchiveOrgImportUrl, + runArchiveOrgImport, +} from "yt-dlp-transcript-common/controller/archiveOrgImport"; import { heldPlatformRefusal, platformCooldownRemainingMs, @@ -24,7 +31,11 @@ import { readChannelConfig, readChannelStat, } from "yt-dlp-transcript-common/controller/channels"; -import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; +import { + destinationExists, + extractVideoId, + runYtdlp, +} from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob"; @@ -496,10 +507,39 @@ export async function importVideoAction( if (!channelConfig) { return { ok: false, error: `Channel "${slug}" not found` }; } - const videoUrl = url.trim(); + let videoUrl = url.trim(); if (!/^https?:\/\//i.test(videoUrl)) { return { ok: false, error: "Enter a valid video URL" }; } + // AN archive.org URL is resolved first (one cached request for the item's + // metadata): an item holding several media files is refused — import a file + // of it, or several with import-archive-org — and the URL becomes the + // canonical page of what is imported (controller/archiveOrgImport.ts). It + // runs on archive.org's own queue whatever the channel's platform, with + // archive.org's yt-dlp args, and a file already on disk is not fetched again. + const archiveOrg = detectPlatform(videoUrl) === "archiveorg"; + let downloadConfig = channelConfig; + if (archiveOrg) { + const resolved = await resolveArchiveOrgImportUrl(videoUrl); + if (!resolved.ok) return { ok: false, error: resolved.error }; + videoUrl = resolved.url; + if ( + await destinationExists( + path.join(paths.channelsDir, slug, "data"), + resolved.id, + channelConfig.handling, + ) + ) { + return { + ok: false, + info: true, + error: `Already downloaded: data/${resolved.id}/ — archive.org is not asked for it again.`, + }; + } + if (channelConfig.platform !== "archiveorg") { + downloadConfig = { ...channelConfig, platform: "archiveorg" }; + } + } // Best-effort canonical id: used only for revalidation/labels. When null, // downloadOneManaged falls back to %(id)s and the reconcile pass repairs the // dir, so we don't hard-fail here. @@ -509,7 +549,10 @@ export async function importVideoAction( const settings = getSettings(); return runManagedFunction({ kind: "import-one", - queueKey: resolveQueueKey(downloadQueueKey(channelConfig), queueKey), + queueKey: resolveQueueKey( + archiveOrg ? platformQueueKey("archiveorg") : downloadQueueKey(channelConfig), + queueKey, + ), paths, channelSlug: slug, videoId, @@ -522,7 +565,7 @@ export async function importVideoAction( try { await downloadOneManaged({ channelSlug: slug, - channelConfig, + channelConfig: downloadConfig, paths, videoUrl, onLog: task.onLog, @@ -557,6 +600,72 @@ export async function importVideoAction( }); } +// IMPORT CHOSEN FILES OF ONE archive.org ITEM (`pnpm ops import-archive-org`): +// one job on archive.org's own queue that imports the files one at a time, +// with a jittered pause between them, skipping any already downloaded, and +// stopping on a rate limit or three failures in a row +// (controller/archiveOrgImport.ts). `files` names exact paths in the item; +// `match` is a case-insensitive regex over them. `dryRun` lists what would be +// fetched and fetches nothing. +export async function importArchiveOrgAction( + slug: string, + opts: { item: string; files?: string[]; match?: string; dryRun?: boolean }, +): Promise<StreamActionResult> { + const paths = getPaths(); + const channelConfig = await readChannelConfig(paths, slug); + if (!channelConfig) { + return { ok: false, error: `Channel "${slug}" not found` }; + } + const item = opts.item.trim(); + if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(item)) { + return { ok: false, error: `"${item}" is not an archive.org identifier` }; + } + if ((opts.files === undefined) === (opts.match === undefined)) { + return { ok: false, error: 'Name the files: exactly one of "files" (a list) or "match" (a regex)' }; + } + if (opts.match !== undefined) { + try { + new RegExp(opts.match, "i"); + } catch (e) { + return { ok: false, error: `"match" is not a valid regex: ${(e as Error).message}` }; + } + } + if (!opts.dryRun) { + const err = await lowDiskError(paths, slug); + if (err) return err; + } + const downloadConfig = + channelConfig.platform === "archiveorg" + ? channelConfig + : { ...channelConfig, platform: "archiveorg" as const }; + return runManagedFunction({ + kind: "import-archive-org", + queueKey: platformQueueKey("archiveorg"), + paths, + channelSlug: slug, + fn: async (onLog, signal, _setProgress, ctx) => { + const result = await runArchiveOrgImport({ + paths, + slug, + channelConfig: downloadConfig, + identifier: item, + selection: opts.files !== undefined ? { files: opts.files } : { match: opts.match! }, + onLog, + signal, + drainSignal: ctx.drainSignal, + dryRun: opts.dryRun, + deps: { + onImported: (id) => safeRevalidate([`/channels/${slug}/videos/${id}`]), + }, + }); + safeRevalidate([`/channels/${slug}`, "/channels"]); + if (result.failed.length > 0 && result.imported.length === 0 && !opts.dryRun) { + throw new Error(result.stopped ?? `${result.failed.length} file(s) failed`); + } + }, + }); +} + // A PODCAST CHANNEL'S RECORDS COMPLETED FROM ITS RSS FEED (the // `feed-metadata` job, controller/feedMetadataBackfill.ts): one fetch of the // channel's url, then title, date, description and duration written into each diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -11,6 +11,8 @@ import { listClipWindows } from "yt-dlp-transcript-common/lib/clipWindow-server" import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server"; import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server"; +import { loadArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg-server"; +import type { ArchiveOrgProvenance } from "yt-dlp-transcript-common/lib/archiveOrg"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { readVideoMetadataForDisplay, @@ -134,6 +136,9 @@ export default async function VideoDetailPage({ const excludedFromTruncatedCheck = await isExcludedFromTruncatedCheck(videoDir); const savedVideo = await loadSavedVideo(videoDir); + // Where an archive.org record came from (lib/archiveOrg-server.ts): the + // item and its torrent, and a mirror's original. Absent everywhere else. + const archiveOrg = await loadArchiveOrgProvenance(videoDir); // The windows another tool asked this editor to fetch. One readdir of // data/<id>/clips/ plus a stat per file — and no per-CHANNEL count anywhere, // because that would be a walk of every video dir to draw one number. @@ -185,6 +190,7 @@ export default async function VideoDetailPage({ doNotClean, excludedFromTruncatedCheck, savedVideo, + archiveOrg, clipWindows, vttProvenance, coverage, @@ -212,6 +218,7 @@ export default async function VideoDetailPage({ doNotClean, excludedFromTruncatedCheck, savedVideo, + archiveOrg, clipWindows, vttProvenance, coverage, @@ -277,6 +284,7 @@ export default async function VideoDetailPage({ </span> )} </div> + {archiveOrg && <ArchiveOrgProvenanceLine prov={archiveOrg} />} {meta.description && ( <details className="text-sm"> <summary className="cursor-pointer text-muted-foreground hover:text-foreground"> @@ -343,6 +351,40 @@ export default async function VideoDetailPage({ ); } +// "Archived on archive.org: <item> · torrent", and for a mirror "Originally on +// YouTube: <url> (uploaded <date>)". Terse, under the header line. +function ArchiveOrgProvenanceLine({ prov }: { prov: ArchiveOrgProvenance }) { + const link = "underline hover:text-foreground"; + return ( + <div aria-label="archive.org provenance" className="flex flex-col gap-1 text-sm text-muted-foreground"> + <span> + Archived on archive.org:{" "} + <a href={prov.fileUrl ?? prov.itemUrl} target="_blank" rel="noreferrer" className={link}> + {prov.item.title ?? prov.identifier} + {prov.file ? ` / ${prov.file}` : ""} + </a> + {prov.torrentUrl && ( + <> + {" · "} + <a href={prov.torrentUrl} target="_blank" rel="noreferrer" className={link}> + torrent + </a> + </> + )} + </span> + {prov.mirror && ( + <span> + Originally on YouTube:{" "} + <a href={prov.mirror.url} target="_blank" rel="noreferrer" className={link}> + {prov.mirror.url} + </a> + {prov.mirror.uploadDate && ` (uploaded ${formatUploadDate(prov.mirror.uploadDate)})`} + </span> + )} + </div> + ); +} + function formatUploadDate(s: string): string { // yt-dlp emits YYYYMMDD. Render as YYYY-MM-DD; pass through anything else. if (/^\d{8}$/.test(s)) { diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx @@ -369,6 +369,7 @@ export function ChannelForm({ <option value="odysee">Odysee</option> <option value="twitch">Twitch</option> <option value="kick">Kick</option> + <option value="archiveorg">archive.org</option> <option value="twitter">X / Twitter (posts)</option> <option value="bluesky">Bluesky (posts)</option> </SeededSelect> @@ -488,6 +489,7 @@ export function ChannelForm({ <option value="odysee">Odysee</option> <option value="twitch">Twitch</option> <option value="kick">Kick</option> + <option value="archiveorg">archive.org</option> <option value="twitter">X / Twitter (posts)</option> <option value="bluesky">Bluesky (posts)</option> </ControlledSelect> diff --git a/editor/app/storage/lib/storeBusy.ts b/editor/app/storage/lib/storeBusy.ts @@ -40,6 +40,7 @@ const STORE_TOUCHING_KINDS = new Set([ "download-missing", "download-missing-subs", "import-one", + "import-archive-org", "redownload-archive", "redownload-incomplete-bucket", "retry-bucket", diff --git a/export/app/components/reports/MomentArticle.tsx b/export/app/components/reports/MomentArticle.tsx @@ -124,9 +124,16 @@ export default function MomentArticle({ view }: { view: MomentPageView }) { <p className="flex flex-wrap gap-x-4 gap-y-1 text-sm"> {r.originalUrl && ( <ExternalLinkText href={r.originalUrl}> - {isSpan && view.start !== undefined ? `Original at ${formatTimestamp(view.start)}` : "Original post"} + {isSpan && view.start !== undefined + ? `${r.originalLabel ?? "Original"} at ${formatTimestamp(view.start)}` + : "Original post"} </ExternalLinkText> )} + {r.downloads?.map((d) => ( + <ExternalLinkText key={d.url} href={d.url}> + {d.label} + </ExternalLinkText> + ))} {r.corpusUrl && ( <a href={r.corpusUrl} className={textLink}> Open in the archive diff --git a/export/app/duplicates/DuplicatesClient.tsx b/export/app/duplicates/DuplicatesClient.tsx @@ -31,6 +31,7 @@ const PLATFORM_LABEL: Record<string, string> = { odysee: "Odysee", twitch: "Twitch", kick: "Kick", + archiveorg: "archive.org", }; const MATCH_LABEL: Record<DuplicateCluster["matchKind"], string> = { diff --git a/mcp/src/reports.ts b/mcp/src/reports.ts @@ -153,7 +153,10 @@ function citationLines(c: CitationView, origin: string | null): string[] { .join(" · "); lines.push(`${n} ${c.kind} ${where} @ ${Math.floor(c.start)}–${Math.floor(c.end)} s`); lines.push(` ${quote}` + (c.speaker ? ` — ${c.speaker}` : "")); - if (c.record.originalUrl) lines.push(` original: ${c.record.originalUrl}`); + if (c.record.originalUrl) { + lines.push(` original${c.record.originalLabel ? ` (${c.record.originalLabel})` : ""}: ${c.record.originalUrl}`); + } + for (const d of c.record.downloads ?? []) lines.push(` ${d.label}: ${d.url}`); lines.push(` moment: ${absolute(origin, c.href)}`); break; } diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -27,6 +27,8 @@ // pnpm ops sync --json '{"slug":"the-quartering"}' --wait // pnpm ops metadata-scan --json '{"slug":"the-quartering"}' // pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait +// pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}' +// pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait // pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}' // pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}' // pnpm ops lane --json '{"lane":"download","held":true}' @@ -96,6 +98,9 @@ const ACTIONS = [ "channel-config", "metadata-scan", "import-video", + // Chosen media files of ONE archive.org item ({slug, item, files: [...] | + // match: "<regex>", dryRun?}): one job, one file at a time, paced. + "import-archive-org", // A podcast channel's records completed from its RSS feed ({slug, dryRun?}): // one fetch of the feed, no media. "feed-metadata",