import { NextResponse } from "next/server"; import { authorizeWorkerRequest } from "yt-dlp-transcript-common/lib/workerToken"; import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels"; import { MAX_CLIP_WINDOW_SECONDS, MAX_FETCH_MAX_HEIGHT, MIN_FETCH_MAX_HEIGHT, isFetchMaxHeight, } from "yt-dlp-transcript-common/lib/clipWindow"; import { sourceVideoQualityForMaxHeight } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { fetchFullSourceAction, fetchWindowAction, type FetchMediaOutcome, } from "../../../channels/[slug]/videos/[id]/videoActions"; export const dynamic = "force-dynamic"; // SOURCE MEDIA FOR ANOTHER TOOL, fetched through the editor's managed download // path. // // umtool's clip bench ran yt-dlp itself. It now posts here, and the bytes land // in the corpus beside the video — cookie policy, per-platform sleeps, 429 // cooldown and backoff included — where the next tool and the next build reuse // them instead of paying for the same seconds twice. // // SAME DOOR AS /api/worker/*: the shared WORKER_TOKEN. Deliberately, and not // laziness — this endpoint spends the source's patience, so it must be no // easier to reach than the executor protocol, and an instance with no token is // not one by accident (503, not an open proxy). // // The download PAUSE does not gate it; see fetchWindowAction for why. // A video id lands in a filesystem path. Anchored — and `.` / `..` are refused // separately, because the class allows a dot and "`..`" alone would otherwise // pass a pattern written to stop traversal. The SLUG uses the repo's own // isValidChannelSlug (controller/channels.ts) rather than a second grammar // here: a slug IS a directory name, and two definitions of what one may // contain is one more than the corpus can have. const ID_RE = /^[\w.-]+$/; const isVideoId = (v: string): boolean => ID_RE.test(v) && v !== "." && v !== ".."; type Body = { channelSlug?: unknown; videoId?: unknown; webpageUrl?: unknown; from?: unknown; to?: unknown; requestedBy?: unknown; manifest?: unknown; clipId?: unknown; reason?: unknown; pad?: unknown; full?: unknown; maxHeight?: unknown; }; const str = (v: unknown): string | undefined => typeof v === "string" && v.trim() !== "" ? v.trim() : undefined; function answer(outcome: FetchMediaOutcome): NextResponse { if (!outcome.ok) { const body: Record = { error: outcome.error }; if (outcome.cooldownMs !== undefined) body.cooldownMs = outcome.cooldownMs; if (outcome.platform !== undefined) body.platform = outcome.platform; return NextResponse.json(body, { status: outcome.status }); } // `started` is the Retry shape (a live StreamActionResult); the route never // asks for it, so reaching it here would be a programming error, not a state // to render. if ("started" in outcome) { return NextResponse.json( { error: "internal: streamed outcome on the HTTP path" }, { status: 500 }, ); } if (outcome.cached) { return NextResponse.json( { cached: true, file: outcome.file, from: outcome.from, to: outcome.to, bytes: outcome.bytes, provenance: outcome.provenance, ...(outcome.height !== undefined ? { height: outcome.height } : {}), }, { status: 200 }, ); } // `existing: true` (release 19, A5): the window was already queued or being // fetched, and `jobId` is THAT job — poll it as you would a new one. return NextResponse.json( { cached: false, jobId: outcome.jobId, file: outcome.file, from: outcome.from, to: outcome.to, ...(outcome.existing ? { existing: true } : {}), }, { status: 202 }, ); } export async function POST(request: Request) { const auth = authorizeWorkerRequest(request.headers.get("authorization")); if (!auth.ok) { return NextResponse.json({ error: auth.error }, { status: auth.status }); } let body: Body; try { body = (await request.json()) as Body; } catch { return NextResponse.json({ error: "malformed JSON body" }, { status: 400 }); } const channelSlug = str(body.channelSlug); const videoId = str(body.videoId); if (!isValidChannelSlug(channelSlug)) { return NextResponse.json( { error: "channelSlug is required and must be a valid channel slug" }, { status: 400 }, ); } if (!videoId || !isVideoId(videoId)) { return NextResponse.json( { error: "videoId is required and must match /^[\\w.-]+$/" }, { status: 400 }, ); } // WHO ASKED IS NOT OPTIONAL. The whole point of routing the fetch through // here is that the bytes carry a reason; an anonymous window is one nobody // can explain in six months. const requestedBy = str(body.requestedBy); if (!requestedBy) { return NextResponse.json( { error: "requestedBy is required (the tool asking for these bytes)" }, { status: 400 }, ); } const provenance = { requestedBy, manifest: str(body.manifest), clipId: str(body.clipId), reason: str(body.reason), pad: typeof body.pad === "number" && Number.isFinite(body.pad) ? body.pad : undefined, requestedAt: new Date().toISOString(), }; // THE HEIGHT CAP, optional. Absent (or null) keeps every default as it was; // anything else must be a whole number of pixels in range, because it ends up // inside a yt-dlp `-f` selector. const maxHeight = body.maxHeight === undefined || body.maxHeight === null ? undefined : body.maxHeight; if (maxHeight !== undefined && !isFetchMaxHeight(maxHeight)) { return NextResponse.json( { error: `maxHeight must be a whole number of pixels from ` + `${MIN_FETCH_MAX_HEIGHT} to ${MAX_FETCH_MAX_HEIGHT}`, }, { status: 400 }, ); } if (body.full === true) { // A cap at or under 720 asks for "video_720"; one above it asks for // "original" in so many words. No cap: the channel's, else the global, // source-video quality, as a persist from the video page would. return answer( await fetchFullSourceAction({ slug: channelSlug, videoId, provenance, quality: maxHeight === undefined ? undefined : sourceVideoQualityForMaxHeight(maxHeight), }), ); } const from = Number(body.from); const to = Number(body.to); if (!Number.isFinite(from) || !Number.isFinite(to) || from < 0) { return NextResponse.json( { error: "from and to must be finite, non-negative seconds" }, { status: 400 }, ); } if (from >= to) { return NextResponse.json( { error: `from (${from}) must be less than to (${to})` }, { status: 400 }, ); } if (to - from > MAX_CLIP_WINDOW_SECONDS) { return NextResponse.json( { error: `a window may be at most ${MAX_CLIP_WINDOW_SECONDS}s ` + `(asked for ${Math.round(to - from)}s); pass full: true for the ` + `whole recording`, }, { status: 400 }, ); } return answer( await fetchWindowAction({ slug: channelSlug, videoId, webpageUrl: str(body.webpageUrl), from, to, provenance, maxHeight, }), ); }