// The social-post corpus: a parallel content layer beside video transcripts. // // Posts are live-chat-shaped, not video-shaped. Like the `subs` layer they get // their own page tree (/posts//{manifest,page-NNNN}.json), their // own LayerScope ("posts") and their own manifest version — and they compose // into the SAME boolean query tree as transcripts, so one search covers both. // // Client-safe: no node imports (same convention as cookiePolicy.ts / // availability.ts). The server-side store lives in posts-server.ts. import { pageFileName } from "./manifest"; // "xenforo" is a forum THREAD read as a posts source: the channel is one // thread, each forum post a Post. It is named for the forum software, not a // host — which forum a thread is on is its URL (and `Post.forum.host`). export type PostPlatform = "twitter" | "bluesky" | "xenforo"; export const POST_PLATFORM_VALUES: ReadonlyArray = [ "twitter", "bluesky", "xenforo", ]; export function isPostPlatform(v: unknown): v is PostPlatform { return v === "twitter" || v === "bluesky" || v === "xenforo"; } // The platform's name as a reader sees it. export function postPlatformLabel(platform: PostPlatform | string | undefined): string { if (platform === "twitter") return "X"; if (platform === "bluesky") return "Bluesky"; if (platform === "xenforo") return "Forum"; return "Post"; } // Media a post carries, as LINKS (the archive keeps the URLs, not the bytes — // a post capture downloads them). Recorded by the forum parser; the X and // Bluesky normalizers count media instead (`mediaCount`). export type PostMediaKind = "image" | "video" | "attachment" | "embed" | "link-card"; export const POST_MEDIA_KINDS: ReadonlyArray = [ "image", "video", "attachment", "embed", "link-card", ]; export type PostMedia = { kind: PostMediaKind; url: string; // A file name (an attachment's), or an image's alt text. name?: string; // An embed's provider ("youtube", "twitter", …), as the forum named it. provider?: string; }; // What a forum post carries beyond the common shape (platform "xenforo"). export type ForumPostInfo = { host: string; // The thread's numeric id, and its title and canonical URL when read. threadId: string; threadTitle?: string; threadUrl?: string; // The thread page the post was read on, and its position in the thread // (the "#N" the forum shows). page?: number; position?: number; // The author's numeric member id. authorId?: string; // When the forum says the post was last edited (ISO-8601). editedAt?: string; // The posts this one quotes, in order: the quoted post's id and author // when the quote names them. quotes?: { postId?: string; author?: string; authorId?: string }[]; }; // A pointer to another post, which may or may not itself be archived. Used for // reply/quote/repost edges so thread structure survives even when the // referenced post is outside the archived account. export type PostRef = { platform: PostPlatform; id: string; url?: string; author?: string; }; export type PostEngagement = { likes?: number; reposts?: number; replies?: number; quotes?: number; }; export type Post = { // Native post id: tweet id, or the atproto record key (rkey). id: string; // `${channelSlug}/${id}` — the same identity convention videos use, so a post // can be addressed by one opaque string across search/AI/viewer layers. slug: string; channelSlug: string; author: string; // handle, e.g. "example.bsky.social" authorName?: string; // display name // ISO-8601, ms precision where the source provides it. THE sort key: it sorts // lexicographically and so drops straight into an LMDB key tuple, and it // fixes the intra-day ordering problem a YYYYMMDD key has. createdAt: string; // YYYYMMDD derived from createdAt. Retained (not derived at read time) so // every existing date filter — the fdf/fdt share params, the lexicographic // bounds in SearchSessionContext.passesFilter and mcp/src/search.ts // passesFilters — keeps working against posts with zero changes. uploadDate: string; text: string; url: string; // canonical permalink platform: PostPlatform; lang?: string; threadId?: string; // root post id, for grouping replyTo?: PostRef; quoted?: PostRef; repostOf?: PostRef; isReply: boolean; isRepost: boolean; links: string[]; // expanded outbound urls mediaCount?: number; // counted, not archived (v1) // The media's URLs, where the source gives them (forum posts). media?: PostMedia[]; // Forum-specific facts (platform "xenforo"). forum?: ForumPostInfo; engagement?: PostEngagement; // Merged in at index time from the channel's availability sidecar (the same // shape videos use: the stored record is the source of truth, the flag on // the served record is derived). Absent = never checked, which is NOT the // same as "still live". isDeleted?: boolean; availability?: PostAvailability; availabilityCheckedAt?: string; }; // Whether an archived post is still live at its source. The post analogue of // video Availability — and arguably the more valuable half of the archive: a // deleted post is exactly the thing a commentary archive exists to preserve. // // Deliberately narrower than the video enum. A post has no members-only or // age-gated equivalent we can distinguish per-item; what we can tell apart is // "the post is gone", "the whole account is gone/protected" (which is not the // post's fault and may reverse), and "the check itself failed". export type PostAvailability = | "available" | "deleted" | "account_unavailable" | "error"; export const POST_AVAILABILITY_VALUES: ReadonlyArray = [ "available", "deleted", "account_unavailable", "error", ]; export function isPostAvailability(v: unknown): v is PostAvailability { return ( typeof v === "string" && (POST_AVAILABILITY_VALUES as string[]).includes(v) ); } // Only `deleted` is treated as "gone for good". `account_unavailable` can // reverse (a protected account reopening, a suspension lifted) and `error` is // usually transient — neither is evidence the post itself was removed, so // neither is surfaced as a deletion. export function isPostGone(a: PostAvailability | null | undefined): boolean { return a === "deleted"; } export type PostAvailabilityRecord = { availability: PostAvailability; checkedAt: string; // Appended only when the availability CHANGES, so the file stays small and // the first-seen-deleted moment is preserved. Mirrors the video record. history?: { availability: PostAvailability; at: string }[]; }; // postId -> record, stored once per channel. Videos keep a file per video dir, // but posts have no per-post directory (they live in month-sharded JSONL), so // the sidecar is per channel. export type PostAvailabilityMap = Record; export type PostPage = Post[]; export type ChannelPostsManifest = { version: number; channelSlug: string; pageCount: number; maxPageBytes: number; generatedAt: string; // postId -> page index. Mirrors ChannelTranscriptsManifest.slugToPage so a // single post can be fetched without scanning every page. slugToPage: Record; }; export const POSTS_MANIFEST_VERSION = 1; // Site-level index of which channels carry posts — the posts analogue of // SubsManifest. Lets the export client render the corpus toggle and the channel // pickers without probing every channel's per-channel manifest. export type PostsChannelEntry = { name: string; slug: string; postCount: number; platform: PostPlatform; groupId?: string; }; export type PostsManifest = { version: number; channels: PostsChannelEntry[]; totalCount: number; generatedAt: string; siteId?: string; }; export const SITE_POSTS_MANIFEST_VERSION = 1; export const postsPageFileName = pageFileName; // --------------------------------------------------------------------------- // Derivations // --------------------------------------------------------------------------- // YYYYMMDD in UTC from an ISO-8601 timestamp. Returns "" for an unparseable // input so a malformed post degrades to "no date" rather than poisoning the // lexicographic date filters with garbage. export function uploadDateFromCreatedAt(createdAt: string): string { const ms = Date.parse(createdAt); if (!Number.isFinite(ms)) return ""; const d = new Date(ms); const y = d.getUTCFullYear(); const m = d.getUTCMonth() + 1; const day = d.getUTCDate(); return `${String(y).padStart(4, "0")}${String(m).padStart(2, "0")}${String(day).padStart(2, "0")}`; } // YYYY-MM shard key from an ISO-8601 timestamp — the on-disk JSONL month shard. // Falls back to "unknown" so a post with a bad timestamp is still stored. export function monthShardFromCreatedAt(createdAt: string): string { const ms = Date.parse(createdAt); if (!Number.isFinite(ms)) return "unknown"; const d = new Date(ms); return `${String(d.getUTCFullYear()).padStart(4, "0")}-${String( d.getUTCMonth() + 1, ).padStart(2, "0")}`; } export function postSlug(channelSlug: string, id: string): string { return `${channelSlug}/${id}`; } // Split a post slug back into its parts. The id may itself contain "/" for no // current platform, but splitting on the FIRST separator keeps that safe. export function parsePostSlug( slug: string, ): { channelSlug: string; id: string } | null { const idx = slug.indexOf("/"); if (idx <= 0 || idx === slug.length - 1) return null; return { channelSlug: slug.slice(0, idx), id: slug.slice(idx + 1) }; } // Canonical permalink for a post. Bluesky needs the handle (its URLs are // /profile//post/); X only needs the numeric id but includes the // handle for readability. A forum post needs its forum's origin (`base`, // "https://forum.example"): XenForo's /posts// resolves to the post in its // thread wherever the thread has moved. export function postPermalink( platform: PostPlatform, author: string, id: string, base?: string, ): string { if (platform === "bluesky") { return `https://bsky.app/profile/${author}/post/${id}`; } if (platform === "xenforo") { return `${(base ?? "").replace(/\/+$/, "")}/posts/${id}/`; } return `https://x.com/${author || "i"}/status/${id}`; } // Runtime validation for a post read back off disk / off the wire. Mirrors // parseChannelConfig's strict-allowlist stance: unknown keys are dropped, and a // record missing any required field is rejected outright rather than repaired. export function parsePost(raw: unknown): Post | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; if (typeof r.id !== "string" || !r.id) return null; if (typeof r.channelSlug !== "string" || !r.channelSlug) return null; if (typeof r.text !== "string") return null; if (typeof r.createdAt !== "string" || !r.createdAt) return null; if (!isPostPlatform(r.platform)) return null; const author = typeof r.author === "string" ? r.author : ""; const post: Post = { id: r.id, slug: typeof r.slug === "string" && r.slug ? r.slug : postSlug(r.channelSlug, r.id), channelSlug: r.channelSlug, author, createdAt: r.createdAt, uploadDate: typeof r.uploadDate === "string" && r.uploadDate ? r.uploadDate : uploadDateFromCreatedAt(r.createdAt), text: r.text, url: typeof r.url === "string" && r.url ? r.url : postPermalink( r.platform, author, r.id, r.platform === "xenforo" ? forumBaseOf(r.forum) : undefined, ), platform: r.platform, isReply: r.isReply === true, isRepost: r.isRepost === true, links: Array.isArray(r.links) && r.links.every((l) => typeof l === "string") ? (r.links as string[]) : [], }; if (typeof r.authorName === "string") post.authorName = r.authorName; if (typeof r.lang === "string") post.lang = r.lang; if (typeof r.threadId === "string") post.threadId = r.threadId; const replyTo = parsePostRef(r.replyTo); if (replyTo) post.replyTo = replyTo; const quoted = parsePostRef(r.quoted); if (quoted) post.quoted = quoted; const repostOf = parsePostRef(r.repostOf); if (repostOf) post.repostOf = repostOf; if (r.isDeleted === true) post.isDeleted = true; if (isPostAvailability(r.availability)) post.availability = r.availability; if (typeof r.availabilityCheckedAt === "string") { post.availabilityCheckedAt = r.availabilityCheckedAt; } if (typeof r.mediaCount === "number" && Number.isFinite(r.mediaCount)) { post.mediaCount = Math.max(0, Math.floor(r.mediaCount)); } const engagement = parseEngagement(r.engagement); if (engagement) post.engagement = engagement; const media = parseMediaList(r.media); if (media) post.media = media; const forum = parseForumInfo(r.forum); if (forum) post.forum = forum; return post; } function forumBaseOf(raw: unknown): string | undefined { const host = (raw as { host?: unknown } | null)?.host; return typeof host === "string" && host ? `https://${host}` : undefined; } function parseMediaList(raw: unknown): PostMedia[] | null { if (!Array.isArray(raw)) return null; const out: PostMedia[] = []; for (const item of raw) { if (!item || typeof item !== "object") continue; const m = item as Record; if (typeof m.url !== "string" || !m.url) continue; if (!(POST_MEDIA_KINDS as string[]).includes(m.kind as string)) continue; const media: PostMedia = { kind: m.kind as PostMediaKind, url: m.url }; if (typeof m.name === "string" && m.name) media.name = m.name; if (typeof m.provider === "string" && m.provider) media.provider = m.provider; out.push(media); } return out.length > 0 ? out : null; } const posInt = (v: unknown): number | undefined => typeof v === "number" && Number.isFinite(v) && v >= 1 ? Math.floor(v) : undefined; const nonBlank = (v: unknown): string | undefined => typeof v === "string" && v ? v : undefined; function parseForumInfo(raw: unknown): ForumPostInfo | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; const host = nonBlank(r.host); const threadId = nonBlank(r.threadId); if (!host || !threadId) return null; const out: ForumPostInfo = { host, threadId }; const threadTitle = nonBlank(r.threadTitle); if (threadTitle) out.threadTitle = threadTitle; const threadUrl = nonBlank(r.threadUrl); if (threadUrl) out.threadUrl = threadUrl; const page = posInt(r.page); if (page) out.page = page; const position = posInt(r.position); if (position) out.position = position; const authorId = nonBlank(r.authorId); if (authorId) out.authorId = authorId; const editedAt = nonBlank(r.editedAt); if (editedAt) out.editedAt = editedAt; if (Array.isArray(r.quotes)) { const quotes: NonNullable = []; for (const q of r.quotes) { if (!q || typeof q !== "object") continue; const qq = q as Record; const one: { postId?: string; author?: string; authorId?: string } = {}; const postId = nonBlank(qq.postId); if (postId) one.postId = postId; const author = nonBlank(qq.author); if (author) one.author = author; const authorId = nonBlank(qq.authorId); if (authorId) one.authorId = authorId; quotes.push(one); } if (quotes.length > 0) out.quotes = quotes; } return out; } function parsePostRef(raw: unknown): PostRef | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; if (typeof r.id !== "string" || !r.id) return null; if (!isPostPlatform(r.platform)) return null; const ref: PostRef = { platform: r.platform, id: r.id }; if (typeof r.url === "string") ref.url = r.url; if (typeof r.author === "string") ref.author = r.author; return ref; } function parseEngagement(raw: unknown): PostEngagement | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; const out: PostEngagement = {}; let any = false; for (const key of ["likes", "reposts", "replies", "quotes"] as const) { const v = r[key]; if (typeof v === "number" && Number.isFinite(v)) { out[key] = Math.max(0, Math.floor(v)); any = true; } } return any ? out : null; } // Newest-first ordering, the order both the page tree and the UI present. // createdAt is ISO-8601 so a plain string compare is a chronological compare; // the id tiebreak keeps the sort stable for same-instant posts. export function comparePostsNewestFirst(a: Post, b: Post): number { if (a.createdAt !== b.createdAt) return a.createdAt < b.createdAt ? 1 : -1; return a.id < b.id ? 1 : a.id > b.id ? -1 : 0; } // The CONVERSATION around a post, oldest first: the "more context" unit the // viewer's thread view and the MCP get_thread tool show. // // For X and Bluesky that is the reply thread (threadId). A forum thread is one // channel of possibly thousands of posts, so a forum post's conversation is // the quote graph instead: the posts it quotes (and theirs, to `depth`), and // the posts that quote it — read in thread order. export function postConversation( post: Post, candidates: ReadonlyArray, depth = 3, ): Post[] { if (post.platform !== "xenforo") { const threadId = post.threadId || post.id; const thread = candidates.filter((p) => (p.threadId || p.id) === threadId); if (!thread.some((p) => p.id === post.id)) thread.push(post); return thread.sort((a, b) => -comparePostsNewestFirst(a, b)); } const byId = new Map(); for (const p of candidates) byId.set(p.id, p); byId.set(post.id, post); const keep = new Map([[post.id, post]]); let frontier: Post[] = [post]; for (let d = 0; d < depth && frontier.length > 0; d++) { const next: Post[] = []; for (const p of frontier) { for (const q of p.forum?.quotes ?? []) { const hit = q.postId ? byId.get(q.postId) : undefined; if (hit && !keep.has(hit.id)) { keep.set(hit.id, hit); next.push(hit); } } } frontier = next; } for (const p of candidates) { if (p.forum?.quotes?.some((q) => q.postId === post.id)) keep.set(p.id, p); } return [...keep.values()].sort(compareForumOrder); } // Thread order for forum posts: position when both have one, else time. function compareForumOrder(a: Post, b: Post): number { const pa = a.forum?.position; const pb = b.forum?.position; if (pa && pb && pa !== pb) return pa - pb; return -comparePostsNewestFirst(a, b); } // Group a flat post list into threads keyed by threadId (falling back to the // post's own id for a root/standalone post). Used by the viewer's thread // context and the MCP get_thread tool. export function groupIntoThreads(posts: ReadonlyArray): Map { const threads = new Map(); for (const post of posts) { const key = post.threadId || post.id; const bucket = threads.get(key); if (bucket) bucket.push(post); else threads.set(key, [post]); } for (const bucket of threads.values()) { // Threads read oldest-first — the opposite of the feed ordering. bucket.sort((a, b) => -comparePostsNewestFirst(a, b)); } return threads; }