#!/usr/bin/env bash # First run, every run: prepare the data volumes, then hand the container over to # one of the four apps. # # editor the admin app (next start, :3001) # site the published archive (serve, :3000) # homepage the project's own site (serve, :3031) — the local deploy in the # builds volume when there is one, else the image's baked build # umtool the clip/report bench (next start, :3050) # shell drop into bash — for `docker compose run --rm editor shell` # # Nothing here publishes a port. Caddy does that; see docker/Caddyfile. # # The app is exec'd, deliberately: armShutdownCancel() in editor/instrumentation.ts # reaps in-flight yt-dlp/whisper children on SIGTERM, and it can only see the # signal if the app is PID 1's own process rather than a grandchild of a wrapper. set -euo pipefail APP="${1:-editor}" TRANSCRIPTS_DIR="${TRANSCRIPTS_DIR:-/data/transcripts}" SETTINGS_FILE="${SETTINGS_FILE:-/data/config/settings.json}" MODELS_DIR="${ARCHILYZER_MODELS_DIR:-/data/models}" BUILDS_DIR="${ARCHILYZER_BUILDS_DIR:-/data/builds}" SITE_OUT="${ARCHILYZER_SITE_OUT:-${BUILDS_DIR}/site}" HOMEPAGE_OUT="${ARCHILYZER_HOMEPAGE_OUT:-${BUILDS_DIR}/homepage}" WHISPER_MODEL="${WHISPER_MODEL:-${MODELS_DIR}/ggml-base.en.bin}" # Which transcription backend THIS IMAGE was built with. Set by the Dockerfile # per target, not by the operator: `runtime`/`runtime-cuda` carry whisper.cpp, # `runtime-vulkan` carries parakeet.cpp on the GPU. It decides which worker the # seed below writes and which model family gets fetched. TRANSCRIBER="${ARCHILYZER_TRANSCRIBER:-whisper-cpp}" export TRANSCRIPTS_DIR SETTINGS_FILE WHISPER_MODEL log() { printf '[entrypoint] %s\n' "$*"; } die() { printf '[entrypoint] %s\n' "$*" >&2; exit 1; } # --------------------------------------------------------------------------- # 1. The data volumes. # --------------------------------------------------------------------------- mkdir -p \ "${TRANSCRIPTS_DIR}/channels" \ "${TRANSCRIPTS_DIR}/sites" \ "$(dirname "${SETTINGS_FILE}")" \ "${MODELS_DIR}" \ "${BUILDS_DIR}" \ "${SITE_OUT}" \ "${HOMEPAGE_OUT}" # The operator's private config dir (compose sets it inside the config volume), # made so `docker compose cp` has somewhere to put the source mirror's rules. # Only made, never filled: what goes in it is the operator's. if [ -n "${ARCHILYZER_CONFIG_DIR:-}" ] && [ ! -d "${ARCHILYZER_CONFIG_DIR}" ]; then mkdir -p "${ARCHILYZER_CONFIG_DIR}" chmod 700 "${ARCHILYZER_CONFIG_DIR}" || true fi # The host's git common dir, mounted read-only by docker-compose.source.yml for # the homepage's source mirror (`archilyzer source publish`). It belongs to the # host's user and the container runs as root, so git would refuse it as # "dubious ownership" without this. Added once — `--add` on every boot would # pile up duplicates in a container that is restarted rather than recreated. SOURCE_REPO_DIR="${ARCHILYZER_SOURCE_REPO:-/data/source.git}" if ! git config --global --get-all safe.directory 2>/dev/null | grep -qxF "${SOURCE_REPO_DIR}"; then git config --global --add safe.directory "${SOURCE_REPO_DIR}" || printf '[entrypoint] WARNING: could not mark %s safe for git\n' "${SOURCE_REPO_DIR}" fi # --------------------------------------------------------------------------- # 2. Seed settings.json — with a worker. # # With no `workers` key (or no file), getSettings() synthesizes # `parallelTranscriptions` (default 2) enabled workers of the default app, # whisper.cpp — two CPU whisper slots, and never parakeet in the Vulkan image. # (defaults() alone has `workers: []`; zero workers — auto-transcribe silently # doing nothing — only happens for a file that says `"workers": []`.) So the # seed carries exactly one local worker for the image's engine. # # Everything else is left out on purpose. getSettings() merges a partial file # over defaults() (every key but `workers`, above), so a short seed is a # FEATURE: keys we don't write here keep # tracking the app's own defaults as those move, instead of being frozen at # whatever they were the day this image was built. # # The worker's config block is empty for the same reason — an empty `bin`/`model` # falls back to WHISPER_BIN / WHISPER_MODEL, which the image already sets. One # source of truth for where the binary and the model are. # --------------------------------------------------------------------------- if [ ! -f "${SETTINGS_FILE}" ]; then case "${TRANSCRIBER}" in parakeet) # The GPU image. `device` is left unset so parakeet.cpp picks the best # backend it can see — Vulkan when /dev/dri is passed through, CPU when # it is not, rather than failing outright. Name a device explicitly # (Vulkan0, cpu, …) on the Workers page when you want to pin it. cat >"${SETTINGS_FILE}" <<'JSON' { "adminTitle": "Archilyzer", "transcriptionApp": "parakeet", "workers": [ { "id": "parakeet", "name": "parakeet (GPU)", "kind": "local", "enabled": true, "priority": 0, "appId": "parakeet", "config": {} } ], "parallelTranscriptions": 1 } JSON log "seeded ${SETTINGS_FILE} with one local parakeet worker" ;; *) cat >"${SETTINGS_FILE}" <<'JSON' { "adminTitle": "Archilyzer", "transcriptionApp": "whisper-cpp", "workers": [ { "id": "whisper-cpp", "name": "whisper.cpp", "kind": "local", "enabled": true, "priority": 0, "appId": "whisper-cpp", "config": {} } ], "parallelTranscriptions": 1 } JSON log "seeded ${SETTINGS_FILE} with one local whisper.cpp worker" ;; esac fi # --------------------------------------------------------------------------- # 3. The speech model. # # Not baked into the image: 142 MB for whisper base.en, 3 GB for large-v3, and # the choice is the operator's. Fetched once into the models volume. Set # ARCHILYZER_FETCH_MODEL= (empty) or =none to skip entirely — e.g. when you mount # your own models directory. # # Two families, because the two images have different engines: # whisper-cpp ggml-.bin from ggerganov/whisper.cpp (base.en, small.en, …) # parakeet .gguf from mudler/parakeet-cpp-gguf (tdt_ctc-110m-q8_0, …) # --------------------------------------------------------------------------- fetch_model() { local name target url if [ "${TRANSCRIBER}" = "parakeet" ]; then name="${ARCHILYZER_FETCH_MODEL-tdt_ctc-110m-q8_0}" target="${MODELS_DIR}/${name}.gguf" url="https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/${name}.gguf" else name="${ARCHILYZER_FETCH_MODEL-base.en}" target="${MODELS_DIR}/ggml-${name}.bin" url="https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-${name}.bin" fi [ -z "${name}" ] && return 0 [ "${name}" = "none" ] && return 0 if [ -s "${target}" ]; then log "speech model present: ${target}" return 0 fi log "fetching ${TRANSCRIBER} model '${name}' -> ${target}" log " (this is a one-time download; set ARCHILYZER_FETCH_MODEL=none to skip)" # Download to a temp name and move: a container killed mid-download must not # leave a truncated file that looks like a model forever after. if curl -fL --retry 3 --retry-delay 2 -o "${target}.partial" "${url}"; then mv "${target}.partial" "${target}" log "model ready: ${target} ($(du -h "${target}" | cut -f1))" else rm -f "${target}.partial" # Not fatal. The editor starts fine without a model; transcription is # what fails, and it fails with a message naming the missing file. log "WARNING: model download failed — transcription will not work until" log " ${target} exists. Retry with: docker compose restart editor" fi } # --------------------------------------------------------------------------- # 4. yt-dlp: which one, and its self-update. # # YTDLP_BIN is the yt-dlp every fetch runs; ARCHILYZER_IMAGE_YTDLP is the one # this image ships. They differ when the operator substituted their own (a # zipapp, or /usr/local/bin/yt-dlp-from-source over a mounted source tree — # RUNNING_IN_DOCKER.md, "Substituting yt-dlp"): an OVERRIDE. # # An image pins yt-dlp at build time, and a stale yt-dlp is the most common # reason downloads suddenly start failing — YouTube changes, yt-dlp ships a fix # within days, and an image from last month has none of them. The self-update is # opt-in because it is a network call on every boot, and it never touches an # override: a patched build is updated where it is built, and `-U` on one either # fails or replaces the patch with a release. # --------------------------------------------------------------------------- YTDLP="${YTDLP_BIN:-yt-dlp}" IMAGE_YTDLP="${ARCHILYZER_IMAGE_YTDLP:-}" # "image" or "override", by where each one actually lands. ytdlp_origin() { local mine theirs if [ -z "${IMAGE_YTDLP}" ]; then echo image return 0 fi mine="$(readlink -f "$(command -v "${YTDLP}" 2>/dev/null || echo "${YTDLP}")" 2>/dev/null || echo "${YTDLP}")" theirs="$(readlink -f "${IMAGE_YTDLP}" 2>/dev/null || echo "${IMAGE_YTDLP}")" if [ "${mine}" = "${theirs}" ]; then echo image; else echo override; fi } update_ytdlp() { case "${YTDLP_AUTO_UPDATE:-0}" in 1 | true | yes | on) ;; *) return 0 ;; esac if [ "$(ytdlp_origin)" = "override" ]; then log "WARNING: YTDLP_AUTO_UPDATE is set, but YTDLP_BIN (${YTDLP}) is not the image's yt-dlp" log " (${IMAGE_YTDLP}) — an override is not self-updated; update it where it is built" return 0 fi log "yt-dlp -U (YTDLP_AUTO_UPDATE is set)" "${YTDLP}" -U || log "WARNING: yt-dlp self-update failed — continuing with $("${YTDLP}" --version 2>/dev/null || echo unknown)" } # One line on every boot, like the vulkan: line — "which yt-dlp is this, and # does it run?" answered before the first download has to. A yt-dlp that is # missing or does not run is a WARNING, never a refusal: the editor is still # useful (reading, building, publishing) without one. ytdlp_line() { local origin version err origin="$(ytdlp_origin)" if ! command -v "${YTDLP}" >/dev/null 2>&1; then log "yt-dlp: MISSING ${YTDLP} (${origin}) — every download will fail until it exists" return 0 fi if version="$(timeout 30 "${YTDLP}" --version 2>/dev/null | head -n 1)" && [ -n "${version}" ]; then log "yt-dlp: ${YTDLP} ${version} (${origin})" else err="$(timeout 30 "${YTDLP}" --version 2>&1 >/dev/null | head -n 1 || true)" log "yt-dlp: MISSING — ${YTDLP} (${origin}) does not run: ${err:-no version printed}" log " every download will fail until it does; see RUNNING_IN_DOCKER.md" fi } # --------------------------------------------------------------------------- # 5. The exposure rail. Shared with the caddy container — one implementation. # --------------------------------------------------------------------------- /repo/docker/guard-exposure.sh # --------------------------------------------------------------------------- # 6. Hand over. # --------------------------------------------------------------------------- next_bin() { local dir="$1" [ -x "${dir}/node_modules/.bin/next" ] || die "next not found in ${dir} — was the image built?" printf '%s' "${dir}/node_modules/.bin/next" } serve_bin() { local dir="$1" [ -x "${dir}/node_modules/.bin/serve" ] || die "serve not found in ${dir} — was the image built?" printf '%s' "${dir}/node_modules/.bin/serve" } case "${APP}" in editor) fetch_model update_ytdlp ytdlp_line log "corpus: ${TRANSCRIPTS_DIR}" log "settings: ${SETTINGS_FILE}" if [ "${TRANSCRIBER}" = "parakeet" ]; then log "engine: parakeet ${PARAKEET_CLI:-parakeet-cli} (model ${PARAKEET_MODEL:-unset})" # One line that answers "is the GPU actually visible in here?" before a # transcription has to answer it the slow way. if [ -e /dev/dri ] && command -v vulkaninfo >/dev/null 2>&1; then log "vulkan: $(vulkaninfo --summary 2>/dev/null | grep -m2 -E 'deviceName' | sed 's/^[[:space:]]*//' | tr '\n' ' ' || echo 'no device reported')" else log "vulkan: NO /dev/dri IN THIS CONTAINER — parakeet will run on CPU." log " Pass the GPU through: see docker-compose.vulkan.yml." fi else log "engine: whisper ${WHISPER_BIN:-whisper-cli} (model ${WHISPER_MODEL})" fi if [ "${ARCHILYZER_IDLE_BOOT:-}" = "1" ]; then log "idle boot: schedulers, auto-queue runners and sweeps stay STOPPED" fi cd /repo/editor # Sixteen threads for Node's filesystem pool instead of four. A call on a # drive that has stalled holds its thread until the drive answers; with four, # four such calls stop the editor answering at all. More threads buy time for # calls already in flight — they isolate nothing (the storage health probe # and its gate are what keep new calls off a stalled drive). export UV_THREADPOOL_SIZE="${UV_THREADPOOL_SIZE:-16}" exec "$(next_bin /repo/editor)" start --port "${EDITOR_PORT:-3001}" ;; site) # The published archive. Built at RUN time, not baked: it is a static render # of a corpus, and there is no corpus in the image. docker/publish-site.sh # fills this directory; until it has, serve a page that says so rather than # a 404 nobody can interpret. if [ -z "$(ls -A "${SITE_OUT}" 2>/dev/null)" ]; then log "no site published yet at ${SITE_OUT} — serving a placeholder" cat >"${SITE_OUT}/index.html" <<'HTML' No site published yet

No site published yet

This is the static archive server. It has nothing to serve because no site has been built into its volume yet.

Add a channel in the editor, sync it, then publish:

docker compose exec editor /repo/docker/publish-site.sh <site-id>

See RUNNING_IN_DOCKER.md.

HTML fi log "serving ${SITE_OUT} on :3000" # -c: the project's OWN serve config — `access-control-allow-origin: *` and a # cache lifetime on the JSON, which is what makes /corpus.json readable by the # MCP server and federatable by a hub. A published archive gets that from the # _headers file compose-site.ts emits, and compose-site.ts says in as many # words that `serve` ignores _headers and reads serve.json instead. It looks # for one in the SERVED directory, which is a volume holding build output, so # the path has to be given. (`serve --cors` looks like the answer and does not # set the header — measured.) # --no-port-switching: if :3000 were taken, serve would silently move to # another port and Caddy would 502 at a healthy-looking container. exec "$(serve_bin /repo/export)" "${SITE_OUT}" -l 3000 \ -c /repo/export/serve.json --no-clipboard --no-port-switching ;; homepage) # Corpus-independent, so a build IS baked into the image — but that build # has no /source mirror and none of the corpus's numbers (there is no corpus # and no .git at image-build time). `archilyzer publish homepage --deploy # --to local` writes a real one into the builds volume; once that directory # is non-empty it is what is served. Chosen at boot: restart this service # after the FIRST local deploy (later ones are served as they land — serve # reads from disk per request). HOMEPAGE_DIR=/repo/homepage/out if [ -n "$(ls -A "${HOMEPAGE_OUT}" 2>/dev/null)" ]; then HOMEPAGE_DIR="${HOMEPAGE_OUT}" log "serving the locally deployed homepage ${HOMEPAGE_DIR} on :3031" else log "nothing deployed at ${HOMEPAGE_OUT} — serving the image's baked homepage ${HOMEPAGE_DIR} on :3031" fi exec "$(serve_bin /repo/homepage)" "${HOMEPAGE_DIR}" -l 3031 \ -c /repo/homepage/serve.json --no-clipboard --no-port-switching ;; umtool) cd /repo/umtool exec "$(next_bin /repo/umtool)" start --port "${UMTOOL_PORT:-3050}" ;; shell) exec bash ;; *) # Anything else is run verbatim, so `docker compose run --rm editor yt-dlp --version` # works without a special case per tool. exec "$@" ;; esac