pediatric-ai-scribe-v3/docker-entrypoint.sh
Daniel 8d0dc968b3 feat: a deploy you can repeat, and prove afterwards
Reproducibility means two things here: the same commit builds the same image,
and the running container can be asked which commit it is.

  - Base images are pinned by digest, not by tag. A tag moves; two builds of one
    commit could otherwise differ. These are manifest-list digests, so buildx
    still picks the right architecture.

  - scripts/build-image.sh also writes ped-ai-local:<revision>, an immutable
    name a deploy can refer to instead of chasing :latest. Its summary goes to
    stderr so stdout stays the Compose invocation.

  - Compose takes the image from PED_AI_IMAGE, so a deploy runs a specific
    revision-tagged image while a local build still uses the local tag.

  - scripts/deploy.sh pins that image in the file Compose interpolates from,
    waits for health, then asks /api/build which revision is actually serving
    and rolls back to the previous image if it does not match. Healthy is not
    the same as running what you asked for. The rollback path was exercised.

  - The entrypoint applies migrations before the app starts, so code and schema
    arrive together. node-pg-migrate takes an advisory lock; losing it is not an
    error, it waits and looks again, so a rolling restart does not fail. A real
    migration failure stops the container rather than serving on a schema that
    does not match the build. RUN_MIGRATIONS=false opts out.

  - The Forgejo workflow builds through that same script, tags by full revision,
    and has an opt-in deploy job. It refuses to run if the deploy directory has
    uncommitted work rather than resetting over it.

The running image was labelled revision=unknown, and /api/build said "unknown",
because `docker compose up --build` never passes GIT_REVISION. That is exactly
the hole this closes.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Dv6sqaY6Vq3ChZHMem3cnU
2026-09-11 00:41:11 +02:00

128 lines
5.4 KiB
Bash
Executable file

#!/bin/sh
# Container entrypoint. Optionally fetches secrets from OpenBao before
# starting the app. Backwards compatible: if OPENBAO_ADDR is unset (e.g. e2e
# container, local dev with a populated .env), the vault step is skipped
# and the process starts with whatever's already in the environment.
#
# When OPENBAO_ADDR is set, OPENBAO_ROLE_ID + OPENBAO_SECRET_ID are required.
# The entrypoint logs in via AppRole, fetches kv/ped-ai/prod, exports each
# key as an env var, and then unsets the auth material before execing the
# real command so the Node process doesn't carry them.
set -eu
if [ -n "${OPENBAO_ADDR:-}" ]; then
if [ -z "${OPENBAO_ROLE_ID:-}" ] || [ -z "${OPENBAO_SECRET_ID:-}" ]; then
echo "[entrypoint] FATAL: OPENBAO_ADDR is set but OPENBAO_ROLE_ID or OPENBAO_SECRET_ID is missing." >&2
exit 1
fi
export BAO_ADDR="${OPENBAO_ADDR}"
echo "[entrypoint] authenticating to OpenBao at ${OPENBAO_ADDR} via AppRole..."
BAO_TOKEN="$(bao write -field=token auth/approle/login \
role_id="${OPENBAO_ROLE_ID}" \
secret_id="${OPENBAO_SECRET_ID}" 2>&1)"
if [ -z "${BAO_TOKEN}" ] || printf '%s' "${BAO_TOKEN}" | grep -qi error; then
echo "[entrypoint] FATAL: AppRole authentication failed:" >&2
echo "${BAO_TOKEN}" >&2
exit 1
fi
export BAO_TOKEN
SECRET_PATH="${OPENBAO_KV_PATH:-kv/ped-ai/prod}"
echo "[entrypoint] fetching secrets from ${SECRET_PATH}..."
SECRET_JSON="$(bao kv get -format=json "${SECRET_PATH}" 2>/dev/null | jq -c '.data.data' 2>/dev/null || true)"
if [ -z "${SECRET_JSON}" ] || [ "${SECRET_JSON}" = "null" ]; then
echo "[entrypoint] FATAL: no secrets returned from ${SECRET_PATH}." >&2
exit 1
fi
# Export each key/value as a shell-safe env var — but ONLY if the key
# isn't already set by docker (env_file / environment: block). This
# lets a docker-compose override win over the OpenBao value, which is
# needed for e2e (TURNSTILE_SECRET_KEY="" / SMTP_HOST="") and any
# environment-specific override.
#
# Pattern: write jq output to a temp file, then while-read in the main
# shell so exports persist (pipes into while run in a subshell and lose
# them). Pre-snapshot env keys and skip those already defined.
_PRESET_KEYS_FILE=$(mktemp)
env | cut -d= -f1 | sort -u > "$_PRESET_KEYS_FILE"
_SECRET_ASSIGNS=$(mktemp)
printf '%s' "${SECRET_JSON}" | jq -r 'to_entries[] | "\(.key)\t\(.value | @sh)"' > "$_SECRET_ASSIGNS"
_APPLIED_COUNT=0
_SKIPPED_COUNT=0
while IFS="$(printf '\t')" read -r _K _VAL_QUOTED; do
if [ -z "$_K" ]; then continue; fi
if grep -qxF "$_K" "$_PRESET_KEYS_FILE"; then
_SKIPPED_COUNT=$((_SKIPPED_COUNT + 1))
else
eval "export $_K=$_VAL_QUOTED"
_APPLIED_COUNT=$((_APPLIED_COUNT + 1))
fi
done < "$_SECRET_ASSIGNS"
rm -f "$_PRESET_KEYS_FILE" "$_SECRET_ASSIGNS"
echo "[entrypoint] applied ${_APPLIED_COUNT} secrets; ${_SKIPPED_COUNT} already set by docker (kept override)"
# Bootstrap credentials are no longer needed in the Node process env.
unset OPENBAO_ROLE_ID OPENBAO_SECRET_ID BAO_TOKEN
SECRET_COUNT="$(printf '%s' "${SECRET_JSON}" | jq -r 'keys | length')"
echo "[entrypoint] ✅ loaded ${SECRET_COUNT} secrets from OpenBao"
else
echo "[entrypoint] OPENBAO_ADDR not set — using existing environment (legacy .env path)"
fi
# ── Schema migrations ────────────────────────────────────────────────
# The code and the schema it needs ship inside the same image, so they have to
# arrive together. Applying them by hand meant a deploy could put new code in
# front of an old schema and only find out at the first request.
#
# node-pg-migrate takes a Postgres advisory lock, so two containers starting at
# once cannot both apply. The one that loses the race is not an error — it
# waits for the winner and looks again — so a rolling restart does not fail.
#
# Set RUN_MIGRATIONS=false to start without touching the schema (a read-only
# replica, or recovering from a bad migration by hand).
if [ "${RUN_MIGRATIONS:-true}" = "true" ]; then
if [ -z "${DATABASE_URL:-}" ]; then
echo "[entrypoint] FATAL: RUN_MIGRATIONS is on but DATABASE_URL is not set." >&2
exit 1
fi
_MIGRATE_ATTEMPT=1
_MIGRATE_MAX=${MIGRATION_ATTEMPTS:-10}
while : ; do
echo "[entrypoint] applying migrations (attempt ${_MIGRATE_ATTEMPT}/${_MIGRATE_MAX})..."
_MIGRATE_OUT="$(node_modules/.bin/node-pg-migrate up 2>&1)" && {
printf '%s\n' "${_MIGRATE_OUT}"
echo "[entrypoint] ✅ schema is up to date"
break
}
printf '%s\n' "${_MIGRATE_OUT}" >&2
# Losing the advisory lock, or racing a database that is still opening its
# listening socket, are both worth another look. Anything else is a real
# migration failure and must stop the deploy rather than serve on a schema
# that does not match the code.
if printf '%s' "${_MIGRATE_OUT}" | grep -qiE "advisory lock|ECONNREFUSED|starting up|Connection terminated"; then
if [ "${_MIGRATE_ATTEMPT}" -ge "${_MIGRATE_MAX}" ]; then
echo "[entrypoint] FATAL: could not apply migrations after ${_MIGRATE_MAX} attempts." >&2
exit 1
fi
_MIGRATE_ATTEMPT=$((_MIGRATE_ATTEMPT + 1))
sleep 3
continue
fi
echo "[entrypoint] FATAL: migration failed. Refusing to start on a schema that does not match this build." >&2
exit 1
done
else
echo "[entrypoint] RUN_MIGRATIONS=false — starting without checking the schema"
fi
exec "$@"