#!/bin/sh # ============================================================== # Strategy Lab — 部署脚本 v2(供 leader 审查/在隔离替身环境演练) # *** 本脚本不会由开发会话在生产执行 *** # 针对 leader 复审意见(2026-09-17)的修正: # R1 每个关键命令显式检查返回码,绝不依赖 set -e 在 && / if ! 上下文的不可靠性。 # R2 竞态窗口:先 systemctl stop 服务 → 重新硬检查 in-flight(服务停止后不再产生 # 新任务)→ 才 swap。停后发现新任务:立即恢复旧服务(发布物未被动过)并退出, # 任务状态原样保留。 # R3 restore 入口也取同一把锁;回滚后验证 ①服务 active ②/api/health JSON # status=="ok" ③恢复后的发布物 sha256 == release-manifest.txt 里的旧哈希。 # 备份在发布目录外、0700;单一 RELEASE_ID;同秒碰撞自动加后缀,绝不覆盖既有备份。 # # 受控替身(演练/审查可注入): # SYSTEMCTL_CMD/DOCKER_CMD/CURL_CMD/RSYNC_CMD/INSTALL_CMD 均可指向替身。 # *不含生产破坏性操作,禁止在无 STAGE_DIR 环境下执行(见 require_prod_paths)。 set -eu SYSTEMCTL_CMD=${SYSTEMCTL_CMD:-systemctl} DOCKER_CMD=${DOCKER_CMD:-docker} CURL_CMD=${CURL_CMD:-curl} INSTALL_CMD=${INSTALL_CMD:-install} SED_CMD=${SED_CMD:-sed} RSYNC_CMD=${RSYNC_CMD:-rsync} CP_CMD=${CP_CMD:-cp} MV_CMD=${MV_CMD:-mv} STATE_DIR=${STATE_DIR:-/home/somhairle/.local/share/strategy-lab-production} RELEASE_DIR=${RELEASE_DIR:-$STATE_DIR/release} BACKUP_ROOT=${BACKUP_ROOT:-$STATE_DIR/release-backups} REPO_ROOT=${REPO_ROOT:?REPO_ROOT must be set (absolute path to strategy-lab checkout)} SERVICE=${SERVICE:-strategy-lab-production} SERVER_BIN_SRC=${SERVER_BIN_SRC:-$REPO_ROOT/server/target/release/strategy-lab-server} DIST_SRC=${DIST_SRC:-$REPO_ROOT/frontend/dist} IMAGE_ID=${IMAGE_ID:?IMAGE_ID must be the immutable image ID, e.g. sha256:...} msg() { printf '[deploy] %s\n' "$1"; } note() { printf '[deploy] NOTE: %s\n' "$1" >&2; } die() { printf '[deploy] FATAL: %s\n' "$1" >&2; exit 1; } # ---------------- explicit-rc helpers (R1) ---------------- run_cp() { "$CP_CMD" -p "$1" "$2" || { note "cp -p $1 -> $2 failed"; return 1; }; } run_sed() { "$SED_CMD" -i "$@" || { note "sed env replacement failed"; return 1; }; } run_install(){ "$INSTALL_CMD" -m 0755 "$1" "$2" || { note "install $1 -> $2 failed"; return 1; }; } run_rsync() { "$RSYNC_CMD" "$@" || { note "rsync $* failed"; return 1; } } run_verchown_mode() { chmod "$1" "$2" || { note "chmod $1 $2 failed"; return 1; } } # manifest is written to a tmp file and atomically moved into place; a failed # mv (=> incomplete manifest) never looks like a finished backup fin_stage() { "$MV_CMD" "$1" "$2" || { note "finalize $2 failed"; return 1; } } db_scalar() { # $1 = db path, $2 = sql (single scalar). sqlite3 if present, else python3 stdlib. if command -v sqlite3 >/dev/null 2>&1; then sqlite3 "$1" "$2" else python3 - "$1" "$2" <<'PY' || die "db_scalar failed on $1" import sqlite3, sys c = sqlite3.connect(sys.argv[1]) row = c.execute(sys.argv[2]).fetchone() print(row[0] if row is not None else 0) PY fi } env_value() { grep -E "^$1=" "$STATE_DIR/runtime.env" | head -1 | cut -d= -f2- } service_active() { "$SYSTEMCTL_CMD" --user is-active "$SERVICE" >/dev/null 2>&1; } service_stop() { "$SYSTEMCTL_CMD" --user stop "$SERVICE" || return 1; } service_start() { "$SYSTEMCTL_CMD" --user start "$SERVICE" || { note "service start failed"; return 1; }; } inflight_datasets_sql() { printf "SELECT COUNT(*) FROM datasets WHERE status IN ('pending','running')" } inflight_runs_sql() { printf "SELECT COUNT(*) FROM runs WHERE status IN ('queued','running')" } # Returns: # 0 = no in-flight work # 1 = in-flight work present (or cannot be ruled out) # 2 = environment/DB error (missing DB_PATH, unreadable DB) — caller MUST # treat this as "unknown state", never as a clean pass. This function # NEVER exits the shell: after a service stop, a deep die() here would # bypass the restore branch and strand the old service in downtime. check_no_inflight() { DB_PATH=$(env_value DB_PATH) || { note "DB_PATH missing in runtime.env"; return 2; } DB_PATH=${DB_PATH:-} [ -n "$DB_PATH" ] || { note "DB_PATH empty in runtime.env"; return 2; } [ -f "$DB_PATH" ] || { note "DB file missing: $DB_PATH"; return 2; } if ! n=$(db_scalar "$DB_PATH" "$(inflight_datasets_sql)"); then note "read datasets failed" return 2 fi [ "$n" = "0" ] || { note "in-flight datasets: $n"; return 1; } if ! n=$(db_scalar "$DB_PATH" "$(inflight_runs_sql)"); then note "read runs failed" return 2 fi [ "$n" = "0" ] || { note "in-flight runs: $n"; return 1; } return 0 } precheck_live() { [ -n "$IMAGE_ID" ] || die "IMAGE_ID required" "$DOCKER_CMD" image inspect "$IMAGE_ID" >/dev/null 2>&1 || die "image not present: $IMAGE_ID" [ "$(env_value WORKER_IMAGE)" != "$IMAGE_ID" ] || die "runtime.env already pins $IMAGE_ID" [ -x "$SERVER_BIN_SRC" ] || die "missing server binary: $SERVER_BIN_SRC" [ -f "$DIST_SRC/index.html" ] || { die "missing frontend dist: $DIST_SRC"; return 1; } service_active || die "$SERVICE not active; resolve before deploy (start it or inspect)" check_no_inflight || die "in-flight work; deploy blocked" msg "precheck(live) OK" } precheck_stopped() { service_active && { note "service still active after stop"; return 1; } check_no_inflight || { note "new in-flight work discovered after stop"; return 2; } msg "precheck(stopped) OK" } release_unique_backup_dir() { rid_base=$(date -u +%Y%m%d-%H%M%S) rid="$rid_base" sfx=a while [ -e "$BACKUP_ROOT/$rid" ]; do rid="$rid_base-$sfx" sfx=$(python3 -c "import sys,random;print(chr(ord('a')+random.randrange(26)))" 2>/dev/null || printf '%s' "$sfx") done RELEASE_ID=$rid printf '%s' "$rid" } backup() { release_unique_backup_dir >/dev/null 2>&1 || die "release id generation failed" [ -n "$RELEASE_ID" ] || die "empty RELEASE_ID" BACKUP_DIR=$BACKUP_ROOT/$RELEASE_ID if [ -e "$BACKUP_DIR" ]; then die "backup would overwrite existing $BACKUP_DIR (refusing)" fi mkdir -p "$BACKUP_DIR" || die "mkdir backup failed" run_verchown_mode 700 "$BACKUP_ROOT" run_verchown_mode 700 "$BACKUP_DIR" run_cp "$STATE_DIR/runtime.env" "$BACKUP_DIR/runtime.env" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: runtime.env copy failed"; } run_verchown_mode 600 "$BACKUP_DIR/runtime.env" run_rsync -a "$RELEASE_DIR/" "$BACKUP_DIR/release/" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: release copy failed"; } { printf 'release_id=%s\ncreated_at=%s\n' "$RELEASE_ID" "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" printf 'image_id_to_use=%s\nprevious_worker_image=%s\nprevious_server=%s\nprevious_frontend=%s\n' \ "$IMAGE_ID" "$(env_value WORKER_IMAGE)" \ "$RELEASE_DIR/strategy-lab-server" "$RELEASE_DIR/frontend" sha256sum "$RELEASE_DIR/strategy-lab-server" 2>/dev/null | awk '{print "previous_server_sha256=" $1}' || printf 'previous_server_sha256=unknown\n' } > "$BACKUP_DIR/release-manifest.txt.tmp" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest write failed"; } # completeness check BEFORE the final marker: restore refuses backups # without the 'complete' marker, so never mark anything half-copied. [ -f "$BACKUP_DIR/runtime.env" ] && [ -f "$BACKUP_DIR/release/strategy-lab-server" ] \ || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup incomplete; refusing to mark"; } grep -q '^previous_server_sha256=' "$BACKUP_DIR/release-manifest.txt.tmp" \ || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup manifest missing previous_server_sha256"; } fin_stage "$BACKUP_DIR/release-manifest.txt.tmp" "$BACKUP_DIR/release-manifest.txt" \ || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest finalize failed"; } run_verchown_mode 600 "$BACKUP_DIR/release-manifest.txt" : > "$BACKUP_DIR/complete" || die "backup marker failed" run_verchown_mode 600 "$BACKUP_DIR/complete" msg "backup $RELEASE_ID -> $BACKUP_DIR (complete)" } set_worker_image_id() { rc=0 run_sed "s|^WORKER_IMAGE=.*|WORKER_IMAGE=${IMAGE_ID}|" "$STATE_DIR/runtime.env" || rc=$? [ "$rc" -eq 0 ] || return "$rc" run_verchown_mode 600 "$STATE_DIR/runtime.env" || return 1 [ "$(env_value WORKER_IMAGE)" = "$IMAGE_ID" ] || { note "runtime.env replacement verification failed" return 1 } return 0 } swap_release() { run_install "$SERVER_BIN_SRC" "$RELEASE_DIR/strategy-lab-server" || return 1 # --checksum: dist contents must win even when size+mtime collide (drill # regression: same-size files with identical mtime were silently skipped) run_rsync -a --delete --checksum "$DIST_SRC/" "$RELEASE_DIR/frontend/" || return 1 return 0 } health_ok() { BIND=$(env_value BIND) || return 1 body=$("$CURL_CMD" -fsS -m 5 "http://$BIND/api/health" 2>/dev/null) || return 1 printf '%s' "$body" | grep -q '"status":"ok"' || return 1 return 0 } wait_health() { i=1 while [ $i -le 20 ]; do if health_ok; then msg "health OK"; return 0; fi sleep 1 i=$((i + 1)) done return 1 } restore_release_id() { rid=${1:?usage: deploy.sh restore } bd=$BACKUP_ROOT/$rid DEPLOY_PHASE=restore; export DEPLOY_PHASE [ -d "$bd" ] || die "no backup for release id: $rid" [ -f "$bd/complete" ] || die "backup incomplete (no complete marker): $bd" [ -f "$bd/runtime.env" ] || die "backup incomplete: runtime.env" [ -f "$bd/release/strategy-lab-server" ] || die "backup incomplete: release" # R3: restore takes the same deploy.lock unless the caller already holds it if [ "${DEPLOY_LOCK_HELD:-0}" != "1" ]; then exec 9>"$STATE_DIR/deploy.lock" flock -n 9 || die "another deploy/rollback holds deploy.lock" fi # --checksum: dist contents must win even when size+mtime collide # touch on-disk artifacts only while the (old/new) binary is not running: # stop first, then replace files, then start again and verify health. # If stop cannot be CONFIRMED (command failed or service still active), # refuse to write ANY file: the running binary may still be using them. # The restore is non-destructive here — the caller can retry after # resolving the stop problem, and all existing files remain recoverable. if ! "$SYSTEMCTL_CMD" --user stop "$SERVICE" 2>/dev/null; then # stop reported failure: only safe to continue if service is verifiably down if service_active; then die "stop failed and $SERVICE still ACTIVE — refusing to restore files (nothing was written; backup $rid intact and recoverable)" fi note "stop reported failure but service verified inactive; continuing" fi service_active && die "could not confirm $SERVICE stopped — refusing to restore files (nothing was written; backup $rid intact and recoverable)" run_cp "$bd/runtime.env" "$STATE_DIR/runtime.env" run_verchown_mode 600 "$STATE_DIR/runtime.env" run_rsync -a --delete --checksum "$bd/release/" "$RELEASE_DIR/" || die "restore release failed" service_start || die "service start after restore failed" sleep 2 service_active || die "service not active after restore" wait_health || die "health failed after restore" verify_restored "$bd" msg "restored $rid" } # verify restored artifacts hash vs manifest verify_restored() { bd=$1 w=$(grep -E '^previous_worker_image=' "$bd/release-manifest.txt" | sed 's/^previous_worker_image=//') || die "manifest corrupt" [ "$(env_value WORKER_IMAGE)" = "$w" ] || die "restored env mismatches manifest" h=$(grep -E '^previous_server_sha256=' "$bd/release-manifest.txt" | sed 's/^previous_server_sha256=//') now=$(sha256sum "$RELEASE_DIR/strategy-lab-server" | awk '{print $1}') [ "$h" = "$now" ] || die "restored server hash mismatch: $now != $h" msg "restored artifacts verified (env + sha256)" } # ---------------- main deploy flow ---------------- main_deploy() { exec 9>"$STATE_DIR/deploy.lock" flock -n 9 || die "another deploy/rollback is running (deploy.lock held)" msg "phase 1/6 precheck (live service)" precheck_live # Backup BEFORE stopping: a backup failure must never turn into downtime — # the service keeps running on untouched artifacts and the deploy simply # refuses. msg "phase 2/6 backup (env + release artifacts, service still up)" DEPLOY_PHASE=backup; export DEPLOY_PHASE backup msg "phase 3/6 stop service (block NEW work)" service_stop || die "service stop failed" # R1: capture the REAL return code of the stopped-state checks — not an # &&-chain / if ! — so a `2` (= new work discovered) is distinguishable # from plain failure and the caller can restore the old service untouched. rc=0 precheck_stopped || rc=$? # A DB/environment error (rc=2) after stop is treated the same as the race # branch: we cannot prove the system is quiet, so restore the old service # untouched rather than swapping files in an unverifiable state. if [ "$rc" -ne 0 ]; then if [ "$rc" = "2" ]; then "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED" note "restored old service WITHOUT replacement (post-stop check errored: DB/env unreadable)" note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID" exit 1 fi # rc=1: new work arrived between checks; the backup above is discarded (it # recorded pre-inflight state and nothing was replaced) "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED" note "restored old service WITHOUT replacement (new tasks preserved)" note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID (release kept; runtime.env already restored bytes)" exit 1 fi msg "phase 4/6 swap (env WORKER_IMAGE + binary + dist)" # R1: each step rc-checked independently (also inside an && chain) DEPLOY_PHASE=swap; export DEPLOY_PHASE if ! set_worker_image_id; then note "env update failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1 fi if ! swap_release; then note "release swap failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1 fi msg "phase 5/6 start service + health" if ! "$SYSTEMCTL_CMD" --user start "$SERVICE"; then note "service start failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1 fi sleep 2 if ! wait_health; then note "health failed; rolling back to $RELEASE_ID" DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID" exit 1 fi msg "phase 6/6 deploy $RELEASE_ID complete (image $IMAGE_ID)" msg "rollback id: $RELEASE_ID (rollback.sh $RELEASE_ID)" } case "${1:-deploy}" in deploy) main_deploy ;; restore) shift; restore_release_id "$@" ;; precheck) precheck_live ;; *) die "usage: deploy.sh [deploy|restore |precheck]" ;; esac