diff options
Diffstat (limited to 'artifacts/etf-recovery-candidate/ops/deploy.sh')
| -rwxr-xr-x | artifacts/etf-recovery-candidate/ops/deploy.sh | 329 |
1 files changed, 329 insertions, 0 deletions
diff --git a/artifacts/etf-recovery-candidate/ops/deploy.sh b/artifacts/etf-recovery-candidate/ops/deploy.sh new file mode 100755 index 0000000..822c90e --- /dev/null +++ b/artifacts/etf-recovery-candidate/ops/deploy.sh @@ -0,0 +1,329 @@ +#!/bin/sh +# ============================================================== +# Strategy Lab — 部署脚本 v2(供 leader 审查/在隔离替身环境演练) +# *** 本脚本不会由开发会话在生产执行 *** +# 针对 leader 复审意见(2026-09-17)的修正: +# R1 每个关键命令显式检查返回码,绝不依赖 set -e 在 && / if ! 上下文的不可靠性。 +# R2 竞态窗口:先 systemctl stop 服务 → 重新硬检查 in-flight(服务停止后不再产生 +# 新任务)→ 才 swap。停后发现新任务:立即恢复旧服务(发布物未被动过)并退出, +# 任务状态原样保留。 +# R3 restore 入口也取同一把锁;回滚后验证 ①服务 active ②/api/health JSON +# status=="ok" ③恢复后的发布物 sha256 == release-manifest.txt 里的旧哈希。 +# 备份在发布目录外、0700;单一 RELEASE_ID;同秒碰撞自动加后缀,绝不覆盖既有备份。 +# +# 受控替身(演练/审查可注入): +# SYSTEMCTL_CMD/DOCKER_CMD/CURL_CMD/RSYNC_CMD/INSTALL_CMD 均可指向替身。 +# *不含生产破坏性操作,禁止在无 STAGE_DIR 环境下执行(见 require_prod_paths)。 +set -eu + +SYSTEMCTL_CMD=${SYSTEMCTL_CMD:-systemctl} +DOCKER_CMD=${DOCKER_CMD:-docker} +CURL_CMD=${CURL_CMD:-curl} +INSTALL_CMD=${INSTALL_CMD:-install} +SED_CMD=${SED_CMD:-sed} +RSYNC_CMD=${RSYNC_CMD:-rsync} +CP_CMD=${CP_CMD:-cp} +MV_CMD=${MV_CMD:-mv} + +STATE_DIR=${STATE_DIR:-/home/somhairle/.local/share/strategy-lab-production} +RELEASE_DIR=${RELEASE_DIR:-$STATE_DIR/release} +BACKUP_ROOT=${BACKUP_ROOT:-$STATE_DIR/release-backups} +REPO_ROOT=${REPO_ROOT:?REPO_ROOT must be set (absolute path to strategy-lab checkout)} +SERVICE=${SERVICE:-strategy-lab-production} +SERVER_BIN_SRC=${SERVER_BIN_SRC:-$REPO_ROOT/server/target/release/strategy-lab-server} +DIST_SRC=${DIST_SRC:-$REPO_ROOT/frontend/dist} +IMAGE_ID=${IMAGE_ID:?IMAGE_ID must be the immutable image ID, e.g. sha256:...} + +msg() { printf '[deploy] %s\n' "$1"; } +note() { printf '[deploy] NOTE: %s\n' "$1" >&2; } +die() { printf '[deploy] FATAL: %s\n' "$1" >&2; exit 1; } + +# ---------------- explicit-rc helpers (R1) ---------------- +run_cp() { "$CP_CMD" -p "$1" "$2" || { note "cp -p $1 -> $2 failed"; return 1; }; } +run_sed() { "$SED_CMD" -i "$@" || { note "sed env replacement failed"; return 1; }; } +run_install(){ "$INSTALL_CMD" -m 0755 "$1" "$2" || { note "install $1 -> $2 failed"; return 1; }; } +run_rsync() { "$RSYNC_CMD" "$@" || { note "rsync $* failed"; return 1; } } +run_verchown_mode() { + chmod "$1" "$2" || { note "chmod $1 $2 failed"; return 1; } +} +# manifest is written to a tmp file and atomically moved into place; a failed +# mv (=> incomplete manifest) never looks like a finished backup +fin_stage() { + "$MV_CMD" "$1" "$2" || { note "finalize $2 failed"; return 1; } +} + +db_scalar() { + # $1 = db path, $2 = sql (single scalar). sqlite3 if present, else python3 stdlib. + if command -v sqlite3 >/dev/null 2>&1; then + sqlite3 "$1" "$2" + else + python3 - "$1" "$2" <<'PY' || die "db_scalar failed on $1" +import sqlite3, sys +c = sqlite3.connect(sys.argv[1]) +row = c.execute(sys.argv[2]).fetchone() +print(row[0] if row is not None else 0) +PY + fi +} + +env_value() { + grep -E "^$1=" "$STATE_DIR/runtime.env" | head -1 | cut -d= -f2- +} + +service_active() { "$SYSTEMCTL_CMD" --user is-active "$SERVICE" >/dev/null 2>&1; } +service_stop() { "$SYSTEMCTL_CMD" --user stop "$SERVICE" || return 1; } +service_start() { "$SYSTEMCTL_CMD" --user start "$SERVICE" || { note "service start failed"; return 1; }; } + +inflight_datasets_sql() { + printf "SELECT COUNT(*) FROM datasets WHERE status IN ('pending','running')" +} +inflight_runs_sql() { + printf "SELECT COUNT(*) FROM runs WHERE status IN ('queued','running')" +} +# Returns: +# 0 = no in-flight work +# 1 = in-flight work present (or cannot be ruled out) +# 2 = environment/DB error (missing DB_PATH, unreadable DB) — caller MUST +# treat this as "unknown state", never as a clean pass. This function +# NEVER exits the shell: after a service stop, a deep die() here would +# bypass the restore branch and strand the old service in downtime. +check_no_inflight() { + DB_PATH=$(env_value DB_PATH) || { note "DB_PATH missing in runtime.env"; return 2; } + DB_PATH=${DB_PATH:-} + [ -n "$DB_PATH" ] || { note "DB_PATH empty in runtime.env"; return 2; } + [ -f "$DB_PATH" ] || { note "DB file missing: $DB_PATH"; return 2; } + if ! n=$(db_scalar "$DB_PATH" "$(inflight_datasets_sql)"); then + note "read datasets failed" + return 2 + fi + [ "$n" = "0" ] || { note "in-flight datasets: $n"; return 1; } + if ! n=$(db_scalar "$DB_PATH" "$(inflight_runs_sql)"); then + note "read runs failed" + return 2 + fi + [ "$n" = "0" ] || { note "in-flight runs: $n"; return 1; } + return 0 +} + +precheck_live() { + [ -n "$IMAGE_ID" ] || die "IMAGE_ID required" + "$DOCKER_CMD" image inspect "$IMAGE_ID" >/dev/null 2>&1 || die "image not present: $IMAGE_ID" + [ "$(env_value WORKER_IMAGE)" != "$IMAGE_ID" ] || die "runtime.env already pins $IMAGE_ID" + [ -x "$SERVER_BIN_SRC" ] || die "missing server binary: $SERVER_BIN_SRC" + [ -f "$DIST_SRC/index.html" ] || { die "missing frontend dist: $DIST_SRC"; return 1; } + service_active || die "$SERVICE not active; resolve before deploy (start it or inspect)" + check_no_inflight || die "in-flight work; deploy blocked" + msg "precheck(live) OK" +} + +precheck_stopped() { + service_active && { note "service still active after stop"; return 1; } + check_no_inflight || { note "new in-flight work discovered after stop"; return 2; } + msg "precheck(stopped) OK" +} + +release_unique_backup_dir() { + rid_base=$(date -u +%Y%m%d-%H%M%S) + rid="$rid_base" + sfx=a + while [ -e "$BACKUP_ROOT/$rid" ]; do + rid="$rid_base-$sfx" + sfx=$(python3 -c "import sys,random;print(chr(ord('a')+random.randrange(26)))" 2>/dev/null || printf '%s' "$sfx") + done + RELEASE_ID=$rid + printf '%s' "$rid" +} + +backup() { + release_unique_backup_dir >/dev/null 2>&1 || die "release id generation failed" + [ -n "$RELEASE_ID" ] || die "empty RELEASE_ID" + BACKUP_DIR=$BACKUP_ROOT/$RELEASE_ID + if [ -e "$BACKUP_DIR" ]; then + die "backup would overwrite existing $BACKUP_DIR (refusing)" + fi + mkdir -p "$BACKUP_DIR" || die "mkdir backup failed" + run_verchown_mode 700 "$BACKUP_ROOT" + run_verchown_mode 700 "$BACKUP_DIR" + run_cp "$STATE_DIR/runtime.env" "$BACKUP_DIR/runtime.env" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: runtime.env copy failed"; } + run_verchown_mode 600 "$BACKUP_DIR/runtime.env" + run_rsync -a "$RELEASE_DIR/" "$BACKUP_DIR/release/" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: release copy failed"; } + { + printf 'release_id=%s\ncreated_at=%s\n' "$RELEASE_ID" "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" + printf 'image_id_to_use=%s\nprevious_worker_image=%s\nprevious_server=%s\nprevious_frontend=%s\n' \ + "$IMAGE_ID" "$(env_value WORKER_IMAGE)" \ + "$RELEASE_DIR/strategy-lab-server" "$RELEASE_DIR/frontend" + sha256sum "$RELEASE_DIR/strategy-lab-server" 2>/dev/null | awk '{print "previous_server_sha256=" $1}' || printf 'previous_server_sha256=unknown\n' + } > "$BACKUP_DIR/release-manifest.txt.tmp" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest write failed"; } + # completeness check BEFORE the final marker: restore refuses backups + # without the 'complete' marker, so never mark anything half-copied. + [ -f "$BACKUP_DIR/runtime.env" ] && [ -f "$BACKUP_DIR/release/strategy-lab-server" ] \ + || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup incomplete; refusing to mark"; } + grep -q '^previous_server_sha256=' "$BACKUP_DIR/release-manifest.txt.tmp" \ + || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup manifest missing previous_server_sha256"; } + + fin_stage "$BACKUP_DIR/release-manifest.txt.tmp" "$BACKUP_DIR/release-manifest.txt" \ + || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest finalize failed"; } + run_verchown_mode 600 "$BACKUP_DIR/release-manifest.txt" + : > "$BACKUP_DIR/complete" || die "backup marker failed" + run_verchown_mode 600 "$BACKUP_DIR/complete" + msg "backup $RELEASE_ID -> $BACKUP_DIR (complete)" +} + +set_worker_image_id() { + rc=0 + run_sed "s|^WORKER_IMAGE=.*|WORKER_IMAGE=${IMAGE_ID}|" "$STATE_DIR/runtime.env" || rc=$? + [ "$rc" -eq 0 ] || return "$rc" + run_verchown_mode 600 "$STATE_DIR/runtime.env" || return 1 + [ "$(env_value WORKER_IMAGE)" = "$IMAGE_ID" ] || { + note "runtime.env replacement verification failed" + return 1 + } + return 0 +} + +swap_release() { + run_install "$SERVER_BIN_SRC" "$RELEASE_DIR/strategy-lab-server" || return 1 + # --checksum: dist contents must win even when size+mtime collide (drill + # regression: same-size files with identical mtime were silently skipped) + run_rsync -a --delete --checksum "$DIST_SRC/" "$RELEASE_DIR/frontend/" || return 1 + return 0 +} + +health_ok() { + BIND=$(env_value BIND) || return 1 + body=$("$CURL_CMD" -fsS -m 5 "http://$BIND/api/health" 2>/dev/null) || return 1 + printf '%s' "$body" | grep -q '"status":"ok"' || return 1 + return 0 +} + +wait_health() { + i=1 + while [ $i -le 20 ]; do + if health_ok; then msg "health OK"; return 0; fi + sleep 1 + i=$((i + 1)) + done + return 1 +} + +restore_release_id() { + rid=${1:?usage: deploy.sh restore <release_id>} + bd=$BACKUP_ROOT/$rid + DEPLOY_PHASE=restore; export DEPLOY_PHASE + [ -d "$bd" ] || die "no backup for release id: $rid" + [ -f "$bd/complete" ] || die "backup incomplete (no complete marker): $bd" + [ -f "$bd/runtime.env" ] || die "backup incomplete: runtime.env" + [ -f "$bd/release/strategy-lab-server" ] || die "backup incomplete: release" + # R3: restore takes the same deploy.lock unless the caller already holds it + if [ "${DEPLOY_LOCK_HELD:-0}" != "1" ]; then + exec 9>"$STATE_DIR/deploy.lock" + flock -n 9 || die "another deploy/rollback holds deploy.lock" + fi + # --checksum: dist contents must win even when size+mtime collide + # touch on-disk artifacts only while the (old/new) binary is not running: + # stop first, then replace files, then start again and verify health. + # If stop cannot be CONFIRMED (command failed or service still active), + # refuse to write ANY file: the running binary may still be using them. + # The restore is non-destructive here — the caller can retry after + # resolving the stop problem, and all existing files remain recoverable. + if ! "$SYSTEMCTL_CMD" --user stop "$SERVICE" 2>/dev/null; then + # stop reported failure: only safe to continue if service is verifiably down + if service_active; then + die "stop failed and $SERVICE still ACTIVE — refusing to restore files (nothing was written; backup $rid intact and recoverable)" + fi + note "stop reported failure but service verified inactive; continuing" + fi + service_active && die "could not confirm $SERVICE stopped — refusing to restore files (nothing was written; backup $rid intact and recoverable)" + run_cp "$bd/runtime.env" "$STATE_DIR/runtime.env" + run_verchown_mode 600 "$STATE_DIR/runtime.env" + run_rsync -a --delete --checksum "$bd/release/" "$RELEASE_DIR/" || die "restore release failed" + service_start || die "service start after restore failed" + sleep 2 + service_active || die "service not active after restore" + wait_health || die "health failed after restore" + verify_restored "$bd" + msg "restored $rid" +} + +# verify restored artifacts hash vs manifest +verify_restored() { + bd=$1 + w=$(grep -E '^previous_worker_image=' "$bd/release-manifest.txt" | sed 's/^previous_worker_image=//') || die "manifest corrupt" + [ "$(env_value WORKER_IMAGE)" = "$w" ] || die "restored env mismatches manifest" + h=$(grep -E '^previous_server_sha256=' "$bd/release-manifest.txt" | sed 's/^previous_server_sha256=//') + now=$(sha256sum "$RELEASE_DIR/strategy-lab-server" | awk '{print $1}') + [ "$h" = "$now" ] || die "restored server hash mismatch: $now != $h" + msg "restored artifacts verified (env + sha256)" +} + +# ---------------- main deploy flow ---------------- +main_deploy() { + exec 9>"$STATE_DIR/deploy.lock" + flock -n 9 || die "another deploy/rollback is running (deploy.lock held)" + + msg "phase 1/6 precheck (live service)" + precheck_live + + # Backup BEFORE stopping: a backup failure must never turn into downtime — + # the service keeps running on untouched artifacts and the deploy simply + # refuses. + msg "phase 2/6 backup (env + release artifacts, service still up)" + DEPLOY_PHASE=backup; export DEPLOY_PHASE + backup + + msg "phase 3/6 stop service (block NEW work)" + service_stop || die "service stop failed" + # R1: capture the REAL return code of the stopped-state checks — not an + # &&-chain / if ! — so a `2` (= new work discovered) is distinguishable + # from plain failure and the caller can restore the old service untouched. + rc=0 + precheck_stopped || rc=$? + # A DB/environment error (rc=2) after stop is treated the same as the race + # branch: we cannot prove the system is quiet, so restore the old service + # untouched rather than swapping files in an unverifiable state. + if [ "$rc" -ne 0 ]; then + if [ "$rc" = "2" ]; then + "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED" + note "restored old service WITHOUT replacement (post-stop check errored: DB/env unreadable)" + note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID" + exit 1 + fi + # rc=1: new work arrived between checks; the backup above is discarded (it + # recorded pre-inflight state and nothing was replaced) + "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED" + note "restored old service WITHOUT replacement (new tasks preserved)" + note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID (release kept; runtime.env already restored bytes)" + exit 1 + fi + + msg "phase 4/6 swap (env WORKER_IMAGE + binary + dist)" + # R1: each step rc-checked independently (also inside an && chain) + DEPLOY_PHASE=swap; export DEPLOY_PHASE + if ! set_worker_image_id; then + note "env update failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1 + fi + if ! swap_release; then + note "release swap failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1 + fi + + msg "phase 5/6 start service + health" + if ! "$SYSTEMCTL_CMD" --user start "$SERVICE"; then + note "service start failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1 + fi + sleep 2 + if ! wait_health; then + note "health failed; rolling back to $RELEASE_ID" + DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID" + exit 1 + fi + + msg "phase 6/6 deploy $RELEASE_ID complete (image $IMAGE_ID)" + msg "rollback id: $RELEASE_ID (rollback.sh $RELEASE_ID)" +} + +case "${1:-deploy}" in + deploy) main_deploy ;; + restore) shift; restore_release_id "$@" ;; + precheck) precheck_live ;; + *) die "usage: deploy.sh [deploy|restore <id>|precheck]" ;; +esac |
