summaryrefslogtreecommitdiff
path: root/artifacts/etf-recovery-candidate/ops/deploy.sh
diff options
context:
space:
mode:
Diffstat (limited to 'artifacts/etf-recovery-candidate/ops/deploy.sh')
-rwxr-xr-xartifacts/etf-recovery-candidate/ops/deploy.sh329
1 files changed, 329 insertions, 0 deletions
diff --git a/artifacts/etf-recovery-candidate/ops/deploy.sh b/artifacts/etf-recovery-candidate/ops/deploy.sh
new file mode 100755
index 0000000..822c90e
--- /dev/null
+++ b/artifacts/etf-recovery-candidate/ops/deploy.sh
@@ -0,0 +1,329 @@
+#!/bin/sh
+# ==============================================================
+# Strategy Lab — 部署脚本 v2(供 leader 审查/在隔离替身环境演练)
+# *** 本脚本不会由开发会话在生产执行 ***
+# 针对 leader 复审意见(2026-09-17)的修正:
+# R1 每个关键命令显式检查返回码,绝不依赖 set -e 在 && / if ! 上下文的不可靠性。
+# R2 竞态窗口:先 systemctl stop 服务 → 重新硬检查 in-flight(服务停止后不再产生
+# 新任务)→ 才 swap。停后发现新任务:立即恢复旧服务(发布物未被动过)并退出,
+# 任务状态原样保留。
+# R3 restore 入口也取同一把锁;回滚后验证 ①服务 active ②/api/health JSON
+# status=="ok" ③恢复后的发布物 sha256 == release-manifest.txt 里的旧哈希。
+# 备份在发布目录外、0700;单一 RELEASE_ID;同秒碰撞自动加后缀,绝不覆盖既有备份。
+#
+# 受控替身(演练/审查可注入):
+# SYSTEMCTL_CMD/DOCKER_CMD/CURL_CMD/RSYNC_CMD/INSTALL_CMD 均可指向替身。
+# *不含生产破坏性操作,禁止在无 STAGE_DIR 环境下执行(见 require_prod_paths)。
+set -eu
+
+SYSTEMCTL_CMD=${SYSTEMCTL_CMD:-systemctl}
+DOCKER_CMD=${DOCKER_CMD:-docker}
+CURL_CMD=${CURL_CMD:-curl}
+INSTALL_CMD=${INSTALL_CMD:-install}
+SED_CMD=${SED_CMD:-sed}
+RSYNC_CMD=${RSYNC_CMD:-rsync}
+CP_CMD=${CP_CMD:-cp}
+MV_CMD=${MV_CMD:-mv}
+
+STATE_DIR=${STATE_DIR:-/home/somhairle/.local/share/strategy-lab-production}
+RELEASE_DIR=${RELEASE_DIR:-$STATE_DIR/release}
+BACKUP_ROOT=${BACKUP_ROOT:-$STATE_DIR/release-backups}
+REPO_ROOT=${REPO_ROOT:?REPO_ROOT must be set (absolute path to strategy-lab checkout)}
+SERVICE=${SERVICE:-strategy-lab-production}
+SERVER_BIN_SRC=${SERVER_BIN_SRC:-$REPO_ROOT/server/target/release/strategy-lab-server}
+DIST_SRC=${DIST_SRC:-$REPO_ROOT/frontend/dist}
+IMAGE_ID=${IMAGE_ID:?IMAGE_ID must be the immutable image ID, e.g. sha256:...}
+
+msg() { printf '[deploy] %s\n' "$1"; }
+note() { printf '[deploy] NOTE: %s\n' "$1" >&2; }
+die() { printf '[deploy] FATAL: %s\n' "$1" >&2; exit 1; }
+
+# ---------------- explicit-rc helpers (R1) ----------------
+run_cp() { "$CP_CMD" -p "$1" "$2" || { note "cp -p $1 -> $2 failed"; return 1; }; }
+run_sed() { "$SED_CMD" -i "$@" || { note "sed env replacement failed"; return 1; }; }
+run_install(){ "$INSTALL_CMD" -m 0755 "$1" "$2" || { note "install $1 -> $2 failed"; return 1; }; }
+run_rsync() { "$RSYNC_CMD" "$@" || { note "rsync $* failed"; return 1; } }
+run_verchown_mode() {
+ chmod "$1" "$2" || { note "chmod $1 $2 failed"; return 1; }
+}
+# manifest is written to a tmp file and atomically moved into place; a failed
+# mv (=> incomplete manifest) never looks like a finished backup
+fin_stage() {
+ "$MV_CMD" "$1" "$2" || { note "finalize $2 failed"; return 1; }
+}
+
+db_scalar() {
+ # $1 = db path, $2 = sql (single scalar). sqlite3 if present, else python3 stdlib.
+ if command -v sqlite3 >/dev/null 2>&1; then
+ sqlite3 "$1" "$2"
+ else
+ python3 - "$1" "$2" <<'PY' || die "db_scalar failed on $1"
+import sqlite3, sys
+c = sqlite3.connect(sys.argv[1])
+row = c.execute(sys.argv[2]).fetchone()
+print(row[0] if row is not None else 0)
+PY
+ fi
+}
+
+env_value() {
+ grep -E "^$1=" "$STATE_DIR/runtime.env" | head -1 | cut -d= -f2-
+}
+
+service_active() { "$SYSTEMCTL_CMD" --user is-active "$SERVICE" >/dev/null 2>&1; }
+service_stop() { "$SYSTEMCTL_CMD" --user stop "$SERVICE" || return 1; }
+service_start() { "$SYSTEMCTL_CMD" --user start "$SERVICE" || { note "service start failed"; return 1; }; }
+
+inflight_datasets_sql() {
+ printf "SELECT COUNT(*) FROM datasets WHERE status IN ('pending','running')"
+}
+inflight_runs_sql() {
+ printf "SELECT COUNT(*) FROM runs WHERE status IN ('queued','running')"
+}
+# Returns:
+# 0 = no in-flight work
+# 1 = in-flight work present (or cannot be ruled out)
+# 2 = environment/DB error (missing DB_PATH, unreadable DB) — caller MUST
+# treat this as "unknown state", never as a clean pass. This function
+# NEVER exits the shell: after a service stop, a deep die() here would
+# bypass the restore branch and strand the old service in downtime.
+check_no_inflight() {
+ DB_PATH=$(env_value DB_PATH) || { note "DB_PATH missing in runtime.env"; return 2; }
+ DB_PATH=${DB_PATH:-}
+ [ -n "$DB_PATH" ] || { note "DB_PATH empty in runtime.env"; return 2; }
+ [ -f "$DB_PATH" ] || { note "DB file missing: $DB_PATH"; return 2; }
+ if ! n=$(db_scalar "$DB_PATH" "$(inflight_datasets_sql)"); then
+ note "read datasets failed"
+ return 2
+ fi
+ [ "$n" = "0" ] || { note "in-flight datasets: $n"; return 1; }
+ if ! n=$(db_scalar "$DB_PATH" "$(inflight_runs_sql)"); then
+ note "read runs failed"
+ return 2
+ fi
+ [ "$n" = "0" ] || { note "in-flight runs: $n"; return 1; }
+ return 0
+}
+
+precheck_live() {
+ [ -n "$IMAGE_ID" ] || die "IMAGE_ID required"
+ "$DOCKER_CMD" image inspect "$IMAGE_ID" >/dev/null 2>&1 || die "image not present: $IMAGE_ID"
+ [ "$(env_value WORKER_IMAGE)" != "$IMAGE_ID" ] || die "runtime.env already pins $IMAGE_ID"
+ [ -x "$SERVER_BIN_SRC" ] || die "missing server binary: $SERVER_BIN_SRC"
+ [ -f "$DIST_SRC/index.html" ] || { die "missing frontend dist: $DIST_SRC"; return 1; }
+ service_active || die "$SERVICE not active; resolve before deploy (start it or inspect)"
+ check_no_inflight || die "in-flight work; deploy blocked"
+ msg "precheck(live) OK"
+}
+
+precheck_stopped() {
+ service_active && { note "service still active after stop"; return 1; }
+ check_no_inflight || { note "new in-flight work discovered after stop"; return 2; }
+ msg "precheck(stopped) OK"
+}
+
+release_unique_backup_dir() {
+ rid_base=$(date -u +%Y%m%d-%H%M%S)
+ rid="$rid_base"
+ sfx=a
+ while [ -e "$BACKUP_ROOT/$rid" ]; do
+ rid="$rid_base-$sfx"
+ sfx=$(python3 -c "import sys,random;print(chr(ord('a')+random.randrange(26)))" 2>/dev/null || printf '%s' "$sfx")
+ done
+ RELEASE_ID=$rid
+ printf '%s' "$rid"
+}
+
+backup() {
+ release_unique_backup_dir >/dev/null 2>&1 || die "release id generation failed"
+ [ -n "$RELEASE_ID" ] || die "empty RELEASE_ID"
+ BACKUP_DIR=$BACKUP_ROOT/$RELEASE_ID
+ if [ -e "$BACKUP_DIR" ]; then
+ die "backup would overwrite existing $BACKUP_DIR (refusing)"
+ fi
+ mkdir -p "$BACKUP_DIR" || die "mkdir backup failed"
+ run_verchown_mode 700 "$BACKUP_ROOT"
+ run_verchown_mode 700 "$BACKUP_DIR"
+ run_cp "$STATE_DIR/runtime.env" "$BACKUP_DIR/runtime.env" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: runtime.env copy failed"; }
+ run_verchown_mode 600 "$BACKUP_DIR/runtime.env"
+ run_rsync -a "$RELEASE_DIR/" "$BACKUP_DIR/release/" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: release copy failed"; }
+ {
+ printf 'release_id=%s\ncreated_at=%s\n' "$RELEASE_ID" "$(date -u '+%Y-%m-%dT%H:%M:%SZ')"
+ printf 'image_id_to_use=%s\nprevious_worker_image=%s\nprevious_server=%s\nprevious_frontend=%s\n' \
+ "$IMAGE_ID" "$(env_value WORKER_IMAGE)" \
+ "$RELEASE_DIR/strategy-lab-server" "$RELEASE_DIR/frontend"
+ sha256sum "$RELEASE_DIR/strategy-lab-server" 2>/dev/null | awk '{print "previous_server_sha256=" $1}' || printf 'previous_server_sha256=unknown\n'
+ } > "$BACKUP_DIR/release-manifest.txt.tmp" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest write failed"; }
+ # completeness check BEFORE the final marker: restore refuses backups
+ # without the 'complete' marker, so never mark anything half-copied.
+ [ -f "$BACKUP_DIR/runtime.env" ] && [ -f "$BACKUP_DIR/release/strategy-lab-server" ] \
+ || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup incomplete; refusing to mark"; }
+ grep -q '^previous_server_sha256=' "$BACKUP_DIR/release-manifest.txt.tmp" \
+ || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup manifest missing previous_server_sha256"; }
+
+ fin_stage "$BACKUP_DIR/release-manifest.txt.tmp" "$BACKUP_DIR/release-manifest.txt" \
+ || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest finalize failed"; }
+ run_verchown_mode 600 "$BACKUP_DIR/release-manifest.txt"
+ : > "$BACKUP_DIR/complete" || die "backup marker failed"
+ run_verchown_mode 600 "$BACKUP_DIR/complete"
+ msg "backup $RELEASE_ID -> $BACKUP_DIR (complete)"
+}
+
+set_worker_image_id() {
+ rc=0
+ run_sed "s|^WORKER_IMAGE=.*|WORKER_IMAGE=${IMAGE_ID}|" "$STATE_DIR/runtime.env" || rc=$?
+ [ "$rc" -eq 0 ] || return "$rc"
+ run_verchown_mode 600 "$STATE_DIR/runtime.env" || return 1
+ [ "$(env_value WORKER_IMAGE)" = "$IMAGE_ID" ] || {
+ note "runtime.env replacement verification failed"
+ return 1
+ }
+ return 0
+}
+
+swap_release() {
+ run_install "$SERVER_BIN_SRC" "$RELEASE_DIR/strategy-lab-server" || return 1
+ # --checksum: dist contents must win even when size+mtime collide (drill
+ # regression: same-size files with identical mtime were silently skipped)
+ run_rsync -a --delete --checksum "$DIST_SRC/" "$RELEASE_DIR/frontend/" || return 1
+ return 0
+}
+
+health_ok() {
+ BIND=$(env_value BIND) || return 1
+ body=$("$CURL_CMD" -fsS -m 5 "http://$BIND/api/health" 2>/dev/null) || return 1
+ printf '%s' "$body" | grep -q '"status":"ok"' || return 1
+ return 0
+}
+
+wait_health() {
+ i=1
+ while [ $i -le 20 ]; do
+ if health_ok; then msg "health OK"; return 0; fi
+ sleep 1
+ i=$((i + 1))
+ done
+ return 1
+}
+
+restore_release_id() {
+ rid=${1:?usage: deploy.sh restore <release_id>}
+ bd=$BACKUP_ROOT/$rid
+ DEPLOY_PHASE=restore; export DEPLOY_PHASE
+ [ -d "$bd" ] || die "no backup for release id: $rid"
+ [ -f "$bd/complete" ] || die "backup incomplete (no complete marker): $bd"
+ [ -f "$bd/runtime.env" ] || die "backup incomplete: runtime.env"
+ [ -f "$bd/release/strategy-lab-server" ] || die "backup incomplete: release"
+ # R3: restore takes the same deploy.lock unless the caller already holds it
+ if [ "${DEPLOY_LOCK_HELD:-0}" != "1" ]; then
+ exec 9>"$STATE_DIR/deploy.lock"
+ flock -n 9 || die "another deploy/rollback holds deploy.lock"
+ fi
+ # --checksum: dist contents must win even when size+mtime collide
+ # touch on-disk artifacts only while the (old/new) binary is not running:
+ # stop first, then replace files, then start again and verify health.
+ # If stop cannot be CONFIRMED (command failed or service still active),
+ # refuse to write ANY file: the running binary may still be using them.
+ # The restore is non-destructive here — the caller can retry after
+ # resolving the stop problem, and all existing files remain recoverable.
+ if ! "$SYSTEMCTL_CMD" --user stop "$SERVICE" 2>/dev/null; then
+ # stop reported failure: only safe to continue if service is verifiably down
+ if service_active; then
+ die "stop failed and $SERVICE still ACTIVE — refusing to restore files (nothing was written; backup $rid intact and recoverable)"
+ fi
+ note "stop reported failure but service verified inactive; continuing"
+ fi
+ service_active && die "could not confirm $SERVICE stopped — refusing to restore files (nothing was written; backup $rid intact and recoverable)"
+ run_cp "$bd/runtime.env" "$STATE_DIR/runtime.env"
+ run_verchown_mode 600 "$STATE_DIR/runtime.env"
+ run_rsync -a --delete --checksum "$bd/release/" "$RELEASE_DIR/" || die "restore release failed"
+ service_start || die "service start after restore failed"
+ sleep 2
+ service_active || die "service not active after restore"
+ wait_health || die "health failed after restore"
+ verify_restored "$bd"
+ msg "restored $rid"
+}
+
+# verify restored artifacts hash vs manifest
+verify_restored() {
+ bd=$1
+ w=$(grep -E '^previous_worker_image=' "$bd/release-manifest.txt" | sed 's/^previous_worker_image=//') || die "manifest corrupt"
+ [ "$(env_value WORKER_IMAGE)" = "$w" ] || die "restored env mismatches manifest"
+ h=$(grep -E '^previous_server_sha256=' "$bd/release-manifest.txt" | sed 's/^previous_server_sha256=//')
+ now=$(sha256sum "$RELEASE_DIR/strategy-lab-server" | awk '{print $1}')
+ [ "$h" = "$now" ] || die "restored server hash mismatch: $now != $h"
+ msg "restored artifacts verified (env + sha256)"
+}
+
+# ---------------- main deploy flow ----------------
+main_deploy() {
+ exec 9>"$STATE_DIR/deploy.lock"
+ flock -n 9 || die "another deploy/rollback is running (deploy.lock held)"
+
+ msg "phase 1/6 precheck (live service)"
+ precheck_live
+
+ # Backup BEFORE stopping: a backup failure must never turn into downtime —
+ # the service keeps running on untouched artifacts and the deploy simply
+ # refuses.
+ msg "phase 2/6 backup (env + release artifacts, service still up)"
+ DEPLOY_PHASE=backup; export DEPLOY_PHASE
+ backup
+
+ msg "phase 3/6 stop service (block NEW work)"
+ service_stop || die "service stop failed"
+ # R1: capture the REAL return code of the stopped-state checks — not an
+ # &&-chain / if ! — so a `2` (= new work discovered) is distinguishable
+ # from plain failure and the caller can restore the old service untouched.
+ rc=0
+ precheck_stopped || rc=$?
+ # A DB/environment error (rc=2) after stop is treated the same as the race
+ # branch: we cannot prove the system is quiet, so restore the old service
+ # untouched rather than swapping files in an unverifiable state.
+ if [ "$rc" -ne 0 ]; then
+ if [ "$rc" = "2" ]; then
+ "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED"
+ note "restored old service WITHOUT replacement (post-stop check errored: DB/env unreadable)"
+ note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID"
+ exit 1
+ fi
+ # rc=1: new work arrived between checks; the backup above is discarded (it
+ # recorded pre-inflight state and nothing was replaced)
+ "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED"
+ note "restored old service WITHOUT replacement (new tasks preserved)"
+ note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID (release kept; runtime.env already restored bytes)"
+ exit 1
+ fi
+
+ msg "phase 4/6 swap (env WORKER_IMAGE + binary + dist)"
+ # R1: each step rc-checked independently (also inside an && chain)
+ DEPLOY_PHASE=swap; export DEPLOY_PHASE
+ if ! set_worker_image_id; then
+ note "env update failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1
+ fi
+ if ! swap_release; then
+ note "release swap failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1
+ fi
+
+ msg "phase 5/6 start service + health"
+ if ! "$SYSTEMCTL_CMD" --user start "$SERVICE"; then
+ note "service start failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1
+ fi
+ sleep 2
+ if ! wait_health; then
+ note "health failed; rolling back to $RELEASE_ID"
+ DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"
+ exit 1
+ fi
+
+ msg "phase 6/6 deploy $RELEASE_ID complete (image $IMAGE_ID)"
+ msg "rollback id: $RELEASE_ID (rollback.sh $RELEASE_ID)"
+}
+
+case "${1:-deploy}" in
+ deploy) main_deploy ;;
+ restore) shift; restore_release_id "$@" ;;
+ precheck) precheck_live ;;
+ *) die "usage: deploy.sh [deploy|restore <id>|precheck]" ;;
+esac