summaryrefslogtreecommitdiff
path: root/artifacts/etf-recovery-candidate/ops/rollback.sh
diff options
context:
space:
mode:
authorSomhairle H. Marisol <[email protected]>2026-09-18 08:27:24 +0800
committerSomhairle H. Marisol <[email protected]>2026-09-18 08:27:24 +0800
commitbf6681eb29ac8b0c80ca17b2b5869f7de3da1198 (patch)
treebd50c535fa6afa5cbb33db3cd60fa76690e05bf1 /artifacts/etf-recovery-candidate/ops/rollback.sh
parentb4f1136457a08c6ad6ef52d9dc1f83fef050d87c (diff)
downloadstrategy-lab-bf6681eb29ac8b0c80ca17b2b5869f7de3da1198.tar.gz
chore(ops): 生产部署/回滚/故障演练脚本入库
[变更性质] 维护性变更:把已通过 84 场景演练的运维脚本纳入版本控制。artifacts/ 目录按仓库政策 ignore(本地证据),本提交按 Leader 指令以 -f 显式 加入,未包含任何密钥(脚本扫无 secret/token/password)。 [维护内容] 此前 ops/deploy.sh、rollback.sh 与 ops-drill2.sh 仅存在于本地 artifacts/(被 .gitignore 排除),生产依赖它们但仓库内无源码可审计。 [实现方案] deploy.sh:deploy.lock 防并发、阶段化备份(env/二进制/前端/DB, manifest 原子化写入 0700 目录)、显式 rc helper、零停机 swap、失败 路径自动 restore、健康与 sha256 校验;restore 兼做手动回滚。 rollback.sh:stop -> 替换 -> start -> 锁 + health + hash 强校验。 ops-drill2.sh:84 场景故障注入演练(备份/swap/restore 阶段注入矩阵、 RA/RB1/RB2 恢复分支),PASSED=84 FAILED=0 exit 0 为执行合约。 用法见 docs/etf-recovery-release-handoff.md 第 3 节。 [影响范围] 脚本本身未改(与演练定稿版本一致),仅入库;回滚目标记录在 manifest(previous_worker_image 等)。
Diffstat (limited to 'artifacts/etf-recovery-candidate/ops/rollback.sh')
-rwxr-xr-xartifacts/etf-recovery-candidate/ops/rollback.sh102
1 files changed, 102 insertions, 0 deletions
diff --git a/artifacts/etf-recovery-candidate/ops/rollback.sh b/artifacts/etf-recovery-candidate/ops/rollback.sh
new file mode 100755
index 0000000..095330e
--- /dev/null
+++ b/artifacts/etf-recovery-candidate/ops/rollback.sh
@@ -0,0 +1,102 @@
+#!/bin/sh
+# Strategy Lab 回滚 v3(供 leader 审查/执行;开发会话不操作生产)。
+# 备份位于 <STATE_DIR>/release-backups/<RELEASE_ID>/,且只在写入了 complete
+# 标记后才可用;没有 complete 标记的备份一律拒绝恢复。
+# R3:取同一把 deploy.lock;顺序:in-flight 检查 → 停服务 → 替换在线文件
+# (运行中绝不写二进制/前端)→ 启动 → is-active + /api/health JSON
+# status=="ok" + sha256 与 runtime.env 较验一致。
+# 用法:
+# rollback.sh <RELEASE_ID> # 精确恢复该 id
+# rollback.sh list # 只读列表(不取锁、不动任何东西)
+set -eu
+
+SYSTEMCTL_CMD=${SYSTEMCTL_CMD:-systemctl}
+CURL_CMD=${CURL_CMD:-curl}
+CP_CMD=${CP_CMD:-cp}
+MV_CMD=${MV_CMD:-mv}
+STATE_DIR=${STATE_DIR:-/home/somhairle/.local/share/strategy-lab-production}
+RELEASE_DIR=${RELEASE_DIR:-$STATE_DIR/release}
+BACKUP_ROOT=${BACKUP_ROOT:-$STATE_DIR/release-backups}
+SERVICE=${SERVICE:-strategy-lab-production}
+
+msg() { printf '[rollback] %s\n' "$1"; }
+die() { printf '[rollback] FATAL: %s\n' "$1" >&2; exit 1; }
+
+env_value() { grep -E "^$1=" "$STATE_DIR/runtime.env" | head -1 | cut -d= -f2-; }
+
+db_scalar() {
+ if command -v sqlite3 >/dev/null 2>&1; then
+ sqlite3 "$1" "$2"
+ else
+ python3 - "$1" "$2" <<'PY' || die "db_scalar failed"
+import sqlite3, sys
+c = sqlite3.connect(sys.argv[1])
+row = c.execute(sys.argv[2]).fetchone()
+print(row[0] if row is not None else 0)
+PY
+ fi
+}
+
+if [ "${1:-}" = "list" ]; then
+ for d in "$BACKUP_ROOT"/*/; do
+ [ -d "$d" ] || continue
+ img=$(grep -E '^image_id_to_use=' "$d/release-manifest.txt" 2>/dev/null | sed 's/^image_id_to_use=//')
+ done_mark=""
+ [ -f "$d/complete" ] && done_mark=" COMPLETE"
+ printf '%s\timage=%s%s\n' "$(basename "$d")" "$img" "$done_mark"
+ done
+ exit 0
+fi
+
+rid=${1:?usage: rollback.sh <RELEASE_ID>|list}
+bd=$BACKUP_ROOT/$rid
+[ -d "$bd" ] || die "no backup for release id: $rid"
+[ -f "$bd/complete" ] || die "backup incomplete (no complete marker): $bd"
+[ -f "$bd/runtime.env" ] || die "backup incomplete: runtime.env"
+[ -f "$bd/release/strategy-lab-server" ] || die "backup incomplete: release"
+
+# R3: same lock as deploy
+exec 9>"$STATE_DIR/deploy.lock"
+flock -n 9 || die "another deploy/rollback is running (deploy.lock held)"
+
+# Refuse while user work is in-flight (a restore must not strand mid-run state)
+DB_PATH=$(grep -E '^DB_PATH=' "$STATE_DIR/runtime.env" | head -1 | cut -s -d= -f2-)
+if [ -n "${DB_PATH:-}" ]; then
+ n=$(db_scalar "$DB_PATH" "SELECT COUNT(*) FROM runs WHERE status IN ('queued','running')")
+ [ "$n" = "0" ] || die "in-flight runs: $n — cancel/finish before rollback"
+fi
+
+old_img=$(grep -E '^previous_worker_image=' "$bd/release-manifest.txt" 2>/dev/null | sed 's/^previous_worker_image=//') || true
+
+# NEVER write on-disk artifacts while a binary is executing it: stop first.
+"$SYSTEMCTL_CMD" --user stop "$SERVICE" || die "stop before restore failed"
+
+cp -p "$bd/runtime.env" "$STATE_DIR/runtime.env" || die "restore env failed"
+chmod 600 "$STATE_DIR/runtime.env"
+# --checksum: dist contents must win even when size+mtime collide
+rsync -a --delete --checksum "$bd/release/" "$RELEASE_DIR/" || {
+ # files may be half-restored — bring back what the backup declares as service
+ "$SYSTEMCTL_CMD" --user start "$SERVICE" 2>/dev/null || true
+ die "restore release failed"
+}
+
+"$SYSTEMCTL_CMD" --user start "$SERVICE" || die "service start after restore failed"
+sleep 2
+if ! "$SYSTEMCTL_CMD" --user is-active "$SERVICE" >/dev/null 2>&1; then
+ die "service not active after restore; check journalctl --user -u $SERVICE"
+fi
+
+# health JSON must report status ok
+BIND=$(grep -E '^BIND=' "$STATE_DIR/runtime.env" | head -1 | cut -d= -f2-)
+body=$("$CURL_CMD" -fsS -m 5 "http://$BIND/api/health" 2>/dev/null) || die "health unreachable"
+printf '%s' "$body" | grep -q '"status":"ok"' || die "health not ok: $body"
+
+# Release-artifact identity: hash must equal the manifest's recorded old value
+want_hash=$(grep -E '^previous_server_sha256=' "$bd/release-manifest.txt" | sed 's/^previous_server_sha256=//')
+now_hash=$(sha256sum "$RELEASE_DIR/strategy-lab-server" | awk '{print $1}')
+[ "$want_hash" = "$now_hash" ] || die "restored server hash mismatch: $now_hash != $want_hash"
+
+# env identity
+[ "$(env_value WORKER_IMAGE)" = "$old_img" ] || die "restored env WORKER_IMAGE mismatch: $(env_value WORKER_IMAGE) != $old_img"
+
+msg "restored $rid: env verified, artifact sha256 verified, health ok."