summaryrefslogtreecommitdiff
path: root/artifacts/etf-recovery-candidate/ops/deploy.sh
blob: 822c90eb45246698e2b7b0ea39484d5b44717061 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
#!/bin/sh
# ==============================================================
# Strategy Lab — 部署脚本 v2(供 leader 审查/在隔离替身环境演练)
#        *** 本脚本不会由开发会话在生产执行 ***
# 针对 leader 复审意见(2026-09-17)的修正:
#  R1 每个关键命令显式检查返回码,绝不依赖 set -e 在 && / if ! 上下文的不可靠性。
#  R2 竞态窗口:先 systemctl stop 服务 → 重新硬检查 in-flight(服务停止后不再产生
#     新任务)→ 才 swap。停后发现新任务:立即恢复旧服务(发布物未被动过)并退出,
#     任务状态原样保留。
#  R3 restore 入口也取同一把锁;回滚后验证 ①服务 active ②/api/health JSON
#     status=="ok" ③恢复后的发布物 sha256 == release-manifest.txt 里的旧哈希。
#  备份在发布目录外、0700;单一 RELEASE_ID;同秒碰撞自动加后缀,绝不覆盖既有备份。
#
# 受控替身(演练/审查可注入):
#   SYSTEMCTL_CMD/DOCKER_CMD/CURL_CMD/RSYNC_CMD/INSTALL_CMD 均可指向替身。
#   *不含生产破坏性操作,禁止在无 STAGE_DIR 环境下执行(见 require_prod_paths)。
set -eu

SYSTEMCTL_CMD=${SYSTEMCTL_CMD:-systemctl}
DOCKER_CMD=${DOCKER_CMD:-docker}
CURL_CMD=${CURL_CMD:-curl}
INSTALL_CMD=${INSTALL_CMD:-install}
SED_CMD=${SED_CMD:-sed}
RSYNC_CMD=${RSYNC_CMD:-rsync}
CP_CMD=${CP_CMD:-cp}
MV_CMD=${MV_CMD:-mv}

STATE_DIR=${STATE_DIR:-/home/somhairle/.local/share/strategy-lab-production}
RELEASE_DIR=${RELEASE_DIR:-$STATE_DIR/release}
BACKUP_ROOT=${BACKUP_ROOT:-$STATE_DIR/release-backups}
REPO_ROOT=${REPO_ROOT:?REPO_ROOT must be set (absolute path to strategy-lab checkout)}
SERVICE=${SERVICE:-strategy-lab-production}
SERVER_BIN_SRC=${SERVER_BIN_SRC:-$REPO_ROOT/server/target/release/strategy-lab-server}
DIST_SRC=${DIST_SRC:-$REPO_ROOT/frontend/dist}
IMAGE_ID=${IMAGE_ID:?IMAGE_ID must be the immutable image ID, e.g. sha256:...}

msg() { printf '[deploy] %s\n' "$1"; }
note() { printf '[deploy] NOTE: %s\n' "$1" >&2; }
die() { printf '[deploy] FATAL: %s\n' "$1" >&2; exit 1; }

# ---------------- explicit-rc helpers (R1) ----------------
run_cp()     { "$CP_CMD" -p "$1" "$2" || { note "cp -p $1 -> $2 failed"; return 1; }; }
run_sed()    { "$SED_CMD" -i "$@" || { note "sed env replacement failed"; return 1; }; }
run_install(){ "$INSTALL_CMD" -m 0755 "$1" "$2" || { note "install $1 -> $2 failed"; return 1; }; }
run_rsync()  { "$RSYNC_CMD" "$@" || { note "rsync $* failed"; return 1; } }
run_verchown_mode() {
  chmod "$1" "$2" || { note "chmod $1 $2 failed"; return 1; }
}
# manifest is written to a tmp file and atomically moved into place; a failed
# mv (=> incomplete manifest) never looks like a finished backup
fin_stage() {
  "$MV_CMD" "$1" "$2" || { note "finalize $2 failed"; return 1; }
}

db_scalar() {
  # $1 = db path, $2 = sql (single scalar). sqlite3 if present, else python3 stdlib.
  if command -v sqlite3 >/dev/null 2>&1; then
    sqlite3 "$1" "$2"
  else
    python3 - "$1" "$2" <<'PY' || die "db_scalar failed on $1"
import sqlite3, sys
c = sqlite3.connect(sys.argv[1])
row = c.execute(sys.argv[2]).fetchone()
print(row[0] if row is not None else 0)
PY
  fi
}

env_value() {
  grep -E "^$1=" "$STATE_DIR/runtime.env" | head -1 | cut -d= -f2-
}

service_active() { "$SYSTEMCTL_CMD" --user is-active "$SERVICE" >/dev/null 2>&1; }
service_stop()   { "$SYSTEMCTL_CMD" --user stop "$SERVICE"       || return 1; }
service_start()  { "$SYSTEMCTL_CMD" --user start "$SERVICE"      || { note "service start failed"; return 1; }; }

inflight_datasets_sql() {
  printf "SELECT COUNT(*) FROM datasets WHERE status IN ('pending','running')"
}
inflight_runs_sql() {
  printf "SELECT COUNT(*) FROM runs WHERE status IN ('queued','running')"
}
# Returns:
#   0 = no in-flight work
#   1 = in-flight work present (or cannot be ruled out)
#   2 = environment/DB error (missing DB_PATH, unreadable DB) — caller MUST
#       treat this as "unknown state", never as a clean pass. This function
#       NEVER exits the shell: after a service stop, a deep die() here would
#       bypass the restore branch and strand the old service in downtime.
check_no_inflight() {
  DB_PATH=$(env_value DB_PATH) || { note "DB_PATH missing in runtime.env"; return 2; }
  DB_PATH=${DB_PATH:-}
  [ -n "$DB_PATH" ] || { note "DB_PATH empty in runtime.env"; return 2; }
  [ -f "$DB_PATH" ] || { note "DB file missing: $DB_PATH"; return 2; }
  if ! n=$(db_scalar "$DB_PATH" "$(inflight_datasets_sql)"); then
    note "read datasets failed"
    return 2
  fi
  [ "$n" = "0" ] || { note "in-flight datasets: $n"; return 1; }
  if ! n=$(db_scalar "$DB_PATH" "$(inflight_runs_sql)"); then
    note "read runs failed"
    return 2
  fi
  [ "$n" = "0" ] || { note "in-flight runs: $n"; return 1; }
  return 0
}

precheck_live() {
  [ -n "$IMAGE_ID" ] || die "IMAGE_ID required"
  "$DOCKER_CMD" image inspect "$IMAGE_ID" >/dev/null 2>&1 || die "image not present: $IMAGE_ID"
  [ "$(env_value WORKER_IMAGE)" != "$IMAGE_ID" ] || die "runtime.env already pins $IMAGE_ID"
  [ -x "$SERVER_BIN_SRC" ] || die "missing server binary: $SERVER_BIN_SRC"
  [ -f "$DIST_SRC/index.html" ] || { die "missing frontend dist: $DIST_SRC"; return 1; }
  service_active || die "$SERVICE not active; resolve before deploy (start it or inspect)"
  check_no_inflight || die "in-flight work; deploy blocked"
  msg "precheck(live) OK"
}

precheck_stopped() {
  service_active && { note "service still active after stop"; return 1; }
  check_no_inflight || { note "new in-flight work discovered after stop"; return 2; }
  msg "precheck(stopped) OK"
}

release_unique_backup_dir() {
  rid_base=$(date -u +%Y%m%d-%H%M%S)
  rid="$rid_base"
  sfx=a
  while [ -e "$BACKUP_ROOT/$rid" ]; do
    rid="$rid_base-$sfx"
    sfx=$(python3 -c "import sys,random;print(chr(ord('a')+random.randrange(26)))" 2>/dev/null || printf '%s' "$sfx")
  done
  RELEASE_ID=$rid
  printf '%s' "$rid"
}

backup() {
  release_unique_backup_dir >/dev/null 2>&1 || die "release id generation failed"
  [ -n "$RELEASE_ID" ] || die "empty RELEASE_ID"
  BACKUP_DIR=$BACKUP_ROOT/$RELEASE_ID
  if [ -e "$BACKUP_DIR" ]; then
    die "backup would overwrite existing $BACKUP_DIR (refusing)"
  fi
  mkdir -p "$BACKUP_DIR" || die "mkdir backup failed"
  run_verchown_mode 700 "$BACKUP_ROOT"
  run_verchown_mode 700 "$BACKUP_DIR"
  run_cp "$STATE_DIR/runtime.env" "$BACKUP_DIR/runtime.env" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: runtime.env copy failed"; }
  run_verchown_mode 600 "$BACKUP_DIR/runtime.env"
  run_rsync -a "$RELEASE_DIR/" "$BACKUP_DIR/release/" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: release copy failed"; }
  {
    printf 'release_id=%s\ncreated_at=%s\n' "$RELEASE_ID" "$(date -u '+%Y-%m-%dT%H:%M:%SZ')"
    printf 'image_id_to_use=%s\nprevious_worker_image=%s\nprevious_server=%s\nprevious_frontend=%s\n' \
      "$IMAGE_ID" "$(env_value WORKER_IMAGE)" \
      "$RELEASE_DIR/strategy-lab-server" "$RELEASE_DIR/frontend"
    sha256sum "$RELEASE_DIR/strategy-lab-server" 2>/dev/null | awk '{print "previous_server_sha256=" $1}' || printf 'previous_server_sha256=unknown\n'
  } > "$BACKUP_DIR/release-manifest.txt.tmp" || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest write failed"; }
  # completeness check BEFORE the final marker: restore refuses backups
  # without the 'complete' marker, so never mark anything half-copied.
  [ -f "$BACKUP_DIR/runtime.env" ] && [ -f "$BACKUP_DIR/release/strategy-lab-server" ] \
    || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup incomplete; refusing to mark"; }
  grep -q '^previous_server_sha256=' "$BACKUP_DIR/release-manifest.txt.tmp" \
    || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup manifest missing previous_server_sha256"; }

  fin_stage "$BACKUP_DIR/release-manifest.txt.tmp" "$BACKUP_DIR/release-manifest.txt" \
    || { mv "$BACKUP_DIR" "$BACKUP_DIR.broken.$RELEASE_ID" 2>/dev/null; die "backup: manifest finalize failed"; }
  run_verchown_mode 600 "$BACKUP_DIR/release-manifest.txt"
  : > "$BACKUP_DIR/complete" || die "backup marker failed"
  run_verchown_mode 600 "$BACKUP_DIR/complete"
  msg "backup $RELEASE_ID -> $BACKUP_DIR (complete)"
}

set_worker_image_id() {
  rc=0
  run_sed "s|^WORKER_IMAGE=.*|WORKER_IMAGE=${IMAGE_ID}|" "$STATE_DIR/runtime.env" || rc=$?
  [ "$rc" -eq 0 ] || return "$rc"
  run_verchown_mode 600 "$STATE_DIR/runtime.env" || return 1
  [ "$(env_value WORKER_IMAGE)" = "$IMAGE_ID" ] || {
    note "runtime.env replacement verification failed"
    return 1
  }
  return 0
}

swap_release() {
  run_install "$SERVER_BIN_SRC" "$RELEASE_DIR/strategy-lab-server" || return 1
  # --checksum: dist contents must win even when size+mtime collide (drill
  # regression: same-size files with identical mtime were silently skipped)
  run_rsync -a --delete --checksum "$DIST_SRC/" "$RELEASE_DIR/frontend/" || return 1
  return 0
}

health_ok() {
  BIND=$(env_value BIND) || return 1
  body=$("$CURL_CMD" -fsS -m 5 "http://$BIND/api/health" 2>/dev/null) || return 1
  printf '%s' "$body" | grep -q '"status":"ok"' || return 1
  return 0
}

wait_health() {
  i=1
  while [ $i -le 20 ]; do
    if health_ok; then msg "health OK"; return 0; fi
    sleep 1
    i=$((i + 1))
  done
  return 1
}

restore_release_id() {
  rid=${1:?usage: deploy.sh restore <release_id>}
  bd=$BACKUP_ROOT/$rid
  DEPLOY_PHASE=restore; export DEPLOY_PHASE
  [ -d "$bd" ] || die "no backup for release id: $rid"
  [ -f "$bd/complete" ] || die "backup incomplete (no complete marker): $bd"
  [ -f "$bd/runtime.env" ] || die "backup incomplete: runtime.env"
  [ -f "$bd/release/strategy-lab-server" ] || die "backup incomplete: release"
  # R3: restore takes the same deploy.lock unless the caller already holds it
  if [ "${DEPLOY_LOCK_HELD:-0}" != "1" ]; then
    exec 9>"$STATE_DIR/deploy.lock"
    flock -n 9 || die "another deploy/rollback holds deploy.lock"
  fi
  # --checksum: dist contents must win even when size+mtime collide
  # touch on-disk artifacts only while the (old/new) binary is not running:
  # stop first, then replace files, then start again and verify health.
  # If stop cannot be CONFIRMED (command failed or service still active),
  # refuse to write ANY file: the running binary may still be using them.
  # The restore is non-destructive here — the caller can retry after
  # resolving the stop problem, and all existing files remain recoverable.
  if ! "$SYSTEMCTL_CMD" --user stop "$SERVICE" 2>/dev/null; then
    # stop reported failure: only safe to continue if service is verifiably down
    if service_active; then
      die "stop failed and $SERVICE still ACTIVE — refusing to restore files (nothing was written; backup $rid intact and recoverable)"
    fi
    note "stop reported failure but service verified inactive; continuing"
  fi
  service_active && die "could not confirm $SERVICE stopped — refusing to restore files (nothing was written; backup $rid intact and recoverable)"
  run_cp "$bd/runtime.env" "$STATE_DIR/runtime.env"
  run_verchown_mode 600 "$STATE_DIR/runtime.env"
  run_rsync -a --delete --checksum "$bd/release/" "$RELEASE_DIR/" || die "restore release failed"
  service_start || die "service start after restore failed"
  sleep 2
  service_active || die "service not active after restore"
  wait_health || die "health failed after restore"
  verify_restored "$bd"
  msg "restored $rid"
}

# verify restored artifacts hash vs manifest
verify_restored() {
  bd=$1
  w=$(grep -E '^previous_worker_image=' "$bd/release-manifest.txt" | sed 's/^previous_worker_image=//') || die "manifest corrupt"
  [ "$(env_value WORKER_IMAGE)" = "$w" ] || die "restored env mismatches manifest"
  h=$(grep -E '^previous_server_sha256=' "$bd/release-manifest.txt" | sed 's/^previous_server_sha256=//')
  now=$(sha256sum "$RELEASE_DIR/strategy-lab-server" | awk '{print $1}')
  [ "$h" = "$now" ] || die "restored server hash mismatch: $now != $h"
  msg "restored artifacts verified (env + sha256)"
}

# ---------------- main deploy flow ----------------
main_deploy() {
  exec 9>"$STATE_DIR/deploy.lock"
  flock -n 9 || die "another deploy/rollback is running (deploy.lock held)"

  msg "phase 1/6 precheck (live service)"
  precheck_live

  # Backup BEFORE stopping: a backup failure must never turn into downtime —
  # the service keeps running on untouched artifacts and the deploy simply
  # refuses.
  msg "phase 2/6 backup (env + release artifacts, service still up)"
  DEPLOY_PHASE=backup; export DEPLOY_PHASE
  backup

  msg "phase 3/6 stop service (block NEW work)"
  service_stop || die "service stop failed"
  # R1: capture the REAL return code of the stopped-state checks — not an
  # &&-chain / if ! — so a `2` (= new work discovered) is distinguishable
  # from plain failure and the caller can restore the old service untouched.
  rc=0
  precheck_stopped || rc=$?
  # A DB/environment error (rc=2) after stop is treated the same as the race
  # branch: we cannot prove the system is quiet, so restore the old service
  # untouched rather than swapping files in an unverifiable state.
  if [ "$rc" -ne 0 ]; then
    if [ "$rc" = "2" ]; then
      "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED"
      note "restored old service WITHOUT replacement (post-stop check errored: DB/env unreadable)"
      note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID"
      exit 1
    fi
    # rc=1: new work arrived between checks; the backup above is discarded (it
    # recorded pre-inflight state and nothing was replaced)
    "$SYSTEMCTL_CMD" --user start "$SERVICE" || note "could NOT re-start old service — MANUAL CHECK NEEDED"
    note "restored old service WITHOUT replacement (new tasks preserved)"
    note "backup dir kept for inspection only: $BACKUP_ROOT/$RELEASE_ID (release kept; runtime.env already restored bytes)"
    exit 1
  fi

  msg "phase 4/6 swap (env WORKER_IMAGE + binary + dist)"
  # R1: each step rc-checked independently (also inside an && chain)
  DEPLOY_PHASE=swap; export DEPLOY_PHASE
  if ! set_worker_image_id; then
    note "env update failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1
  fi
  if ! swap_release; then
    note "release swap failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1
  fi

  msg "phase 5/6 start service + health"
  if ! "$SYSTEMCTL_CMD" --user start "$SERVICE"; then
    note "service start failed"; DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"; exit 1
  fi
  sleep 2
  if ! wait_health; then
    note "health failed; rolling back to $RELEASE_ID"
    DEPLOY_LOCK_HELD=1 restore_release_id "$RELEASE_ID"
    exit 1
  fi

  msg "phase 6/6 deploy $RELEASE_ID complete (image $IMAGE_ID)"
  msg "rollback id: $RELEASE_ID  (rollback.sh $RELEASE_ID)"
}

case "${1:-deploy}" in
  deploy)   main_deploy ;;
  restore)  shift; restore_release_id "$@" ;;
  precheck) precheck_live ;;
  *) die "usage: deploy.sh [deploy|restore <id>|precheck]" ;;
esac