515 lines
20 KiB
Bash
Executable File
515 lines
20 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
set -Eeuo pipefail
|
|
|
|
PATH=/usr/local/bin:/usr/bin:/usr/sbin:/sbin:/bin
|
|
export PATH
|
|
umask 077
|
|
|
|
PREFIX=${TALLYNOTE_INSTALL_PREFIX:-/opt/tallynote}
|
|
DATA_DIR=${TALLYNOTE_DATA_DIR:-/var/lib/tallynote}
|
|
CONFIG_DIR=${TALLYNOTE_CONFIG_DIR:-/etc/tallynote}
|
|
CONFIG_FILE="$CONFIG_DIR/tallynote.env"
|
|
REQUEST_FILE="$DATA_DIR/update-request.json"
|
|
CURRENT_LINK="$PREFIX/current"
|
|
STATE_FILE="$PREFIX/.update-state"
|
|
LOCK_FILE="$PREFIX/.update-runner.lock"
|
|
RUNNER_LOG="$PREFIX/.update-runner.log"
|
|
SERVICE_NAME=${TALLYNOTE_SERVICE_NAME:-tallynote.service}
|
|
HOST=${TALLYNOTE_HOST:-127.0.0.1}
|
|
PORT=${TALLYNOTE_PORT:-3000}
|
|
HEALTH_HOST=$HOST
|
|
if [[ "$HEALTH_HOST" == 0.0.0.0 ]]; then HEALTH_HOST=127.0.0.1; fi
|
|
if [[ "$HEALTH_HOST" == :: ]]; then HEALTH_HOST=::1; fi
|
|
if [[ "$HEALTH_HOST" == *:* && "$HEALTH_HOST" != \[* ]]; then HEALTH_HOST="[$HEALTH_HOST]"; fi
|
|
|
|
die() { printf 'tallynote update runner: %s\n' "$*" >&2; exit 1; }
|
|
|
|
node_is_usable() {
|
|
local candidate=$1 major
|
|
[[ -n "$candidate" && -x "$candidate" ]] || return 1
|
|
major=$("$candidate" -p 'process.versions.node.split(".")[0]' 2>/dev/null || true)
|
|
[[ "$major" =~ ^[0-9]+$ && "$major" -ge 24 ]]
|
|
}
|
|
|
|
resolve_node() {
|
|
local candidate=${TALLYNOTE_NODE:-}
|
|
if [[ -z "$candidate" && -f "$CONFIG_FILE" && ! -L "$CONFIG_FILE" ]]; then
|
|
candidate=$(sed -n 's/^TALLYNOTE_NODE=//p' "$CONFIG_FILE" | head -n 1)
|
|
fi
|
|
if node_is_usable "$candidate"; then
|
|
printf '%s' "$candidate"
|
|
return 0
|
|
fi
|
|
candidate=$(command -v node || true)
|
|
node_is_usable "$candidate" || return 1
|
|
printf '%s' "$candidate"
|
|
}
|
|
|
|
# The runner may exit during any of the checks below. Install its EXIT cleanup
|
|
# before doing privileged preflight so a partial invocation never leaves a
|
|
# heartbeat or lock behind.
|
|
STATE_CREATED=0
|
|
heartbeat_pid=''
|
|
heartbeat_owner=$$
|
|
RUNNER_LOCK_FD=9
|
|
RUNNER_LOCK_MODE=''
|
|
stop_heartbeat() {
|
|
if [[ -n "$heartbeat_pid" ]]; then
|
|
kill "$heartbeat_pid" 2>/dev/null || true
|
|
wait "$heartbeat_pid" 2>/dev/null || true
|
|
heartbeat_pid=''
|
|
fi
|
|
}
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
|
|
release_runner_lock() {
|
|
if [[ "$RUNNER_LOCK_MODE" == flock ]]; then
|
|
flock -u "$RUNNER_LOCK_FD" 2>/dev/null || true
|
|
eval "exec ${RUNNER_LOCK_FD}>&-" 2>/dev/null || true
|
|
elif [[ "$RUNNER_LOCK_MODE" == mkdir ]]; then
|
|
rmdir -- "$LOCK_FILE.d" 2>/dev/null || true
|
|
fi
|
|
}
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
|
|
early_cleanup() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
if (( result != 0 )); then
|
|
# A preflight failure happens before the normal phase-specific trap is
|
|
# installed. Remove only the one-shot request marker; never remove an
|
|
# existing recovery marker unless this invocation created it.
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
if (( STATE_CREATED == 1 )); then rm -f -- "$STATE_FILE" 2>/dev/null || true; fi
|
|
fi
|
|
release_runner_lock
|
|
return "$result"
|
|
}
|
|
trap early_cleanup EXIT
|
|
|
|
[[ ${EUID:-$(id -u)} -eq 0 ]] || die 'must run as root'
|
|
[[ -d "$PREFIX" ]] || die 'install prefix is missing'
|
|
if command -v flock >/dev/null 2>&1; then
|
|
exec 9>"$LOCK_FILE" || die '无法打开更新运行锁'
|
|
flock -n "$RUNNER_LOCK_FD" || exit 0
|
|
RUNNER_LOCK_MODE=flock
|
|
else
|
|
# macOS development fixtures do not ship util-linux; retain an atomic lock
|
|
# fallback there while Linux production uses flock above.
|
|
mkdir "$LOCK_FILE.d" 2>/dev/null || exit 0
|
|
RUNNER_LOCK_MODE='mkdir'
|
|
fi
|
|
[[ -f "$REQUEST_FILE" || -f "$STATE_FILE" ]] || exit 0
|
|
[[ -L "$CURRENT_LINK" ]] || die 'current release link is missing'
|
|
|
|
old_target=$(readlink -f -- "$CURRENT_LINK")
|
|
[[ "$old_target" == "$PREFIX/releases/"* && -d "$old_target" ]] || die 'current release target is invalid'
|
|
|
|
# Capture the service state before any download/apply work. The value is
|
|
# persisted in the recovery marker so a later runner process can restore the
|
|
# operator's original state after a crash (the service is normally inactive by
|
|
# the time recovery starts).
|
|
was_active=0
|
|
if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
|
|
|
|
request_operation='apply'
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
request_operation=$(sed -n 's/.*"operation"[[:space:]]*:[[:space:]]*"\(download\|apply\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
|
[[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply'
|
|
fi
|
|
|
|
# Capture the request id before any privileged preflight can fail. The
|
|
# request file is an application-owned one-shot marker; removing it on an
|
|
# early runner failure lets the server-side lease reaper release the DB row.
|
|
job_id=''
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
|
fi
|
|
write_recovery_state() {
|
|
local phase=$1 temporary
|
|
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
|
|
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
|
|
printf 'job_id=%s\nold_target=%s\nphase=%s\ninitial_active=%s\n' "$job_id" "$old_target" "$phase" "$was_active" > "$temporary"
|
|
chmod 600 "$temporary"
|
|
mv -Tf -- "$temporary" "$STATE_FILE"
|
|
STATE_CREATED=1
|
|
}
|
|
|
|
clear_recovery_state() {
|
|
[[ ! -L "$STATE_FILE" ]] || return 1
|
|
rm -f -- "$STATE_FILE"
|
|
STATE_CREATED=0
|
|
}
|
|
|
|
heartbeat() {
|
|
# Keep the lease fresh during long downloads/backups, but stop on a hard
|
|
# runner kill so an orphaned child cannot keep the recovery marker alive.
|
|
while kill -0 "$heartbeat_owner" 2>/dev/null; do
|
|
sleep 10 || exit 0
|
|
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0
|
|
touch "$STATE_FILE" 2>/dev/null || exit 0
|
|
done
|
|
}
|
|
|
|
start_heartbeat() {
|
|
stop_heartbeat
|
|
heartbeat &
|
|
heartbeat_pid=$!
|
|
}
|
|
|
|
# This trap covers failures before the normal apply cleanup trap is installed,
|
|
# including a missing runtime, an invalid current link, and a failed service
|
|
# stop. It deliberately does not remove a pre-existing recovery marker.
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
|
|
preflight_cleanup() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
release_runner_lock
|
|
if (( result != 0 )); then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
|
|
fi
|
|
return "$result"
|
|
}
|
|
trap preflight_cleanup EXIT
|
|
|
|
DOWNLOAD_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_DOWNLOAD_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_DOWNLOAD_TIMEOUT_SECONDS:-1800}}
|
|
APPLY_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_APPLY_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_APPLY_TIMEOUT_SECONDS:-1800}}
|
|
FINALIZE_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_FINALIZE_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_FINALIZE_TIMEOUT_SECONDS:-30}}
|
|
TIMEOUT_BIN=$(command -v timeout || true)
|
|
|
|
run_update_cli() {
|
|
local node=$1 timeout_seconds=$2 label=$3 result
|
|
shift 3
|
|
[[ "$timeout_seconds" =~ ^[1-9][0-9]*$ ]] || die "${label} timeout must be a positive integer"
|
|
{
|
|
printf '\n[%s] %s (timeout=%ss)\ncommand:' "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$label" "$timeout_seconds"
|
|
printf ' %q' "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@"
|
|
printf '\n'
|
|
} >>"$RUNNER_LOG"
|
|
if [[ -n "$TIMEOUT_BIN" ]]; then
|
|
"$TIMEOUT_BIN" --foreground --signal=TERM --kill-after=10s "${timeout_seconds}s" \
|
|
"$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" >>"$RUNNER_LOG" 2>&1
|
|
result=$?
|
|
elif "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" >>"$RUNNER_LOG" 2>&1; then
|
|
result=0
|
|
else
|
|
result=$?
|
|
fi
|
|
printf '[%s] %s exited with status %s\n' "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$label" "$result" >>"$RUNNER_LOG"
|
|
return "$result"
|
|
}
|
|
|
|
# Downloading is intentionally handled while the main service remains up.
|
|
# The CLI persists the validated payload under the root-owned workspace and
|
|
# leaves the job staged for a later apply request.
|
|
if [[ "$request_operation" == download ]]; then
|
|
# A previous download runner may have been interrupted after creating its
|
|
# marker. Clear only that download marker and retry the idempotent request.
|
|
if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then
|
|
clear_recovery_state || die '无法清理上一次下载状态'
|
|
fi
|
|
write_recovery_state download || die '无法写入更新恢复状态'
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
|
|
cleanup_download() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
if (( result != 0 )); then
|
|
# The CLI normally records failed itself. If it died before opening the
|
|
# database, the expired marker/request will be reconciled by the app.
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
fi
|
|
clear_recovery_state || true
|
|
release_runner_lock
|
|
return "$result"
|
|
}
|
|
trap cleanup_download EXIT
|
|
trap 'exit 143' TERM
|
|
trap 'exit 130' INT
|
|
start_heartbeat
|
|
node_bin=$(resolve_node) || die 'Node.js 24+ not found'
|
|
cli="$CURRENT_LINK/dist/server/cli/update.js"
|
|
[[ -f "$cli" ]] || die 'update CLI not found in current release'
|
|
set +e
|
|
run_update_cli "$node_bin" "$DOWNLOAD_TIMEOUT_SECONDS" download --request-file "$REQUEST_FILE"
|
|
download_result=$?
|
|
set -e
|
|
if (( download_result != 0 )); then
|
|
# The CLI normally records failed itself. Retry the explicit finalization
|
|
# for failures that happen before its catch handler can persist the row,
|
|
# then remove the one-shot request so a failed download cannot keep the
|
|
# path unit in a permanently triggered state.
|
|
download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
|
if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then
|
|
for _ in 1 2 3; do
|
|
if run_update_cli "$node_bin" "$FINALIZE_TIMEOUT_SECONDS" finalize-download --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败'; then break; fi
|
|
sleep 1
|
|
done
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
exit "$download_result"
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
exit 0
|
|
fi
|
|
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
|
|
restore_initial_service() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
release_runner_lock
|
|
if (( result != 0 )); then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
|
|
fi
|
|
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
|
|
return "$result"
|
|
}
|
|
trap restore_initial_service EXIT
|
|
old_node=$(resolve_node) || die 'Node.js 24+ not found'
|
|
handled=0
|
|
|
|
write_update_state() { write_recovery_state "$1"; }
|
|
clear_update_state() { clear_recovery_state; }
|
|
|
|
finalize_state_job() {
|
|
local node=$1 status=$2 state_job=$3
|
|
[[ "$state_job" =~ ^[0-9a-f-]{36}$ && -n "$node" ]] || return 1
|
|
[[ -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
|
|
run_update_cli "$node" "$FINALIZE_TIMEOUT_SECONDS" finalize-recovery --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本'
|
|
}
|
|
|
|
recover_stale_state() {
|
|
local state_job state_old state_phase state_initial_active current_target recovery_node rollback_link state_mode state_uid
|
|
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || die 'update state file is invalid'
|
|
state_uid=$(stat -c '%u' "$STATE_FILE" 2>/dev/null || stat -f '%u' "$STATE_FILE")
|
|
state_mode=$(stat -c '%a' "$STATE_FILE" 2>/dev/null || stat -f '%Lp' "$STATE_FILE")
|
|
[[ "$state_uid" == 0 && "$state_mode" =~ ^[0-7]+$ && $((8#$state_mode & 077)) -eq 0 ]] || die 'update state file permissions are invalid'
|
|
state_job=$(sed -n 's/^job_id=//p' "$STATE_FILE" | head -n 1)
|
|
state_old=$(sed -n 's/^old_target=//p' "$STATE_FILE" | head -n 1)
|
|
state_phase=$(sed -n 's/^phase=//p' "$STATE_FILE" | head -n 1)
|
|
state_initial_active=$(sed -n 's/^initial_active=//p' "$STATE_FILE" | head -n 1)
|
|
[[ "$state_job" =~ ^[0-9a-f-]{36}$ ]] || die 'update state job id is invalid'
|
|
[[ "$state_old" == "$PREFIX/releases/"* && -d "$state_old" && ! -L "$state_old" ]] || die 'update state target is invalid'
|
|
if [[ -z "$state_initial_active" ]]; then
|
|
# Markers from older releases did not persist this field. Preserve their
|
|
# historical conservative behavior instead of rejecting recovery.
|
|
state_initial_active=0
|
|
fi
|
|
[[ "$state_initial_active" == 0 || "$state_initial_active" == 1 ]] || die 'update state initial service state is invalid'
|
|
was_active=$state_initial_active
|
|
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
|
|
if [[ "$state_phase" == download && "$current_target" == "$state_old" ]]; then
|
|
# Downloading never changes the active release. If the runner was killed
|
|
# after the CLI staged its payload but before it removed the recovery
|
|
# marker, keep the request available for an idempotent retry. Treating
|
|
# every stale download marker as a failed apply would discard a usable
|
|
# staged payload and leave the browser showing a misleading failure.
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
if [[ "$state_phase" == finalizing && "$current_target" != "$state_old" ]]; then
|
|
recovery_node=$(resolve_node) || die 'Node.js 24+ not found'
|
|
for _ in 1 2 3; do
|
|
if finalize_state_job "$recovery_node" completed "$state_job"; then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
clear_update_state || true
|
|
return 10
|
|
fi
|
|
sleep 1
|
|
done
|
|
return 1
|
|
fi
|
|
if [[ "$current_target" == "$state_old" ]]; then
|
|
# The process may have restored the old release before it was killed. In
|
|
# that case the old link is already safe to serve, but the database row
|
|
# can still be `applying`; finish it as failed before clearing recovery
|
|
# markers so the UI does not poll forever.
|
|
recovery_node=$(resolve_node) || die 'Node.js 24+ not found'
|
|
if finalize_state_job "$recovery_node" failed "$state_job"; then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
clear_update_state || true
|
|
return 11
|
|
fi
|
|
# A crash before the CLI created its job row is safe to retry. Preserve
|
|
# the request while dropping only the stale state marker.
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
if [[ "$current_target" != "$state_old" ]]; then
|
|
rollback_link="$PREFIX/.current-recovery-$$-${RANDOM}.tmp"
|
|
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
|
|
ln -s -- "$state_old" "$rollback_link" || return 1
|
|
if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then
|
|
rm -f -- "$rollback_link" 2>/dev/null || true
|
|
return 1
|
|
fi
|
|
recovery_node=$(resolve_node) || die 'Node.js 24+ not found'
|
|
if ! finalize_state_job "$recovery_node" failed "$state_job"; then
|
|
# If the original queue is still present, retry it from the restored old
|
|
# release; a crash before the CLI wrote its job row is recoverable this
|
|
# way. Without a queue there is no safe operation to replay.
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
return 1
|
|
fi
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
clear_update_state || true
|
|
return 11
|
|
fi
|
|
clear_update_state || true
|
|
return 0
|
|
}
|
|
|
|
if [[ -e "$STATE_FILE" ]]; then
|
|
recovery_result=0
|
|
set +e
|
|
recover_stale_state
|
|
recovery_result=$?
|
|
set -e
|
|
case "$recovery_result" in
|
|
10) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 0 ;;
|
|
11) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;;
|
|
0) : ;;
|
|
*) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;;
|
|
esac
|
|
fi
|
|
|
|
[[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0
|
|
|
|
# Only create the marker for this invocation after any marker from a previous
|
|
# interrupted run has been reconciled. Otherwise the freshly-created `running`
|
|
# marker is indistinguishable from stale recovery state and the runner can
|
|
# finalize its own queued job as failed before the update CLI starts.
|
|
if [[ ! -e "$STATE_FILE" ]]; then
|
|
write_recovery_state running || die '无法写入更新恢复状态'
|
|
fi
|
|
start_heartbeat
|
|
if ! systemctl stop "$SERVICE_NAME"; then
|
|
die '无法停止 TallyNote 服务'
|
|
fi
|
|
|
|
rollback_current() {
|
|
local current_target rollback_link
|
|
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
|
|
if [[ "$current_target" == "$old_target" ]]; then
|
|
# An earlier failure branch may already have restored the link. Keep the
|
|
# marker truthful so the EXIT trap can still finalize the job.
|
|
return 0
|
|
fi
|
|
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
|
|
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
|
|
ln -s -- "$old_target" "$rollback_link" || return 1
|
|
if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then
|
|
rm -f -- "$rollback_link" 2>/dev/null || true
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
finalize_failed_job() {
|
|
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
|
|
[[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
|
|
# Give SQLite a moment to release a transient lock before declaring the
|
|
# recovery itself failed.
|
|
for _ in 1 2 3; do
|
|
if run_update_cli "$old_node" "$FINALIZE_TIMEOUT_SECONDS" finalize-failed --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本'; then
|
|
return 0
|
|
fi
|
|
sleep 1
|
|
done
|
|
return 1
|
|
}
|
|
|
|
finalize_completed_job() {
|
|
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
|
|
[[ -n "$final_node" ]] || return 1
|
|
run_update_cli "$final_node" "$FINALIZE_TIMEOUT_SECONDS" finalize-completed --finalize-job "$job_id" --finalize-status completed
|
|
}
|
|
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
|
|
cleanup_after_update() {
|
|
local result=$? rollback_ok=1
|
|
stop_heartbeat
|
|
if (( result != 0 && handled == 0 )); then
|
|
if ! rollback_current; then rollback_ok=0; fi
|
|
# Once the old release is active again, always try to close the job. The
|
|
# previous marker could remain set when an earlier branch had already
|
|
# rolled back before entering this EXIT trap, leaving `applying` forever.
|
|
if (( rollback_ok == 1 )); then
|
|
if finalize_failed_job; then
|
|
rm -f -- "$REQUEST_FILE"
|
|
clear_update_state || true
|
|
fi
|
|
fi
|
|
fi
|
|
if (( was_active )); then
|
|
systemctl start "$SERVICE_NAME" || true
|
|
else
|
|
systemctl stop "$SERVICE_NAME" || true
|
|
fi
|
|
release_runner_lock
|
|
return "$result"
|
|
}
|
|
trap cleanup_after_update EXIT
|
|
|
|
node_bin=$(resolve_node) || die 'Node.js 24+ not found'
|
|
cli="$CURRENT_LINK/dist/server/cli/update.js"
|
|
[[ -f "$cli" ]] || die 'update CLI not found in current release'
|
|
|
|
set +e
|
|
run_update_cli "$node_bin" "$APPLY_TIMEOUT_SECONDS" apply --request-file "$REQUEST_FILE" --defer-completion
|
|
update_result=$?
|
|
set -e
|
|
if (( update_result != 0 )); then
|
|
exit "$update_result"
|
|
fi
|
|
|
|
write_update_state health-check || exit 1
|
|
|
|
systemctl start "$SERVICE_NAME"
|
|
healthy=0
|
|
for _ in $(seq 1 30); do
|
|
if curl --proto '=http' --max-time 2 --silent --show-error "http://$HEALTH_HOST:$PORT/health" >/dev/null 2>&1; then healthy=1; break; fi
|
|
sleep 1
|
|
done
|
|
|
|
if (( healthy == 0 )); then
|
|
systemctl stop "$SERVICE_NAME" || true
|
|
rollback_current || die '无法恢复上一版本链接'
|
|
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
|
|
if ! finalize_failed_job; then
|
|
exit 1
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
clear_update_state || true
|
|
handled=1
|
|
trap - EXIT
|
|
exit 1
|
|
fi
|
|
|
|
# Preserve an operator's intentionally stopped service after validating the
|
|
# new release in a temporary start.
|
|
if (( was_active == 0 )); then
|
|
systemctl stop "$SERVICE_NAME"
|
|
fi
|
|
|
|
write_update_state finalizing || exit 1
|
|
final_node=$(resolve_node) || die 'Node.js 24+ not found'
|
|
if [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]]; then
|
|
finalized=0
|
|
for _ in 1 2 3; do
|
|
if finalize_completed_job; then finalized=1; break; fi
|
|
sleep 1
|
|
done
|
|
(( finalized == 1 )) || exit 1
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
clear_update_state || true
|
|
handled=1
|
|
trap - EXIT
|
|
exit 0
|