This commit is contained in:
@@ -32,10 +32,96 @@ if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
||||
[[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply'
|
||||
fi
|
||||
|
||||
# Capture the request id before any privileged preflight can fail. The
|
||||
# request file is an application-owned one-shot marker; removing it on an
|
||||
# early runner failure lets the server-side lease reaper release the DB row.
|
||||
job_id=''
|
||||
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
||||
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
||||
fi
|
||||
STATE_CREATED=0
|
||||
heartbeat_pid=''
|
||||
heartbeat_owner=$$
|
||||
|
||||
write_recovery_state() {
|
||||
local phase=$1 temporary
|
||||
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
|
||||
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
|
||||
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
|
||||
chmod 600 "$temporary"
|
||||
mv -Tf -- "$temporary" "$STATE_FILE"
|
||||
STATE_CREATED=1
|
||||
}
|
||||
|
||||
clear_recovery_state() {
|
||||
[[ ! -L "$STATE_FILE" ]] || return 1
|
||||
rm -f -- "$STATE_FILE"
|
||||
STATE_CREATED=0
|
||||
}
|
||||
|
||||
stop_heartbeat() {
|
||||
if [[ -n "$heartbeat_pid" ]]; then
|
||||
kill "$heartbeat_pid" 2>/dev/null || true
|
||||
wait "$heartbeat_pid" 2>/dev/null || true
|
||||
heartbeat_pid=''
|
||||
fi
|
||||
}
|
||||
|
||||
heartbeat() {
|
||||
# Keep the lease fresh during long downloads/backups, but stop on a hard
|
||||
# runner kill so an orphaned child cannot keep the recovery marker alive.
|
||||
while kill -0 "$heartbeat_owner" 2>/dev/null; do
|
||||
sleep 10 || exit 0
|
||||
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0
|
||||
touch "$STATE_FILE" 2>/dev/null || exit 0
|
||||
done
|
||||
}
|
||||
|
||||
start_heartbeat() {
|
||||
stop_heartbeat
|
||||
heartbeat &
|
||||
heartbeat_pid=$!
|
||||
}
|
||||
|
||||
# This trap covers failures before the normal apply cleanup trap is installed,
|
||||
# including a missing runtime, an invalid current link, and a failed service
|
||||
# stop. It deliberately does not remove a pre-existing recovery marker.
|
||||
preflight_cleanup() {
|
||||
local result=$?
|
||||
stop_heartbeat
|
||||
if (( result != 0 )); then
|
||||
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
||||
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
|
||||
fi
|
||||
return "$result"
|
||||
}
|
||||
trap preflight_cleanup EXIT
|
||||
|
||||
# Downloading is intentionally handled while the main service remains up.
|
||||
# The CLI persists the validated payload under the root-owned workspace and
|
||||
# leaves the job staged for a later apply request.
|
||||
if [[ "$request_operation" == download ]]; then
|
||||
# A previous download runner may have been interrupted after creating its
|
||||
# marker. Clear only that download marker and retry the idempotent request.
|
||||
if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then
|
||||
clear_recovery_state || die '无法清理上一次下载状态'
|
||||
fi
|
||||
write_recovery_state download || die '无法写入更新恢复状态'
|
||||
cleanup_download() {
|
||||
local result=$?
|
||||
stop_heartbeat
|
||||
if (( result != 0 )); then
|
||||
# The CLI normally records failed itself. If it died before opening the
|
||||
# database, the expired marker/request will be reconciled by the app.
|
||||
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
||||
fi
|
||||
clear_recovery_state || true
|
||||
return "$result"
|
||||
}
|
||||
trap cleanup_download EXIT
|
||||
trap 'exit 143' TERM
|
||||
trap 'exit 130' INT
|
||||
start_heartbeat
|
||||
node_bin="$CURRENT_LINK/runtime/bin/node"
|
||||
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
|
||||
[[ -n "$node_bin" ]] || die 'node runtime not found'
|
||||
@@ -69,34 +155,21 @@ if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
|
||||
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
|
||||
restore_initial_service() {
|
||||
local result=$?
|
||||
stop_heartbeat
|
||||
if (( result != 0 )); then
|
||||
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
||||
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
|
||||
fi
|
||||
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
|
||||
return "$result"
|
||||
}
|
||||
trap restore_initial_service EXIT
|
||||
systemctl stop "$SERVICE_NAME"
|
||||
|
||||
job_id=''
|
||||
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
||||
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
||||
fi
|
||||
old_node="$CURRENT_LINK/runtime/bin/node"
|
||||
[[ -x "$old_node" ]] || old_node=$(command -v node || true)
|
||||
switched=0
|
||||
handled=0
|
||||
|
||||
write_update_state() {
|
||||
local phase=$1 temporary
|
||||
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
|
||||
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
|
||||
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
|
||||
chmod 600 "$temporary"
|
||||
mv -Tf -- "$temporary" "$STATE_FILE"
|
||||
}
|
||||
|
||||
clear_update_state() {
|
||||
[[ ! -L "$STATE_FILE" ]] || return 1
|
||||
rm -f -- "$STATE_FILE"
|
||||
}
|
||||
write_update_state() { write_recovery_state "$1"; }
|
||||
clear_update_state() { clear_recovery_state; }
|
||||
|
||||
finalize_state_job() {
|
||||
local node=$1 status=$2 state_job=$3
|
||||
@@ -195,13 +268,24 @@ fi
|
||||
|
||||
[[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0
|
||||
|
||||
# Only create the marker for this invocation after any marker from a previous
|
||||
# interrupted run has been reconciled. Otherwise the freshly-created `running`
|
||||
# marker is indistinguishable from stale recovery state and the runner can
|
||||
# finalize its own queued job as failed before the update CLI starts.
|
||||
if [[ ! -e "$STATE_FILE" ]]; then
|
||||
write_recovery_state running || die '无法写入更新恢复状态'
|
||||
fi
|
||||
start_heartbeat
|
||||
if ! systemctl stop "$SERVICE_NAME"; then
|
||||
die '无法停止 TallyNote 服务'
|
||||
fi
|
||||
|
||||
rollback_current() {
|
||||
local current_target rollback_link
|
||||
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
|
||||
if [[ "$current_target" == "$old_target" ]]; then
|
||||
# An earlier failure branch may already have restored the link. Keep the
|
||||
# marker truthful so the EXIT trap can still finalize the job.
|
||||
switched=0
|
||||
return 0
|
||||
fi
|
||||
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
|
||||
@@ -211,7 +295,6 @@ rollback_current() {
|
||||
rm -f -- "$rollback_link" 2>/dev/null || true
|
||||
return 1
|
||||
fi
|
||||
switched=0
|
||||
}
|
||||
|
||||
finalize_failed_job() {
|
||||
@@ -237,6 +320,7 @@ finalize_completed_job() {
|
||||
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
|
||||
cleanup_after_update() {
|
||||
local result=$? rollback_ok=1
|
||||
stop_heartbeat
|
||||
if (( result != 0 && handled == 0 )); then
|
||||
if ! rollback_current; then rollback_ok=0; fi
|
||||
# Once the old release is active again, always try to close the job. The
|
||||
@@ -258,7 +342,6 @@ cleanup_after_update() {
|
||||
}
|
||||
trap cleanup_after_update EXIT
|
||||
|
||||
write_update_state running || exit 1
|
||||
node_bin="$CURRENT_LINK/runtime/bin/node"
|
||||
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
|
||||
[[ -n "$node_bin" ]] || die 'node runtime not found'
|
||||
@@ -273,9 +356,6 @@ if (( update_result != 0 )); then
|
||||
exit "$update_result"
|
||||
fi
|
||||
|
||||
if [[ "$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)" != "$old_target" ]]; then
|
||||
switched=1
|
||||
fi
|
||||
write_update_state health-check || exit 1
|
||||
|
||||
systemctl start "$SERVICE_NAME"
|
||||
|
||||
Reference in New Issue
Block a user