fix: prevent stuck background updates
TallyNote release / linux-x64 (push) Failing after 2m51s

This commit is contained in:
Qiufeng
2026-09-03 07:51:35 +08:00
parent ee89e04aae
commit 69a4b482ec
7 changed files with 266 additions and 47 deletions
+106 -26
View File
@@ -32,10 +32,96 @@ if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
[[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply'
fi
# Capture the request id before any privileged preflight can fail. The
# request file is an application-owned one-shot marker; removing it on an
# early runner failure lets the server-side lease reaper release the DB row.
job_id=''
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
fi
STATE_CREATED=0
heartbeat_pid=''
heartbeat_owner=$$
write_recovery_state() {
local phase=$1 temporary
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
chmod 600 "$temporary"
mv -Tf -- "$temporary" "$STATE_FILE"
STATE_CREATED=1
}
clear_recovery_state() {
[[ ! -L "$STATE_FILE" ]] || return 1
rm -f -- "$STATE_FILE"
STATE_CREATED=0
}
stop_heartbeat() {
if [[ -n "$heartbeat_pid" ]]; then
kill "$heartbeat_pid" 2>/dev/null || true
wait "$heartbeat_pid" 2>/dev/null || true
heartbeat_pid=''
fi
}
heartbeat() {
# Keep the lease fresh during long downloads/backups, but stop on a hard
# runner kill so an orphaned child cannot keep the recovery marker alive.
while kill -0 "$heartbeat_owner" 2>/dev/null; do
sleep 10 || exit 0
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0
touch "$STATE_FILE" 2>/dev/null || exit 0
done
}
start_heartbeat() {
stop_heartbeat
heartbeat &
heartbeat_pid=$!
}
# This trap covers failures before the normal apply cleanup trap is installed,
# including a missing runtime, an invalid current link, and a failed service
# stop. It deliberately does not remove a pre-existing recovery marker.
preflight_cleanup() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
fi
return "$result"
}
trap preflight_cleanup EXIT
# Downloading is intentionally handled while the main service remains up.
# The CLI persists the validated payload under the root-owned workspace and
# leaves the job staged for a later apply request.
if [[ "$request_operation" == download ]]; then
# A previous download runner may have been interrupted after creating its
# marker. Clear only that download marker and retry the idempotent request.
if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then
clear_recovery_state || die '无法清理上一次下载状态'
fi
write_recovery_state download || die '无法写入更新恢复状态'
cleanup_download() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
# The CLI normally records failed itself. If it died before opening the
# database, the expired marker/request will be reconciled by the app.
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
fi
clear_recovery_state || true
return "$result"
}
trap cleanup_download EXIT
trap 'exit 143' TERM
trap 'exit 130' INT
start_heartbeat
node_bin="$CURRENT_LINK/runtime/bin/node"
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
[[ -n "$node_bin" ]] || die 'node runtime not found'
@@ -69,34 +155,21 @@ if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
restore_initial_service() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
fi
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
return "$result"
}
trap restore_initial_service EXIT
systemctl stop "$SERVICE_NAME"
job_id=''
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
fi
old_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$old_node" ]] || old_node=$(command -v node || true)
switched=0
handled=0
write_update_state() {
local phase=$1 temporary
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
chmod 600 "$temporary"
mv -Tf -- "$temporary" "$STATE_FILE"
}
clear_update_state() {
[[ ! -L "$STATE_FILE" ]] || return 1
rm -f -- "$STATE_FILE"
}
write_update_state() { write_recovery_state "$1"; }
clear_update_state() { clear_recovery_state; }
finalize_state_job() {
local node=$1 status=$2 state_job=$3
@@ -195,13 +268,24 @@ fi
[[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0
# Only create the marker for this invocation after any marker from a previous
# interrupted run has been reconciled. Otherwise the freshly-created `running`
# marker is indistinguishable from stale recovery state and the runner can
# finalize its own queued job as failed before the update CLI starts.
if [[ ! -e "$STATE_FILE" ]]; then
write_recovery_state running || die '无法写入更新恢复状态'
fi
start_heartbeat
if ! systemctl stop "$SERVICE_NAME"; then
die '无法停止 TallyNote 服务'
fi
rollback_current() {
local current_target rollback_link
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
if [[ "$current_target" == "$old_target" ]]; then
# An earlier failure branch may already have restored the link. Keep the
# marker truthful so the EXIT trap can still finalize the job.
switched=0
return 0
fi
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
@@ -211,7 +295,6 @@ rollback_current() {
rm -f -- "$rollback_link" 2>/dev/null || true
return 1
fi
switched=0
}
finalize_failed_job() {
@@ -237,6 +320,7 @@ finalize_completed_job() {
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
cleanup_after_update() {
local result=$? rollback_ok=1
stop_heartbeat
if (( result != 0 && handled == 0 )); then
if ! rollback_current; then rollback_ok=0; fi
# Once the old release is active again, always try to close the job. The
@@ -258,7 +342,6 @@ cleanup_after_update() {
}
trap cleanup_after_update EXIT
write_update_state running || exit 1
node_bin="$CURRENT_LINK/runtime/bin/node"
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
[[ -n "$node_bin" ]] || die 'node runtime not found'
@@ -273,9 +356,6 @@ if (( update_result != 0 )); then
exit "$update_result"
fi
if [[ "$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)" != "$old_target" ]]; then
switched=1
fi
write_update_state health-check || exit 1
systemctl start "$SERVICE_NAME"