fix: make online updates recoverable
TallyNote release / linux-x64 (push) Successful in 7m45s

This commit is contained in:
Qiufeng
2026-09-05 14:56:15 +08:00
parent a080f531cd
commit ed0b492461
12 changed files with 526 additions and 165 deletions
+110 -22
View File
@@ -10,6 +10,8 @@ DATA_DIR=${TALLYNOTE_DATA_DIR:-/var/lib/tallynote}
REQUEST_FILE="$DATA_DIR/update-request.json"
CURRENT_LINK="$PREFIX/current"
STATE_FILE="$PREFIX/.update-state"
LOCK_FILE="$PREFIX/.update-runner.lock"
RUNNER_LOG="$PREFIX/.update-runner.log"
SERVICE_NAME=${TALLYNOTE_SERVICE_NAME:-tallynote.service}
HOST=${TALLYNOTE_HOST:-127.0.0.1}
PORT=${TALLYNOTE_PORT:-3000}
@@ -19,13 +21,72 @@ if [[ "$HEALTH_HOST" == :: ]]; then HEALTH_HOST=::1; fi
if [[ "$HEALTH_HOST" == *:* && "$HEALTH_HOST" != \[* ]]; then HEALTH_HOST="[$HEALTH_HOST]"; fi
die() { printf 'tallynote update runner: %s\n' "$*" >&2; exit 1; }
# The runner may exit during any of the checks below. Install its EXIT cleanup
# before doing privileged preflight so a partial invocation never leaves a
# heartbeat or lock behind.
STATE_CREATED=0
heartbeat_pid=''
heartbeat_owner=$$
RUNNER_LOCK_FD=9
RUNNER_LOCK_MODE=''
stop_heartbeat() {
if [[ -n "$heartbeat_pid" ]]; then
kill "$heartbeat_pid" 2>/dev/null || true
wait "$heartbeat_pid" 2>/dev/null || true
heartbeat_pid=''
fi
}
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
release_runner_lock() {
if [[ "$RUNNER_LOCK_MODE" == flock ]]; then
flock -u "$RUNNER_LOCK_FD" 2>/dev/null || true
eval "exec ${RUNNER_LOCK_FD}>&-" 2>/dev/null || true
elif [[ "$RUNNER_LOCK_MODE" == mkdir ]]; then
rmdir -- "$LOCK_FILE.d" 2>/dev/null || true
fi
}
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
early_cleanup() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
# A preflight failure happens before the normal phase-specific trap is
# installed. Remove only the one-shot request marker; never remove an
# existing recovery marker unless this invocation created it.
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then rm -f -- "$STATE_FILE" 2>/dev/null || true; fi
fi
release_runner_lock
return "$result"
}
trap early_cleanup EXIT
[[ ${EUID:-$(id -u)} -eq 0 ]] || die 'must run as root'
[[ -d "$PREFIX" ]] || die 'install prefix is missing'
if command -v flock >/dev/null 2>&1; then
exec 9>"$LOCK_FILE" || die '无法打开更新运行锁'
flock -n "$RUNNER_LOCK_FD" || exit 0
RUNNER_LOCK_MODE=flock
else
# macOS development fixtures do not ship util-linux; retain an atomic lock
# fallback there while Linux production uses flock above.
mkdir "$LOCK_FILE.d" 2>/dev/null || exit 0
RUNNER_LOCK_MODE='mkdir'
fi
[[ -f "$REQUEST_FILE" || -f "$STATE_FILE" ]] || exit 0
[[ -L "$CURRENT_LINK" ]] || die 'current release link is missing'
old_target=$(readlink -f -- "$CURRENT_LINK")
[[ "$old_target" == "$PREFIX/releases/"* && -d "$old_target" ]] || die 'current release target is invalid'
# Capture the service state before any download/apply work. The value is
# persisted in the recovery marker so a later runner process can restore the
# operator's original state after a crash (the service is normally inactive by
# the time recovery starts).
was_active=0
if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
request_operation='apply'
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
request_operation=$(sed -n 's/.*"operation"[[:space:]]*:[[:space:]]*"\(download\|apply\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
@@ -39,15 +100,11 @@ job_id=''
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
fi
STATE_CREATED=0
heartbeat_pid=''
heartbeat_owner=$$
write_recovery_state() {
local phase=$1 temporary
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
printf 'job_id=%s\nold_target=%s\nphase=%s\ninitial_active=%s\n' "$job_id" "$old_target" "$phase" "$was_active" > "$temporary"
chmod 600 "$temporary"
mv -Tf -- "$temporary" "$STATE_FILE"
STATE_CREATED=1
@@ -59,14 +116,6 @@ clear_recovery_state() {
STATE_CREATED=0
}
stop_heartbeat() {
if [[ -n "$heartbeat_pid" ]]; then
kill "$heartbeat_pid" 2>/dev/null || true
wait "$heartbeat_pid" 2>/dev/null || true
heartbeat_pid=''
fi
}
heartbeat() {
# Keep the lease fresh during long downloads/backups, but stop on a hard
# runner kill so an orphaned child cannot keep the recovery marker alive.
@@ -86,9 +135,11 @@ start_heartbeat() {
# This trap covers failures before the normal apply cleanup trap is installed,
# including a missing runtime, an invalid current link, and a failed service
# stop. It deliberately does not remove a pre-existing recovery marker.
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
preflight_cleanup() {
local result=$?
stop_heartbeat
release_runner_lock
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
@@ -97,6 +148,33 @@ preflight_cleanup() {
}
trap preflight_cleanup EXIT
DOWNLOAD_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_DOWNLOAD_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_DOWNLOAD_TIMEOUT_SECONDS:-1800}}
APPLY_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_APPLY_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_APPLY_TIMEOUT_SECONDS:-1800}}
FINALIZE_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_FINALIZE_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_FINALIZE_TIMEOUT_SECONDS:-30}}
TIMEOUT_BIN=$(command -v timeout || true)
run_update_cli() {
local node=$1 timeout_seconds=$2 label=$3 result
shift 3
[[ "$timeout_seconds" =~ ^[1-9][0-9]*$ ]] || die "${label} timeout must be a positive integer"
{
printf '\n[%s] %s (timeout=%ss)\ncommand:' "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$label" "$timeout_seconds"
printf ' %q' "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@"
printf '\n'
} >>"$RUNNER_LOG"
if [[ -n "$TIMEOUT_BIN" ]]; then
"$TIMEOUT_BIN" --foreground --signal=TERM --kill-after=10s "${timeout_seconds}s" \
"$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" >>"$RUNNER_LOG" 2>&1
result=$?
elif "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" >>"$RUNNER_LOG" 2>&1; then
result=0
else
result=$?
fi
printf '[%s] %s exited with status %s\n' "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$label" "$result" >>"$RUNNER_LOG"
return "$result"
}
# Downloading is intentionally handled while the main service remains up.
# The CLI persists the validated payload under the root-owned workspace and
# leaves the job staged for a later apply request.
@@ -107,6 +185,7 @@ if [[ "$request_operation" == download ]]; then
clear_recovery_state || die '无法清理上一次下载状态'
fi
write_recovery_state download || die '无法写入更新恢复状态'
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap
cleanup_download() {
local result=$?
stop_heartbeat
@@ -116,6 +195,7 @@ if [[ "$request_operation" == download ]]; then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
fi
clear_recovery_state || true
release_runner_lock
return "$result"
}
trap cleanup_download EXIT
@@ -128,7 +208,7 @@ if [[ "$request_operation" == download ]]; then
cli="$CURRENT_LINK/dist/server/cli/update.js"
[[ -f "$cli" ]] || die 'update CLI not found in current release'
set +e
"$node_bin" "$cli" --request-file "$REQUEST_FILE"
run_update_cli "$node_bin" "$DOWNLOAD_TIMEOUT_SECONDS" download --request-file "$REQUEST_FILE"
download_result=$?
set -e
if (( download_result != 0 )); then
@@ -139,7 +219,7 @@ if [[ "$request_operation" == download ]]; then
download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then
for _ in 1 2 3; do
if "$node_bin" "$cli" --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败' >/dev/null 2>&1; then break; fi
if run_update_cli "$node_bin" "$FINALIZE_TIMEOUT_SECONDS" finalize-download --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败'; then break; fi
sleep 1
done
fi
@@ -150,12 +230,11 @@ if [[ "$request_operation" == download ]]; then
exit 0
fi
was_active=0
if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
restore_initial_service() {
local result=$?
stop_heartbeat
release_runner_lock
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
@@ -175,11 +254,11 @@ finalize_state_job() {
local node=$1 status=$2 state_job=$3
[[ "$state_job" =~ ^[0-9a-f-]{36}$ && -n "$node" ]] || return 1
[[ -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
"$node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1
run_update_cli "$node" "$FINALIZE_TIMEOUT_SECONDS" finalize-recovery --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本'
}
recover_stale_state() {
local state_job state_old state_phase current_target recovery_node rollback_link state_mode state_uid
local state_job state_old state_phase state_initial_active current_target recovery_node rollback_link state_mode state_uid
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || die 'update state file is invalid'
state_uid=$(stat -c '%u' "$STATE_FILE" 2>/dev/null || stat -f '%u' "$STATE_FILE")
state_mode=$(stat -c '%a' "$STATE_FILE" 2>/dev/null || stat -f '%Lp' "$STATE_FILE")
@@ -187,8 +266,16 @@ recover_stale_state() {
state_job=$(sed -n 's/^job_id=//p' "$STATE_FILE" | head -n 1)
state_old=$(sed -n 's/^old_target=//p' "$STATE_FILE" | head -n 1)
state_phase=$(sed -n 's/^phase=//p' "$STATE_FILE" | head -n 1)
state_initial_active=$(sed -n 's/^initial_active=//p' "$STATE_FILE" | head -n 1)
[[ "$state_job" =~ ^[0-9a-f-]{36}$ ]] || die 'update state job id is invalid'
[[ "$state_old" == "$PREFIX/releases/"* && -d "$state_old" && ! -L "$state_old" ]] || die 'update state target is invalid'
if [[ -z "$state_initial_active" ]]; then
# Markers from older releases did not persist this field. Preserve their
# historical conservative behavior instead of rejecting recovery.
state_initial_active=0
fi
[[ "$state_initial_active" == 0 || "$state_initial_active" == 1 ]] || die 'update state initial service state is invalid'
was_active=$state_initial_active
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
if [[ "$state_phase" == download && "$current_target" == "$state_old" ]]; then
# Downloading never changes the active release. If the runner was killed
@@ -312,7 +399,7 @@ finalize_failed_job() {
# Give SQLite a moment to release a transient lock before declaring the
# recovery itself failed.
for _ in 1 2 3; do
if "$old_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1; then
if run_update_cli "$old_node" "$FINALIZE_TIMEOUT_SECONDS" finalize-failed --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本'; then
return 0
fi
sleep 1
@@ -323,7 +410,7 @@ finalize_failed_job() {
finalize_completed_job() {
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
[[ -n "$final_node" ]] || return 1
"$final_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status completed >/dev/null 2>&1
run_update_cli "$final_node" "$FINALIZE_TIMEOUT_SECONDS" finalize-completed --finalize-job "$job_id" --finalize-status completed
}
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
@@ -347,6 +434,7 @@ cleanup_after_update() {
else
systemctl stop "$SERVICE_NAME" || true
fi
release_runner_lock
return "$result"
}
trap cleanup_after_update EXIT
@@ -358,7 +446,7 @@ cli="$CURRENT_LINK/dist/server/cli/update.js"
[[ -f "$cli" ]] || die 'update CLI not found in current release'
set +e
"$node_bin" "$cli" --request-file "$REQUEST_FILE" --defer-completion
run_update_cli "$node_bin" "$APPLY_TIMEOUT_SECONDS" apply --request-file "$REQUEST_FILE" --defer-completion
update_result=$?
set -e
if (( update_result != 0 )); then