fix: recover stuck background updates
TallyNote release / linux-x64 (push) Successful in 6m3s

This commit is contained in:
Qiufeng
2026-09-03 06:49:48 +08:00
parent b431fe167e
commit ee89e04aae
11 changed files with 331 additions and 36 deletions
+60 -5
View File
@@ -41,7 +41,25 @@ if [[ "$request_operation" == download ]]; then
[[ -n "$node_bin" ]] || die 'node runtime not found'
cli="$CURRENT_LINK/dist/server/cli/update.js"
[[ -f "$cli" ]] || die 'update CLI not found in current release'
"$node_bin" "$cli" --request-file "$REQUEST_FILE" || exit $?
set +e
"$node_bin" "$cli" --request-file "$REQUEST_FILE"
download_result=$?
set -e
if (( download_result != 0 )); then
# The CLI normally records failed itself. Retry the explicit finalization
# for failures that happen before its catch handler can persist the row,
# then remove the one-shot request so a failed download cannot keep the
# path unit in a permanently triggered state.
download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then
for _ in 1 2 3; do
if "$node_bin" "$cli" --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败' >/dev/null 2>&1; then break; fi
sleep 1
done
fi
rm -f -- "$REQUEST_FILE"
exit "$download_result"
fi
rm -f -- "$REQUEST_FILE"
exit 0
fi
@@ -112,6 +130,27 @@ recover_stale_state() {
done
return 1
fi
if [[ "$current_target" == "$state_old" ]]; then
# The process may have restored the old release before it was killed. In
# that case the old link is already safe to serve, but the database row
# can still be `applying`; finish it as failed before clearing recovery
# markers so the UI does not poll forever.
recovery_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
if finalize_state_job "$recovery_node" failed "$state_job"; then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
clear_update_state || true
return 11
fi
# A crash before the CLI created its job row is safe to retry. Preserve
# the request while dropping only the stale state marker.
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
clear_update_state || true
return 0
fi
clear_update_state || true
return 0
fi
if [[ "$current_target" != "$state_old" ]]; then
rollback_link="$PREFIX/.current-recovery-$$-${RANDOM}.tmp"
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
@@ -159,7 +198,12 @@ fi
rollback_current() {
local current_target rollback_link
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
[[ "$current_target" == "$old_target" ]] && return 0
if [[ "$current_target" == "$old_target" ]]; then
# An earlier failure branch may already have restored the link. Keep the
# marker truthful so the EXIT trap can still finalize the job.
switched=0
return 0
fi
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
ln -s -- "$old_target" "$rollback_link" || return 1
@@ -172,8 +216,16 @@ rollback_current() {
finalize_failed_job() {
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
[[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 0
"$old_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1
[[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
# Give SQLite a moment to release a transient lock before declaring the
# recovery itself failed.
for _ in 1 2 3; do
if "$old_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1; then
return 0
fi
sleep 1
done
return 1
}
finalize_completed_job() {
@@ -187,7 +239,10 @@ cleanup_after_update() {
local result=$? rollback_ok=1
if (( result != 0 && handled == 0 )); then
if ! rollback_current; then rollback_ok=0; fi
if (( rollback_ok == 1 && switched == 0 )); then
# Once the old release is active again, always try to close the job. The
# previous marker could remain set when an earlier branch had already
# rolled back before entering this EXIT trap, leaving `applying` forever.
if (( rollback_ok == 1 )); then
if finalize_failed_job; then
rm -f -- "$REQUEST_FILE"
clear_update_state || true