Files
TallyNote/scripts/tallynote-update-runner.sh
T
Qiufeng 91621df6c4
TallyNote release / linux-x64 (push) Successful in 6m34s
fix: recover stuck update queue
2026-09-03 12:39:44 +08:00

413 lines
15 KiB
Bash
Executable File

#!/usr/bin/env bash
set -Eeuo pipefail
PATH=/usr/sbin:/usr/bin:/sbin:/bin
export PATH
umask 077
PREFIX=${TALLYNOTE_INSTALL_PREFIX:-/opt/tallynote}
DATA_DIR=${TALLYNOTE_DATA_DIR:-/var/lib/tallynote}
REQUEST_FILE="$DATA_DIR/update-request.json"
CURRENT_LINK="$PREFIX/current"
STATE_FILE="$PREFIX/.update-state"
SERVICE_NAME=${TALLYNOTE_SERVICE_NAME:-tallynote.service}
HOST=${TALLYNOTE_HOST:-127.0.0.1}
PORT=${TALLYNOTE_PORT:-3000}
HEALTH_HOST=$HOST
if [[ "$HEALTH_HOST" == 0.0.0.0 ]]; then HEALTH_HOST=127.0.0.1; fi
if [[ "$HEALTH_HOST" == :: ]]; then HEALTH_HOST=::1; fi
if [[ "$HEALTH_HOST" == *:* && "$HEALTH_HOST" != \[* ]]; then HEALTH_HOST="[$HEALTH_HOST]"; fi
die() { printf 'tallynote update runner: %s\n' "$*" >&2; exit 1; }
[[ ${EUID:-$(id -u)} -eq 0 ]] || die 'must run as root'
[[ -f "$REQUEST_FILE" || -f "$STATE_FILE" ]] || exit 0
[[ -L "$CURRENT_LINK" ]] || die 'current release link is missing'
old_target=$(readlink -f -- "$CURRENT_LINK")
[[ "$old_target" == "$PREFIX/releases/"* && -d "$old_target" ]] || die 'current release target is invalid'
request_operation='apply'
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
request_operation=$(sed -n 's/.*"operation"[[:space:]]*:[[:space:]]*"\(download\|apply\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
[[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply'
fi
# Capture the request id before any privileged preflight can fail. The
# request file is an application-owned one-shot marker; removing it on an
# early runner failure lets the server-side lease reaper release the DB row.
job_id=''
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
fi
STATE_CREATED=0
heartbeat_pid=''
heartbeat_owner=$$
write_recovery_state() {
local phase=$1 temporary
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
chmod 600 "$temporary"
mv -Tf -- "$temporary" "$STATE_FILE"
STATE_CREATED=1
}
clear_recovery_state() {
[[ ! -L "$STATE_FILE" ]] || return 1
rm -f -- "$STATE_FILE"
STATE_CREATED=0
}
stop_heartbeat() {
if [[ -n "$heartbeat_pid" ]]; then
kill "$heartbeat_pid" 2>/dev/null || true
wait "$heartbeat_pid" 2>/dev/null || true
heartbeat_pid=''
fi
}
heartbeat() {
# Keep the lease fresh during long downloads/backups, but stop on a hard
# runner kill so an orphaned child cannot keep the recovery marker alive.
while kill -0 "$heartbeat_owner" 2>/dev/null; do
sleep 10 || exit 0
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0
touch "$STATE_FILE" 2>/dev/null || exit 0
done
}
start_heartbeat() {
stop_heartbeat
heartbeat &
heartbeat_pid=$!
}
# This trap covers failures before the normal apply cleanup trap is installed,
# including a missing runtime, an invalid current link, and a failed service
# stop. It deliberately does not remove a pre-existing recovery marker.
preflight_cleanup() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
fi
return "$result"
}
trap preflight_cleanup EXIT
# Downloading is intentionally handled while the main service remains up.
# The CLI persists the validated payload under the root-owned workspace and
# leaves the job staged for a later apply request.
if [[ "$request_operation" == download ]]; then
# A previous download runner may have been interrupted after creating its
# marker. Clear only that download marker and retry the idempotent request.
if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then
clear_recovery_state || die '无法清理上一次下载状态'
fi
write_recovery_state download || die '无法写入更新恢复状态'
cleanup_download() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
# The CLI normally records failed itself. If it died before opening the
# database, the expired marker/request will be reconciled by the app.
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
fi
clear_recovery_state || true
return "$result"
}
trap cleanup_download EXIT
trap 'exit 143' TERM
trap 'exit 130' INT
start_heartbeat
node_bin="$CURRENT_LINK/runtime/bin/node"
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
[[ -n "$node_bin" ]] || die 'node runtime not found'
cli="$CURRENT_LINK/dist/server/cli/update.js"
[[ -f "$cli" ]] || die 'update CLI not found in current release'
set +e
"$node_bin" "$cli" --request-file "$REQUEST_FILE"
download_result=$?
set -e
if (( download_result != 0 )); then
# The CLI normally records failed itself. Retry the explicit finalization
# for failures that happen before its catch handler can persist the row,
# then remove the one-shot request so a failed download cannot keep the
# path unit in a permanently triggered state.
download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then
for _ in 1 2 3; do
if "$node_bin" "$cli" --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败' >/dev/null 2>&1; then break; fi
sleep 1
done
fi
rm -f -- "$REQUEST_FILE"
exit "$download_result"
fi
rm -f -- "$REQUEST_FILE"
exit 0
fi
was_active=0
if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
restore_initial_service() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
fi
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
return "$result"
}
trap restore_initial_service EXIT
old_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$old_node" ]] || old_node=$(command -v node || true)
handled=0
write_update_state() { write_recovery_state "$1"; }
clear_update_state() { clear_recovery_state; }
finalize_state_job() {
local node=$1 status=$2 state_job=$3
[[ "$state_job" =~ ^[0-9a-f-]{36}$ && -n "$node" ]] || return 1
[[ -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
"$node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1
}
recover_stale_state() {
local state_job state_old state_phase current_target recovery_node rollback_link state_mode state_uid
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || die 'update state file is invalid'
state_uid=$(stat -c '%u' "$STATE_FILE" 2>/dev/null || stat -f '%u' "$STATE_FILE")
state_mode=$(stat -c '%a' "$STATE_FILE" 2>/dev/null || stat -f '%Lp' "$STATE_FILE")
[[ "$state_uid" == 0 && "$state_mode" =~ ^[0-7]+$ && $((8#$state_mode & 077)) -eq 0 ]] || die 'update state file permissions are invalid'
state_job=$(sed -n 's/^job_id=//p' "$STATE_FILE" | head -n 1)
state_old=$(sed -n 's/^old_target=//p' "$STATE_FILE" | head -n 1)
state_phase=$(sed -n 's/^phase=//p' "$STATE_FILE" | head -n 1)
[[ "$state_job" =~ ^[0-9a-f-]{36}$ ]] || die 'update state job id is invalid'
[[ "$state_old" == "$PREFIX/releases/"* && -d "$state_old" && ! -L "$state_old" ]] || die 'update state target is invalid'
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
if [[ "$state_phase" == download && "$current_target" == "$state_old" ]]; then
# Downloading never changes the active release. If the runner was killed
# after the CLI staged its payload but before it removed the recovery
# marker, keep the request available for an idempotent retry. Treating
# every stale download marker as a failed apply would discard a usable
# staged payload and leave the browser showing a misleading failure.
clear_update_state || true
return 0
fi
if [[ "$state_phase" == finalizing && "$current_target" != "$state_old" ]]; then
recovery_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
for _ in 1 2 3; do
if finalize_state_job "$recovery_node" completed "$state_job"; then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
clear_update_state || true
return 10
fi
sleep 1
done
return 1
fi
if [[ "$current_target" == "$state_old" ]]; then
# The process may have restored the old release before it was killed. In
# that case the old link is already safe to serve, but the database row
# can still be `applying`; finish it as failed before clearing recovery
# markers so the UI does not poll forever.
recovery_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
if finalize_state_job "$recovery_node" failed "$state_job"; then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
clear_update_state || true
return 11
fi
# A crash before the CLI created its job row is safe to retry. Preserve
# the request while dropping only the stale state marker.
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
clear_update_state || true
return 0
fi
clear_update_state || true
return 0
fi
if [[ "$current_target" != "$state_old" ]]; then
rollback_link="$PREFIX/.current-recovery-$$-${RANDOM}.tmp"
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
ln -s -- "$state_old" "$rollback_link" || return 1
if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then
rm -f -- "$rollback_link" 2>/dev/null || true
return 1
fi
recovery_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
if ! finalize_state_job "$recovery_node" failed "$state_job"; then
# If the original queue is still present, retry it from the restored old
# release; a crash before the CLI wrote its job row is recoverable this
# way. Without a queue there is no safe operation to replay.
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
clear_update_state || true
return 0
fi
return 1
fi
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
clear_update_state || true
return 11
fi
clear_update_state || true
return 0
}
if [[ -e "$STATE_FILE" ]]; then
recovery_result=0
set +e
recover_stale_state
recovery_result=$?
set -e
case "$recovery_result" in
10) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 0 ;;
11) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;;
0) : ;;
*) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;;
esac
fi
[[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0
# Only create the marker for this invocation after any marker from a previous
# interrupted run has been reconciled. Otherwise the freshly-created `running`
# marker is indistinguishable from stale recovery state and the runner can
# finalize its own queued job as failed before the update CLI starts.
if [[ ! -e "$STATE_FILE" ]]; then
write_recovery_state running || die '无法写入更新恢复状态'
fi
start_heartbeat
if ! systemctl stop "$SERVICE_NAME"; then
die '无法停止 TallyNote 服务'
fi
rollback_current() {
local current_target rollback_link
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
if [[ "$current_target" == "$old_target" ]]; then
# An earlier failure branch may already have restored the link. Keep the
# marker truthful so the EXIT trap can still finalize the job.
return 0
fi
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
ln -s -- "$old_target" "$rollback_link" || return 1
if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then
rm -f -- "$rollback_link" 2>/dev/null || true
return 1
fi
}
finalize_failed_job() {
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
[[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
# Give SQLite a moment to release a transient lock before declaring the
# recovery itself failed.
for _ in 1 2 3; do
if "$old_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1; then
return 0
fi
sleep 1
done
return 1
}
finalize_completed_job() {
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
[[ -n "$final_node" ]] || return 1
"$final_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status completed >/dev/null 2>&1
}
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
cleanup_after_update() {
local result=$? rollback_ok=1
stop_heartbeat
if (( result != 0 && handled == 0 )); then
if ! rollback_current; then rollback_ok=0; fi
# Once the old release is active again, always try to close the job. The
# previous marker could remain set when an earlier branch had already
# rolled back before entering this EXIT trap, leaving `applying` forever.
if (( rollback_ok == 1 )); then
if finalize_failed_job; then
rm -f -- "$REQUEST_FILE"
clear_update_state || true
fi
fi
fi
if (( was_active )); then
systemctl start "$SERVICE_NAME" || true
else
systemctl stop "$SERVICE_NAME" || true
fi
return "$result"
}
trap cleanup_after_update EXIT
node_bin="$CURRENT_LINK/runtime/bin/node"
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
[[ -n "$node_bin" ]] || die 'node runtime not found'
cli="$CURRENT_LINK/dist/server/cli/update.js"
[[ -f "$cli" ]] || die 'update CLI not found in current release'
set +e
"$node_bin" "$cli" --request-file "$REQUEST_FILE" --defer-completion
update_result=$?
set -e
if (( update_result != 0 )); then
exit "$update_result"
fi
write_update_state health-check || exit 1
systemctl start "$SERVICE_NAME"
healthy=0
for _ in $(seq 1 30); do
if curl --proto '=http' --max-time 2 --silent --show-error "http://$HEALTH_HOST:$PORT/health" >/dev/null 2>&1; then healthy=1; break; fi
sleep 1
done
if (( healthy == 0 )); then
systemctl stop "$SERVICE_NAME" || true
rollback_current || die '无法恢复上一版本链接'
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
if ! finalize_failed_job; then
exit 1
fi
rm -f -- "$REQUEST_FILE"
clear_update_state || true
handled=1
trap - EXIT
exit 1
fi
# Preserve an operator's intentionally stopped service after validating the
# new release in a temporary start.
if (( was_active == 0 )); then
systemctl stop "$SERVICE_NAME"
fi
write_update_state finalizing || exit 1
final_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$final_node" ]] || final_node=$(command -v node || true)
if [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]]; then
finalized=0
for _ in 1 2 3; do
if finalize_completed_job; then finalized=1; break; fi
sleep 1
done
(( finalized == 1 )) || exit 1
fi
rm -f -- "$REQUEST_FILE"
clear_update_state || true
handled=1
trap - EXIT
exit 0