#!/usr/bin/env bash set -Eeuo pipefail PATH=/usr/sbin:/usr/bin:/sbin:/bin export PATH umask 077 PREFIX=${TALLYNOTE_INSTALL_PREFIX:-/opt/tallynote} DATA_DIR=${TALLYNOTE_DATA_DIR:-/var/lib/tallynote} REQUEST_FILE="$DATA_DIR/update-request.json" CURRENT_LINK="$PREFIX/current" STATE_FILE="$PREFIX/.update-state" SERVICE_NAME=${TALLYNOTE_SERVICE_NAME:-tallynote.service} HOST=${TALLYNOTE_HOST:-127.0.0.1} PORT=${TALLYNOTE_PORT:-3000} HEALTH_HOST=$HOST if [[ "$HEALTH_HOST" == 0.0.0.0 ]]; then HEALTH_HOST=127.0.0.1; fi if [[ "$HEALTH_HOST" == :: ]]; then HEALTH_HOST=::1; fi if [[ "$HEALTH_HOST" == *:* && "$HEALTH_HOST" != \[* ]]; then HEALTH_HOST="[$HEALTH_HOST]"; fi die() { printf 'tallynote update runner: %s\n' "$*" >&2; exit 1; } [[ ${EUID:-$(id -u)} -eq 0 ]] || die 'must run as root' [[ -f "$REQUEST_FILE" || -f "$STATE_FILE" ]] || exit 0 [[ -L "$CURRENT_LINK" ]] || die 'current release link is missing' old_target=$(readlink -f -- "$CURRENT_LINK") [[ "$old_target" == "$PREFIX/releases/"* && -d "$old_target" ]] || die 'current release target is invalid' request_operation='apply' if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then request_operation=$(sed -n 's/.*"operation"[[:space:]]*:[[:space:]]*"\(download\|apply\)".*/\1/p' "$REQUEST_FILE" | head -n 1) [[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply' fi # Capture the request id before any privileged preflight can fail. The # request file is an application-owned one-shot marker; removing it on an # early runner failure lets the server-side lease reaper release the DB row. job_id='' if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1) fi STATE_CREATED=0 heartbeat_pid='' heartbeat_owner=$$ write_recovery_state() { local phase=$1 temporary temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp" [[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1 printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary" chmod 600 "$temporary" mv -Tf -- "$temporary" "$STATE_FILE" STATE_CREATED=1 } clear_recovery_state() { [[ ! -L "$STATE_FILE" ]] || return 1 rm -f -- "$STATE_FILE" STATE_CREATED=0 } stop_heartbeat() { if [[ -n "$heartbeat_pid" ]]; then kill "$heartbeat_pid" 2>/dev/null || true wait "$heartbeat_pid" 2>/dev/null || true heartbeat_pid='' fi } heartbeat() { # Keep the lease fresh during long downloads/backups, but stop on a hard # runner kill so an orphaned child cannot keep the recovery marker alive. while kill -0 "$heartbeat_owner" 2>/dev/null; do sleep 10 || exit 0 [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0 touch "$STATE_FILE" 2>/dev/null || exit 0 done } start_heartbeat() { stop_heartbeat heartbeat & heartbeat_pid=$! } # This trap covers failures before the normal apply cleanup trap is installed, # including a missing runtime, an invalid current link, and a failed service # stop. It deliberately does not remove a pre-existing recovery marker. preflight_cleanup() { local result=$? stop_heartbeat if (( result != 0 )); then rm -f -- "$REQUEST_FILE" 2>/dev/null || true if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi fi return "$result" } trap preflight_cleanup EXIT # Downloading is intentionally handled while the main service remains up. # The CLI persists the validated payload under the root-owned workspace and # leaves the job staged for a later apply request. if [[ "$request_operation" == download ]]; then # A previous download runner may have been interrupted after creating its # marker. Clear only that download marker and retry the idempotent request. if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then clear_recovery_state || die '无法清理上一次下载状态' fi write_recovery_state download || die '无法写入更新恢复状态' cleanup_download() { local result=$? stop_heartbeat if (( result != 0 )); then # The CLI normally records failed itself. If it died before opening the # database, the expired marker/request will be reconciled by the app. rm -f -- "$REQUEST_FILE" 2>/dev/null || true fi clear_recovery_state || true return "$result" } trap cleanup_download EXIT trap 'exit 143' TERM trap 'exit 130' INT start_heartbeat node_bin="$CURRENT_LINK/runtime/bin/node" [[ -x "$node_bin" ]] || node_bin=$(command -v node || true) [[ -n "$node_bin" ]] || die 'node runtime not found' cli="$CURRENT_LINK/dist/server/cli/update.js" [[ -f "$cli" ]] || die 'update CLI not found in current release' set +e "$node_bin" "$cli" --request-file "$REQUEST_FILE" download_result=$? set -e if (( download_result != 0 )); then # The CLI normally records failed itself. Retry the explicit finalization # for failures that happen before its catch handler can persist the row, # then remove the one-shot request so a failed download cannot keep the # path unit in a permanently triggered state. download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1) if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then for _ in 1 2 3; do if "$node_bin" "$cli" --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败' >/dev/null 2>&1; then break; fi sleep 1 done fi rm -f -- "$REQUEST_FILE" exit "$download_result" fi rm -f -- "$REQUEST_FILE" exit 0 fi was_active=0 if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below restore_initial_service() { local result=$? stop_heartbeat if (( result != 0 )); then rm -f -- "$REQUEST_FILE" 2>/dev/null || true if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi fi if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi return "$result" } trap restore_initial_service EXIT old_node="$CURRENT_LINK/runtime/bin/node" [[ -x "$old_node" ]] || old_node=$(command -v node || true) handled=0 write_update_state() { write_recovery_state "$1"; } clear_update_state() { clear_recovery_state; } finalize_state_job() { local node=$1 status=$2 state_job=$3 [[ "$state_job" =~ ^[0-9a-f-]{36}$ && -n "$node" ]] || return 1 [[ -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1 "$node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1 } recover_stale_state() { local state_job state_old state_phase current_target recovery_node rollback_link state_mode state_uid [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || die 'update state file is invalid' state_uid=$(stat -c '%u' "$STATE_FILE" 2>/dev/null || stat -f '%u' "$STATE_FILE") state_mode=$(stat -c '%a' "$STATE_FILE" 2>/dev/null || stat -f '%Lp' "$STATE_FILE") [[ "$state_uid" == 0 && "$state_mode" =~ ^[0-7]+$ && $((8#$state_mode & 077)) -eq 0 ]] || die 'update state file permissions are invalid' state_job=$(sed -n 's/^job_id=//p' "$STATE_FILE" | head -n 1) state_old=$(sed -n 's/^old_target=//p' "$STATE_FILE" | head -n 1) state_phase=$(sed -n 's/^phase=//p' "$STATE_FILE" | head -n 1) [[ "$state_job" =~ ^[0-9a-f-]{36}$ ]] || die 'update state job id is invalid' [[ "$state_old" == "$PREFIX/releases/"* && -d "$state_old" && ! -L "$state_old" ]] || die 'update state target is invalid' current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true) if [[ "$state_phase" == finalizing && "$current_target" != "$state_old" ]]; then recovery_node="$CURRENT_LINK/runtime/bin/node" [[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true) for _ in 1 2 3; do if finalize_state_job "$recovery_node" completed "$state_job"; then rm -f -- "$REQUEST_FILE" 2>/dev/null || true clear_update_state || true return 10 fi sleep 1 done return 1 fi if [[ "$current_target" == "$state_old" ]]; then # The process may have restored the old release before it was killed. In # that case the old link is already safe to serve, but the database row # can still be `applying`; finish it as failed before clearing recovery # markers so the UI does not poll forever. recovery_node="$CURRENT_LINK/runtime/bin/node" [[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true) if finalize_state_job "$recovery_node" failed "$state_job"; then rm -f -- "$REQUEST_FILE" 2>/dev/null || true clear_update_state || true return 11 fi # A crash before the CLI created its job row is safe to retry. Preserve # the request while dropping only the stale state marker. if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then clear_update_state || true return 0 fi clear_update_state || true return 0 fi if [[ "$current_target" != "$state_old" ]]; then rollback_link="$PREFIX/.current-recovery-$$-${RANDOM}.tmp" [[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1 ln -s -- "$state_old" "$rollback_link" || return 1 if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then rm -f -- "$rollback_link" 2>/dev/null || true return 1 fi recovery_node="$CURRENT_LINK/runtime/bin/node" [[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true) if ! finalize_state_job "$recovery_node" failed "$state_job"; then # If the original queue is still present, retry it from the restored old # release; a crash before the CLI wrote its job row is recoverable this # way. Without a queue there is no safe operation to replay. if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then clear_update_state || true return 0 fi return 1 fi rm -f -- "$REQUEST_FILE" 2>/dev/null || true clear_update_state || true return 11 fi clear_update_state || true return 0 } if [[ -e "$STATE_FILE" ]]; then recovery_result=0 set +e recover_stale_state recovery_result=$? set -e case "$recovery_result" in 10) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 0 ;; 11) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;; 0) : ;; *) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;; esac fi [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0 # Only create the marker for this invocation after any marker from a previous # interrupted run has been reconciled. Otherwise the freshly-created `running` # marker is indistinguishable from stale recovery state and the runner can # finalize its own queued job as failed before the update CLI starts. if [[ ! -e "$STATE_FILE" ]]; then write_recovery_state running || die '无法写入更新恢复状态' fi start_heartbeat if ! systemctl stop "$SERVICE_NAME"; then die '无法停止 TallyNote 服务' fi rollback_current() { local current_target rollback_link current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true) if [[ "$current_target" == "$old_target" ]]; then # An earlier failure branch may already have restored the link. Keep the # marker truthful so the EXIT trap can still finalize the job. return 0 fi rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp" [[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1 ln -s -- "$old_target" "$rollback_link" || return 1 if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then rm -f -- "$rollback_link" 2>/dev/null || true return 1 fi } finalize_failed_job() { [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0 [[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1 # Give SQLite a moment to release a transient lock before declaring the # recovery itself failed. for _ in 1 2 3; do if "$old_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1; then return 0 fi sleep 1 done return 1 } finalize_completed_job() { [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0 [[ -n "$final_node" ]] || return 1 "$final_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status completed >/dev/null 2>&1 } # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below cleanup_after_update() { local result=$? rollback_ok=1 stop_heartbeat if (( result != 0 && handled == 0 )); then if ! rollback_current; then rollback_ok=0; fi # Once the old release is active again, always try to close the job. The # previous marker could remain set when an earlier branch had already # rolled back before entering this EXIT trap, leaving `applying` forever. if (( rollback_ok == 1 )); then if finalize_failed_job; then rm -f -- "$REQUEST_FILE" clear_update_state || true fi fi fi if (( was_active )); then systemctl start "$SERVICE_NAME" || true else systemctl stop "$SERVICE_NAME" || true fi return "$result" } trap cleanup_after_update EXIT node_bin="$CURRENT_LINK/runtime/bin/node" [[ -x "$node_bin" ]] || node_bin=$(command -v node || true) [[ -n "$node_bin" ]] || die 'node runtime not found' cli="$CURRENT_LINK/dist/server/cli/update.js" [[ -f "$cli" ]] || die 'update CLI not found in current release' set +e "$node_bin" "$cli" --request-file "$REQUEST_FILE" --defer-completion update_result=$? set -e if (( update_result != 0 )); then exit "$update_result" fi write_update_state health-check || exit 1 systemctl start "$SERVICE_NAME" healthy=0 for _ in $(seq 1 30); do if curl --proto '=http' --max-time 2 --silent --show-error "http://$HEALTH_HOST:$PORT/health" >/dev/null 2>&1; then healthy=1; break; fi sleep 1 done if (( healthy == 0 )); then systemctl stop "$SERVICE_NAME" || true rollback_current || die '无法恢复上一版本链接' if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi if ! finalize_failed_job; then exit 1 fi rm -f -- "$REQUEST_FILE" clear_update_state || true handled=1 trap - EXIT exit 1 fi # Preserve an operator's intentionally stopped service after validating the # new release in a temporary start. if (( was_active == 0 )); then systemctl stop "$SERVICE_NAME" fi write_update_state finalizing || exit 1 final_node="$CURRENT_LINK/runtime/bin/node" [[ -x "$final_node" ]] || final_node=$(command -v node || true) if [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]]; then finalized=0 for _ in 1 2 3; do if finalize_completed_job; then finalized=1; break; fi sleep 1 done (( finalized == 1 )) || exit 1 fi rm -f -- "$REQUEST_FILE" clear_update_state || true handled=1 trap - EXIT exit 0