#!/usr/bin/env bash set -Eeuo pipefail PATH=/usr/local/bin:/usr/bin:/usr/sbin:/sbin:/bin export PATH umask 077 PREFIX=${TALLYNOTE_INSTALL_PREFIX:-/opt/tallynote} DATA_DIR=${TALLYNOTE_DATA_DIR:-/var/lib/tallynote} CONFIG_DIR=${TALLYNOTE_CONFIG_DIR:-/etc/tallynote} CONFIG_FILE="$CONFIG_DIR/tallynote.env" REQUEST_FILE="$DATA_DIR/update-request.json" CURRENT_LINK="$PREFIX/current" STATE_FILE="$PREFIX/.update-state" LOCK_FILE="$PREFIX/.update-runner.lock" RUNNER_LOG="$PREFIX/.update-runner.log" SERVICE_NAME=${TALLYNOTE_SERVICE_NAME:-tallynote.service} HOST=${TALLYNOTE_HOST:-127.0.0.1} PORT=${TALLYNOTE_PORT:-3000} HEALTH_HOST=$HOST if [[ "$HEALTH_HOST" == 0.0.0.0 ]]; then HEALTH_HOST=127.0.0.1; fi if [[ "$HEALTH_HOST" == :: ]]; then HEALTH_HOST=::1; fi if [[ "$HEALTH_HOST" == *:* && "$HEALTH_HOST" != \[* ]]; then HEALTH_HOST="[$HEALTH_HOST]"; fi die() { printf 'tallynote update runner: %s\n' "$*" >&2; exit 1; } node_is_usable() { local candidate=$1 major [[ -n "$candidate" && -x "$candidate" ]] || return 1 major=$("$candidate" -p 'process.versions.node.split(".")[0]' 2>/dev/null || true) [[ "$major" =~ ^[0-9]+$ && "$major" -ge 24 ]] } resolve_node() { local candidate=${TALLYNOTE_NODE:-} if [[ -z "$candidate" && -f "$CONFIG_FILE" && ! -L "$CONFIG_FILE" ]]; then candidate=$(sed -n 's/^TALLYNOTE_NODE=//p' "$CONFIG_FILE" | head -n 1) fi if node_is_usable "$candidate"; then printf '%s' "$candidate" return 0 fi candidate=$(command -v node || true) node_is_usable "$candidate" || return 1 printf '%s' "$candidate" } # The runner may exit during any of the checks below. Install its EXIT cleanup # before doing privileged preflight so a partial invocation never leaves a # heartbeat or lock behind. STATE_CREATED=0 heartbeat_pid='' heartbeat_owner=$$ RUNNER_LOCK_FD=9 RUNNER_LOCK_MODE='' stop_heartbeat() { if [[ -n "$heartbeat_pid" ]]; then kill "$heartbeat_pid" 2>/dev/null || true wait "$heartbeat_pid" 2>/dev/null || true heartbeat_pid='' fi } # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap release_runner_lock() { if [[ "$RUNNER_LOCK_MODE" == flock ]]; then flock -u "$RUNNER_LOCK_FD" 2>/dev/null || true eval "exec ${RUNNER_LOCK_FD}>&-" 2>/dev/null || true elif [[ "$RUNNER_LOCK_MODE" == mkdir ]]; then rmdir -- "$LOCK_FILE.d" 2>/dev/null || true fi } # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap early_cleanup() { local result=$? stop_heartbeat if (( result != 0 )); then # A preflight failure happens before the normal phase-specific trap is # installed. Remove only the one-shot request marker; never remove an # existing recovery marker unless this invocation created it. rm -f -- "$REQUEST_FILE" 2>/dev/null || true if (( STATE_CREATED == 1 )); then rm -f -- "$STATE_FILE" 2>/dev/null || true; fi fi release_runner_lock return "$result" } trap early_cleanup EXIT [[ ${EUID:-$(id -u)} -eq 0 ]] || die 'must run as root' [[ -d "$PREFIX" ]] || die 'install prefix is missing' if command -v flock >/dev/null 2>&1; then exec 9>"$LOCK_FILE" || die '无法打开更新运行锁' flock -n "$RUNNER_LOCK_FD" || exit 0 RUNNER_LOCK_MODE=flock else # macOS development fixtures do not ship util-linux; retain an atomic lock # fallback there while Linux production uses flock above. mkdir "$LOCK_FILE.d" 2>/dev/null || exit 0 RUNNER_LOCK_MODE='mkdir' fi [[ -f "$REQUEST_FILE" || -f "$STATE_FILE" ]] || exit 0 [[ -L "$CURRENT_LINK" ]] || die 'current release link is missing' old_target=$(readlink -f -- "$CURRENT_LINK") [[ "$old_target" == "$PREFIX/releases/"* && -d "$old_target" ]] || die 'current release target is invalid' # Capture the service state before any download/apply work. The value is # persisted in the recovery marker so a later runner process can restore the # operator's original state after a crash (the service is normally inactive by # the time recovery starts). was_active=0 if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi request_operation='apply' if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then request_operation=$(sed -n 's/.*"operation"[[:space:]]*:[[:space:]]*"\(download\|apply\)".*/\1/p' "$REQUEST_FILE" | head -n 1) [[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply' fi # Capture the request id before any privileged preflight can fail. The # request file is an application-owned one-shot marker; removing it on an # early runner failure lets the server-side lease reaper release the DB row. job_id='' if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1) fi write_recovery_state() { local phase=$1 temporary temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp" [[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1 printf 'job_id=%s\nold_target=%s\nphase=%s\ninitial_active=%s\n' "$job_id" "$old_target" "$phase" "$was_active" > "$temporary" chmod 600 "$temporary" mv -Tf -- "$temporary" "$STATE_FILE" STATE_CREATED=1 } clear_recovery_state() { [[ ! -L "$STATE_FILE" ]] || return 1 rm -f -- "$STATE_FILE" STATE_CREATED=0 } heartbeat() { # Keep the lease fresh during long downloads/backups, but stop on a hard # runner kill so an orphaned child cannot keep the recovery marker alive. while kill -0 "$heartbeat_owner" 2>/dev/null; do sleep 10 || exit 0 [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0 touch "$STATE_FILE" 2>/dev/null || exit 0 done } start_heartbeat() { stop_heartbeat heartbeat & heartbeat_pid=$! } # This trap covers failures before the normal apply cleanup trap is installed, # including a missing runtime, an invalid current link, and a failed service # stop. It deliberately does not remove a pre-existing recovery marker. # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap preflight_cleanup() { local result=$? stop_heartbeat release_runner_lock if (( result != 0 )); then rm -f -- "$REQUEST_FILE" 2>/dev/null || true if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi fi return "$result" } trap preflight_cleanup EXIT DOWNLOAD_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_DOWNLOAD_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_DOWNLOAD_TIMEOUT_SECONDS:-1800}} APPLY_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_APPLY_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_APPLY_TIMEOUT_SECONDS:-1800}} FINALIZE_TIMEOUT_SECONDS=${TALLYNOTE_UPDATE_FINALIZE_TIMEOUT_SECONDS:-${TALLYNOTE_UPDATE_RUNNER_FINALIZE_TIMEOUT_SECONDS:-30}} TIMEOUT_BIN=$(command -v timeout || true) run_update_cli() { local node=$1 timeout_seconds=$2 label=$3 result shift 3 [[ "$timeout_seconds" =~ ^[1-9][0-9]*$ ]] || die "${label} timeout must be a positive integer" { printf '\n[%s] %s (timeout=%ss)\ncommand:' "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$label" "$timeout_seconds" printf ' %q' "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" printf '\n' } >>"$RUNNER_LOG" if [[ -n "$TIMEOUT_BIN" ]]; then "$TIMEOUT_BIN" --foreground --signal=TERM --kill-after=10s "${timeout_seconds}s" \ "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" >>"$RUNNER_LOG" 2>&1 result=$? elif "$node" "$CURRENT_LINK/dist/server/cli/update.js" "$@" >>"$RUNNER_LOG" 2>&1; then result=0 else result=$? fi printf '[%s] %s exited with status %s\n' "$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$label" "$result" >>"$RUNNER_LOG" return "$result" } # Downloading is intentionally handled while the main service remains up. # The CLI persists the validated payload under the root-owned workspace and # leaves the job staged for a later apply request. if [[ "$request_operation" == download ]]; then # A previous download runner may have been interrupted after creating its # marker. Clear only that download marker and retry the idempotent request. if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then clear_recovery_state || die '无法清理上一次下载状态' fi write_recovery_state download || die '无法写入更新恢复状态' # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap cleanup_download() { local result=$? stop_heartbeat if (( result != 0 )); then # The CLI normally records failed itself. If it died before opening the # database, the expired marker/request will be reconciled by the app. rm -f -- "$REQUEST_FILE" 2>/dev/null || true fi clear_recovery_state || true release_runner_lock return "$result" } trap cleanup_download EXIT trap 'exit 143' TERM trap 'exit 130' INT start_heartbeat node_bin=$(resolve_node) || die 'Node.js 24+ not found' cli="$CURRENT_LINK/dist/server/cli/update.js" [[ -f "$cli" ]] || die 'update CLI not found in current release' set +e run_update_cli "$node_bin" "$DOWNLOAD_TIMEOUT_SECONDS" download --request-file "$REQUEST_FILE" download_result=$? set -e if (( download_result != 0 )); then # The CLI normally records failed itself. Retry the explicit finalization # for failures that happen before its catch handler can persist the row, # then remove the one-shot request so a failed download cannot keep the # path unit in a permanently triggered state. download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1) if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then for _ in 1 2 3; do if run_update_cli "$node_bin" "$FINALIZE_TIMEOUT_SECONDS" finalize-download --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败'; then break; fi sleep 1 done fi rm -f -- "$REQUEST_FILE" exit "$download_result" fi rm -f -- "$REQUEST_FILE" exit 0 fi # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below restore_initial_service() { local result=$? stop_heartbeat release_runner_lock if (( result != 0 )); then rm -f -- "$REQUEST_FILE" 2>/dev/null || true if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi fi if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi return "$result" } trap restore_initial_service EXIT old_node=$(resolve_node) || die 'Node.js 24+ not found' handled=0 write_update_state() { write_recovery_state "$1"; } clear_update_state() { clear_recovery_state; } finalize_state_job() { local node=$1 status=$2 state_job=$3 [[ "$state_job" =~ ^[0-9a-f-]{36}$ && -n "$node" ]] || return 1 [[ -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1 run_update_cli "$node" "$FINALIZE_TIMEOUT_SECONDS" finalize-recovery --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本' } recover_stale_state() { local state_job state_old state_phase state_initial_active current_target recovery_node rollback_link state_mode state_uid [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || die 'update state file is invalid' state_uid=$(stat -c '%u' "$STATE_FILE" 2>/dev/null || stat -f '%u' "$STATE_FILE") state_mode=$(stat -c '%a' "$STATE_FILE" 2>/dev/null || stat -f '%Lp' "$STATE_FILE") [[ "$state_uid" == 0 && "$state_mode" =~ ^[0-7]+$ && $((8#$state_mode & 077)) -eq 0 ]] || die 'update state file permissions are invalid' state_job=$(sed -n 's/^job_id=//p' "$STATE_FILE" | head -n 1) state_old=$(sed -n 's/^old_target=//p' "$STATE_FILE" | head -n 1) state_phase=$(sed -n 's/^phase=//p' "$STATE_FILE" | head -n 1) state_initial_active=$(sed -n 's/^initial_active=//p' "$STATE_FILE" | head -n 1) [[ "$state_job" =~ ^[0-9a-f-]{36}$ ]] || die 'update state job id is invalid' [[ "$state_old" == "$PREFIX/releases/"* && -d "$state_old" && ! -L "$state_old" ]] || die 'update state target is invalid' if [[ -z "$state_initial_active" ]]; then # Markers from older releases did not persist this field. Preserve their # historical conservative behavior instead of rejecting recovery. state_initial_active=0 fi [[ "$state_initial_active" == 0 || "$state_initial_active" == 1 ]] || die 'update state initial service state is invalid' was_active=$state_initial_active current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true) if [[ "$state_phase" == download && "$current_target" == "$state_old" ]]; then # Downloading never changes the active release. If the runner was killed # after the CLI staged its payload but before it removed the recovery # marker, keep the request available for an idempotent retry. Treating # every stale download marker as a failed apply would discard a usable # staged payload and leave the browser showing a misleading failure. clear_update_state || true return 0 fi if [[ "$state_phase" == finalizing && "$current_target" != "$state_old" ]]; then recovery_node=$(resolve_node) || die 'Node.js 24+ not found' for _ in 1 2 3; do if finalize_state_job "$recovery_node" completed "$state_job"; then rm -f -- "$REQUEST_FILE" 2>/dev/null || true clear_update_state || true return 10 fi sleep 1 done return 1 fi if [[ "$current_target" == "$state_old" ]]; then # The process may have restored the old release before it was killed. In # that case the old link is already safe to serve, but the database row # can still be `applying`; finish it as failed before clearing recovery # markers so the UI does not poll forever. recovery_node=$(resolve_node) || die 'Node.js 24+ not found' if finalize_state_job "$recovery_node" failed "$state_job"; then rm -f -- "$REQUEST_FILE" 2>/dev/null || true clear_update_state || true return 11 fi # A crash before the CLI created its job row is safe to retry. Preserve # the request while dropping only the stale state marker. if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then clear_update_state || true return 0 fi clear_update_state || true return 0 fi if [[ "$current_target" != "$state_old" ]]; then rollback_link="$PREFIX/.current-recovery-$$-${RANDOM}.tmp" [[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1 ln -s -- "$state_old" "$rollback_link" || return 1 if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then rm -f -- "$rollback_link" 2>/dev/null || true return 1 fi recovery_node=$(resolve_node) || die 'Node.js 24+ not found' if ! finalize_state_job "$recovery_node" failed "$state_job"; then # If the original queue is still present, retry it from the restored old # release; a crash before the CLI wrote its job row is recoverable this # way. Without a queue there is no safe operation to replay. if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then clear_update_state || true return 0 fi return 1 fi rm -f -- "$REQUEST_FILE" 2>/dev/null || true clear_update_state || true return 11 fi clear_update_state || true return 0 } if [[ -e "$STATE_FILE" ]]; then recovery_result=0 set +e recover_stale_state recovery_result=$? set -e case "$recovery_result" in 10) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 0 ;; 11) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;; 0) : ;; *) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;; esac fi [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0 # Only create the marker for this invocation after any marker from a previous # interrupted run has been reconciled. Otherwise the freshly-created `running` # marker is indistinguishable from stale recovery state and the runner can # finalize its own queued job as failed before the update CLI starts. if [[ ! -e "$STATE_FILE" ]]; then write_recovery_state running || die '无法写入更新恢复状态' fi start_heartbeat if ! systemctl stop "$SERVICE_NAME"; then die '无法停止 TallyNote 服务' fi rollback_current() { local current_target rollback_link current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true) if [[ "$current_target" == "$old_target" ]]; then # An earlier failure branch may already have restored the link. Keep the # marker truthful so the EXIT trap can still finalize the job. return 0 fi rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp" [[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1 ln -s -- "$old_target" "$rollback_link" || return 1 if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then rm -f -- "$rollback_link" 2>/dev/null || true return 1 fi } finalize_failed_job() { [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0 [[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1 # Give SQLite a moment to release a transient lock before declaring the # recovery itself failed. for _ in 1 2 3; do if run_update_cli "$old_node" "$FINALIZE_TIMEOUT_SECONDS" finalize-failed --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本'; then return 0 fi sleep 1 done return 1 } finalize_completed_job() { [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0 [[ -n "$final_node" ]] || return 1 run_update_cli "$final_node" "$FINALIZE_TIMEOUT_SECONDS" finalize-completed --finalize-job "$job_id" --finalize-status completed } # shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below cleanup_after_update() { local result=$? rollback_ok=1 stop_heartbeat if (( result != 0 && handled == 0 )); then if ! rollback_current; then rollback_ok=0; fi # Once the old release is active again, always try to close the job. The # previous marker could remain set when an earlier branch had already # rolled back before entering this EXIT trap, leaving `applying` forever. if (( rollback_ok == 1 )); then if finalize_failed_job; then rm -f -- "$REQUEST_FILE" clear_update_state || true fi fi fi if (( was_active )); then systemctl start "$SERVICE_NAME" || true else systemctl stop "$SERVICE_NAME" || true fi release_runner_lock return "$result" } trap cleanup_after_update EXIT node_bin=$(resolve_node) || die 'Node.js 24+ not found' cli="$CURRENT_LINK/dist/server/cli/update.js" [[ -f "$cli" ]] || die 'update CLI not found in current release' set +e run_update_cli "$node_bin" "$APPLY_TIMEOUT_SECONDS" apply --request-file "$REQUEST_FILE" --defer-completion update_result=$? set -e if (( update_result != 0 )); then exit "$update_result" fi write_update_state health-check || exit 1 systemctl start "$SERVICE_NAME" healthy=0 for _ in $(seq 1 30); do if curl --proto '=http' --max-time 2 --silent --show-error "http://$HEALTH_HOST:$PORT/health" >/dev/null 2>&1; then healthy=1; break; fi sleep 1 done if (( healthy == 0 )); then systemctl stop "$SERVICE_NAME" || true rollback_current || die '无法恢复上一版本链接' if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi if ! finalize_failed_job; then exit 1 fi rm -f -- "$REQUEST_FILE" clear_update_state || true handled=1 trap - EXIT exit 1 fi # Preserve an operator's intentionally stopped service after validating the # new release in a temporary start. if (( was_active == 0 )); then systemctl stop "$SERVICE_NAME" fi write_update_state finalizing || exit 1 final_node=$(resolve_node) || die 'Node.js 24+ not found' if [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]]; then finalized=0 for _ in 1 2 3; do if finalize_completed_job; then finalized=1; break; fi sleep 1 done (( finalized == 1 )) || exit 1 fi rm -f -- "$REQUEST_FILE" clear_update_state || true handled=1 trap - EXIT exit 0