404 lines
15 KiB
Bash
Executable File
404 lines
15 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
set -Eeuo pipefail
|
|
|
|
PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
|
export PATH
|
|
umask 077
|
|
|
|
PREFIX=${TALLYNOTE_INSTALL_PREFIX:-/opt/tallynote}
|
|
DATA_DIR=${TALLYNOTE_DATA_DIR:-/var/lib/tallynote}
|
|
REQUEST_FILE="$DATA_DIR/update-request.json"
|
|
CURRENT_LINK="$PREFIX/current"
|
|
STATE_FILE="$PREFIX/.update-state"
|
|
SERVICE_NAME=${TALLYNOTE_SERVICE_NAME:-tallynote.service}
|
|
HOST=${TALLYNOTE_HOST:-127.0.0.1}
|
|
PORT=${TALLYNOTE_PORT:-3000}
|
|
HEALTH_HOST=$HOST
|
|
if [[ "$HEALTH_HOST" == 0.0.0.0 ]]; then HEALTH_HOST=127.0.0.1; fi
|
|
if [[ "$HEALTH_HOST" == :: ]]; then HEALTH_HOST=::1; fi
|
|
if [[ "$HEALTH_HOST" == *:* && "$HEALTH_HOST" != \[* ]]; then HEALTH_HOST="[$HEALTH_HOST]"; fi
|
|
|
|
die() { printf 'tallynote update runner: %s\n' "$*" >&2; exit 1; }
|
|
[[ ${EUID:-$(id -u)} -eq 0 ]] || die 'must run as root'
|
|
[[ -f "$REQUEST_FILE" || -f "$STATE_FILE" ]] || exit 0
|
|
[[ -L "$CURRENT_LINK" ]] || die 'current release link is missing'
|
|
|
|
old_target=$(readlink -f -- "$CURRENT_LINK")
|
|
[[ "$old_target" == "$PREFIX/releases/"* && -d "$old_target" ]] || die 'current release target is invalid'
|
|
|
|
request_operation='apply'
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
request_operation=$(sed -n 's/.*"operation"[[:space:]]*:[[:space:]]*"\(download\|apply\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
|
[[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply'
|
|
fi
|
|
|
|
# Capture the request id before any privileged preflight can fail. The
|
|
# request file is an application-owned one-shot marker; removing it on an
|
|
# early runner failure lets the server-side lease reaper release the DB row.
|
|
job_id=''
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
|
fi
|
|
STATE_CREATED=0
|
|
heartbeat_pid=''
|
|
heartbeat_owner=$$
|
|
|
|
write_recovery_state() {
|
|
local phase=$1 temporary
|
|
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
|
|
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
|
|
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
|
|
chmod 600 "$temporary"
|
|
mv -Tf -- "$temporary" "$STATE_FILE"
|
|
STATE_CREATED=1
|
|
}
|
|
|
|
clear_recovery_state() {
|
|
[[ ! -L "$STATE_FILE" ]] || return 1
|
|
rm -f -- "$STATE_FILE"
|
|
STATE_CREATED=0
|
|
}
|
|
|
|
stop_heartbeat() {
|
|
if [[ -n "$heartbeat_pid" ]]; then
|
|
kill "$heartbeat_pid" 2>/dev/null || true
|
|
wait "$heartbeat_pid" 2>/dev/null || true
|
|
heartbeat_pid=''
|
|
fi
|
|
}
|
|
|
|
heartbeat() {
|
|
# Keep the lease fresh during long downloads/backups, but stop on a hard
|
|
# runner kill so an orphaned child cannot keep the recovery marker alive.
|
|
while kill -0 "$heartbeat_owner" 2>/dev/null; do
|
|
sleep 10 || exit 0
|
|
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0
|
|
touch "$STATE_FILE" 2>/dev/null || exit 0
|
|
done
|
|
}
|
|
|
|
start_heartbeat() {
|
|
stop_heartbeat
|
|
heartbeat &
|
|
heartbeat_pid=$!
|
|
}
|
|
|
|
# This trap covers failures before the normal apply cleanup trap is installed,
|
|
# including a missing runtime, an invalid current link, and a failed service
|
|
# stop. It deliberately does not remove a pre-existing recovery marker.
|
|
preflight_cleanup() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
if (( result != 0 )); then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
|
|
fi
|
|
return "$result"
|
|
}
|
|
trap preflight_cleanup EXIT
|
|
|
|
# Downloading is intentionally handled while the main service remains up.
|
|
# The CLI persists the validated payload under the root-owned workspace and
|
|
# leaves the job staged for a later apply request.
|
|
if [[ "$request_operation" == download ]]; then
|
|
# A previous download runner may have been interrupted after creating its
|
|
# marker. Clear only that download marker and retry the idempotent request.
|
|
if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then
|
|
clear_recovery_state || die '无法清理上一次下载状态'
|
|
fi
|
|
write_recovery_state download || die '无法写入更新恢复状态'
|
|
cleanup_download() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
if (( result != 0 )); then
|
|
# The CLI normally records failed itself. If it died before opening the
|
|
# database, the expired marker/request will be reconciled by the app.
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
fi
|
|
clear_recovery_state || true
|
|
return "$result"
|
|
}
|
|
trap cleanup_download EXIT
|
|
trap 'exit 143' TERM
|
|
trap 'exit 130' INT
|
|
start_heartbeat
|
|
node_bin="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
|
|
[[ -n "$node_bin" ]] || die 'node runtime not found'
|
|
cli="$CURRENT_LINK/dist/server/cli/update.js"
|
|
[[ -f "$cli" ]] || die 'update CLI not found in current release'
|
|
set +e
|
|
"$node_bin" "$cli" --request-file "$REQUEST_FILE"
|
|
download_result=$?
|
|
set -e
|
|
if (( download_result != 0 )); then
|
|
# The CLI normally records failed itself. Retry the explicit finalization
|
|
# for failures that happen before its catch handler can persist the row,
|
|
# then remove the one-shot request so a failed download cannot keep the
|
|
# path unit in a permanently triggered state.
|
|
download_job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
|
|
if [[ "$download_job_id" =~ ^[0-9a-f-]{36}$ ]]; then
|
|
for _ in 1 2 3; do
|
|
if "$node_bin" "$cli" --finalize-job "$download_job_id" --finalize-status failed --message '更新下载失败' >/dev/null 2>&1; then break; fi
|
|
sleep 1
|
|
done
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
exit "$download_result"
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
exit 0
|
|
fi
|
|
|
|
was_active=0
|
|
if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
|
|
restore_initial_service() {
|
|
local result=$?
|
|
stop_heartbeat
|
|
if (( result != 0 )); then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
|
|
fi
|
|
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
|
|
return "$result"
|
|
}
|
|
trap restore_initial_service EXIT
|
|
old_node="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$old_node" ]] || old_node=$(command -v node || true)
|
|
handled=0
|
|
|
|
write_update_state() { write_recovery_state "$1"; }
|
|
clear_update_state() { clear_recovery_state; }
|
|
|
|
finalize_state_job() {
|
|
local node=$1 status=$2 state_job=$3
|
|
[[ "$state_job" =~ ^[0-9a-f-]{36}$ && -n "$node" ]] || return 1
|
|
[[ -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
|
|
"$node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$state_job" --finalize-status "$status" --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1
|
|
}
|
|
|
|
recover_stale_state() {
|
|
local state_job state_old state_phase current_target recovery_node rollback_link state_mode state_uid
|
|
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || die 'update state file is invalid'
|
|
state_uid=$(stat -c '%u' "$STATE_FILE" 2>/dev/null || stat -f '%u' "$STATE_FILE")
|
|
state_mode=$(stat -c '%a' "$STATE_FILE" 2>/dev/null || stat -f '%Lp' "$STATE_FILE")
|
|
[[ "$state_uid" == 0 && "$state_mode" =~ ^[0-7]+$ && $((8#$state_mode & 077)) -eq 0 ]] || die 'update state file permissions are invalid'
|
|
state_job=$(sed -n 's/^job_id=//p' "$STATE_FILE" | head -n 1)
|
|
state_old=$(sed -n 's/^old_target=//p' "$STATE_FILE" | head -n 1)
|
|
state_phase=$(sed -n 's/^phase=//p' "$STATE_FILE" | head -n 1)
|
|
[[ "$state_job" =~ ^[0-9a-f-]{36}$ ]] || die 'update state job id is invalid'
|
|
[[ "$state_old" == "$PREFIX/releases/"* && -d "$state_old" && ! -L "$state_old" ]] || die 'update state target is invalid'
|
|
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
|
|
if [[ "$state_phase" == finalizing && "$current_target" != "$state_old" ]]; then
|
|
recovery_node="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
|
|
for _ in 1 2 3; do
|
|
if finalize_state_job "$recovery_node" completed "$state_job"; then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
clear_update_state || true
|
|
return 10
|
|
fi
|
|
sleep 1
|
|
done
|
|
return 1
|
|
fi
|
|
if [[ "$current_target" == "$state_old" ]]; then
|
|
# The process may have restored the old release before it was killed. In
|
|
# that case the old link is already safe to serve, but the database row
|
|
# can still be `applying`; finish it as failed before clearing recovery
|
|
# markers so the UI does not poll forever.
|
|
recovery_node="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
|
|
if finalize_state_job "$recovery_node" failed "$state_job"; then
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
clear_update_state || true
|
|
return 11
|
|
fi
|
|
# A crash before the CLI created its job row is safe to retry. Preserve
|
|
# the request while dropping only the stale state marker.
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
if [[ "$current_target" != "$state_old" ]]; then
|
|
rollback_link="$PREFIX/.current-recovery-$$-${RANDOM}.tmp"
|
|
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
|
|
ln -s -- "$state_old" "$rollback_link" || return 1
|
|
if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then
|
|
rm -f -- "$rollback_link" 2>/dev/null || true
|
|
return 1
|
|
fi
|
|
recovery_node="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$recovery_node" ]] || recovery_node=$(command -v node || true)
|
|
if ! finalize_state_job "$recovery_node" failed "$state_job"; then
|
|
# If the original queue is still present, retry it from the restored old
|
|
# release; a crash before the CLI wrote its job row is recoverable this
|
|
# way. Without a queue there is no safe operation to replay.
|
|
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
|
|
clear_update_state || true
|
|
return 0
|
|
fi
|
|
return 1
|
|
fi
|
|
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
|
|
clear_update_state || true
|
|
return 11
|
|
fi
|
|
clear_update_state || true
|
|
return 0
|
|
}
|
|
|
|
if [[ -e "$STATE_FILE" ]]; then
|
|
recovery_result=0
|
|
set +e
|
|
recover_stale_state
|
|
recovery_result=$?
|
|
set -e
|
|
case "$recovery_result" in
|
|
10) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 0 ;;
|
|
11) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;;
|
|
0) : ;;
|
|
*) if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi; exit 1 ;;
|
|
esac
|
|
fi
|
|
|
|
[[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0
|
|
|
|
# Only create the marker for this invocation after any marker from a previous
|
|
# interrupted run has been reconciled. Otherwise the freshly-created `running`
|
|
# marker is indistinguishable from stale recovery state and the runner can
|
|
# finalize its own queued job as failed before the update CLI starts.
|
|
if [[ ! -e "$STATE_FILE" ]]; then
|
|
write_recovery_state running || die '无法写入更新恢复状态'
|
|
fi
|
|
start_heartbeat
|
|
if ! systemctl stop "$SERVICE_NAME"; then
|
|
die '无法停止 TallyNote 服务'
|
|
fi
|
|
|
|
rollback_current() {
|
|
local current_target rollback_link
|
|
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
|
|
if [[ "$current_target" == "$old_target" ]]; then
|
|
# An earlier failure branch may already have restored the link. Keep the
|
|
# marker truthful so the EXIT trap can still finalize the job.
|
|
return 0
|
|
fi
|
|
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
|
|
[[ ! -e "$rollback_link" && ! -L "$rollback_link" ]] || return 1
|
|
ln -s -- "$old_target" "$rollback_link" || return 1
|
|
if ! mv -Tf -- "$rollback_link" "$CURRENT_LINK"; then
|
|
rm -f -- "$rollback_link" 2>/dev/null || true
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
finalize_failed_job() {
|
|
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
|
|
[[ -n "$old_node" && -f "$CURRENT_LINK/dist/server/cli/update.js" ]] || return 1
|
|
# Give SQLite a moment to release a transient lock before declaring the
|
|
# recovery itself failed.
|
|
for _ in 1 2 3; do
|
|
if "$old_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status failed --message '新版本健康检查失败,已恢复上一版本' >/dev/null 2>&1; then
|
|
return 0
|
|
fi
|
|
sleep 1
|
|
done
|
|
return 1
|
|
}
|
|
|
|
finalize_completed_job() {
|
|
[[ "$job_id" =~ ^[0-9a-f-]{36}$ ]] || return 0
|
|
[[ -n "$final_node" ]] || return 1
|
|
"$final_node" "$CURRENT_LINK/dist/server/cli/update.js" --finalize-job "$job_id" --finalize-status completed >/dev/null 2>&1
|
|
}
|
|
|
|
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
|
|
cleanup_after_update() {
|
|
local result=$? rollback_ok=1
|
|
stop_heartbeat
|
|
if (( result != 0 && handled == 0 )); then
|
|
if ! rollback_current; then rollback_ok=0; fi
|
|
# Once the old release is active again, always try to close the job. The
|
|
# previous marker could remain set when an earlier branch had already
|
|
# rolled back before entering this EXIT trap, leaving `applying` forever.
|
|
if (( rollback_ok == 1 )); then
|
|
if finalize_failed_job; then
|
|
rm -f -- "$REQUEST_FILE"
|
|
clear_update_state || true
|
|
fi
|
|
fi
|
|
fi
|
|
if (( was_active )); then
|
|
systemctl start "$SERVICE_NAME" || true
|
|
else
|
|
systemctl stop "$SERVICE_NAME" || true
|
|
fi
|
|
return "$result"
|
|
}
|
|
trap cleanup_after_update EXIT
|
|
|
|
node_bin="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
|
|
[[ -n "$node_bin" ]] || die 'node runtime not found'
|
|
cli="$CURRENT_LINK/dist/server/cli/update.js"
|
|
[[ -f "$cli" ]] || die 'update CLI not found in current release'
|
|
|
|
set +e
|
|
"$node_bin" "$cli" --request-file "$REQUEST_FILE" --defer-completion
|
|
update_result=$?
|
|
set -e
|
|
if (( update_result != 0 )); then
|
|
exit "$update_result"
|
|
fi
|
|
|
|
write_update_state health-check || exit 1
|
|
|
|
systemctl start "$SERVICE_NAME"
|
|
healthy=0
|
|
for _ in $(seq 1 30); do
|
|
if curl --proto '=http' --max-time 2 --silent --show-error "http://$HEALTH_HOST:$PORT/health" >/dev/null 2>&1; then healthy=1; break; fi
|
|
sleep 1
|
|
done
|
|
|
|
if (( healthy == 0 )); then
|
|
systemctl stop "$SERVICE_NAME" || true
|
|
rollback_current || die '无法恢复上一版本链接'
|
|
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
|
|
if ! finalize_failed_job; then
|
|
exit 1
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
clear_update_state || true
|
|
handled=1
|
|
trap - EXIT
|
|
exit 1
|
|
fi
|
|
|
|
# Preserve an operator's intentionally stopped service after validating the
|
|
# new release in a temporary start.
|
|
if (( was_active == 0 )); then
|
|
systemctl stop "$SERVICE_NAME"
|
|
fi
|
|
|
|
write_update_state finalizing || exit 1
|
|
final_node="$CURRENT_LINK/runtime/bin/node"
|
|
[[ -x "$final_node" ]] || final_node=$(command -v node || true)
|
|
if [[ "$job_id" =~ ^[0-9a-f-]{36}$ ]]; then
|
|
finalized=0
|
|
for _ in 1 2 3; do
|
|
if finalize_completed_job; then finalized=1; break; fi
|
|
sleep 1
|
|
done
|
|
(( finalized == 1 )) || exit 1
|
|
fi
|
|
rm -f -- "$REQUEST_FILE"
|
|
clear_update_state || true
|
|
handled=1
|
|
trap - EXIT
|
|
exit 0
|