Compare commits

...
1 Commits
Author SHA1 Message Date
Qiufeng 69a4b482ec fix: prevent stuck background updates
TallyNote release / linux-x64 (push) Failing after 2m51s
2026-09-03 07:51:35 +08:00
7 changed files with 266 additions and 47 deletions
+1
View File
@@ -30,6 +30,7 @@ jobs:
pnpm install --frozen-lockfile
pnpm check
pnpm test
pnpm test:installer
- name: Build Linux release
run: pnpm release:build "${GITHUB_REF_NAME#v}" ./release
- name: Create and publish Gitea Release
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "tallynote",
"version": "1.1.14",
"version": "1.1.15",
"private": true,
"type": "module",
"packageManager": "pnpm@9.0.6",
+106 -26
View File
@@ -32,10 +32,96 @@ if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
[[ "$request_operation" == download || "$request_operation" == apply ]] || request_operation='apply'
fi
# Capture the request id before any privileged preflight can fail. The
# request file is an application-owned one-shot marker; removing it on an
# early runner failure lets the server-side lease reaper release the DB row.
job_id=''
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
fi
STATE_CREATED=0
heartbeat_pid=''
heartbeat_owner=$$
write_recovery_state() {
local phase=$1 temporary
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
chmod 600 "$temporary"
mv -Tf -- "$temporary" "$STATE_FILE"
STATE_CREATED=1
}
clear_recovery_state() {
[[ ! -L "$STATE_FILE" ]] || return 1
rm -f -- "$STATE_FILE"
STATE_CREATED=0
}
stop_heartbeat() {
if [[ -n "$heartbeat_pid" ]]; then
kill "$heartbeat_pid" 2>/dev/null || true
wait "$heartbeat_pid" 2>/dev/null || true
heartbeat_pid=''
fi
}
heartbeat() {
# Keep the lease fresh during long downloads/backups, but stop on a hard
# runner kill so an orphaned child cannot keep the recovery marker alive.
while kill -0 "$heartbeat_owner" 2>/dev/null; do
sleep 10 || exit 0
[[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] || exit 0
touch "$STATE_FILE" 2>/dev/null || exit 0
done
}
start_heartbeat() {
stop_heartbeat
heartbeat &
heartbeat_pid=$!
}
# This trap covers failures before the normal apply cleanup trap is installed,
# including a missing runtime, an invalid current link, and a failed service
# stop. It deliberately does not remove a pre-existing recovery marker.
preflight_cleanup() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
fi
return "$result"
}
trap preflight_cleanup EXIT
# Downloading is intentionally handled while the main service remains up.
# The CLI persists the validated payload under the root-owned workspace and
# leaves the job staged for a later apply request.
if [[ "$request_operation" == download ]]; then
# A previous download runner may have been interrupted after creating its
# marker. Clear only that download marker and retry the idempotent request.
if [[ -f "$STATE_FILE" && ! -L "$STATE_FILE" ]] && grep -q '^phase=download$' "$STATE_FILE"; then
clear_recovery_state || die '无法清理上一次下载状态'
fi
write_recovery_state download || die '无法写入更新恢复状态'
cleanup_download() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
# The CLI normally records failed itself. If it died before opening the
# database, the expired marker/request will be reconciled by the app.
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
fi
clear_recovery_state || true
return "$result"
}
trap cleanup_download EXIT
trap 'exit 143' TERM
trap 'exit 130' INT
start_heartbeat
node_bin="$CURRENT_LINK/runtime/bin/node"
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
[[ -n "$node_bin" ]] || die 'node runtime not found'
@@ -69,34 +155,21 @@ if systemctl is-active --quiet "$SERVICE_NAME"; then was_active=1; fi
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
restore_initial_service() {
local result=$?
stop_heartbeat
if (( result != 0 )); then
rm -f -- "$REQUEST_FILE" 2>/dev/null || true
if (( STATE_CREATED == 1 )); then clear_recovery_state || true; fi
fi
if (( was_active )); then systemctl start "$SERVICE_NAME" || true; fi
return "$result"
}
trap restore_initial_service EXIT
systemctl stop "$SERVICE_NAME"
job_id=''
if [[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]]; then
job_id=$(sed -n 's/.*"jobId"[[:space:]]*:[[:space:]]*"\([0-9a-f-]*\)".*/\1/p' "$REQUEST_FILE" | head -n 1)
fi
old_node="$CURRENT_LINK/runtime/bin/node"
[[ -x "$old_node" ]] || old_node=$(command -v node || true)
switched=0
handled=0
write_update_state() {
local phase=$1 temporary
temporary="$PREFIX/.update-state-$$-${RANDOM}.tmp"
[[ ! -e "$temporary" && ! -L "$temporary" ]] || return 1
printf 'job_id=%s\nold_target=%s\nphase=%s\n' "$job_id" "$old_target" "$phase" > "$temporary"
chmod 600 "$temporary"
mv -Tf -- "$temporary" "$STATE_FILE"
}
clear_update_state() {
[[ ! -L "$STATE_FILE" ]] || return 1
rm -f -- "$STATE_FILE"
}
write_update_state() { write_recovery_state "$1"; }
clear_update_state() { clear_recovery_state; }
finalize_state_job() {
local node=$1 status=$2 state_job=$3
@@ -195,13 +268,24 @@ fi
[[ -f "$REQUEST_FILE" && ! -L "$REQUEST_FILE" ]] || exit 0
# Only create the marker for this invocation after any marker from a previous
# interrupted run has been reconciled. Otherwise the freshly-created `running`
# marker is indistinguishable from stale recovery state and the runner can
# finalize its own queued job as failed before the update CLI starts.
if [[ ! -e "$STATE_FILE" ]]; then
write_recovery_state running || die '无法写入更新恢复状态'
fi
start_heartbeat
if ! systemctl stop "$SERVICE_NAME"; then
die '无法停止 TallyNote 服务'
fi
rollback_current() {
local current_target rollback_link
current_target=$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)
if [[ "$current_target" == "$old_target" ]]; then
# An earlier failure branch may already have restored the link. Keep the
# marker truthful so the EXIT trap can still finalize the job.
switched=0
return 0
fi
rollback_link="$PREFIX/.current-rollback-$$-${RANDOM}.tmp"
@@ -211,7 +295,6 @@ rollback_current() {
rm -f -- "$rollback_link" 2>/dev/null || true
return 1
fi
switched=0
}
finalize_failed_job() {
@@ -237,6 +320,7 @@ finalize_completed_job() {
# shellcheck disable=SC2329 # invoked indirectly by the EXIT trap below
cleanup_after_update() {
local result=$? rollback_ok=1
stop_heartbeat
if (( result != 0 && handled == 0 )); then
if ! rollback_current; then rollback_ok=0; fi
# Once the old release is active again, always try to close the job. The
@@ -258,7 +342,6 @@ cleanup_after_update() {
}
trap cleanup_after_update EXIT
write_update_state running || exit 1
node_bin="$CURRENT_LINK/runtime/bin/node"
[[ -x "$node_bin" ]] || node_bin=$(command -v node || true)
[[ -n "$node_bin" ]] || die 'node runtime not found'
@@ -273,9 +356,6 @@ if (( update_result != 0 )); then
exit "$update_result"
fi
if [[ "$(readlink -f -- "$CURRENT_LINK" 2>/dev/null || true)" != "$old_target" ]]; then
switched=1
fi
write_update_state health-check || exit 1
systemctl start "$SERVICE_NAME"
+29
View File
@@ -160,6 +160,35 @@ bash -c '
wait_for_service_health 0.0.0.0 3011
' _ "$installer_lib"
# Exercise the privileged runner's normal apply hand-off with portable command
# shims. In particular, the freshly-created running marker must not be treated
# as stale state before the update CLI gets a chance to process the request.
runner_root="$tmp/runner"
runner_prefix="$runner_root/prefix"
runner_data="$runner_root/data"
runner_tools="$runner_root/tools"
mkdir -p "$runner_prefix/releases/1.0.0/runtime/bin" "$runner_prefix/releases/1.0.0/dist/server/cli" "$runner_data" "$runner_tools"
ln -s "$runner_prefix/releases/1.0.0" "$runner_prefix/current"
printf '%s\n' '{"jobId":"00000000-0000-4000-8000-000000000001","operation":"apply"}' > "$runner_data/update-request.json"
printf '%s\n' '#!/usr/bin/env bash' 'printf "%s\\n" "$*" >> "$TALLYNOTE_NODE_TRACE"' 'exit 0' > "$runner_prefix/releases/1.0.0/runtime/bin/node"
printf '%s\n' cli > "$runner_prefix/releases/1.0.0/dist/server/cli/update.js"
printf '%s\n' '#!/usr/bin/env bash' 'case "${1:-}" in is-active) exit 0;; *) exit 0;; esac' > "$runner_tools/systemctl"
printf '%s\n' '#!/usr/bin/env bash' 'if [[ "${1:-}" == "-f" ]]; then shift; [[ "${1:-}" == "--" ]] && shift; /bin/realpath "$1"; else /usr/bin/readlink "$@"; fi' > "$runner_tools/readlink"
printf '%s\n' '#!/usr/bin/env bash' 'if [[ "${1:-}" == "-Tf" ]]; then shift; /bin/mv -f "$@"; else /bin/mv "$@"; fi' > "$runner_tools/mv"
printf '%s\n' '#!/usr/bin/env bash' 'exit 0' > "$runner_tools/curl"
chmod 755 "$runner_prefix/releases/1.0.0/runtime/bin/node" "$runner_tools/systemctl" "$runner_tools/readlink" "$runner_tools/mv" "$runner_tools/curl"
runner_script="$runner_root/runner.sh"
runner_path="$runner_tools:/usr/sbin:/usr/bin:/sbin:/bin"
sed "s#PATH=/usr/sbin:/usr/bin:/sbin:/bin#PATH=$runner_path#" "$root/scripts/tallynote-update-runner.sh" > "$runner_script"
chmod 755 "$runner_script"
runner_prefix_physical=$(cd "$runner_prefix" && pwd -P)
runner_data_physical=$(cd "$runner_data" && pwd -P)
env EUID=0 TALLYNOTE_INSTALL_PREFIX="$runner_prefix_physical" TALLYNOTE_DATA_DIR="$runner_data_physical" TALLYNOTE_NODE_TRACE="$runner_root/node.log" bash "$runner_script"
grep -q -- '--request-file' "$runner_root/node.log"
grep -q -- '--finalize-job' "$runner_root/node.log"
[[ ! -e "$runner_data/update-request.json" ]]
[[ ! -e "$runner_prefix/.update-state" ]]
# A RETURN trap installed by install_release must be cleared while its local
# temporary variables still exist; otherwise set -u fails at the end of main.
release_fixture="$tmp/release-fixture"
+3
View File
@@ -3,6 +3,7 @@ import { loadConfig, prepareDataDirectories, acquireInstanceLock } from "./confi
import { openDatabase } from "./db/index.js";
import { buildApp } from "./app.js";
import { cleanupOrphanedExports, expireExports, resumeExports } from "./exporter.js";
import { reconcileOrphanedUpdateJobs } from "./update-service.js";
const config = loadConfig();
prepareDataDirectories(config);
@@ -18,6 +19,7 @@ async function start() {
await expireExports(database.sqlite, config);
await cleanupOrphanedExports(database.sqlite, config);
await resumeExports(database.sqlite, config);
reconcileOrphanedUpdateJobs(database.sqlite, config);
const app = await buildApp(database, config);
const janitor = setInterval(() => {
void cleanupStaging(config);
@@ -27,6 +29,7 @@ async function start() {
void processFileDeletions(database.sqlite, config);
void expireExports(database.sqlite, config);
void cleanupOrphanedExports(database.sqlite, config);
reconcileOrphanedUpdateJobs(database.sqlite, config);
}, 60_000);
const shutdown = async () => {
clearInterval(janitor);
+74 -17
View File
@@ -1,4 +1,4 @@
import { lstatSync, realpathSync } from "node:fs";
import { lstatSync, realpathSync, unlinkSync } from "node:fs";
import { chmod, mkdir, rename, writeFile } from "node:fs/promises";
import path from "node:path";
import { createPublicKey, randomUUID, verify as verifySignature } from "node:crypto";
@@ -33,9 +33,9 @@ export const ACTIVE_UPDATE_STATUSES: readonly UpdateJobStatus[] = [
];
// A queued job normally starts within seconds and an applying job completes
// after the service health check. This grace period only applies when both
// hand-off markers are gone, so an active runner is never reclaimed midway
// through a download, backup, or switch.
// after the service health check. The runner refreshes its recovery marker as
// a lease while doing long downloads/backups; only an expired lease permits
// the server to reclaim an active row.
export const ORPHANED_UPDATE_TIMEOUT_MS = 5 * 60 * 1000;
export type CachedRelease = {
@@ -364,12 +364,24 @@ export function publicUpdateJob(row: Record<string, unknown> | undefined): Recor
};
}
function markerExists(filePath: string): boolean {
function markerMtime(filePath: string): number | null {
try {
const info = lstatSync(filePath);
return info.isFile() || info.isSymbolicLink();
return info.isFile() ? info.mtimeMs : null;
} catch {
return false;
return null;
}
}
function removeExpiredRequest(filePath: string, now: number): void {
try {
const info = lstatSync(filePath);
if (!info.isFile() && !info.isSymbolicLink()) return;
if (now - info.mtimeMs < ORPHANED_UPDATE_TIMEOUT_MS) return;
unlinkSync(filePath);
} catch {
// The root runner may own the marker during a recovery race. The DB
// transition below is still enough to release the browser queue.
}
}
@@ -385,28 +397,64 @@ function currentReleaseVersion(config: AppConfig): string | null {
}
/**
* Release an update row left behind after both privileged hand-off markers
* disappeared. This is deliberately conservative: staged downloads remain
* available for an explicit apply, and any visible marker means the runner
* still owns recovery.
* Release an update row left behind after its privileged runner lease expired.
* This is deliberately conservative: staged downloads remain available for an
* explicit apply, and a fresh request/state marker means the runner still owns
* recovery.
*/
export function reconcileOrphanedUpdateJobs(database: Database.Database, config: AppConfig, now = Date.now()): number {
const placeholders = ACTIVE_UPDATE_STATUSES.map(() => "?").join(",");
const rows = database.prepare(`
SELECT id, status, version, admin_id AS adminId, request_id AS requestId,
SELECT id, status, operation, version, admin_id AS adminId, request_id AS requestId,
updated_at AS updatedAt
FROM update_jobs
WHERE status IN (${placeholders})
ORDER BY updated_at ASC
`).all(...ACTIVE_UPDATE_STATUSES) as Array<{ id: string; status: UpdateJobStatus; version: string; adminId: string | null; requestId: string | null; updatedAt: number | null }>;
`).all(...ACTIVE_UPDATE_STATUSES) as Array<{ id: string; status: UpdateJobStatus; operation: "download" | "apply"; version: string; adminId: string | null; requestId: string | null; updatedAt: number | null }>;
if (rows.length === 0) return 0;
const requestPresent = markerExists(config.updateRequestPath);
const statePresent = markerExists(path.join(config.installPrefix, ".update-state"));
if (requestPresent || statePresent) return 0;
const statePath = path.join(config.installPrefix, ".update-state");
const requestMtime = markerMtime(config.updateRequestPath);
const stateMtime = markerMtime(statePath);
const requestPresent = requestMtime !== null;
const statePresent = stateMtime !== null;
const requestFresh = requestPresent && now - (requestMtime ?? 0) < ORPHANED_UPDATE_TIMEOUT_MS;
const stateFresh = statePresent && now - (stateMtime ?? 0) < ORPHANED_UPDATE_TIMEOUT_MS;
// A staged download is normally kept for an explicit apply. The one
// exception is the hand-off window where the API has already changed the
// operation to `apply` but crashed before writing the request file. That
// row is still safe to retry and must not block the queue forever.
const releaseVersion = currentReleaseVersion(config);
let reconciled = 0;
for (const row of rows) {
if (row.status === "staged" || typeof row.updatedAt !== "number" || now - row.updatedAt < ORPHANED_UPDATE_TIMEOUT_MS) continue;
if (typeof row.updatedAt !== "number" || now - row.updatedAt < ORPHANED_UPDATE_TIMEOUT_MS) continue;
// The runner refreshes the state marker while a download is in flight.
// A stale request/state marker therefore no longer protects an orphaned
// row forever, while a fresh marker remains owned by the runner.
if (row.status === "staged") {
if (row.operation !== "apply" || requestFresh || stateFresh) continue;
const changed = database.transaction(() => {
const result = database.prepare(`
UPDATE update_jobs
SET operation='download', error_message=NULL, updated_at=?
WHERE id=? AND status='staged' AND operation='apply' AND updated_at=?
`).run(now, row.id, row.updatedAt);
if (result.changes !== 1) return false;
writeAudit(database, {
requestId: row.requestId || randomUUID(),
actorAdminId: row.adminId,
action: "update.reconciled",
targetType: "update",
targetId: row.id,
outcome: "success",
before: { status: row.status, operation: row.operation, version: row.version },
after: { status: "staged", operation: "download", version: row.version, reason: "apply_request_missing" },
});
return true;
})();
if (changed) reconciled += 1;
continue;
}
if (requestFresh || stateFresh) continue;
const status: "completed" | "failed" = row.status === "applying" && releaseVersion === row.version ? "completed" : "failed";
const errorMessage = status === "failed" ? "更新任务超时,已释放更新队列" : null;
const changed = database.transaction(() => {
@@ -430,5 +478,14 @@ export function reconcileOrphanedUpdateJobs(database: Database.Database, config:
})();
if (changed) reconciled += 1;
}
// Prevent a stale request from being replayed after its DB row has been
// marked failed. The path is fixed by the server configuration and the
// operation is safe even when a root runner is racing with this call.
// A download runner refreshes the state marker while it is still using the
// request. Keep the request until that lease also expires; otherwise a
// long download can lose its job id and fail to finalize its row.
if (!stateFresh && (!requestPresent || (requestMtime !== null && now - requestMtime >= ORPHANED_UPDATE_TIMEOUT_MS))) {
removeExpiredRequest(config.updateRequestPath, now);
}
return reconciled;
}
+52 -3
View File
@@ -1,5 +1,5 @@
import { afterEach, describe, expect, it } from "vitest";
import { mkdir, readlink, symlink, writeFile, readFile, stat, readdir } from "node:fs/promises";
import { mkdir, readlink, symlink, writeFile, readFile, stat, readdir, utimes } from "node:fs/promises";
import { mkdtemp, rm } from "node:fs/promises";
import { tmpdir } from "node:os";
import path from "node:path";
@@ -309,14 +309,63 @@ describe("更新安全工具", () => {
const queuedId = randomUUID();
const applyingId = randomUUID();
const stagedId = randomUUID();
const stagedApplyId = randomUUID();
insert.run(queuedId, "apply", "queued", "1.2.0", "linux-x64", "https://updates.example/queued.tar.gz", staleAt, staleAt);
insert.run(applyingId, "apply", "applying", config.appVersion, "linux-x64", "https://updates.example/applying.tar.gz", staleAt, staleAt);
insert.run(stagedId, "download", "staged", "1.2.0", "linux-x64", "https://updates.example/staged.tar.gz", staleAt, staleAt);
insert.run(stagedApplyId, "apply", "staged", "1.2.0", "linux-x64", "https://updates.example/staged-apply.tar.gz", staleAt, staleAt);
const now = Date.now();
expect(reconcileOrphanedUpdateJobs(database.sqlite, config, now)).toBe(2);
expect(reconcileOrphanedUpdateJobs(database.sqlite, config, now)).toBe(3);
expect(database.sqlite.prepare("SELECT status FROM update_jobs WHERE id=?").get(queuedId)).toEqual({ status: "failed" });
expect(database.sqlite.prepare("SELECT status FROM update_jobs WHERE id=?").get(applyingId)).toEqual({ status: "completed" });
expect(database.sqlite.prepare("SELECT status FROM update_jobs WHERE id=?").get(stagedId)).toEqual({ status: "staged" });
expect(database.sqlite.prepare("SELECT status, operation FROM update_jobs WHERE id=?").get(stagedId)).toEqual({ status: "staged", operation: "download" });
expect(database.sqlite.prepare("SELECT status, operation FROM update_jobs WHERE id=?").get(stagedApplyId)).toEqual({ status: "staged", operation: "download" });
} finally {
if (database) database.sqlite.close();
await rm(root, { recursive: true, force: true });
}
});
it("下载心跳有效时不回收任务或删除仍在使用的请求文件", async () => {
const root = await mkdtemp(path.join(tmpdir(), "tallynote-update-heartbeat-"));
let database: ReturnType<typeof openDatabase> | undefined;
try {
const dataDir = path.join(root, "data");
const installPrefix = path.join(root, "install");
process.env.TALLYNOTE_DATA_DIR = dataDir;
process.env.TALLYNOTE_INSTALL_PREFIX = installPrefix;
process.env.TALLYNOTE_PUBLIC_ORIGIN = "http://127.0.0.1:3998";
process.env.TALLYNOTE_COOKIE_SECURE = "false";
process.env.TALLYNOTE_UPDATE_STRATEGY = "systemd";
process.env.TALLYNOTE_UPDATE_METADATA_URL = "https://updates.example/latest";
process.env.TALLYNOTE_UPDATE_ALLOWED_HOSTS = "updates.example";
process.env.TALLYNOTE_UPDATE_REQUIRE_SIGNATURE = "false";
const config = loadConfig();
prepareDataDirectories(config);
await mkdir(path.join(config.releasesDir, config.appVersion, "dist"), { recursive: true });
await symlink(path.join(config.releasesDir, config.appVersion), config.currentLink);
database = openDatabase(config);
const staleAt = Date.now() - ORPHANED_UPDATE_TIMEOUT_MS - 1;
const jobId = randomUUID();
database.sqlite.prepare(`
INSERT INTO update_jobs(id, operation, status, version, platform, asset_url, created_at, updated_at)
VALUES (?, 'download', 'downloading', '1.2.0', 'linux-x64', ?, ?, ?)
`).run(jobId, "https://updates.example/download.tar.gz", staleAt, staleAt);
await writeFile(config.updateRequestPath, JSON.stringify({ jobId, operation: "download" }));
const statePath = path.join(config.installPrefix, ".update-state");
await writeFile(statePath, `job_id=${jobId}\nold_target=${path.join(config.releasesDir, config.appVersion)}\nphase=download\n`);
const now = Date.now();
await utimes(config.updateRequestPath, new Date(staleAt), new Date(staleAt));
await utimes(statePath, new Date(now), new Date(now));
expect(reconcileOrphanedUpdateJobs(database.sqlite, config, now)).toBe(0);
expect(database.sqlite.prepare("SELECT status FROM update_jobs WHERE id=?").get(jobId)).toEqual({ status: "downloading" });
expect(await stat(config.updateRequestPath)).toBeTruthy();
const expiredNow = now + ORPHANED_UPDATE_TIMEOUT_MS + 1;
await utimes(statePath, new Date(staleAt), new Date(staleAt));
expect(reconcileOrphanedUpdateJobs(database.sqlite, config, expiredNow)).toBe(1);
expect(database.sqlite.prepare("SELECT status FROM update_jobs WHERE id=?").get(jobId)).toEqual({ status: "failed" });
await expect(stat(config.updateRequestPath)).rejects.toThrow();
} finally {
if (database) database.sqlite.close();
await rm(root, { recursive: true, force: true });