This commit is contained in:
@@ -3,6 +3,7 @@ import { loadConfig, prepareDataDirectories, acquireInstanceLock } from "./confi
|
||||
import { openDatabase } from "./db/index.js";
|
||||
import { buildApp } from "./app.js";
|
||||
import { cleanupOrphanedExports, expireExports, resumeExports } from "./exporter.js";
|
||||
import { reconcileOrphanedUpdateJobs } from "./update-service.js";
|
||||
|
||||
const config = loadConfig();
|
||||
prepareDataDirectories(config);
|
||||
@@ -18,6 +19,7 @@ async function start() {
|
||||
await expireExports(database.sqlite, config);
|
||||
await cleanupOrphanedExports(database.sqlite, config);
|
||||
await resumeExports(database.sqlite, config);
|
||||
reconcileOrphanedUpdateJobs(database.sqlite, config);
|
||||
const app = await buildApp(database, config);
|
||||
const janitor = setInterval(() => {
|
||||
void cleanupStaging(config);
|
||||
@@ -27,6 +29,7 @@ async function start() {
|
||||
void processFileDeletions(database.sqlite, config);
|
||||
void expireExports(database.sqlite, config);
|
||||
void cleanupOrphanedExports(database.sqlite, config);
|
||||
reconcileOrphanedUpdateJobs(database.sqlite, config);
|
||||
}, 60_000);
|
||||
const shutdown = async () => {
|
||||
clearInterval(janitor);
|
||||
|
||||
+74
-17
@@ -1,4 +1,4 @@
|
||||
import { lstatSync, realpathSync } from "node:fs";
|
||||
import { lstatSync, realpathSync, unlinkSync } from "node:fs";
|
||||
import { chmod, mkdir, rename, writeFile } from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { createPublicKey, randomUUID, verify as verifySignature } from "node:crypto";
|
||||
@@ -33,9 +33,9 @@ export const ACTIVE_UPDATE_STATUSES: readonly UpdateJobStatus[] = [
|
||||
];
|
||||
|
||||
// A queued job normally starts within seconds and an applying job completes
|
||||
// after the service health check. This grace period only applies when both
|
||||
// hand-off markers are gone, so an active runner is never reclaimed midway
|
||||
// through a download, backup, or switch.
|
||||
// after the service health check. The runner refreshes its recovery marker as
|
||||
// a lease while doing long downloads/backups; only an expired lease permits
|
||||
// the server to reclaim an active row.
|
||||
export const ORPHANED_UPDATE_TIMEOUT_MS = 5 * 60 * 1000;
|
||||
|
||||
export type CachedRelease = {
|
||||
@@ -364,12 +364,24 @@ export function publicUpdateJob(row: Record<string, unknown> | undefined): Recor
|
||||
};
|
||||
}
|
||||
|
||||
function markerExists(filePath: string): boolean {
|
||||
function markerMtime(filePath: string): number | null {
|
||||
try {
|
||||
const info = lstatSync(filePath);
|
||||
return info.isFile() || info.isSymbolicLink();
|
||||
return info.isFile() ? info.mtimeMs : null;
|
||||
} catch {
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function removeExpiredRequest(filePath: string, now: number): void {
|
||||
try {
|
||||
const info = lstatSync(filePath);
|
||||
if (!info.isFile() && !info.isSymbolicLink()) return;
|
||||
if (now - info.mtimeMs < ORPHANED_UPDATE_TIMEOUT_MS) return;
|
||||
unlinkSync(filePath);
|
||||
} catch {
|
||||
// The root runner may own the marker during a recovery race. The DB
|
||||
// transition below is still enough to release the browser queue.
|
||||
}
|
||||
}
|
||||
|
||||
@@ -385,28 +397,64 @@ function currentReleaseVersion(config: AppConfig): string | null {
|
||||
}
|
||||
|
||||
/**
|
||||
* Release an update row left behind after both privileged hand-off markers
|
||||
* disappeared. This is deliberately conservative: staged downloads remain
|
||||
* available for an explicit apply, and any visible marker means the runner
|
||||
* still owns recovery.
|
||||
* Release an update row left behind after its privileged runner lease expired.
|
||||
* This is deliberately conservative: staged downloads remain available for an
|
||||
* explicit apply, and a fresh request/state marker means the runner still owns
|
||||
* recovery.
|
||||
*/
|
||||
export function reconcileOrphanedUpdateJobs(database: Database.Database, config: AppConfig, now = Date.now()): number {
|
||||
const placeholders = ACTIVE_UPDATE_STATUSES.map(() => "?").join(",");
|
||||
const rows = database.prepare(`
|
||||
SELECT id, status, version, admin_id AS adminId, request_id AS requestId,
|
||||
SELECT id, status, operation, version, admin_id AS adminId, request_id AS requestId,
|
||||
updated_at AS updatedAt
|
||||
FROM update_jobs
|
||||
WHERE status IN (${placeholders})
|
||||
ORDER BY updated_at ASC
|
||||
`).all(...ACTIVE_UPDATE_STATUSES) as Array<{ id: string; status: UpdateJobStatus; version: string; adminId: string | null; requestId: string | null; updatedAt: number | null }>;
|
||||
`).all(...ACTIVE_UPDATE_STATUSES) as Array<{ id: string; status: UpdateJobStatus; operation: "download" | "apply"; version: string; adminId: string | null; requestId: string | null; updatedAt: number | null }>;
|
||||
if (rows.length === 0) return 0;
|
||||
const requestPresent = markerExists(config.updateRequestPath);
|
||||
const statePresent = markerExists(path.join(config.installPrefix, ".update-state"));
|
||||
if (requestPresent || statePresent) return 0;
|
||||
const statePath = path.join(config.installPrefix, ".update-state");
|
||||
const requestMtime = markerMtime(config.updateRequestPath);
|
||||
const stateMtime = markerMtime(statePath);
|
||||
const requestPresent = requestMtime !== null;
|
||||
const statePresent = stateMtime !== null;
|
||||
const requestFresh = requestPresent && now - (requestMtime ?? 0) < ORPHANED_UPDATE_TIMEOUT_MS;
|
||||
const stateFresh = statePresent && now - (stateMtime ?? 0) < ORPHANED_UPDATE_TIMEOUT_MS;
|
||||
// A staged download is normally kept for an explicit apply. The one
|
||||
// exception is the hand-off window where the API has already changed the
|
||||
// operation to `apply` but crashed before writing the request file. That
|
||||
// row is still safe to retry and must not block the queue forever.
|
||||
const releaseVersion = currentReleaseVersion(config);
|
||||
let reconciled = 0;
|
||||
for (const row of rows) {
|
||||
if (row.status === "staged" || typeof row.updatedAt !== "number" || now - row.updatedAt < ORPHANED_UPDATE_TIMEOUT_MS) continue;
|
||||
if (typeof row.updatedAt !== "number" || now - row.updatedAt < ORPHANED_UPDATE_TIMEOUT_MS) continue;
|
||||
// The runner refreshes the state marker while a download is in flight.
|
||||
// A stale request/state marker therefore no longer protects an orphaned
|
||||
// row forever, while a fresh marker remains owned by the runner.
|
||||
if (row.status === "staged") {
|
||||
if (row.operation !== "apply" || requestFresh || stateFresh) continue;
|
||||
const changed = database.transaction(() => {
|
||||
const result = database.prepare(`
|
||||
UPDATE update_jobs
|
||||
SET operation='download', error_message=NULL, updated_at=?
|
||||
WHERE id=? AND status='staged' AND operation='apply' AND updated_at=?
|
||||
`).run(now, row.id, row.updatedAt);
|
||||
if (result.changes !== 1) return false;
|
||||
writeAudit(database, {
|
||||
requestId: row.requestId || randomUUID(),
|
||||
actorAdminId: row.adminId,
|
||||
action: "update.reconciled",
|
||||
targetType: "update",
|
||||
targetId: row.id,
|
||||
outcome: "success",
|
||||
before: { status: row.status, operation: row.operation, version: row.version },
|
||||
after: { status: "staged", operation: "download", version: row.version, reason: "apply_request_missing" },
|
||||
});
|
||||
return true;
|
||||
})();
|
||||
if (changed) reconciled += 1;
|
||||
continue;
|
||||
}
|
||||
if (requestFresh || stateFresh) continue;
|
||||
const status: "completed" | "failed" = row.status === "applying" && releaseVersion === row.version ? "completed" : "failed";
|
||||
const errorMessage = status === "failed" ? "更新任务超时,已释放更新队列" : null;
|
||||
const changed = database.transaction(() => {
|
||||
@@ -430,5 +478,14 @@ export function reconcileOrphanedUpdateJobs(database: Database.Database, config:
|
||||
})();
|
||||
if (changed) reconciled += 1;
|
||||
}
|
||||
// Prevent a stale request from being replayed after its DB row has been
|
||||
// marked failed. The path is fixed by the server configuration and the
|
||||
// operation is safe even when a root runner is racing with this call.
|
||||
// A download runner refreshes the state marker while it is still using the
|
||||
// request. Keep the request until that lease also expires; otherwise a
|
||||
// long download can lose its job id and fail to finalize its row.
|
||||
if (!stateFresh && (!requestPresent || (requestMtime !== null && now - requestMtime >= ORPHANED_UPDATE_TIMEOUT_MS))) {
|
||||
removeExpiredRequest(config.updateRequestPath, now);
|
||||
}
|
||||
return reconciled;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user