Harden archives, snapshots, locks, and search against corruption and races
Template tests / tests (pull_request) Failing after 33s
Template tests / tests (pull_request) Failing after 33s
Phase 1/2 of the improvement plan (PR 6 of the sequence). Recovery and
resource-limit hardening for the storage-adjacent modules.
ZIP resource limits (core/zip.js):
- unzipSync now enforces entry-count, total compressed, total inflated, and
per-entry inflated budgets, and caps inflation with inflateRawSync
maxOutputLength so a deflate bomb can't exhaust memory. Exact inflated-size
match (not "at least") and CRC verification are kept. Import uses the
default limits.
Transactional archive import (core/archive.js):
- The import validates the guide AND every step before writing anything, then
stages the whole guide in a temp directory and publishes it with a single
atomic rename. A corrupt step no longer leaves a partial guide in the
library; a failure cleans up the staging directory.
Atomic snapshot restore (core/snapshots.js):
- Restore extracts and validates into a temp directory first; only then does
it swap content in, moving live content aside so a mid-swap failure rolls
back. A corrupt/truncated snapshot can no longer destroy the live guide
(the old restore deleted live content before extracting).
- Fixed snapshot filename collisions: names kept milliseconds so two backups
in the same second no longer overwrite each other.
Automatic backups (core/snapshots.js):
- Implemented the previously-dead backups.automatic/everyNSaves/keepLast
settings: autoSnapshotIfDue snapshots every N saves and prunes to keepLast,
wired into the save choke point in main.js. Never throws — a backup failure
cannot break the save that triggered it.
Exclusive locks (core/locks.js):
- acquireLock uses O_CREAT|O_EXCL (flag 'wx') so only one writer wins the
race; the old read-then-write left a window where two writers both believed
they held the lock. Added a per-acquisition token so release only removes
the exact lock it took (never one a force-steal replaced). Same-process
re-acquire still succeeds; cross-process fresh locks conflict.
Search reconciliation (core/search.js):
- New reconcile(store) rebuilds/repairs the index against the library at
startup using per-guide fingerprints (updatedAt+revision): reindexes new/
changed guides, drops entries for deleted ones, and exposes a recovery
status ('ok'|'reset'|'reconciled'). A missing/corrupt/version-mismatched
index recovers instead of silently returning nothing. Wired into startup.
Recovery surface:
- New recovery:status IPC + preload method returns quarantined files (this
session) and the search index status so the UI can surface data issues.
Tests: ZIP bomb/limits, transactional import abort with no partial guide,
atomic snapshot restore preserving the live guide on corruption, exclusive
lock conflict/steal/release-by-token, same-process re-acquire, search
reconcile (rebuild/drop/reindex/corrupt-reset), and automatic backup
cadence/pruning. 255 unit tests pass; startup smoke and workflow E2E pass.
Co-Authored-By: Claude Fable 5 <[email protected]>
This commit is contained in:
+32
-8
@@ -124,18 +124,40 @@ function importGuideArchive(store, file, { mode = 'copy' } = {}) {
|
||||
}
|
||||
|
||||
function finalizeImport(store, newGuide, idMap, stepJsons, stepFiles) {
|
||||
// Transactional import: validate the guide and EVERY step first, then write
|
||||
// the whole guide into a temporary staging directory, and only publish it
|
||||
// with a single atomic rename. Previously guide.json was written before the
|
||||
// steps validated, so a bad step left a partial guide in the library.
|
||||
validateGuide(newGuide);
|
||||
writeJsonSync(path.join(store.guideDir(newGuide.guideId), 'guide.json'), newGuide);
|
||||
|
||||
const normalizedSteps = [];
|
||||
for (const [stepId, { raw }] of stepJsons) {
|
||||
const step = normalizeStep({ ...raw, stepId });
|
||||
step.parentStepId = raw.parentStepId ? idMap.get(raw.parentStepId) || null : null;
|
||||
validateStep(step);
|
||||
const dir = store.stepDir(newGuide.guideId, stepId);
|
||||
writeJsonSync(path.join(dir, 'step.json'), step);
|
||||
for (const { name, data } of stepFiles.get(stepId) || []) {
|
||||
atomicWriteFileSync(path.join(dir, name), data);
|
||||
validateStep(step); // throws before anything is written on a bad step
|
||||
normalizedSteps.push([stepId, step]);
|
||||
}
|
||||
|
||||
const finalDir = store.guideDir(newGuide.guideId);
|
||||
if (fs.existsSync(finalDir)) throw new Error(`guide already exists: ${newGuide.guideId}`);
|
||||
const stagingDir = `${finalDir}.importing-${Date.now()}`;
|
||||
fs.rmSync(stagingDir, { recursive: true, force: true });
|
||||
try {
|
||||
fs.mkdirSync(stagingDir, { recursive: true });
|
||||
writeJsonSync(path.join(stagingDir, 'guide.json'), newGuide);
|
||||
for (const [stepId, step] of normalizedSteps) {
|
||||
const dir = path.join(stagingDir, 'steps', stepId);
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
writeJsonSync(path.join(dir, 'step.json'), step);
|
||||
for (const { name, data } of stepFiles.get(stepId) || []) {
|
||||
atomicWriteFileSync(path.join(dir, name), data);
|
||||
}
|
||||
}
|
||||
// Publish atomically. If the final dir appeared meanwhile, fail cleanly.
|
||||
if (fs.existsSync(finalDir)) throw new Error(`guide already exists: ${newGuide.guideId}`);
|
||||
fs.renameSync(stagingDir, finalDir);
|
||||
} catch (err) {
|
||||
fs.rmSync(stagingDir, { recursive: true, force: true });
|
||||
throw err;
|
||||
}
|
||||
return store.getGuide(newGuide.guideId);
|
||||
}
|
||||
@@ -160,7 +182,9 @@ function saveLinkedGuide(store, guideId, { force = false } = {}) {
|
||||
store.saveGuide(guide, { touch: false });
|
||||
return { saved: true, path: target };
|
||||
} finally {
|
||||
releaseLock(target);
|
||||
// Release by our acquisition token so we never remove a lock a concurrent
|
||||
// force-steal replaced with theirs.
|
||||
releaseLock(target, { lock: result.lock });
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user