@compr/opscontext-mcp 2.10.0 → 2.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/audit.js CHANGED
@@ -46,7 +46,7 @@
46
46
  // Records every state-changing operation. Each line carries the SHA-256 hash
47
47
  // of the previous line's canonical content, so mutation of any historical
48
48
  // record breaks chain verification at that index.
49
- import { existsSync, mkdirSync, readFileSync, appendFileSync, openSync, closeSync, unlinkSync, statSync, writeSync, readSync, fsyncSync, renameSync, linkSync, readdirSync, constants, } from "fs";
49
+ import { existsSync, mkdirSync, readFileSync, writeFileSync, appendFileSync, openSync, closeSync, unlinkSync, statSync, writeSync, readSync, fsyncSync, renameSync, linkSync, readdirSync, ftruncateSync, fstatSync, constants, } from "fs";
50
50
  import { basename, join } from "path";
51
51
  import { homedir } from "os";
52
52
  import { createHash } from "crypto";
@@ -102,14 +102,10 @@ function acquireLockSync() {
102
102
  // Lockfile exists. Check if it's stale.
103
103
  try {
104
104
  const st = statSync(path);
105
- if (Date.now() - st.mtimeMs > STALE_LOCK_MS) {
106
- // Orphaned — force-unlink and retry.
107
- try {
108
- unlinkSync(path);
109
- }
110
- catch {
111
- /* another process just cleaned it; retry */
112
- }
105
+ // [LOCK] [A-DEAD-HOLDER-LOSES-THE-LOCK-AT-ONCE]
106
+ if (lockHolderIsGone(path, st) || Date.now() - st.mtimeMs > STALE_LOCK_MS) {
107
+ // Orphaned — force-unlink and retry, only if it is still the file we judged.
108
+ unlinkIfUnchanged(path, st);
113
109
  continue;
114
110
  }
115
111
  }
@@ -121,6 +117,51 @@ function acquireLockSync() {
121
117
  }
122
118
  throw new Error(`Failed to acquire audit lock at ${path} within ${LOCK_TIMEOUT_MS}ms`);
123
119
  }
120
+ /**
121
+ * [LOCKED] [A-DEAD-HOLDER-LOSES-THE-LOCK-AT-ONCE] - 2026-09-27
122
+ * [NEVER] make writers wait out STALE_LOCK_MS for a holder that is provably gone.
123
+ * WHY: a writer killed while holding the append lock left it for 10 s; every other writer waited 2 s
124
+ * per entry and lost it (4 per writer per crash, 3 of 3 runs, E2E_REVIEW_2026-09 B1-2; 448 such
125
+ * losses in the launchd log before 2.5.8). It happened again on 2026-09-27 at 17:01Z: a chat
126
+ * server died holding the lock and a new server lost two entries. The lock file names its
127
+ * holder's pid; nothing read it.
128
+ * FIX: a lock whose pid no longer exists, or an empty lock older than a second (its holder died
129
+ * between creating it and writing its pid; a live holder writes it at once), is broken now. A
130
+ * live pid, even a reused one, keeps today's 10 s rule. The file is removed only if it is still
131
+ * the one judged (same inode and time), which narrows the gap where two waiters both break it.
132
+ * [LOCK] [AUDIT-001-WRITE-RACE-FIX]: the lock stays; only the stale test is sharper.
133
+ */
134
+ function lockHolderIsGone(path, st) {
135
+ let body = "";
136
+ try {
137
+ body = readFileSync(path, "utf-8");
138
+ }
139
+ catch {
140
+ return false;
141
+ }
142
+ const pid = parseInt(body.split("\n")[0], 10);
143
+ if (!Number.isInteger(pid) || pid <= 0)
144
+ return Date.now() - st.mtimeMs > 1000;
145
+ if (pid === process.pid)
146
+ return false;
147
+ try {
148
+ process.kill(pid, 0);
149
+ return false; // alive (or not ours to signal): wait as before
150
+ }
151
+ catch (e) {
152
+ return e.code === "ESRCH";
153
+ }
154
+ }
155
+ function unlinkIfUnchanged(path, judged) {
156
+ try {
157
+ const now = statSync(path);
158
+ if (now.ino === judged.ino && now.mtimeMs === judged.mtimeMs)
159
+ unlinkSync(path);
160
+ }
161
+ catch {
162
+ /* already gone: another waiter broke it */
163
+ }
164
+ }
124
165
  function auditDir() {
125
166
  // CONTEXTENGINE_HOME lets tests run against a temp dir without touching ~/.contextengine
126
167
  return process.env.CONTEXTENGINE_HOME || join(homedir(), ".contextengine");
@@ -194,6 +235,10 @@ function readLastHash() {
194
235
  * was rendered as the specific, plausible claim "there is no history".
195
236
  * FIX: throw. appendAudit() must surface problems loudly (see [AUDIT-CHAIN]); call sites
196
237
  * that need isolation already use safeAppend(), which logs to stderr and continues.
238
+ * 2026-09-27: a last record cut short (no final newline) is no longer left to block every append
239
+ * for good: it is set aside and noted on the chain first. [LOCK] [TORN-TAIL-IS-KEPT-AND-CHAINED]
240
+ * This throw remains for a complete last line that is not a record, and for a log that holds
241
+ * no complete record at all; neither ever chains onto genesis.
197
242
  */
198
243
  function parseHeadOrThrow(line) {
199
244
  let rec;
@@ -201,9 +246,9 @@ function parseHeadOrThrow(line) {
201
246
  rec = JSON.parse(line);
202
247
  }
203
248
  catch {
204
- throw new Error("Audit log tail is not valid JSON — refusing to append onto an unknown head. " +
205
- "Inspect the last line of ~/.contextengine/audit.log; a partial final record can be " +
206
- "removed by hand, which verifyChain() will then confirm.");
249
+ throw new Error("Audit log tail is not valid JSON, refusing to append onto an unknown head. " +
250
+ "The last line of ~/.contextengine/audit.log is complete but is not a record (a record cut " +
251
+ "short is set aside automatically); inspect it, and 'contextengine audit-verify' says where it is.");
207
252
  }
208
253
  if (typeof rec.hash !== "string" || rec.hash.length !== 64) {
209
254
  throw new Error("Audit log tail has no usable hash — refusing to append onto an unknown head.");
@@ -263,39 +308,239 @@ export function appendAudit(event, payload, actor = "system") {
263
308
  ensureDir();
264
309
  const release = acquireLockSync();
265
310
  try {
266
- const path = auditPath();
267
- // 🔒 LOCKED [AUDIT-HEAD-FROM-DISK] — 2026-08-17
268
- // ⛔ NEVER derive the head hash from an in-process cache again.
269
- // WHY: the previous code trusted `cachedLastHash` whenever `statSync().size` matched
270
- // a locally-tracked `cachedSize` that was ARITHMETIC (`cachedSize += byteLength`),
271
- // not observed. Any divergence between bytes-we-think-we-wrote and bytes-on-disk
272
- // — a partial write, a concurrent writer whose bytes happened to sum the same, an
273
- // externally rotated/truncated log — left us hashing onto a head that is not the
274
- // real tail, forking the chain. It was a correctness guarantee resting on a
275
- // perf cache, which the file's own [audit-001-write-race] LOCK explicitly warns
276
- // against ("the in-process chain cache is a perf optimization, NOT a correctness
277
- // guarantee").
278
- // FIX: with [AUDIT-TAIL-READ-IS-O1] the true head costs ~0 ms, so there is nothing left
279
- // to optimise. Read it from disk under the lock, every time. The cache is gone.
280
- const prevHash = readLastHash();
281
- const ts = new Date().toISOString();
282
- const hash = computeHash(prevHash, ts, event, actor, payload);
283
- const record = {
284
- ts,
285
- event,
286
- actor,
287
- payload,
288
- prev_hash: prevHash,
289
- hash,
290
- };
291
- const line = JSON.stringify(record) + "\n";
292
- appendFileSync(path, line);
293
- return record;
311
+ return appendHoldingLock(event, payload, actor);
294
312
  }
295
313
  finally {
296
314
  release();
297
315
  }
298
316
  }
317
+ /** appendAudit() for a caller that already holds the append lock (the scrub of the live log). */
318
+ function appendHoldingLock(event, payload, actor) {
319
+ const path = auditPath();
320
+ settleTail(path);
321
+ // [LOCK] [A-REFUSED-APPEND-IS-COUNTED-AND-CHAINED]
322
+ const refused = takeRefusals();
323
+ if (refused) {
324
+ try {
325
+ writeRecord(path, "audit.append_failed", refused.summary, "system");
326
+ refused.done();
327
+ }
328
+ catch (e) {
329
+ refused.putBack();
330
+ throw e;
331
+ }
332
+ }
333
+ return writeRecord(path, event, payload, actor);
334
+ }
335
+ /** Under the append lock: a last record cut short is set aside and noted before anything else is
336
+ * written, so the log ends with a newline afterwards. [LOCK] [TORN-TAIL-IS-KEPT-AND-CHAINED] */
337
+ function settleTail(path) {
338
+ const torn = repairTornTail(path);
339
+ if (torn) {
340
+ writeRecord(path, "audit.torn_tail", {
341
+ kept: torn.kept,
342
+ bytes: torn.bytes,
343
+ sha256: torn.sha256,
344
+ note: "the last record was cut short (full disk or crash); its bytes were moved to the kept file and the log continues from the last complete record",
345
+ }, "system");
346
+ }
347
+ }
348
+ function writeRecord(path, event, payload, actor) {
349
+ // 🔒 LOCKED [AUDIT-HEAD-FROM-DISK] — 2026-08-17
350
+ // ⛔ NEVER derive the head hash from an in-process cache again.
351
+ // WHY: the previous code trusted `cachedLastHash` whenever `statSync().size` matched
352
+ // a locally-tracked `cachedSize` that was ARITHMETIC (`cachedSize += byteLength`),
353
+ // not observed. Any divergence between bytes-we-think-we-wrote and bytes-on-disk
354
+ // — a partial write, a concurrent writer whose bytes happened to sum the same, an
355
+ // externally rotated/truncated log — left us hashing onto a head that is not the
356
+ // real tail, forking the chain. It was a correctness guarantee resting on a
357
+ // perf cache, which the file's own [audit-001-write-race] LOCK explicitly warns
358
+ // against ("the in-process chain cache is a perf optimization, NOT a correctness
359
+ // guarantee").
360
+ // FIX: with [AUDIT-TAIL-READ-IS-O1] the true head costs ~0 ms, so there is nothing left
361
+ // to optimise. Read it from disk under the lock, every time. The cache is gone.
362
+ const prevHash = readLastHash();
363
+ const ts = new Date().toISOString();
364
+ const hash = computeHash(prevHash, ts, event, actor, payload);
365
+ const record = {
366
+ ts,
367
+ event,
368
+ actor,
369
+ payload,
370
+ prev_hash: prevHash,
371
+ hash,
372
+ };
373
+ const line = JSON.stringify(record) + "\n";
374
+ appendFileSync(path, line);
375
+ return record;
376
+ }
377
+ /**
378
+ * [LOCKED] [TORN-TAIL-IS-KEPT-AND-CHAINED] - 2026-09-27
379
+ * [NEVER] leave a cut-short final record in place (every later append is refused for good), delete
380
+ * its bytes, or chain the next record onto anything but the last complete record.
381
+ * WHY: a full disk cut one record in half. [UNREADABLE-HEAD-IS-NOT-GENESIS] then refused every
382
+ * later append, correctly never chaining onto genesis, but for good: 194,875 refusals in the
383
+ * replay, 5 of 5 still refused once space was back, each reported only on stderr, while the
384
+ * receiver answered {"ok":true} and `emit-event` printed "Appended". audit-verify said "0
385
+ * record(s) checked" about 3,564 intact ones (E2E_REVIEW_2026-09 B1-1, 3 of 3 runs). A torn
386
+ * write has one signature: the file does not end with a newline. `kill -9` never produced one
387
+ * (10 of 10 kills during 2 MB appends); a full disk did every time.
388
+ * FIX: under the append lock, when the file does not end with a newline: a fragment that is a
389
+ * whole record only lost its newline, which is added back. Anything else is copied to
390
+ * audit.torn-<time>.partial (fsynced, never deleted), the log is cut back to its last complete
391
+ * record, and an `audit.torn_tail` record naming the kept file, its length and SHA-256 is
392
+ * chained onto that record before the caller's. A file holding no complete record at all is
393
+ * left alone and the append throws, as before. A last line that ends with a newline and is
394
+ * not JSON is not a torn write: it still throws. [LOCK] [UNREADABLE-HEAD-IS-NOT-GENESIS]
395
+ */
396
+ function repairTornTail(path) {
397
+ // One open and one fstat on the common path (runs before every append).
398
+ let fd;
399
+ try {
400
+ fd = openSync(path, constants.O_RDWR);
401
+ }
402
+ catch (e) {
403
+ if (e.code === "ENOENT")
404
+ return null;
405
+ throw e;
406
+ }
407
+ try {
408
+ const size = fstatSync(fd).size;
409
+ if (size === 0)
410
+ return null;
411
+ const last = Buffer.alloc(1);
412
+ readSync(fd, last, 0, 1, size - 1);
413
+ if (last[0] === 10)
414
+ return null; // ends with a complete line: nothing was cut short
415
+ // Find the last newline, one window at a time from the end.
416
+ let nl = -1;
417
+ let pos = size;
418
+ const win = Buffer.alloc(TAIL_READ_BYTES);
419
+ while (pos > 0 && nl === -1) {
420
+ const start = Math.max(0, pos - win.length);
421
+ const n = readSync(fd, win, 0, pos - start, start);
422
+ const i = win.subarray(0, n).lastIndexOf(10);
423
+ if (i !== -1)
424
+ nl = start + i;
425
+ pos = start;
426
+ }
427
+ const fragStart = nl + 1;
428
+ const frag = Buffer.alloc(size - fragStart);
429
+ readSync(fd, frag, 0, frag.length, fragStart);
430
+ let whole = false;
431
+ try {
432
+ const r = JSON.parse(frag.toString("utf-8"));
433
+ whole = typeof r?.hash === "string" && r.hash.length === 64;
434
+ }
435
+ catch {
436
+ whole = false;
437
+ }
438
+ if (whole) {
439
+ writeSync(fd, "\n", size); // a complete record that only lost its newline
440
+ return null;
441
+ }
442
+ if (nl === -1)
443
+ return null; // no complete record to continue from: the head read throws
444
+ const kept = `audit.torn-${new Date().toISOString().replace(/[:.]/g, "-")}.partial`;
445
+ try {
446
+ writeFileAndSync(join(auditDir(), kept), frag);
447
+ }
448
+ catch (e) {
449
+ safeUnlink(join(auditDir(), kept)); // still a full disk: no half-written copy per attempt; the log is untouched
450
+ throw e;
451
+ }
452
+ ftruncateSync(fd, fragStart);
453
+ fsyncSync(fd);
454
+ return { kept, bytes: frag.length, sha256: createHash("sha256").update(frag).digest("hex") };
455
+ }
456
+ finally {
457
+ closeSync(fd);
458
+ }
459
+ }
460
+ /**
461
+ * [LOCKED] [A-REFUSED-APPEND-IS-COUNTED-AND-CHAINED] - 2026-09-27
462
+ * [NEVER] let safeAppend() drop an entry with a stderr line as the only trace, or report success
463
+ * for an append that failed.
464
+ * WHY: safeAppend() isolates hot paths from audit failures, and its only surface was stderr. A
465
+ * writer that crashed inside the append lock cost every other writer ~4 entries over 10 s
466
+ * (B1-2, 3 of 3 runs; 448 such lines in the real launchd log before 2.5.8), a torn tail cost
467
+ * all of them (B1-1), and nothing on the chain or in fleet health said so. The receiver and
468
+ * `emit-event` told their senders "written" regardless.
469
+ * FIX: safeAppend() returns whether it wrote. A refusal adds one line to audit-refused.jsonl (best
470
+ * effort: on a full disk nothing can be written anywhere). The next append that succeeds takes
471
+ * the file (rename, under the append lock) and chains one `audit.append_failed` record with
472
+ * the count, the time span, the kinds and the errors, so the gap is on the chain; if that
473
+ * record cannot be written the lines go back. Fleet health reads both. [LOCK] [HEALTH-SEES-THE-CHAIN]
474
+ */
475
+ function refusedPath() {
476
+ return join(auditDir(), "audit-refused.jsonl");
477
+ }
478
+ function recordRefusal(event, error) {
479
+ try {
480
+ appendFileSync(refusedPath(), JSON.stringify({ ts: new Date().toISOString(), pid: process.pid, event, error: error.slice(0, 200) }) + "\n");
481
+ }
482
+ catch {
483
+ /* a full disk: nothing can be written anywhere, stderr already has it */
484
+ }
485
+ }
486
+ /** Pending refusals, taken out of the way; `done` drops them once chained, `putBack` restores them. */
487
+ function takeRefusals() {
488
+ const path = refusedPath();
489
+ if (!existsSync(path))
490
+ return null;
491
+ const taking = `${path}.${process.pid}.taking`;
492
+ let body;
493
+ try {
494
+ renameSync(path, taking);
495
+ body = readFileSync(taking, "utf-8");
496
+ }
497
+ catch {
498
+ return null; // another writer took it, or it vanished: nothing to chain from here
499
+ }
500
+ const events = {};
501
+ const errors = {};
502
+ const pids = new Set();
503
+ let count = 0;
504
+ let first = "";
505
+ let last = "";
506
+ for (const line of body.split("\n")) {
507
+ if (!line)
508
+ continue;
509
+ let r;
510
+ try {
511
+ r = JSON.parse(line);
512
+ }
513
+ catch {
514
+ continue;
515
+ }
516
+ count++;
517
+ if (r.ts && (!first || r.ts < first))
518
+ first = r.ts;
519
+ if (r.ts && r.ts > last)
520
+ last = r.ts;
521
+ if (r.event)
522
+ events[r.event] = (events[r.event] ?? 0) + 1;
523
+ if (r.error)
524
+ errors[r.error] = (errors[r.error] ?? 0) + 1;
525
+ if (typeof r.pid === "number")
526
+ pids.add(r.pid);
527
+ }
528
+ if (count === 0) {
529
+ safeUnlink(taking);
530
+ return null;
531
+ }
532
+ return {
533
+ summary: { count, first, last, events, errors, pids: [...pids] },
534
+ done: () => safeUnlink(taking),
535
+ putBack: () => {
536
+ try {
537
+ appendFileSync(path, body);
538
+ safeUnlink(taking);
539
+ }
540
+ catch { /* the taking file stays for the next run */ }
541
+ },
542
+ };
543
+ }
299
544
  /**
300
545
  * 🔒 LOCKED [ROTATION-MUST-NOT-ORPHAN-THE-CHAIN] — 2026-08-20
301
546
  * ⛔ NEVER rotate by truncating, moving or deleting audit.log. NEVER let a rotated log
@@ -399,32 +644,53 @@ function placeWithoutOverwrite(tmp, target) {
399
644
  safeUnlink(tmp);
400
645
  return true;
401
646
  }
402
- function parseLines(data, label) {
403
- return data
404
- .split("\n")
405
- .filter(Boolean)
406
- .map((line, i) => {
647
+ function parseLines(data, label, unreadable, baseIndex = 0) {
648
+ const out = [];
649
+ const lines = data.split("\n");
650
+ for (let i = 0; i < lines.length; i++) {
651
+ const line = lines[i];
652
+ if (!line)
653
+ continue;
407
654
  try {
408
- return JSON.parse(line);
655
+ out.push(JSON.parse(line));
409
656
  }
410
657
  catch {
411
- throw new Error(`Corrupt audit line ${i + 1} in ${label}: not valid JSON`);
658
+ if (!unreadable)
659
+ throw new Error(`Corrupt audit line ${i + 1} in ${label}: not valid JSON`);
660
+ unreadable.push({ file: label, line: i + 1, beforeIndex: baseIndex + out.length });
412
661
  }
413
- });
662
+ }
663
+ return out;
414
664
  }
415
665
  export function readAuditLog(opts = {}) {
416
- const includeArchives = opts.includeArchives !== false;
666
+ return readHistory(opts.includeArchives !== false);
667
+ }
668
+ /**
669
+ * The history, segments then live log. With `unreadable`, a line that is not JSON is listed there
670
+ * and skipped instead of aborting the read.
671
+ *
672
+ * [LOCKED] [VERIFY-READS-PAST-AN-UNREADABLE-LINE] - 2026-09-27
673
+ * [NEVER] let one unreadable line stop the verifier from checking every other record.
674
+ * WHY: verifyChain() read through readAuditLog(), which throws on the first line that is not JSON,
675
+ * so one cut-short record made audit-verify print "FAILED, 0 record(s) checked" and "treat the
676
+ * affected records as unverified" about 3,564 intact ones (E2E_REVIEW_2026-09 B1-1): the
677
+ * [VERIFY-FORK-IS-NOT-TAMPER] failure again, a verdict on everything from one bad line.
678
+ * FIX: the verifier reads tolerantly: an unreadable line is reported with its file and line number,
679
+ * makes the report fail, and every other record is still checked. Every other reader keeps the
680
+ * strict read, which throws.
681
+ */
682
+ function readHistory(includeArchives, unreadable) {
417
683
  const path = auditPath();
418
- const live = existsSync(path) ? parseLines(readFileSync(path, "utf-8"), "audit.log") : [];
684
+ const liveText = existsSync(path) ? readFileSync(path, "utf-8") : "";
419
685
  if (!includeArchives)
420
- return live;
686
+ return parseLines(liveText, "audit.log", unreadable);
421
687
  const segments = listSegments();
422
688
  if (segments.length === 0)
423
- return live;
689
+ return parseLines(liveText, "audit.log", unreadable);
424
690
  const history = [];
425
691
  let lastSegmentHashes = new Set();
426
692
  for (const f of segments) {
427
- const recs = parseLines(readFileSync(join(archiveDir(), f), "utf-8"), f);
693
+ const recs = parseLines(readFileSync(join(archiveDir(), f), "utf-8"), f, unreadable, history.length);
428
694
  // 🔒 LOCKED [NO-SPREAD-OVER-A-SEGMENT] — 2026-08-20
429
695
  // ⛔ NEVER use push(...records) on a segment. Found on the first real rotation:
430
696
  // a 494,152-record segment threw "Maximum call stack size exceeded" because the
@@ -434,6 +700,7 @@ export function readAuditLog(opts = {}) {
434
700
  history.push(r);
435
701
  lastSegmentHashes = new Set(recs.map((r) => r.hash));
436
702
  }
703
+ const live = parseLines(liveText, "audit.log", unreadable, history.length);
437
704
  // Seam de-dup — see [ROTATE-ARCHIVE-BEFORE-TRUNCATE]. A crash after the segment was
438
705
  // renamed but before the live log was truncated leaves the archived prefix present in
439
706
  // both files. Drop only the LEADING run of live records already in the last segment;
@@ -441,6 +708,12 @@ export function readAuditLog(opts = {}) {
441
708
  let start = 0;
442
709
  while (start < live.length && lastSegmentHashes.has(live[start].hash))
443
710
  start++;
711
+ if (unreadable && start > 0) {
712
+ const base = history.length;
713
+ for (const u of unreadable)
714
+ if (u.file === "audit.log")
715
+ u.beforeIndex = Math.max(base, u.beforeIndex - start);
716
+ }
444
717
  // [LOCK] [NO-SPREAD-OVER-A-SEGMENT] — same reason.
445
718
  for (let i = start; i < live.length; i++)
446
719
  history.push(live[i]);
@@ -522,6 +795,7 @@ export function rotateAuditLog(opts = {}) {
522
795
  };
523
796
  }
524
797
  try {
798
+ finishInterruptedMoves(); // [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED]
525
799
  return rotateHoldingLock(opts);
526
800
  }
527
801
  finally {
@@ -575,8 +849,20 @@ function rotateHoldingLock(opts) {
575
849
  mkdirSync(adir, { recursive: true });
576
850
  // Snapshot outside the lock: parsing 500k records is far too slow to hold the append
577
851
  // lock for, and acquireLockSync() force-breaks locks older than STALE_LOCK_MS.
578
- const snapshotSize = statSync(path).size;
579
- const live = parseLines(readFileSync(path, "utf-8"), "audit.log");
852
+ //
853
+ // [LOCKED] [ROTATION-SNAPSHOT-IS-THE-BYTES-READ] - 2026-09-27
854
+ // [NEVER] take the snapshot size from statSync() and the records from a separate read.
855
+ // WHY: Node 20's readFileSync(path, "utf-8") reads to the end of the file, past the size a
856
+ // statSync() just before it returned, whenever a writer appends during the read (10 of 10
857
+ // reads in a replay). The records appended in between were then parsed into the remainder
858
+ // AND copied again as raw bytes from the old size: written twice. It happened for real:
859
+ // 20 learning.import records of 2026-09-25 19:45:46 sit twice in audit-0062.jsonl, and the
860
+ // verifier called it a concurrent-append fork. Replayed 6 of 6 (E2E_REVIEW_2026-09 B2-1).
861
+ // FIX: one read, cut at its last newline; that byte length IS the snapshot, so the raw copy
862
+ // below starts exactly where the parsed records end, whatever the read returned.
863
+ const snapshotBuf = readFileSync(path);
864
+ const snapshotSize = snapshotBuf.lastIndexOf(10) + 1;
865
+ const live = parseLines(snapshotBuf.subarray(0, snapshotSize).toString("utf-8"), "audit.log");
580
866
  // [LOCK] [ROTATION-HOLDS-THE-LOCK-BEFORE-IT-PLANS]: archiveCount is a count from the plan's
581
867
  // read. Appends at the tail since then are fine; a different head means someone else cut or
582
868
  // rewrote the log, and slicing it with an old count archives the wrong records.
@@ -591,9 +877,20 @@ function rotateHoldingLock(opts) {
591
877
  const segName = plan.segmentFile;
592
878
  const segTmp = join(adir, `.${segName}.tmp`);
593
879
  const segBody = archived.map((r) => JSON.stringify(r)).join("\n") + "\n";
880
+ const rotateRecord = {
881
+ segment: segName,
882
+ archived_records: archived.length,
883
+ first_hash: archived[0].hash,
884
+ last_hash: archived[archived.length - 1].hash,
885
+ cutoff: plan.cutoff,
886
+ };
887
+ // [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED]: from here to the audit.rotate record, a crash leaves
888
+ // this note, and the next rotation, restore or scrub finishes the job from it.
889
+ writeIntent(ROTATE_INTENT, rotateRecord);
594
890
  writeFileAndSync(segTmp, segBody);
595
891
  // [LOCK] [SEGMENT-IS-NEVER-OVERWRITTEN]
596
892
  if (!placeWithoutOverwrite(segTmp, join(adir, segName))) {
893
+ clearIntent(ROTATE_INTENT);
597
894
  return { ...empty, refusedReason: `segment ${segName} already exists; refusing to overwrite archived history` };
598
895
  }
599
896
  // [ROTATE-ARCHIVE-BEFORE-TRUNCATE]: the segment is durable from here on. Only now may
@@ -610,6 +907,7 @@ function rotateHoldingLock(opts) {
610
907
  // Our segment then only duplicates records that other rotation archived, so it goes.
611
908
  if (currentSize < snapshotSize || readFirstRecordHash(path) !== plan.firstLiveHash) {
612
909
  safeUnlink(join(adir, segName));
910
+ clearIntent(ROTATE_INTENT);
613
911
  return {
614
912
  ...empty,
615
913
  refusedReason: "the live log was cut by another writer during this rotation; our segment was removed and nothing else was written",
@@ -638,13 +936,8 @@ function rotateHoldingLock(opts) {
638
936
  }
639
937
  // Self-documenting evidence: the rotation itself is an audited event, chained onto the
640
938
  // new head like any other record.
641
- appendAudit("audit.rotate", {
642
- segment: segName,
643
- archived_records: archived.length,
644
- first_hash: archived[0].hash,
645
- last_hash: archived[archived.length - 1].hash,
646
- cutoff: plan.cutoff,
647
- }, "system");
939
+ appendAudit("audit.rotate", rotateRecord, "system");
940
+ clearIntent(ROTATE_INTENT);
648
941
  return {
649
942
  ...plan,
650
943
  rotated: true,
@@ -692,16 +985,18 @@ function acquireRotateLock() {
692
985
  catch (e) {
693
986
  if (e.code !== "EEXIST")
694
987
  throw e;
695
- let age;
988
+ let st;
696
989
  try {
697
- age = Date.now() - statSync(lock).mtimeMs;
990
+ st = statSync(lock);
698
991
  }
699
992
  catch {
700
993
  continue; // released between our open and our stat: try again
701
994
  }
702
- if (age < ROTATE_LOCK_STALE_MS)
995
+ const age = Date.now() - st.mtimeMs;
996
+ // [LOCK] [A-DEAD-HOLDER-LOSES-THE-LOCK-AT-ONCE]: a crashed rotation no longer blocks the next for 10 minutes.
997
+ if (age < ROTATE_LOCK_STALE_MS && !lockHolderIsGone(lock, st))
703
998
  return { heldMs: age };
704
- safeUnlink(lock);
999
+ unlinkIfUnchanged(lock, st);
705
1000
  continue;
706
1001
  }
707
1002
  try {
@@ -716,6 +1011,172 @@ function acquireRotateLock() {
716
1011
  // Another process broke the same stale lock and won the create.
717
1012
  return { heldMs: 0 };
718
1013
  }
1014
+ /**
1015
+ * [LOCKED] [AN-INTERRUPTED-MOVE-IS-FINISHED] - 2026-09-27
1016
+ * [NEVER] let a rotation or a restore that stopped halfway be archived again, stay unrecorded on the
1017
+ * chain, or leave a temp file that outlives a scrub.
1018
+ * WHY: replayed in real processes killed at each write (E2E_REVIEW_2026-09 B2-2, B2-3, B3-3):
1019
+ * - segment placed, live log not yet cut (9 of 9 kills, and a full disk 3 of 3): the seam de-dup
1020
+ * hid the overlap until the NEXT rotation archived the same 70,000 records again, and the
1021
+ * verifier then said "190,011 records verified" for 120,011. [ROTATE-ARCHIVE-BEFORE-TRUNCATE]
1022
+ * promised this state "loses nothing and is de-duplicated": true only until the next rotation;
1023
+ * - live log cut, or a restore placed, and the process gone before its record (9 of 9): history
1024
+ * moved or came back with no audit.rotate / audit.restore saying who, when and why, and a
1025
+ * retried restore is refused because its records are already there;
1026
+ * - segment linked, its temp name not yet removed (3 of 3): the hidden second name kept 700
1027
+ * planted keys through a scrub that reported success.
1028
+ * FIX: before a rotation writes its segment, and before a restore places its block, a small note
1029
+ * (.rotate-intent.json, .restore-intent.json in audit-archive/) names the move. Every holder of
1030
+ * the rotate lock (rotation, auto-rotation even below its trigger, restore, scrub) first runs
1031
+ * finishInterruptedMoves(): temp files are removed (every writer of them holds this lock, and
1032
+ * each is a copy of records present elsewhere); for a noted rotation whose segment exists, the
1033
+ * leading live records already in that segment are dropped under the append lock and the
1034
+ * missing audit.rotate record is chained, marked completed_after_interruption; for a noted
1035
+ * restore whose segment exists, its audit.restore record, reason included, is chained the same
1036
+ * way. A record already in the live log is never written twice.
1037
+ */
1038
+ const ROTATE_INTENT = ".rotate-intent.json";
1039
+ const RESTORE_INTENT = ".restore-intent.json";
1040
+ const LIVE_TEMPS = [".audit.log.tmp", ".audit.log.scrub.tmp"];
1041
+ function writeIntent(name, data) {
1042
+ mkdirSync(archiveDir(), { recursive: true });
1043
+ writeFileAndSync(join(archiveDir(), name), JSON.stringify(data));
1044
+ }
1045
+ /** The note, or null. A note that exists but cannot be read (cut short by the same crash) is
1046
+ * removed and reported: it names nothing we can finish, and left in place it would keep every
1047
+ * hourly auto-rotation taking the lock for nothing. */
1048
+ function readIntent(name) {
1049
+ const path = join(archiveDir(), name);
1050
+ if (!existsSync(path))
1051
+ return null;
1052
+ try {
1053
+ const v = JSON.parse(readFileSync(path, "utf-8"));
1054
+ if (v && typeof v.segment === "string")
1055
+ return v;
1056
+ }
1057
+ catch { /* unreadable: dropped below */ }
1058
+ console.error(`[ContextEngine] audit: ${name} could not be read and was removed; audit-verify reports anything it left unfinished`);
1059
+ safeUnlink(path);
1060
+ return null;
1061
+ }
1062
+ function clearIntent(name) {
1063
+ safeUnlink(join(archiveDir(), name));
1064
+ }
1065
+ function leftoverTemps() {
1066
+ const out = [];
1067
+ if (existsSync(archiveDir())) {
1068
+ for (const f of readdirSync(archiveDir()))
1069
+ if (f.startsWith(".") && f.endsWith(".tmp"))
1070
+ out.push(join(archiveDir(), f));
1071
+ }
1072
+ for (const f of LIVE_TEMPS)
1073
+ if (existsSync(join(auditDir(), f)))
1074
+ out.push(join(auditDir(), f));
1075
+ return out;
1076
+ }
1077
+ /** Cheap: is there anything for finishInterruptedMoves() to do? */
1078
+ export function interruptedMovePending() {
1079
+ return existsSync(join(archiveDir(), ROTATE_INTENT)) || existsSync(join(archiveDir(), RESTORE_INTENT)) || leftoverTemps().length > 0;
1080
+ }
1081
+ function describeFinish(f) {
1082
+ const parts = [];
1083
+ if (f.rotation)
1084
+ parts.push(`rotation to ${f.rotation.segment} finished (${f.rotation.duplicatesDropped} record(s) already archived dropped from the live log, record ${f.rotation.recorded ? "chained" : "already there"})`);
1085
+ if (f.restore)
1086
+ parts.push(`restore of ${f.restore.segment} recorded ${f.restore.recorded ? "late" : "already"}`);
1087
+ if (f.tempsRemoved.length > 0)
1088
+ parts.push(`${f.tempsRemoved.length} leftover temp file(s) removed`);
1089
+ return parts.length > 0 ? parts.join("; ") : "nothing to finish";
1090
+ }
1091
+ /** Does the live log text hold a record of this kind naming this segment? */
1092
+ function liveNames(text, event, segment) {
1093
+ const hint = `"event":"${event}"`;
1094
+ for (const line of text.split("\n")) {
1095
+ if (!line.includes(hint) || !line.includes(segment))
1096
+ continue;
1097
+ try {
1098
+ const r = JSON.parse(line);
1099
+ if (r.event === event && r.payload.segment === segment)
1100
+ return true;
1101
+ }
1102
+ catch { /* not a record */ }
1103
+ }
1104
+ return false;
1105
+ }
1106
+ /** Run with the rotate lock held. [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED] */
1107
+ export function finishInterruptedMoves() {
1108
+ const report = { tempsRemoved: [], rotation: null, restore: null };
1109
+ for (const t of leftoverTemps()) {
1110
+ safeUnlink(t);
1111
+ if (!existsSync(t))
1112
+ report.tempsRemoved.push(basename(t));
1113
+ }
1114
+ const ri = readIntent(ROTATE_INTENT);
1115
+ if (ri && typeof ri.segment === "string") {
1116
+ const seg = join(archiveDir(), ri.segment);
1117
+ if (existsSync(seg)) {
1118
+ const segHashes = new Set(parseLines(readFileSync(seg, "utf-8"), ri.segment).map((r) => r.hash));
1119
+ const path = auditPath();
1120
+ const release = acquireLockSync();
1121
+ try {
1122
+ settleTail(path);
1123
+ const text = existsSync(path) ? readFileSync(path, "utf-8") : "";
1124
+ const lines = text.split("\n");
1125
+ let dropped = 0;
1126
+ while (dropped < lines.length && lines[dropped]) {
1127
+ let h;
1128
+ try {
1129
+ h = JSON.parse(lines[dropped]).hash;
1130
+ }
1131
+ catch {
1132
+ break;
1133
+ }
1134
+ if (!segHashes.has(h))
1135
+ break;
1136
+ dropped++;
1137
+ }
1138
+ if (dropped > 0) {
1139
+ const tmp = join(auditDir(), ".audit.log.tmp");
1140
+ writeFileAndSync(tmp, lines.slice(dropped).join("\n"));
1141
+ renameSync(tmp, path);
1142
+ }
1143
+ const recorded = !liveNames(text, "audit.rotate", ri.segment);
1144
+ if (recorded) {
1145
+ writeRecord(path, "audit.rotate", { ...ri, completed_after_interruption: true, duplicates_dropped: dropped }, "system");
1146
+ }
1147
+ report.rotation = { segment: ri.segment, duplicatesDropped: dropped, recorded };
1148
+ }
1149
+ finally {
1150
+ release();
1151
+ }
1152
+ }
1153
+ clearIntent(ROTATE_INTENT);
1154
+ }
1155
+ const si = readIntent(RESTORE_INTENT);
1156
+ if (si && typeof si.segment === "string") {
1157
+ if (existsSync(join(archiveDir(), si.segment))) {
1158
+ const path = auditPath();
1159
+ const release = acquireLockSync();
1160
+ try {
1161
+ settleTail(path);
1162
+ const text = existsSync(path) ? readFileSync(path, "utf-8") : "";
1163
+ const recorded = !liveNames(text, "audit.restore", si.segment);
1164
+ const { actor, ...payload } = si;
1165
+ if (recorded)
1166
+ writeRecord(path, "audit.restore", { ...payload, completed_after_interruption: true }, typeof actor === "string" ? actor : "system");
1167
+ report.restore = { segment: si.segment, recorded };
1168
+ }
1169
+ finally {
1170
+ release();
1171
+ }
1172
+ }
1173
+ clearIntent(RESTORE_INTENT);
1174
+ }
1175
+ if (report.tempsRemoved.length > 0 || report.rotation || report.restore) {
1176
+ console.error(`[ContextEngine] audit: finished an interrupted move: ${describeFinish(report)}`);
1177
+ }
1178
+ return report;
1179
+ }
719
1180
  /** Count newline-terminated lines without parsing. The live log is small by construction. */
720
1181
  export function countLiveRecords() {
721
1182
  const path = auditPath();
@@ -736,6 +1197,23 @@ export function autoRotateAuditLog(opts = {}) {
736
1197
  }
737
1198
  const liveRecords = countLiveRecords();
738
1199
  if (liveRecords <= trigger) {
1200
+ // A move interrupted after the live log was cut leaves the log below the trigger: finish it now,
1201
+ // not when the log next grows past it. [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED]
1202
+ if (interruptedMovePending()) {
1203
+ const lock = acquireRotateLock();
1204
+ if ("heldMs" in lock)
1205
+ return { action: "in_progress", liveRecords, detail: "another rotation holds the lock" };
1206
+ try {
1207
+ const f = finishInterruptedMoves();
1208
+ return { action: "finished", liveRecords, detail: describeFinish(f) };
1209
+ }
1210
+ catch (e) {
1211
+ return { action: "error", liveRecords, detail: e.message };
1212
+ }
1213
+ finally {
1214
+ lock.release();
1215
+ }
1216
+ }
739
1217
  return { action: "below_trigger", liveRecords, detail: `${liveRecords} live record(s), trigger is ${trigger}` };
740
1218
  }
741
1219
  // One runner at a time: rotateAuditLog() takes the rotate lock itself, so this path and the
@@ -760,9 +1238,13 @@ export function autoRotateAuditLog(opts = {}) {
760
1238
  }
761
1239
  }
762
1240
  function writeFileAndSync(target, body) {
1241
+ const buf = typeof body === "string" ? Buffer.from(body, "utf-8") : body;
763
1242
  const fd = openSync(target, "w");
764
1243
  try {
765
- writeSync(fd, body);
1244
+ // writeSync may write fewer bytes than asked (a disk filling up): loop until all are down.
1245
+ let off = 0;
1246
+ while (off < buf.length)
1247
+ off += writeSync(fd, buf, off, buf.length - off);
766
1248
  fsyncSync(fd);
767
1249
  }
768
1250
  finally {
@@ -791,11 +1273,17 @@ function writeFileAndSync(target, body) {
791
1273
  * `ok` is true when there are no tampered and no orphan records. Forks are surfaced
792
1274
  * with counts and indices so the report stays honest in both directions — it must
793
1275
  * never claim a forked log is pristine either.
1276
+ * 2026-09-27: a fourth class. A record whose hash was already seen is a DUPLICATE (a second copy
1277
+ * of the same record), counted once and skipped for linkage, so the record after a copied
1278
+ * block links to the original. Before, the first copy read as a "fork" and the total counted
1279
+ * every copy: 190,011 records "verified" for 120,011 real ones after an interrupted rotation
1280
+ * (E2E_REVIEW_2026-09 B2-1, B2-2). A copy's content is still checked against its own hash.
794
1281
  */
795
1282
  export function verifyChain() {
796
1283
  let records;
1284
+ const unreadable = [];
797
1285
  try {
798
- records = readAuditLog();
1286
+ records = readHistory(true, unreadable); // [LOCK] [VERIFY-READS-PAST-AN-UNREADABLE-LINE]
799
1287
  }
800
1288
  catch (e) {
801
1289
  return {
@@ -808,6 +1296,7 @@ export function verifyChain() {
808
1296
  const tampered = [];
809
1297
  const orphans = [];
810
1298
  const forks = [];
1299
+ const duplicates = [];
811
1300
  // Every hash observed so far, so a fork (parent = a known earlier head) can be told
812
1301
  // apart from an orphan (parent never existed in this log).
813
1302
  const seen = new Set([GENESIS_HASH]);
@@ -820,6 +1309,12 @@ export function verifyChain() {
820
1309
  const expected = computeHash(r.prev_hash, r.ts, r.event, r.actor, r.payload);
821
1310
  if (r.hash !== expected)
822
1311
  tampered.push(i);
1312
+ // 1b. A second copy of a record already in the history: counted once, never relinked, so
1313
+ // the record after a copied block still links to the original. [LOCK] [VERIFY-FORK-IS-NOT-TAMPER]
1314
+ if (seen.has(r.hash)) {
1315
+ duplicates.push(i);
1316
+ continue;
1317
+ }
823
1318
  // 2. Linkage — fork vs orphan.
824
1319
  if (r.prev_hash !== prev) {
825
1320
  if (seen.has(r.prev_hash))
@@ -846,27 +1341,40 @@ export function verifyChain() {
846
1341
  continue;
847
1342
  for (const e of list) {
848
1343
  if (typeof e.hash === "string" && typeof e.content_hash === "string")
849
- acks.set(e.hash, e.content_hash);
1344
+ acks.set(e.hash, { contentHash: e.content_hash, ackIndex: i });
850
1345
  }
851
1346
  }
852
1347
  const redacted = [];
853
1348
  const stillTampered = [];
1349
+ const usedAcks = new Map(); // ack record index -> records it covers here
854
1350
  for (const i of tampered) {
855
1351
  const r = records[i];
856
1352
  const bound = acks.get(r.hash);
857
- if (bound && bound === computeHash(r.prev_hash, r.ts, r.event, r.actor, r.payload))
1353
+ if (bound && bound.contentHash === computeHash(r.prev_hash, r.ts, r.event, r.actor, r.payload)) {
858
1354
  redacted.push(i);
1355
+ usedAcks.set(bound.ackIndex, (usedAcks.get(bound.ackIndex) ?? 0) + 1);
1356
+ }
859
1357
  else
860
1358
  stillTampered.push(i);
861
1359
  }
1360
+ const acknowledgements = [...usedAcks.entries()].sort((a, b) => a[0] - b[0]).map(([index, n]) => ({
1361
+ index,
1362
+ ts: records[index].ts,
1363
+ actor: records[index].actor,
1364
+ reason: String(records[index].payload.reason ?? ""),
1365
+ records: n,
1366
+ }));
862
1367
  tampered.length = 0;
863
1368
  tampered.push(...stillTampered);
864
- const ok = tampered.length === 0 && orphans.length === 0;
865
- const firstProblem = tampered.length > 0 ? tampered[0] : orphans.length > 0 ? orphans[0] : null;
1369
+ const ok = tampered.length === 0 && orphans.length === 0 && unreadable.length === 0;
1370
+ const firstProblem = tampered.length > 0 ? tampered[0] : unreadable.length > 0 ? unreadable[0].beforeIndex : orphans.length > 0 ? orphans[0] : null;
866
1371
  let reason = null;
867
1372
  if (tampered.length > 0) {
868
1373
  reason = `${tampered.length} record(s) with altered content — first at index ${tampered[0]}`;
869
1374
  }
1375
+ else if (unreadable.length > 0) {
1376
+ reason = `${unreadable.length} line(s) that are not records, first at ${unreadable[0].file} line ${unreadable[0].line}; every other record was checked, and the record after such a line cannot be linked`;
1377
+ }
870
1378
  else if (orphans.length > 0) {
871
1379
  reason = `${orphans.length} record(s) whose parent is absent from the log (deleted or truncated history) — first at index ${orphans[0]}`;
872
1380
  }
@@ -878,9 +1386,112 @@ export function verifyChain() {
878
1386
  tamperedIndices: tampered,
879
1387
  orphanIndices: orphans,
880
1388
  forkIndices: forks,
1389
+ duplicateIndices: duplicates,
1390
+ acknowledgements,
1391
+ unreadable,
881
1392
  redactedIndices: redacted,
882
1393
  };
883
1394
  }
1395
+ export function verifyStatePath() {
1396
+ return join(auditDir(), "audit-verify.json");
1397
+ }
1398
+ /** Keep the result of a full check. Best effort: a check that cannot record still printed its verdict. */
1399
+ export function recordVerifyState(report, ms, by) {
1400
+ const dups = (report.duplicateIndices ?? []).length;
1401
+ const state = {
1402
+ checkedAt: new Date().toISOString(),
1403
+ ms,
1404
+ by,
1405
+ ok: report.ok,
1406
+ total: report.total,
1407
+ unique: report.total - dups,
1408
+ altered: (report.tamperedIndices ?? []).length,
1409
+ orphans: (report.orphanIndices ?? []).length,
1410
+ unreadable: (report.unreadable ?? []).length,
1411
+ duplicates: dups,
1412
+ forks: (report.forkIndices ?? []).length,
1413
+ redacted: (report.redactedIndices ?? []).length,
1414
+ reason: report.breakReason,
1415
+ };
1416
+ try {
1417
+ ensureDir();
1418
+ const tmp = `${verifyStatePath()}.tmp-${process.pid}`;
1419
+ writeFileSync(tmp, JSON.stringify(state, null, 2) + "\n");
1420
+ renameSync(tmp, verifyStatePath());
1421
+ }
1422
+ catch {
1423
+ /* the verdict was printed; health will say the chain was not checked recently */
1424
+ }
1425
+ return state;
1426
+ }
1427
+ export function readVerifyState() {
1428
+ try {
1429
+ const s = JSON.parse(readFileSync(verifyStatePath(), "utf-8"));
1430
+ return typeof s.checkedAt === "string" && typeof s.ok === "boolean" ? s : null;
1431
+ }
1432
+ catch {
1433
+ return null;
1434
+ }
1435
+ }
1436
+ /** Refusals written by safeAppend() and not chained yet: count, and the newest one. */
1437
+ export function pendingRefusals() {
1438
+ let body = "";
1439
+ try {
1440
+ body = readFileSync(refusedPath(), "utf-8");
1441
+ }
1442
+ catch {
1443
+ return { count: 0, last: null, error: null };
1444
+ }
1445
+ let count = 0;
1446
+ let last = null;
1447
+ let error = null;
1448
+ for (const line of body.split("\n")) {
1449
+ if (!line)
1450
+ continue;
1451
+ try {
1452
+ const r = JSON.parse(line);
1453
+ count++;
1454
+ if (r.ts && (!last || r.ts > last)) {
1455
+ last = r.ts;
1456
+ error = r.error ?? null;
1457
+ }
1458
+ }
1459
+ catch { /* a line being written */ }
1460
+ }
1461
+ return { count, last, error };
1462
+ }
1463
+ /** One scheduled full check at a time across processes: O_EXCL, stale after two hours. */
1464
+ export function acquireVerifyLock() {
1465
+ const lock = join(auditDir(), "audit-verify.lock");
1466
+ ensureDir();
1467
+ for (let attempt = 0; attempt < 2; attempt++) {
1468
+ try {
1469
+ const fd = openSync(lock, constants.O_CREAT | constants.O_EXCL | constants.O_WRONLY, 0o600);
1470
+ try {
1471
+ writeSync(fd, `${process.pid}\n`);
1472
+ }
1473
+ catch { /* courtesy */ }
1474
+ closeSync(fd);
1475
+ return () => safeUnlink(lock);
1476
+ }
1477
+ catch (e) {
1478
+ if (e.code !== "EEXIST")
1479
+ return null;
1480
+ let st;
1481
+ try {
1482
+ st = statSync(lock);
1483
+ }
1484
+ catch {
1485
+ continue;
1486
+ }
1487
+ // [LOCK] [A-DEAD-HOLDER-LOSES-THE-LOCK-AT-ONCE]
1488
+ if (Date.now() - st.mtimeMs < 2 * 3_600_000 && !lockHolderIsGone(lock, st))
1489
+ return null;
1490
+ unlinkIfUnchanged(lock, st);
1491
+ }
1492
+ }
1493
+ return null;
1494
+ }
884
1495
  /**
885
1496
  * Acknowledge that records were deliberately redacted (a secret removed from their content).
886
1497
  *
@@ -1028,11 +1639,26 @@ export function restoreSegment(file, opts = {}) {
1028
1639
  return refuse(`the verifier does not report the record after ${prev.name} as an orphan; nothing to restore`, base);
1029
1640
  }
1030
1641
  const adir = archiveDir();
1642
+ const restorePayload = {
1643
+ segment: segName,
1644
+ after: prev.name,
1645
+ records: block.length,
1646
+ first_hash: block[0].hash,
1647
+ last_hash: block[block.length - 1].hash,
1648
+ first_ts: block[0].ts,
1649
+ last_ts: block[block.length - 1].ts,
1650
+ source: basename(file),
1651
+ reason: opts.reason.trim(),
1652
+ };
1653
+ // [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED]: a crash from here to the record leaves this note.
1654
+ writeIntent(RESTORE_INTENT, { ...restorePayload, actor: opts.actor ?? "system" });
1031
1655
  const tmp = join(adir, `.${segName}.tmp`);
1032
1656
  writeFileAndSync(tmp, block.map((x) => JSON.stringify(x)).join("\n") + "\n");
1033
1657
  const target = join(adir, segName);
1034
- if (!placeWithoutOverwrite(tmp, target))
1658
+ if (!placeWithoutOverwrite(tmp, target)) {
1659
+ clearIntent(RESTORE_INTENT);
1035
1660
  return refuse(`segment ${segName} already exists; refusing to overwrite it`, base);
1661
+ }
1036
1662
  const afterReport = verifyChain();
1037
1663
  // `>=` on the total: live appends keep landing during two full verifies (seconds on a real
1038
1664
  // log), and they only ever add records.
@@ -1041,19 +1667,11 @@ export function restoreSegment(file, opts = {}) {
1041
1667
  afterReport.total >= beforeReport.total + block.length;
1042
1668
  if (!improved) {
1043
1669
  safeUnlink(target);
1670
+ clearIntent(RESTORE_INTENT);
1044
1671
  return refuse("the chain did not come out exactly one orphan better; the restored segment was removed again", base);
1045
1672
  }
1046
- const record = appendAudit("audit.restore", {
1047
- segment: segName,
1048
- after: prev.name,
1049
- records: block.length,
1050
- first_hash: block[0].hash,
1051
- last_hash: block[block.length - 1].hash,
1052
- first_ts: block[0].ts,
1053
- last_ts: block[block.length - 1].ts,
1054
- source: basename(file),
1055
- reason: opts.reason.trim(),
1056
- }, opts.actor ?? "system");
1673
+ const record = appendAudit("audit.restore", restorePayload, opts.actor ?? "system");
1674
+ clearIntent(RESTORE_INTENT);
1057
1675
  return { ...plan, restored: true, record };
1058
1676
  };
1059
1677
  if (!opts.apply)
@@ -1063,6 +1681,7 @@ export function restoreSegment(file, opts = {}) {
1063
1681
  if ("heldMs" in lock)
1064
1682
  return refuse("a rotation is in progress; try again in a minute", base);
1065
1683
  try {
1684
+ finishInterruptedMoves(); // [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED]
1066
1685
  return planAndMaybeWrite();
1067
1686
  }
1068
1687
  finally {
@@ -1127,6 +1746,8 @@ function scrubText(text, redact) {
1127
1746
  * the live log under the append lock (appends wait, none is lost), then one audit.redact
1128
1747
  * record per 100 rewrites names each original hash and its new content hash
1129
1748
  * ([REDACTION-IS-A-CHAINED-RECORD]). Running it again changes nothing.
1749
+ * 2026-09-27 (Yan's GO): the order is now acknowledgement first, rewrite second, for every file.
1750
+ * [LOCK] [SCRUB-ACKNOWLEDGES-BEFORE-IT-REWRITES]
1130
1751
  */
1131
1752
  export function scrubAuditLog(opts) {
1132
1753
  const report = { applied: false, refusedReason: null, files: [], redactedRecords: 0, counts: {}, acknowledgements: [] };
@@ -1138,13 +1759,23 @@ export function scrubAuditLog(opts) {
1138
1759
  for (const [k, n] of Object.entries(s.counts))
1139
1760
  report.counts[k] = (report.counts[k] ?? 0) + n;
1140
1761
  };
1141
- // Acknowledge each file right after it is rewritten, so a scrub that stops halfway leaves no
1142
- // rewritten record unacknowledged.
1143
- const acknowledge = (entries) => {
1762
+ // [LOCKED] [SCRUB-ACKNOWLEDGES-BEFORE-IT-REWRITES] - 2026-09-27
1763
+ // [NEVER] rewrite a file before the acknowledgements for its rewrites are on the chain.
1764
+ // WHY: the scrub renamed each rewritten segment into place, then acknowledged it. Killed between the
1765
+ // two (3 of 3 runs, E2E_REVIEW_2026-09 B3-1), 100 records read as "altered content: this is
1766
+ // tampering", and a rerun could not repair it: it found nothing left to change in them. The
1767
+ // product's own act was reported as an attack, recoverable only by listing 100 indices by hand.
1768
+ // FIX: acknowledge first, rewrite second. A crash in between leaves an acknowledgement for a rewrite
1769
+ // that did not happen: it binds content that is not there, so it is simply unused, and a rerun
1770
+ // rewrites and acknowledges again. The live log is acknowledged and rewritten under one hold of
1771
+ // the append lock: its acknowledgements are appended, then the rewritten records plus exactly
1772
+ // the bytes appended since go in place. [LOCK] [SCRUB-IS-ACKNOWLEDGED-REDACTION]
1773
+ const acknowledge = (entries, write) => {
1144
1774
  for (let i = 0; i < entries.length; i += SCRUB_ACK_BATCH) {
1145
- report.acknowledgements.push(appendAudit("audit.redact", { reason: opts.reason.trim(), redacted: entries.slice(i, i + SCRUB_ACK_BATCH) }, opts.actor ?? "system"));
1775
+ report.acknowledgements.push(write({ reason: opts.reason.trim(), redacted: entries.slice(i, i + SCRUB_ACK_BATCH) }));
1146
1776
  }
1147
1777
  };
1778
+ const actor = opts.actor ?? "system";
1148
1779
  const run = () => {
1149
1780
  const adir = archiveDir();
1150
1781
  for (const name of listSegments()) {
@@ -1154,8 +1785,8 @@ export function scrubAuditLog(opts) {
1154
1785
  if (opts.apply && s.redacted.length > 0) {
1155
1786
  const tmp = join(adir, `.${name}.scrub.tmp`);
1156
1787
  writeFileAndSync(tmp, s.body);
1788
+ acknowledge(s.redacted, (payload) => appendAudit("audit.redact", payload, actor));
1157
1789
  renameSync(tmp, path);
1158
- acknowledge(s.redacted);
1159
1790
  }
1160
1791
  }
1161
1792
  const live = auditPath();
@@ -1164,22 +1795,23 @@ export function scrubAuditLog(opts) {
1164
1795
  tally("audit.log", scrubText(readFileSync(live, "utf-8"), opts.redact));
1165
1796
  }
1166
1797
  else {
1167
- let liveAcks = [];
1168
1798
  const release = acquireLockSync();
1169
1799
  try {
1170
- const s = scrubText(readFileSync(live, "utf-8"), opts.redact);
1800
+ settleTail(live); // the log ends with a newline from here on
1801
+ const before = readFileSync(live);
1802
+ const s = scrubText(before.toString("utf-8"), opts.redact);
1171
1803
  tally("audit.log", s);
1172
1804
  if (s.redacted.length > 0) {
1805
+ acknowledge(s.redacted, (payload) => writeRecord(live, "audit.redact", payload, actor));
1806
+ const grown = readFileSync(live).subarray(before.length); // our acknowledgements, nothing else: we hold the lock
1173
1807
  const tmp = join(auditDir(), ".audit.log.scrub.tmp");
1174
- writeFileAndSync(tmp, s.body);
1808
+ writeFileAndSync(tmp, Buffer.concat([Buffer.from(s.body, "utf-8"), grown]));
1175
1809
  renameSync(tmp, live);
1176
- liveAcks = s.redacted;
1177
1810
  }
1178
1811
  }
1179
1812
  finally {
1180
1813
  release();
1181
1814
  }
1182
- acknowledge(liveAcks); // after the release: appendAudit takes the same lock
1183
1815
  }
1184
1816
  }
1185
1817
  return { ...report, applied: !!opts.apply };
@@ -1190,6 +1822,7 @@ export function scrubAuditLog(opts) {
1190
1822
  if ("heldMs" in lock)
1191
1823
  return { ...report, refusedReason: "a rotation is in progress; try again in a minute" };
1192
1824
  try {
1825
+ finishInterruptedMoves(); // [LOCK] [AN-INTERRUPTED-MOVE-IS-FINISHED]: no leftover temp keeps a secret
1193
1826
  return run();
1194
1827
  }
1195
1828
  finally {
@@ -1221,13 +1854,19 @@ export function resetCacheForTest() {
1221
1854
  // Safe wrapper that never throws into hot paths. Use this from production
1222
1855
  // call sites so a failed audit append cannot break a learning save or
1223
1856
  // session write.
1857
+ // Returns whether the entry was written. A refusal is counted in audit-refused.jsonl and chained
1858
+ // as `audit.append_failed` by the next append that succeeds. [LOCK] [A-REFUSED-APPEND-IS-COUNTED-AND-CHAINED]
1224
1859
  export function safeAppend(event, payload, actor = "system") {
1225
1860
  try {
1226
1861
  appendAudit(event, payload, actor);
1862
+ return true;
1227
1863
  }
1228
1864
  catch (e) {
1229
- // Last-resort surface — stderr only, never throw upward.
1230
- process.stderr.write(`[ContextEngine] audit append failed: ${e instanceof Error ? e.message : String(e)}\n`);
1865
+ const message = e instanceof Error ? e.message : String(e);
1866
+ // Never throw upward: stderr, plus the count the chain and fleet health will carry.
1867
+ process.stderr.write(`[ContextEngine] audit append failed: ${message}\n`);
1868
+ recordRefusal(event, message);
1869
+ return false;
1231
1870
  }
1232
1871
  }
1233
1872
  //# sourceMappingURL=audit.js.map