@kontextmind/kxm 0.7.50 → 0.7.51
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/CHANGELOG.md +42 -0
- package/package.json +1 -1
- package/plugins/kxm/.claude-plugin/plugin.json +1 -1
- package/plugins/kxm/dist/mcp-server.js +1 -1
- package/plugins/kxm/dist/runtime-supervisor.js +74 -3
- package/plugins/kxm/dist/runtime.js +76 -3
- package/plugins/kxm/dist/server.js +66 -2
- package/plugins/kxm/package.json +1 -1
- package/plugins/kxm/src/database.ts +206 -1
- package/plugins/kxm/src/intake.ts +86 -35
- package/plugins/kxm/src/mcp-server.ts +1 -1
- package/plugins/kxm/src/runtime-store.ts +8 -1
package/CHANGELOG.md
CHANGED
|
@@ -53,6 +53,48 @@ All notable user-facing changes are documented here. The project follows [Semant
|
|
|
53
53
|
arrival order, so a backdated timestamp cannot jump the queue; persisted records
|
|
54
54
|
are cross-checked against every duplicated column on read; and the intake schema
|
|
55
55
|
no longer admits contradictory states.
|
|
56
|
+
- **A coordinator rebind could widen a tool ceiling by clearing its allow list.**
|
|
57
|
+
An absent or empty allow list imposes no restriction, so dropping a populated one
|
|
58
|
+
is now refused (`coordinator_rebind_clears_allowlist`). Resume drains held intent
|
|
59
|
+
to exhaustion instead of stopping at a page cap, and fails loudly rather than
|
|
60
|
+
half-resuming. Coordinator fingerprints are computed over the normalised
|
|
61
|
+
authority, so identities bound by 0.7.46 with unordered effect lists no longer
|
|
62
|
+
require a policy rebind after upgrade — and the equivalence test is shared, so a
|
|
63
|
+
row written by 0.7.46 is not a `coordinator_write_lost` conflict just because it
|
|
64
|
+
was found by losing an insert race instead of reading the slot.
|
|
65
|
+
- **A failed `BEGIN` poisoned the database connection.** The transaction marker was
|
|
66
|
+
claimed before `BEGIN` and the statement sat outside the `try/finally`, so a
|
|
67
|
+
`BEGIN` that gave up on a busy writer left every later transaction failing with a
|
|
68
|
+
misleading "nested transactions are not allowed". The marker is now claimed only
|
|
69
|
+
after a successful `BEGIN`, genuine lock contention surfaces as
|
|
70
|
+
`runtime_transaction_busy` — decided by SQLite's **result code**, on both runtimes
|
|
71
|
+
this ships on: Node's `errcode` and `bun:sqlite`'s `errno` (extended codes land on
|
|
72
|
+
their primaries, so `SQLITE_BUSY_RECOVERY`, `SQLITE_BUSY_SNAPSHOT` and
|
|
73
|
+
`SQLITE_LOCKED_SHAREDCACHE` all count), then a symbolic `SQLITE_BUSY*` /
|
|
74
|
+
`SQLITE_LOCKED*` / `SQLITE_PROTOCOL*` name, with anchored message text used only
|
|
75
|
+
when an error carries neither. A code **or** a SQLite result name wins over the text
|
|
76
|
+
in both directions, so a permanent error — `SQLITE_FULL`, `SQLITE_CANTOPEN` — quoting
|
|
77
|
+
"database is locked" is not mistaken for contention, and `bun:sqlite`'s symbolic
|
|
78
|
+
`code` is read as the result name it is while Node's own `ERR_SQLITE_ERROR` is not.
|
|
79
|
+
Any other `BEGIN` failure keeps its own error instead of looking retryable. A contended connection then refuses further write-mode `BEGIN`s for one
|
|
80
|
+
second (`TRANSACTION_BUSY_BACKOFF_MS`), so retries **inside that window** fail fast
|
|
81
|
+
rather than paying the 5-second busy timeout once per attempt; a retry after the
|
|
82
|
+
window can pay it again. The window is measured with `process.hrtime` and belongs to
|
|
83
|
+
the clock that armed it, so neither a system clock change nor an injected test clock
|
|
84
|
+
can extend, shorten or clear another caller's throttle, and a clock that returns a
|
|
85
|
+
non-finite number is refused rather than trusted (`runtime_transaction_clock_invalid`)
|
|
86
|
+
wherever a reading is taken — checking a pending deadline, and arming a fresh one. The
|
|
87
|
+
clock is read at two call sites, not on every `BEGIN`. Zero reads are limited to
|
|
88
|
+
successful write-mode transactions with no pending deadline, successful `DEFERRED`
|
|
89
|
+
transactions with or without one, permanent `BEGIN` failures with no pending deadline, and
|
|
90
|
+
a nested-transaction rejection; a `DEFERRED` transaction skips the deadline check, but a
|
|
91
|
+
contended `BEGIN DEFERRED` failure still reads the clock and arms a deadline. Two further
|
|
92
|
+
cases are measured rather than asserted by the committed suite — a clean success, and a
|
|
93
|
+
permanent failure, each following an expired deadline, read it once. So this guards the
|
|
94
|
+
seam rather than every transaction, and the counts we assert are named separately from the
|
|
95
|
+
counts we observed. The throttle is per connection
|
|
96
|
+
object in this process — it is not cross-process, and it does not leak to another
|
|
97
|
+
connection to the same database.
|
|
56
98
|
|
|
57
99
|
## 0.7.0 - 2026-09-11
|
|
58
100
|
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
|
|
3
3
|
"name": "kxm",
|
|
4
4
|
"displayName": "KXM",
|
|
5
|
-
"version": "0.7.
|
|
5
|
+
"version": "0.7.51",
|
|
6
6
|
"description": "Headless multi-agent orchestration, durable workflows, and a live operator dashboard for Pi and Claude Code",
|
|
7
7
|
"author": {
|
|
8
8
|
"name": "KontextMind",
|
|
@@ -17121,7 +17121,7 @@ async function deliverInboxNotification(messageId, delivered, notify) {
|
|
|
17121
17121
|
}
|
|
17122
17122
|
|
|
17123
17123
|
// plugins/kxm/src/mcp-server.ts
|
|
17124
|
-
var VERSION = "0.7.
|
|
17124
|
+
var VERSION = "0.7.51";
|
|
17125
17125
|
var inbox = /* @__PURE__ */ new Map();
|
|
17126
17126
|
var notifiedInbox = /* @__PURE__ */ new Set();
|
|
17127
17127
|
var meshClient;
|
|
@@ -17645,12 +17645,76 @@ function openDatabase(file, description, spec) {
|
|
|
17645
17645
|
}
|
|
17646
17646
|
}
|
|
17647
17647
|
var activeTransactions = /* @__PURE__ */ new WeakSet();
|
|
17648
|
-
|
|
17648
|
+
var TRANSACTION_BUSY_BACKOFF_MS = 1e3;
|
|
17649
|
+
function monotonicNowMs() {
|
|
17650
|
+
return Number(process.hrtime.bigint() / 1000000n);
|
|
17651
|
+
}
|
|
17652
|
+
var transactionThrottles = /* @__PURE__ */ new WeakMap();
|
|
17653
|
+
function finiteNow(clock, label) {
|
|
17654
|
+
const now = clock();
|
|
17655
|
+
if (typeof now !== "number" || !Number.isFinite(now)) {
|
|
17656
|
+
throw databaseError(
|
|
17657
|
+
"runtime_transaction_clock_invalid",
|
|
17658
|
+
"transaction",
|
|
17659
|
+
`${label} must return a finite monotonic number; got ${String(now)}`
|
|
17660
|
+
);
|
|
17661
|
+
}
|
|
17662
|
+
return now;
|
|
17663
|
+
}
|
|
17664
|
+
var CONTENTION_PRIMARY_CODES = [5, 6, 15];
|
|
17665
|
+
var CONTENTION_SYMBOLIC_NAMES = /^SQLITE_(?:BUSY|LOCKED|PROTOCOL)(?:_[A-Z0-9]+)?$/;
|
|
17666
|
+
var SQLITE_RESULT_NAMES = /^SQLITE_[A-Z][A-Z0-9_]*$/;
|
|
17667
|
+
var CONTENTION_MESSAGES = /^(?:database is locked|database table is locked|locking protocol|SQLITE_BUSY|SQLITE_LOCKED|SQLITE_PROTOCOL)(?:$|[\s.:])/i;
|
|
17668
|
+
function isTransactionContention(error) {
|
|
17669
|
+
const carrier = error;
|
|
17670
|
+
for (const value of [carrier?.errcode, carrier?.errCode, carrier?.errno]) {
|
|
17671
|
+
if (typeof value === "number" && Number.isInteger(value)) return CONTENTION_PRIMARY_CODES.includes(value & 255);
|
|
17672
|
+
}
|
|
17673
|
+
for (const value of [carrier?.code, carrier?.name]) {
|
|
17674
|
+
if (typeof value === "string" && SQLITE_RESULT_NAMES.test(value)) {
|
|
17675
|
+
return CONTENTION_SYMBOLIC_NAMES.test(value);
|
|
17676
|
+
}
|
|
17677
|
+
}
|
|
17678
|
+
return CONTENTION_MESSAGES.test(error instanceof Error ? error.message : String(error));
|
|
17679
|
+
}
|
|
17680
|
+
function withDatabaseTransaction(database, work, mode = "IMMEDIATE", clock = monotonicNowMs) {
|
|
17649
17681
|
if (activeTransactions.has(database)) {
|
|
17650
17682
|
throw databaseError("runtime_transaction_nested", "transaction", "nested transactions are not allowed");
|
|
17651
17683
|
}
|
|
17684
|
+
if (mode !== "DEFERRED") {
|
|
17685
|
+
const deadlines = transactionThrottles.get(database);
|
|
17686
|
+
const until = deadlines?.get(clock);
|
|
17687
|
+
if (until !== void 0) {
|
|
17688
|
+
const remaining = until - finiteNow(clock, "the transaction clock");
|
|
17689
|
+
if (remaining > 0) {
|
|
17690
|
+
throw databaseError(
|
|
17691
|
+
"runtime_transaction_busy",
|
|
17692
|
+
"transaction",
|
|
17693
|
+
`a previous BEGIN was blocked on this database; retry deferred ${String(remaining)}ms`
|
|
17694
|
+
);
|
|
17695
|
+
}
|
|
17696
|
+
deadlines?.delete(clock);
|
|
17697
|
+
}
|
|
17698
|
+
}
|
|
17699
|
+
try {
|
|
17700
|
+
database.exec(`BEGIN ${mode}`);
|
|
17701
|
+
} catch (error) {
|
|
17702
|
+
if (!isTransactionContention(error)) throw error;
|
|
17703
|
+
const now = finiteNow(clock, "the transaction clock");
|
|
17704
|
+
let deadlines = transactionThrottles.get(database);
|
|
17705
|
+
if (deadlines === void 0) {
|
|
17706
|
+
deadlines = /* @__PURE__ */ new WeakMap();
|
|
17707
|
+
transactionThrottles.set(database, deadlines);
|
|
17708
|
+
}
|
|
17709
|
+
deadlines.set(clock, now + TRANSACTION_BUSY_BACKOFF_MS);
|
|
17710
|
+
throw databaseError(
|
|
17711
|
+
"runtime_transaction_busy",
|
|
17712
|
+
"transaction",
|
|
17713
|
+
`BEGIN ${mode} blocked by another transaction: ${error instanceof Error ? error.message : String(error)}`
|
|
17714
|
+
);
|
|
17715
|
+
}
|
|
17652
17716
|
activeTransactions.add(database);
|
|
17653
|
-
|
|
17717
|
+
if (mode !== "DEFERRED") transactionThrottles.get(database)?.delete(clock);
|
|
17654
17718
|
try {
|
|
17655
17719
|
const result = work();
|
|
17656
17720
|
database.exec("COMMIT");
|
|
@@ -18558,7 +18622,14 @@ var KxmRunEventStore = class {
|
|
|
18558
18622
|
`).get(projectId, coordinatorId, idempotencyKey);
|
|
18559
18623
|
return row ? intakeFromRow(row) : void 0;
|
|
18560
18624
|
}
|
|
18561
|
-
/**
|
|
18625
|
+
/**
|
|
18626
|
+
* Intake rows in the given dispatch states, in arrival order (replay-safe).
|
|
18627
|
+
*
|
|
18628
|
+
* `rowid` gives same-store arrival order, which is what queue priority needs
|
|
18629
|
+
* here. It is **not** a durable sequence: this repository backs stores up with
|
|
18630
|
+
* `VACUUM INTO`, and a vacuum may renumber implicit rowids. An explicit
|
|
18631
|
+
* immutable arrival sequence is tracked in the plan's schema-v6 follow-ups.
|
|
18632
|
+
*/
|
|
18562
18633
|
intakeInStates(projectId, states, limit = 100) {
|
|
18563
18634
|
if (states.length === 0) return [];
|
|
18564
18635
|
const placeholders = states.map(() => "?").join(", ");
|
|
@@ -17816,12 +17816,76 @@ function openDatabase(file, description, spec) {
|
|
|
17816
17816
|
}
|
|
17817
17817
|
}
|
|
17818
17818
|
var activeTransactions = /* @__PURE__ */ new WeakSet();
|
|
17819
|
-
|
|
17819
|
+
var TRANSACTION_BUSY_BACKOFF_MS = 1e3;
|
|
17820
|
+
function monotonicNowMs() {
|
|
17821
|
+
return Number(process.hrtime.bigint() / 1000000n);
|
|
17822
|
+
}
|
|
17823
|
+
var transactionThrottles = /* @__PURE__ */ new WeakMap();
|
|
17824
|
+
function finiteNow(clock, label) {
|
|
17825
|
+
const now = clock();
|
|
17826
|
+
if (typeof now !== "number" || !Number.isFinite(now)) {
|
|
17827
|
+
throw databaseError(
|
|
17828
|
+
"runtime_transaction_clock_invalid",
|
|
17829
|
+
"transaction",
|
|
17830
|
+
`${label} must return a finite monotonic number; got ${String(now)}`
|
|
17831
|
+
);
|
|
17832
|
+
}
|
|
17833
|
+
return now;
|
|
17834
|
+
}
|
|
17835
|
+
var CONTENTION_PRIMARY_CODES = [5, 6, 15];
|
|
17836
|
+
var CONTENTION_SYMBOLIC_NAMES = /^SQLITE_(?:BUSY|LOCKED|PROTOCOL)(?:_[A-Z0-9]+)?$/;
|
|
17837
|
+
var SQLITE_RESULT_NAMES = /^SQLITE_[A-Z][A-Z0-9_]*$/;
|
|
17838
|
+
var CONTENTION_MESSAGES = /^(?:database is locked|database table is locked|locking protocol|SQLITE_BUSY|SQLITE_LOCKED|SQLITE_PROTOCOL)(?:$|[\s.:])/i;
|
|
17839
|
+
function isTransactionContention(error) {
|
|
17840
|
+
const carrier = error;
|
|
17841
|
+
for (const value of [carrier?.errcode, carrier?.errCode, carrier?.errno]) {
|
|
17842
|
+
if (typeof value === "number" && Number.isInteger(value)) return CONTENTION_PRIMARY_CODES.includes(value & 255);
|
|
17843
|
+
}
|
|
17844
|
+
for (const value of [carrier?.code, carrier?.name]) {
|
|
17845
|
+
if (typeof value === "string" && SQLITE_RESULT_NAMES.test(value)) {
|
|
17846
|
+
return CONTENTION_SYMBOLIC_NAMES.test(value);
|
|
17847
|
+
}
|
|
17848
|
+
}
|
|
17849
|
+
return CONTENTION_MESSAGES.test(error instanceof Error ? error.message : String(error));
|
|
17850
|
+
}
|
|
17851
|
+
function withDatabaseTransaction(database, work, mode = "IMMEDIATE", clock = monotonicNowMs) {
|
|
17820
17852
|
if (activeTransactions.has(database)) {
|
|
17821
17853
|
throw databaseError("runtime_transaction_nested", "transaction", "nested transactions are not allowed");
|
|
17822
17854
|
}
|
|
17855
|
+
if (mode !== "DEFERRED") {
|
|
17856
|
+
const deadlines = transactionThrottles.get(database);
|
|
17857
|
+
const until = deadlines?.get(clock);
|
|
17858
|
+
if (until !== void 0) {
|
|
17859
|
+
const remaining = until - finiteNow(clock, "the transaction clock");
|
|
17860
|
+
if (remaining > 0) {
|
|
17861
|
+
throw databaseError(
|
|
17862
|
+
"runtime_transaction_busy",
|
|
17863
|
+
"transaction",
|
|
17864
|
+
`a previous BEGIN was blocked on this database; retry deferred ${String(remaining)}ms`
|
|
17865
|
+
);
|
|
17866
|
+
}
|
|
17867
|
+
deadlines?.delete(clock);
|
|
17868
|
+
}
|
|
17869
|
+
}
|
|
17870
|
+
try {
|
|
17871
|
+
database.exec(`BEGIN ${mode}`);
|
|
17872
|
+
} catch (error) {
|
|
17873
|
+
if (!isTransactionContention(error)) throw error;
|
|
17874
|
+
const now = finiteNow(clock, "the transaction clock");
|
|
17875
|
+
let deadlines = transactionThrottles.get(database);
|
|
17876
|
+
if (deadlines === void 0) {
|
|
17877
|
+
deadlines = /* @__PURE__ */ new WeakMap();
|
|
17878
|
+
transactionThrottles.set(database, deadlines);
|
|
17879
|
+
}
|
|
17880
|
+
deadlines.set(clock, now + TRANSACTION_BUSY_BACKOFF_MS);
|
|
17881
|
+
throw databaseError(
|
|
17882
|
+
"runtime_transaction_busy",
|
|
17883
|
+
"transaction",
|
|
17884
|
+
`BEGIN ${mode} blocked by another transaction: ${error instanceof Error ? error.message : String(error)}`
|
|
17885
|
+
);
|
|
17886
|
+
}
|
|
17823
17887
|
activeTransactions.add(database);
|
|
17824
|
-
|
|
17888
|
+
if (mode !== "DEFERRED") transactionThrottles.get(database)?.delete(clock);
|
|
17825
17889
|
try {
|
|
17826
17890
|
const result = work();
|
|
17827
17891
|
database.exec("COMMIT");
|
|
@@ -18998,7 +19062,14 @@ var KxmRunEventStore = class {
|
|
|
18998
19062
|
`).get(projectId, coordinatorId, idempotencyKey);
|
|
18999
19063
|
return row ? intakeFromRow(row) : void 0;
|
|
19000
19064
|
}
|
|
19001
|
-
/**
|
|
19065
|
+
/**
|
|
19066
|
+
* Intake rows in the given dispatch states, in arrival order (replay-safe).
|
|
19067
|
+
*
|
|
19068
|
+
* `rowid` gives same-store arrival order, which is what queue priority needs
|
|
19069
|
+
* here. It is **not** a durable sequence: this repository backs stores up with
|
|
19070
|
+
* `VACUUM INTO`, and a vacuum may renumber implicit rowids. An explicit
|
|
19071
|
+
* immutable arrival sequence is tracked in the plan's schema-v6 follow-ups.
|
|
19072
|
+
*/
|
|
19002
19073
|
intakeInStates(projectId, states, limit = 100) {
|
|
19003
19074
|
if (states.length === 0) return [];
|
|
19004
19075
|
const placeholders = states.map(() => "?").join(", ");
|
|
@@ -30289,6 +30360,7 @@ export {
|
|
|
30289
30360
|
SUBAGENT_TYPES,
|
|
30290
30361
|
SteelClient,
|
|
30291
30362
|
SubagentManager,
|
|
30363
|
+
TRANSACTION_BUSY_BACKOFF_MS,
|
|
30292
30364
|
VIEWPORT_PRESETS,
|
|
30293
30365
|
WIN_NPM_INNER_EXE,
|
|
30294
30366
|
acceptKxmRun,
|
|
@@ -30344,6 +30416,7 @@ export {
|
|
|
30344
30416
|
hashKxmTokenProof,
|
|
30345
30417
|
isKnownHarnessId,
|
|
30346
30418
|
isKxmRuntimeContextClosed,
|
|
30419
|
+
isTransactionContention,
|
|
30347
30420
|
isWindowsHarnessShim,
|
|
30348
30421
|
kxmDeclaredExecutorIds,
|
|
30349
30422
|
kxmDeclaredRepositoryIds,
|
|
@@ -11537,12 +11537,76 @@ function openDatabase(file, description, spec) {
|
|
|
11537
11537
|
}
|
|
11538
11538
|
}
|
|
11539
11539
|
var activeTransactions = /* @__PURE__ */ new WeakSet();
|
|
11540
|
-
|
|
11540
|
+
var TRANSACTION_BUSY_BACKOFF_MS = 1e3;
|
|
11541
|
+
function monotonicNowMs() {
|
|
11542
|
+
return Number(process.hrtime.bigint() / 1000000n);
|
|
11543
|
+
}
|
|
11544
|
+
var transactionThrottles = /* @__PURE__ */ new WeakMap();
|
|
11545
|
+
function finiteNow(clock, label) {
|
|
11546
|
+
const now = clock();
|
|
11547
|
+
if (typeof now !== "number" || !Number.isFinite(now)) {
|
|
11548
|
+
throw databaseError(
|
|
11549
|
+
"runtime_transaction_clock_invalid",
|
|
11550
|
+
"transaction",
|
|
11551
|
+
`${label} must return a finite monotonic number; got ${String(now)}`
|
|
11552
|
+
);
|
|
11553
|
+
}
|
|
11554
|
+
return now;
|
|
11555
|
+
}
|
|
11556
|
+
var CONTENTION_PRIMARY_CODES = [5, 6, 15];
|
|
11557
|
+
var CONTENTION_SYMBOLIC_NAMES = /^SQLITE_(?:BUSY|LOCKED|PROTOCOL)(?:_[A-Z0-9]+)?$/;
|
|
11558
|
+
var SQLITE_RESULT_NAMES = /^SQLITE_[A-Z][A-Z0-9_]*$/;
|
|
11559
|
+
var CONTENTION_MESSAGES = /^(?:database is locked|database table is locked|locking protocol|SQLITE_BUSY|SQLITE_LOCKED|SQLITE_PROTOCOL)(?:$|[\s.:])/i;
|
|
11560
|
+
function isTransactionContention(error) {
|
|
11561
|
+
const carrier = error;
|
|
11562
|
+
for (const value of [carrier?.errcode, carrier?.errCode, carrier?.errno]) {
|
|
11563
|
+
if (typeof value === "number" && Number.isInteger(value)) return CONTENTION_PRIMARY_CODES.includes(value & 255);
|
|
11564
|
+
}
|
|
11565
|
+
for (const value of [carrier?.code, carrier?.name]) {
|
|
11566
|
+
if (typeof value === "string" && SQLITE_RESULT_NAMES.test(value)) {
|
|
11567
|
+
return CONTENTION_SYMBOLIC_NAMES.test(value);
|
|
11568
|
+
}
|
|
11569
|
+
}
|
|
11570
|
+
return CONTENTION_MESSAGES.test(error instanceof Error ? error.message : String(error));
|
|
11571
|
+
}
|
|
11572
|
+
function withDatabaseTransaction(database, work, mode = "IMMEDIATE", clock = monotonicNowMs) {
|
|
11541
11573
|
if (activeTransactions.has(database)) {
|
|
11542
11574
|
throw databaseError("runtime_transaction_nested", "transaction", "nested transactions are not allowed");
|
|
11543
11575
|
}
|
|
11576
|
+
if (mode !== "DEFERRED") {
|
|
11577
|
+
const deadlines = transactionThrottles.get(database);
|
|
11578
|
+
const until = deadlines?.get(clock);
|
|
11579
|
+
if (until !== void 0) {
|
|
11580
|
+
const remaining = until - finiteNow(clock, "the transaction clock");
|
|
11581
|
+
if (remaining > 0) {
|
|
11582
|
+
throw databaseError(
|
|
11583
|
+
"runtime_transaction_busy",
|
|
11584
|
+
"transaction",
|
|
11585
|
+
`a previous BEGIN was blocked on this database; retry deferred ${String(remaining)}ms`
|
|
11586
|
+
);
|
|
11587
|
+
}
|
|
11588
|
+
deadlines?.delete(clock);
|
|
11589
|
+
}
|
|
11590
|
+
}
|
|
11591
|
+
try {
|
|
11592
|
+
database.exec(`BEGIN ${mode}`);
|
|
11593
|
+
} catch (error) {
|
|
11594
|
+
if (!isTransactionContention(error)) throw error;
|
|
11595
|
+
const now = finiteNow(clock, "the transaction clock");
|
|
11596
|
+
let deadlines = transactionThrottles.get(database);
|
|
11597
|
+
if (deadlines === void 0) {
|
|
11598
|
+
deadlines = /* @__PURE__ */ new WeakMap();
|
|
11599
|
+
transactionThrottles.set(database, deadlines);
|
|
11600
|
+
}
|
|
11601
|
+
deadlines.set(clock, now + TRANSACTION_BUSY_BACKOFF_MS);
|
|
11602
|
+
throw databaseError(
|
|
11603
|
+
"runtime_transaction_busy",
|
|
11604
|
+
"transaction",
|
|
11605
|
+
`BEGIN ${mode} blocked by another transaction: ${error instanceof Error ? error.message : String(error)}`
|
|
11606
|
+
);
|
|
11607
|
+
}
|
|
11544
11608
|
activeTransactions.add(database);
|
|
11545
|
-
|
|
11609
|
+
if (mode !== "DEFERRED") transactionThrottles.get(database)?.delete(clock);
|
|
11546
11610
|
try {
|
|
11547
11611
|
const result = work();
|
|
11548
11612
|
database.exec("COMMIT");
|
package/plugins/kxm/package.json
CHANGED
|
@@ -213,21 +213,226 @@ export function openDatabase(file: string, description: string, spec: DatabaseSc
|
|
|
213
213
|
|
|
214
214
|
const activeTransactions = new WeakSet<DatabaseSync>();
|
|
215
215
|
|
|
216
|
+
/**
|
|
217
|
+
* How long a connection refuses to retry a `BEGIN` that lost the write race.
|
|
218
|
+
*
|
|
219
|
+
* SQLite's busy timeout is per connection, so a `BEGIN` against a locked
|
|
220
|
+
* database waits the full timeout before failing. Callers that recover from a
|
|
221
|
+
* lost write open several transactions in a row; without this window each of
|
|
222
|
+
* them pays the timeout, and a suite run showed that turning one 4-second test
|
|
223
|
+
* into 17 minutes. The old code got that speed by accident — it left the
|
|
224
|
+
* connection permanently marked as in-transaction after a failed `BEGIN`, which
|
|
225
|
+
* is the defect fixed below — so the backoff replaces the fast-fail without
|
|
226
|
+
* re-introducing the poison.
|
|
227
|
+
*
|
|
228
|
+
* Two limits, both deliberate:
|
|
229
|
+
* - It is a **throttle, not a queue**. A caller whose lock cleared 1 ms later is
|
|
230
|
+
* still refused for the rest of the window; the refusal is explicit
|
|
231
|
+
* (`runtime_transaction_busy`, message says `retry deferred`) and bounded by
|
|
232
|
+
* this constant. A retry after the window may pay the busy timeout again.
|
|
233
|
+
* - It is **per connection object, in this process**. It is not a cross-process
|
|
234
|
+
* backoff and does not leak to another connection to the same file.
|
|
235
|
+
*/
|
|
236
|
+
export const TRANSACTION_BUSY_BACKOFF_MS = 1_000;
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* Monotonic elapsed time, deliberately not `Date.now()`.
|
|
240
|
+
*
|
|
241
|
+
* A wall-clock step backwards would otherwise keep a long-gone write lock
|
|
242
|
+
* refusing transactions until real time caught up, and a step forward would end
|
|
243
|
+
* the throttle early. `hrtime.bigint()` is relative to an arbitrary past origin
|
|
244
|
+
* and never moves backwards, so the window is bounded by
|
|
245
|
+
* {@link TRANSACTION_BUSY_BACKOFF_MS} no matter what the system clock does.
|
|
246
|
+
*
|
|
247
|
+
* Injectable: production callers never pass it, and a test drives it directly so
|
|
248
|
+
* the window is stepped rather than raced or slept through.
|
|
249
|
+
*/
|
|
250
|
+
export type MonotonicClock = () => number;
|
|
251
|
+
|
|
252
|
+
function monotonicNowMs(): number {
|
|
253
|
+
return Number(process.hrtime.bigint() / 1_000_000n);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* Pending throttle per connection, **keyed by the clock that armed it**.
|
|
258
|
+
*
|
|
259
|
+
* The nesting is the fix, not tidiness. A deadline is a number from *some* clock,
|
|
260
|
+
* and the injectable `clock` exists only so a test can step the window; comparing
|
|
261
|
+
* a deadline armed by one clock against a reading taken from another is how an
|
|
262
|
+
* injected "one hour from now" throttles a production caller that never passed a
|
|
263
|
+
* clock at all — and how that same caller could clear a deadline it never armed.
|
|
264
|
+
* Each clock domain therefore gets its own deadline and can only read, expire or
|
|
265
|
+
* replace its own. The inner map is **weak in the clock**: it keeps no otherwise
|
|
266
|
+
* unreachable clock function alive, so an attempt that builds a fresh closure per
|
|
267
|
+
* call leaves nothing behind once that closure is collected. A clock that *is*
|
|
268
|
+
* retained keeps its entry until it expires or is replaced — collection is neither
|
|
269
|
+
* immediate nor size-bounded, and no committed test measures any of this, because no
|
|
270
|
+
* production caller passes a clock.
|
|
271
|
+
*/
|
|
272
|
+
const transactionThrottles = new WeakMap<DatabaseSync, WeakMap<MonotonicClock, number>>();
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* A clock reading this helper can reason about.
|
|
276
|
+
*
|
|
277
|
+
* Scope, stated as measured rather than as a slogan. The clock is read at **two call
|
|
278
|
+
* sites**: checking an existing deadline on a non-`DEFERRED` attempt, and arming after a
|
|
279
|
+
* contended `BEGIN` failure. Reads per call, each reproduced by an assertion in
|
|
280
|
+
* `test/core/intake.test.ts`: clean success with no pending entry 0; successful
|
|
281
|
+
* `DEFERRED`, pending entry present or not, 0; fresh contention 1; refusal inside a window
|
|
282
|
+
* 1; an expired deadline followed by renewed contention 2 in that call; a permanent
|
|
283
|
+
* `BEGIN` failure with no pending entry 0; a nested-transaction rejection 0.
|
|
284
|
+
*
|
|
285
|
+
* Which means it is wrong in both directions to say either "every `BEGIN` validates the
|
|
286
|
+
* clock" or "the *only* transaction that skips it is a clean uncontended success" — the
|
|
287
|
+
* second was mine, twice, and the second correction to it was still a slogan. The tests
|
|
288
|
+
* count; the prose only points at them. Production cannot reach the guard at all, because
|
|
289
|
+
* the default is `hrtime`; it exists so the injectable seam cannot become a silent
|
|
290
|
+
* bypass.
|
|
291
|
+
*/
|
|
292
|
+
function finiteNow(clock: MonotonicClock, label: string): number {
|
|
293
|
+
const now = clock();
|
|
294
|
+
if (typeof now !== "number" || !Number.isFinite(now)) {
|
|
295
|
+
throw databaseError(
|
|
296
|
+
"runtime_transaction_clock_invalid",
|
|
297
|
+
"transaction",
|
|
298
|
+
`${label} must return a finite monotonic number; got ${String(now)}`,
|
|
299
|
+
);
|
|
300
|
+
}
|
|
301
|
+
return now;
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* Is this `BEGIN` failure contention for the write lock, as opposed to a
|
|
306
|
+
* programming or environment error?
|
|
307
|
+
*
|
|
308
|
+
* Only contention may be retried, and only contention earns the throttle window.
|
|
309
|
+
* Anything else — "cannot start a transaction within a transaction", a closed
|
|
310
|
+
* connection, a miscompiled statement — must surface unchanged, or a permanent
|
|
311
|
+
* bug looks like a transient one and the caller retries forever.
|
|
312
|
+
*
|
|
313
|
+
* SQLite's **numeric result code decides** where one exists, because it is stable
|
|
314
|
+
* across versions while message text is not: the primary code is `code & 0xff`, so
|
|
315
|
+
* extended forms land on their primaries — `SQLITE_BUSY_RECOVERY` (261) and
|
|
316
|
+
* `SQLITE_BUSY_SNAPSHOT` (517) on `SQLITE_BUSY` (5), `SQLITE_LOCKED_SHAREDCACHE`
|
|
317
|
+
* (262) on `SQLITE_LOCKED` (6). `SQLITE_PROTOCOL` (15) is included deliberately:
|
|
318
|
+
* SQLite raises it when repeated attempts to start a transaction under WAL exhaust
|
|
319
|
+
* the retry count, which is a lock-acquisition retry condition, not a broken
|
|
320
|
+
* database.
|
|
321
|
+
*
|
|
322
|
+
* Where both runtimes expose one, **the number wins over the text**: Node spells it
|
|
323
|
+
* `errcode`, `bun:sqlite` spells it `errno` (verified on Bun 1.3.14, where a shared-
|
|
324
|
+
* cache `BEGIN` arrives as `errno: 262`, `code: "SQLITE_LOCKED_SHAREDCACHE"`,
|
|
325
|
+
* message "database schema is locked: shared"). A symbolic `code`/`name` of the form
|
|
326
|
+
* `SQLITE_BUSY*`/`SQLITE_LOCKED*`/`SQLITE_PROTOCOL*` is accepted next, and bare
|
|
327
|
+
* message matching is the last resort — applied only when neither exists, so a
|
|
328
|
+
* wrapper that merely quotes "database is locked" alongside a permanent code is not
|
|
329
|
+
* mistaken for contention.
|
|
330
|
+
*
|
|
331
|
+
* Deliberately **not** contention: `SQLITE_FULL` / "database or disk is full",
|
|
332
|
+
* "unable to open database file", and WAL shared-memory I/O failures. Those stay
|
|
333
|
+
* broken until something outside this connection changes.
|
|
334
|
+
*/
|
|
335
|
+
/** Node: `errcode`. Bun: `errno`. Both spell the extended code as a number. */
|
|
336
|
+
const CONTENTION_PRIMARY_CODES: readonly number[] = [5, 6, 15];
|
|
337
|
+
/** `bun:sqlite` puts the symbolic name in `code`; Node puts its own kind there. */
|
|
338
|
+
const CONTENTION_SYMBOLIC_NAMES = /^SQLITE_(?:BUSY|LOCKED|PROTOCOL)(?:_[A-Z0-9]+)?$/;
|
|
339
|
+
/** Any SQLite result name. If one is present it decides, so text cannot argue. */
|
|
340
|
+
const SQLITE_RESULT_NAMES = /^SQLITE_[A-Z][A-Z0-9_]*$/;
|
|
341
|
+
const CONTENTION_MESSAGES =
|
|
342
|
+
/^(?:database is locked|database table is locked|locking protocol|SQLITE_BUSY|SQLITE_LOCKED|SQLITE_PROTOCOL)(?:$|[\s.:])/i;
|
|
343
|
+
|
|
344
|
+
export function isTransactionContention(error: unknown): boolean {
|
|
345
|
+
const carrier = error as { errcode?: unknown; errCode?: unknown; errno?: unknown; code?: unknown; name?: unknown }
|
|
346
|
+
| undefined;
|
|
347
|
+
// A numeric code wins first, then a symbolic result name, and both win **over the
|
|
348
|
+
// message**: `errno` is what `bun:sqlite` exposes for the extended result code while
|
|
349
|
+
// its `code` field holds the symbolic name, and Node spells the number `errcode`.
|
|
350
|
+
// All three properties are read; what the list orders is which **integer value
|
|
351
|
+
// decides** — a present `errcode` outranks `errCode`, which outranks `errno`, and a
|
|
352
|
+
// later number is never consulted. Text is consulted only when the error carries
|
|
353
|
+
// neither a number nor a SQLite result name, so no wrapper quoting an older
|
|
354
|
+
// "database is locked" can outvote a code on either runtime.
|
|
355
|
+
for (const value of [carrier?.errcode, carrier?.errCode, carrier?.errno]) {
|
|
356
|
+
if (typeof value === "number" && Number.isInteger(value)) return CONTENTION_PRIMARY_CODES.includes(value & 0xff);
|
|
357
|
+
}
|
|
358
|
+
for (const value of [carrier?.code, carrier?.name]) {
|
|
359
|
+
// A result **name** is as authoritative as a number, and in both directions:
|
|
360
|
+
// `SQLITE_FULL` wearing a "database is locked" message is not contention. Node
|
|
361
|
+
// spells its own error kind `ERR_SQLITE_ERROR`, which is deliberately not a
|
|
362
|
+
// SQLite result name and so never reaches a verdict here.
|
|
363
|
+
if (typeof value === "string" && SQLITE_RESULT_NAMES.test(value)) {
|
|
364
|
+
return CONTENTION_SYMBOLIC_NAMES.test(value);
|
|
365
|
+
}
|
|
366
|
+
}
|
|
367
|
+
return CONTENTION_MESSAGES.test(error instanceof Error ? error.message : String(error));
|
|
368
|
+
}
|
|
369
|
+
|
|
216
370
|
export function withDatabaseTransaction<T>(
|
|
217
371
|
database: DatabaseSync,
|
|
218
372
|
work: () => T,
|
|
219
373
|
mode: "IMMEDIATE" | "DEFERRED" | "EXCLUSIVE" = "IMMEDIATE",
|
|
374
|
+
clock: MonotonicClock = monotonicNowMs,
|
|
220
375
|
): T {
|
|
221
376
|
if (activeTransactions.has(database)) {
|
|
222
377
|
throw databaseError("runtime_transaction_nested", "transaction", "nested transactions are not allowed");
|
|
223
378
|
}
|
|
379
|
+
// A DEFERRED BEGIN takes no write lock, so it is exempt from the *check*: refusing
|
|
380
|
+
// it would deny legitimate work over a contention it did not ask for. Exempt from
|
|
381
|
+
// the check only — shared-cache schema locks can still make a deferred `BEGIN`
|
|
382
|
+
// fail, and that failure arms a deadline like any other, because the next attempt
|
|
383
|
+
// would stall the same way.
|
|
384
|
+
if (mode !== "DEFERRED") {
|
|
385
|
+
const deadlines = transactionThrottles.get(database);
|
|
386
|
+
const until = deadlines?.get(clock);
|
|
387
|
+
if (until !== undefined) {
|
|
388
|
+
const remaining = until - finiteNow(clock, "the transaction clock");
|
|
389
|
+
if (remaining > 0) {
|
|
390
|
+
throw databaseError(
|
|
391
|
+
"runtime_transaction_busy",
|
|
392
|
+
"transaction",
|
|
393
|
+
`a previous BEGIN was blocked on this database; retry deferred ${String(remaining)}ms`,
|
|
394
|
+
);
|
|
395
|
+
}
|
|
396
|
+
deadlines?.delete(clock);
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
// Claim the marker only once BEGIN has succeeded. If BEGIN throws — another
|
|
400
|
+
// writer held the database past the busy timeout — a connection that never
|
|
401
|
+
// entered a transaction must not be left permanently marked as inside one,
|
|
402
|
+
// which would fail every later transaction on it with a misleading
|
|
403
|
+
// "nested" error. `finally` cannot cover this: it only runs after the try.
|
|
404
|
+
try {
|
|
405
|
+
database.exec(`BEGIN ${mode}`);
|
|
406
|
+
} catch (error) {
|
|
407
|
+
if (!isTransactionContention(error)) throw error;
|
|
408
|
+
const now = finiteNow(clock, "the transaction clock");
|
|
409
|
+
let deadlines = transactionThrottles.get(database);
|
|
410
|
+
if (deadlines === undefined) {
|
|
411
|
+
deadlines = new WeakMap<MonotonicClock, number>();
|
|
412
|
+
transactionThrottles.set(database, deadlines);
|
|
413
|
+
}
|
|
414
|
+
deadlines.set(clock, now + TRANSACTION_BUSY_BACKOFF_MS);
|
|
415
|
+
throw databaseError(
|
|
416
|
+
"runtime_transaction_busy",
|
|
417
|
+
"transaction",
|
|
418
|
+
`BEGIN ${mode} blocked by another transaction: ${error instanceof Error ? error.message : String(error)}`,
|
|
419
|
+
);
|
|
420
|
+
}
|
|
224
421
|
activeTransactions.add(database);
|
|
225
|
-
|
|
422
|
+
// A successful write-mode BEGIN means this connection is holding the slot, so any
|
|
423
|
+
// earlier contention is over. A DEFERRED success proves nothing about the write
|
|
424
|
+
// lock and must not clear a throttle that another caller's contention armed.
|
|
425
|
+
if (mode !== "DEFERRED") transactionThrottles.get(database)?.delete(clock);
|
|
226
426
|
try {
|
|
227
427
|
const result = work();
|
|
228
428
|
database.exec("COMMIT");
|
|
229
429
|
return result;
|
|
230
430
|
} catch (error) {
|
|
431
|
+
// A failed ROLLBACK usually means the connection is gone. Clearing the marker
|
|
432
|
+
// is bookkeeping, not proof: it does **not** show SQLite exited the
|
|
433
|
+
// transaction, and nothing here invalidates a handle whose rollback failed.
|
|
434
|
+
// Pre-existing, named rather than papered over — the caller receives the
|
|
435
|
+
// original error.
|
|
231
436
|
try { database.exec("ROLLBACK"); } catch { /* ignore rollback error if connection dead */ }
|
|
232
437
|
throw error;
|
|
233
438
|
} finally {
|
|
@@ -98,7 +98,45 @@ const ACTOR_ID_RE = /^.{1,200}$/;
|
|
|
98
98
|
|
|
99
99
|
/** Canonical SHA-256 over the authority ceiling — the coordinator's fingerprint. */
|
|
100
100
|
export function kxmCeilingHash(authority: KxmCoordinatorAuthority): string {
|
|
101
|
-
return `sha256:${createHash("sha256").update(stableStringify(authority), "utf8").digest("hex")}`;
|
|
101
|
+
return `sha256:${createHash("sha256").update(stableStringify(normalizeAuthority(authority)), "utf8").digest("hex")}`;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Does an already-stored coordinator express this ceiling?
|
|
106
|
+
*
|
|
107
|
+
* A row written before set normalisation existed carries a fingerprint that
|
|
108
|
+
* `kxmCeilingHash` no longer reproduces, so comparing the stored hash alone is
|
|
109
|
+
* not enough: recompute over the authority it kept. Every path that asks this
|
|
110
|
+
* question — the initial slot lookup and **both** lost-write read-backs — must
|
|
111
|
+
* go through here, or an upgrade makes the same row equivalent on lookup and a
|
|
112
|
+
* `coordinator_write_lost` conflict on the race path.
|
|
113
|
+
*
|
|
114
|
+
* A legacy row is returned as stored, so its `ceilingHash` is historical: a
|
|
115
|
+
* caller must not assume every persisted hash uses today's algorithm.
|
|
116
|
+
*/
|
|
117
|
+
function ceilingsMatch(stored: KxmCoordinatorRecord, ceilingHash: string): boolean {
|
|
118
|
+
return stored.ceilingHash === ceilingHash || kxmCeilingHash(stored.authority) === ceilingHash;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function normalizeAuthority(authority: KxmCoordinatorAuthority): KxmCoordinatorAuthority {
|
|
122
|
+
const tools = authority.tools;
|
|
123
|
+
return {
|
|
124
|
+
repositoryAccess: authority.repositoryAccess,
|
|
125
|
+
effects: canonicalSet(authority.effects ?? []),
|
|
126
|
+
...(tools !== undefined
|
|
127
|
+
? {
|
|
128
|
+
tools: {
|
|
129
|
+
...(tools.preset !== undefined ? { preset: tools.preset } : {}),
|
|
130
|
+
...(tools.allow !== undefined ? { allow: canonicalSet(tools.allow) } : {}),
|
|
131
|
+
...(tools.deny !== undefined ? { deny: canonicalSet(tools.deny) } : {}),
|
|
132
|
+
},
|
|
133
|
+
}
|
|
134
|
+
: {}),
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function canonicalSet(values: readonly string[]): string[] {
|
|
139
|
+
return [...new Set(values)].sort();
|
|
102
140
|
}
|
|
103
141
|
|
|
104
142
|
/**
|
|
@@ -131,7 +169,7 @@ export function bindKxmCoordinator(
|
|
|
131
169
|
|
|
132
170
|
if (existing) {
|
|
133
171
|
const current = JSON.parse(existing.record) as KxmCoordinatorRecord;
|
|
134
|
-
if (current
|
|
172
|
+
if (ceilingsMatch(current, ceilingHash)) {
|
|
135
173
|
return { coordinator: current, created: false };
|
|
136
174
|
}
|
|
137
175
|
if (!input.rebind) {
|
|
@@ -169,11 +207,11 @@ export function bindKxmCoordinator(
|
|
|
169
207
|
record: kxmCanonicalJson(record as unknown as JsonValue),
|
|
170
208
|
});
|
|
171
209
|
if (!replaced) {
|
|
172
|
-
// Another process
|
|
173
|
-
//
|
|
210
|
+
// Another process reached the same ceiling first. Return its identity when it
|
|
211
|
+
// is the ceiling we asked for; anything else is a real conflict.
|
|
174
212
|
const winner = context.eventStore.coordinatorInSlot(context.projectId, role, channel);
|
|
175
213
|
const record2 = winner ? (JSON.parse(winner.record) as KxmCoordinatorRecord) : undefined;
|
|
176
|
-
if (record2 && record2
|
|
214
|
+
if (record2 && ceilingsMatch(record2, ceilingHash)) {
|
|
177
215
|
return { coordinator: record2, created: false };
|
|
178
216
|
}
|
|
179
217
|
throw runtimeError("coordinator_write_lost", existing.coordinatorId, "the coordinator slot changed underneath this rebind");
|
|
@@ -196,7 +234,7 @@ export function bindKxmCoordinator(
|
|
|
196
234
|
configRevision: context.configRevision,
|
|
197
235
|
ceilingHash,
|
|
198
236
|
};
|
|
199
|
-
return
|
|
237
|
+
return persistCoordinator(context, record);
|
|
200
238
|
}
|
|
201
239
|
|
|
202
240
|
/** Resolve a coordinator by id; unknown identity fails closed. */
|
|
@@ -210,10 +248,15 @@ export function resolveKxmCoordinator(context: KxmRuntimeContext, coordinatorId:
|
|
|
210
248
|
* Accept one internal message at the intake boundary.
|
|
211
249
|
*
|
|
212
250
|
* Duplicate ingress (same idempotency key, same content) returns the original
|
|
213
|
-
* record and cannot
|
|
251
|
+
* record and cannot produce a second admission record. This layer deduplicates
|
|
252
|
+
* *intake and admission*; creating a task is the M2 consumer's job, so nothing here can
|
|
253
|
+
* promise anything about tasks. The same key with **different** content
|
|
214
254
|
* is rejected: a reused key must not smuggle a different payload. Payloads
|
|
215
|
-
* classified `secret` are never persisted — only their hash, so
|
|
216
|
-
*
|
|
255
|
+
* classified `secret` are never persisted — only their hash, so a caller that
|
|
256
|
+
* classifies honestly gets a store that will not hold that payload. The scope is
|
|
257
|
+
* exactly that: classification is caller-asserted, so this is a storage decision
|
|
258
|
+
* under a label, not secret detection, and intake is still not a place to keep
|
|
259
|
+
* credentials. While the project is paused the message is durable with
|
|
217
260
|
* a held dispatch intent instead of being dropped.
|
|
218
261
|
*/
|
|
219
262
|
export function acceptKxmIntakeMessage(
|
|
@@ -429,28 +472,45 @@ export function isKxmProjectPaused(context: KxmRuntimeContext): boolean {
|
|
|
429
472
|
return context.eventStore.projectControl(context.projectId)?.paused === true;
|
|
430
473
|
}
|
|
431
474
|
|
|
475
|
+
/** How many held rows one drain page reads. Bounds a page, not the whole drain. */
|
|
476
|
+
const INTAKE_DRAIN_PAGE_ROWS = 500;
|
|
477
|
+
|
|
432
478
|
function releaseHeldIntake(context: KxmRuntimeContext, now: string): KxmIntakeMessage[] {
|
|
433
479
|
const released: KxmIntakeMessage[] = [];
|
|
434
480
|
// Drain every held row. A paging loop, not a single capped page: stranding the
|
|
435
481
|
// 501st message behind a "resume releases held intent" claim is a lie of omission.
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
482
|
+
//
|
|
483
|
+
// What is still true after that fix: the loop holds one write transaction and
|
|
484
|
+
// retains every released message, so total work and memory grow with the held
|
|
485
|
+
// backlog even though each read is bounded. Measured on an in-memory store: 150k
|
|
486
|
+
// held rows with 64-byte payloads took 4.5 s synchronously and ~95 MiB of heap;
|
|
487
|
+
// 500k maximum-size payloads would retain ~7.6 GiB before any database overhead,
|
|
488
|
+
// and an allocation failure will not reliably arrive as `intake_drain_stalled`.
|
|
489
|
+
// Resume is finite because the write lock keeps ingress out of the loop, so this
|
|
490
|
+
// is a throughput and memory limit, not a correctness hole. Bounding it is a
|
|
491
|
+
// pre-condition of M2 sustained traffic, not of this contract — see "Still open".
|
|
492
|
+
for (;;) {
|
|
493
|
+
const held = context.eventStore.intakeInStates(context.projectId, ["held_paused"], INTAKE_DRAIN_PAGE_ROWS);
|
|
494
|
+
if (held.length === 0) return released;
|
|
495
|
+
let progressed = false;
|
|
439
496
|
for (const row of held) {
|
|
440
497
|
const message = JSON.parse(row.record) as KxmIntakeMessage;
|
|
441
498
|
const next: KxmIntakeMessage = { ...message, dispatch: { state: "ready", updatedAt: now } };
|
|
442
499
|
validateIntakeMessage(next, next.messageId);
|
|
443
500
|
if (context.eventStore.updateIntakeDispatch(row.messageId, "held_paused", { state: "ready", record: kxmCanonicalJson(next as unknown as JsonValue) })) {
|
|
444
501
|
released.push(next);
|
|
502
|
+
progressed = true;
|
|
445
503
|
}
|
|
446
504
|
}
|
|
447
|
-
|
|
448
|
-
|
|
505
|
+
if (!progressed) {
|
|
506
|
+
// Cannot move anything: fail loudly inside the caller's transaction, which
|
|
507
|
+
// rolls the resume back, rather than half-resuming and leaving rows held.
|
|
508
|
+
throw runtimeError("intake_drain_stalled", context.projectId, `held intake did not drain; ${String(held.length)} row(s) still held`);
|
|
509
|
+
}
|
|
449
510
|
}
|
|
450
|
-
return released;
|
|
451
511
|
}
|
|
452
512
|
|
|
453
|
-
function persistCoordinator(context: KxmRuntimeContext, record: KxmCoordinatorRecord): { coordinator: KxmCoordinatorRecord } {
|
|
513
|
+
function persistCoordinator(context: KxmRuntimeContext, record: KxmCoordinatorRecord): { coordinator: KxmCoordinatorRecord; created: boolean } {
|
|
454
514
|
validateCoordinator(record, record.coordinatorId);
|
|
455
515
|
const inserted = context.eventStore.insertCoordinatorIfAbsent({
|
|
456
516
|
coordinatorId: record.coordinatorId,
|
|
@@ -462,13 +522,13 @@ function persistCoordinator(context: KxmRuntimeContext, record: KxmCoordinatorRe
|
|
|
462
522
|
boundAt: record.boundAt,
|
|
463
523
|
record: kxmCanonicalJson(record as unknown as JsonValue),
|
|
464
524
|
});
|
|
465
|
-
if (inserted) return { coordinator: record };
|
|
525
|
+
if (inserted) return { coordinator: record, created: true };
|
|
466
526
|
// Lost the race for an empty slot: binding is create-once, so the winner is the
|
|
467
527
|
// answer whenever it reached the same ceiling. Only a different ceiling is a
|
|
468
|
-
// conflict worth reporting.
|
|
528
|
+
// conflict worth reporting. The loser must not report that it created anything.
|
|
469
529
|
const winner = context.eventStore.coordinatorInSlot(record.projectId, record.role, record.channel);
|
|
470
530
|
const won = winner ? (JSON.parse(winner.record) as KxmCoordinatorRecord) : undefined;
|
|
471
|
-
if (won && won
|
|
531
|
+
if (won && ceilingsMatch(won, record.ceilingHash)) return { coordinator: won, created: false };
|
|
472
532
|
throw runtimeError("coordinator_write_lost", record.coordinatorId, "the coordinator slot was claimed by a different ceiling");
|
|
473
533
|
}
|
|
474
534
|
|
|
@@ -502,6 +562,12 @@ function assertCeilingNotWidened(
|
|
|
502
562
|
if (newlyAllowed.length > 0) {
|
|
503
563
|
throw runtimeError("coordinator_rebind_widens_tools", coordinatorId, `a rebind may not allow new tools: ${newlyAllowed.join(", ")}`);
|
|
504
564
|
}
|
|
565
|
+
// Tool evaluation applies no allowlist restriction when the list is empty or
|
|
566
|
+
// absent (`commands.ts` gates only on a non-empty allow list), so dropping a
|
|
567
|
+
// populated list is a widening even though every entry it named is gone.
|
|
568
|
+
if (allowedBefore.size > 0 && (nextTools.allow ?? []).length === 0) {
|
|
569
|
+
throw runtimeError("coordinator_rebind_clears_allowlist", coordinatorId, "a rebind may not drop or empty a populated allow list; an absent allow list imposes no restriction");
|
|
570
|
+
}
|
|
505
571
|
const deniedBefore = new Set(previousTools.deny ?? []);
|
|
506
572
|
const undenied = [...deniedBefore].filter((tool) => !(nextTools.deny ?? []).includes(tool));
|
|
507
573
|
if (undenied.length > 0) {
|
|
@@ -540,24 +606,9 @@ function validateAuthority(authority: KxmCoordinatorAuthority): KxmCoordinatorAu
|
|
|
540
606
|
}
|
|
541
607
|
// Sets are stored canonically: order and repeats carry no authority, and leaving
|
|
542
608
|
// them as supplied would let an equivalent ceiling masquerade as a rebind.
|
|
543
|
-
return
|
|
544
|
-
repositoryAccess: authority.repositoryAccess,
|
|
545
|
-
effects: canonicalSet(effects),
|
|
546
|
-
...(tools !== undefined
|
|
547
|
-
? {
|
|
548
|
-
tools: {
|
|
549
|
-
...(tools.preset !== undefined ? { preset: tools.preset } : {}),
|
|
550
|
-
...(tools.allow !== undefined ? { allow: canonicalSet(tools.allow) } : {}),
|
|
551
|
-
...(tools.deny !== undefined ? { deny: canonicalSet(tools.deny) } : {}),
|
|
552
|
-
},
|
|
553
|
-
}
|
|
554
|
-
: {}),
|
|
555
|
-
};
|
|
609
|
+
return normalizeAuthority(authority);
|
|
556
610
|
}
|
|
557
611
|
|
|
558
|
-
function canonicalSet(values: readonly string[]): string[] {
|
|
559
|
-
return [...new Set(values)].sort();
|
|
560
|
-
}
|
|
561
612
|
|
|
562
613
|
function validateActor<T extends { kind: KxmCoordinatorRecord["boundBy"]["kind"]; id: string }>(actor: T, field: string): T {
|
|
563
614
|
const kinds: string[] = ["human", "runtime", "hub", "agent", "adapter"];
|
|
@@ -8,7 +8,7 @@ import { AGENT_COMMANDS_MAP, enforceToolPolicy, getMcpTools, reconcileInbox } fr
|
|
|
8
8
|
import { deliverInboxNotification } from "./inbox.ts";
|
|
9
9
|
import type { HubEvent, MessageRecord } from "./protocol.ts";
|
|
10
10
|
|
|
11
|
-
const VERSION = "0.7.
|
|
11
|
+
const VERSION = "0.7.51";
|
|
12
12
|
const inbox = new Map<string, MessageRecord>();
|
|
13
13
|
const notifiedInbox = new Set<string>();
|
|
14
14
|
let meshClient: HubClient | undefined;
|
|
@@ -1255,7 +1255,14 @@ export class KxmRunEventStore {
|
|
|
1255
1255
|
return row ? intakeFromRow(row) : undefined;
|
|
1256
1256
|
}
|
|
1257
1257
|
|
|
1258
|
-
/**
|
|
1258
|
+
/**
|
|
1259
|
+
* Intake rows in the given dispatch states, in arrival order (replay-safe).
|
|
1260
|
+
*
|
|
1261
|
+
* `rowid` gives same-store arrival order, which is what queue priority needs
|
|
1262
|
+
* here. It is **not** a durable sequence: this repository backs stores up with
|
|
1263
|
+
* `VACUUM INTO`, and a vacuum may renumber implicit rowids. An explicit
|
|
1264
|
+
* immutable arrival sequence is tracked in the plan's schema-v6 follow-ups.
|
|
1265
|
+
*/
|
|
1259
1266
|
intakeInStates(projectId: string, states: readonly KxmIntakeMessageRow["dispatchState"][], limit = 100): KxmIntakeMessageRow[] {
|
|
1260
1267
|
if (states.length === 0) return [];
|
|
1261
1268
|
const placeholders = states.map(() => "?").join(", ");
|