@gamaze/hicortex 0.23.2 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,440 @@
1
+ "use strict";
2
+ /**
3
+ * The durable distill inbox + its drain (#529).
4
+ *
5
+ * Delivery and compute are separated: `POST /distill` in queue mode stores the
6
+ * REDACTED segment in the `distill_queue` table (migration v24) and answers
7
+ * 201 immediately — no LLM call, no probe, no chunk-size detection, no token
8
+ * gate (all of those move to the drain). The nightly's drain stage then
9
+ * distills queued items oldest-first, round-robin per client, before
10
+ * consolidation, so all distill LLM traffic happens inside the owner's
11
+ * scheduled runs instead of on the clients' drifting capture grids.
12
+ *
13
+ * This module holds BOTH halves as pure units:
14
+ * - the inbox STORE (enqueue, dedup mirrors, depth/age stats, delete) — real
15
+ * functions the /distill handler calls directly;
16
+ * - the DRAIN — dependency-injected (LlmClient-like object, embed fn,
17
+ * yield-probe, token-budget gate, run deadline, sleep) so the whole stage
18
+ * is unit-testable with no HTTP and no real LLM. Every LLM call goes
19
+ * through `distillSession` (distiller.ts) on the injected client — never a
20
+ * bespoke fetch — so the drain inherits the #337 undici dispatcher, #355
21
+ * single-flight, the retry ladder, and the circuit breaker by construction.
22
+ *
23
+ * Ordering contract (spec #529): oldest-first (rowid = arrival order),
24
+ * round-robin per client key (`source_machine` + `source_agent`) so one
25
+ * client's giant backlog cannot starve the others, and a session's segments
26
+ * in arrival order (natural — a client POSTs its segments sequentially, so
27
+ * ascending rowid within a client is per-session ascending).
28
+ *
29
+ * Checkpointing contract: per item, ALL memory inserts and the inbox-row
30
+ * delete commit in ONE transaction. An interrupted run (crash, deadline,
31
+ * endpoint outage) leaves processed items done and the rest queued — the
32
+ * segment-exact dedup keys (`<sid>#<segment_id>#<i>`) make the retry
33
+ * idempotent, dup-over-loss.
34
+ */
35
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
36
+ if (k2 === undefined) k2 = k;
37
+ var desc = Object.getOwnPropertyDescriptor(m, k);
38
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
39
+ desc = { enumerable: true, get: function() { return m[k]; } };
40
+ }
41
+ Object.defineProperty(o, k2, desc);
42
+ }) : (function(o, m, k, k2) {
43
+ if (k2 === undefined) k2 = k;
44
+ o[k2] = m[k];
45
+ }));
46
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
47
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
48
+ }) : function(o, v) {
49
+ o["default"] = v;
50
+ });
51
+ var __importStar = (this && this.__importStar) || (function () {
52
+ var ownKeys = function(o) {
53
+ ownKeys = Object.getOwnPropertyNames || function (o) {
54
+ var ar = [];
55
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
56
+ return ar;
57
+ };
58
+ return ownKeys(o);
59
+ };
60
+ return function (mod) {
61
+ if (mod && mod.__esModule) return mod;
62
+ var result = {};
63
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
64
+ __setModuleDefault(result, mod);
65
+ return result;
66
+ };
67
+ })();
68
+ Object.defineProperty(exports, "__esModule", { value: true });
69
+ exports.enqueueDistill = enqueueDistill;
70
+ exports.countQueuedSegment = countQueuedSegment;
71
+ exports.countQueuedSession = countQueuedSession;
72
+ exports.queueDepth = queueDepth;
73
+ exports.oldestQueueAgeMs = oldestQueueAgeMs;
74
+ exports.readQueueStats = readQueueStats;
75
+ exports.isFreshBrain = isFreshBrain;
76
+ exports.resolveQueueMode = resolveQueueMode;
77
+ exports.drainDistillQueue = drainDistillQueue;
78
+ const distiller_js_1 = require("./distiller.js");
79
+ const calibration_js_1 = require("./calibration.js");
80
+ const storage = __importStar(require("./storage.js"));
81
+ const dedup_js_1 = require("./dedup.js");
82
+ /** Normalize an unknown wire value into a nullable string column. */
83
+ function str(v) {
84
+ return typeof v === "string" && v.length > 0 ? v : null;
85
+ }
86
+ /**
87
+ * Insert one delivery into the inbox. Idempotent on the UNIQUE(session_id,
88
+ * segment_id) index — a re-POST of the SAME key (only reachable through a
89
+ * race: the handler's prechecks already answered the sequential re-POST with
90
+ * a 200 skip) reports `{ duplicate: true }` and inserts nothing. Posts with
91
+ * no session_id have no dedup key by construction and always insert (SQLite
92
+ * unique indexes treat NULLs as distinct).
93
+ */
94
+ function enqueueDistill(db, input) {
95
+ const result = db
96
+ .prepare(`INSERT OR IGNORE INTO distill_queue
97
+ (session_id, segment_id, source_agent, source_agent_id, source_domain,
98
+ source_machine, project, session_date, privacy, text, arrived_at)
99
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`)
100
+ .run(str(input.session_id), typeof input.segment_id === "string" ? input.segment_id : "", str(input.source_agent), str(input.source_agent_id), str(input.source_domain), str(input.source_machine), str(input.project),
101
+ // Resolve the date ONCE at delivery (mirrors the sync handler): absent
102
+ // or non-string becomes today, and the drain reuses the stored value so
103
+ // a queued segment keeps its session date.
104
+ str(input.session_date) ?? new Date().toISOString().slice(0, 10), str(input.privacy), input.text, new Date().toISOString());
105
+ return { duplicate: result.changes === 0 };
106
+ }
107
+ /**
108
+ * Inbox mirror of the segment-exact delivery precheck: how many QUEUED rows
109
+ * match this exact `session_id` + `segment_id`. The handler adds this to
110
+ * `countExistingSegment` so a queued-but-undistilled segment's re-POST
111
+ * answers 200 skipped — the client's cursor advances and the segment is
112
+ * never queued twice.
113
+ */
114
+ function countQueuedSegment(db, sessionId, segmentId) {
115
+ return db
116
+ .prepare("SELECT COUNT(*) AS c FROM distill_queue WHERE session_id = ? AND segment_id = ?")
117
+ .get(sessionId, segmentId).c;
118
+ }
119
+ /**
120
+ * Inbox mirror of the legacy session-level precheck: how many QUEUED rows
121
+ * belong to this session (whole-session OR any segment) — a legacy
122
+ * whole-session re-POST while any part of the session sits queued skips.
123
+ */
124
+ function countQueuedSession(db, sessionId) {
125
+ return db
126
+ .prepare("SELECT COUNT(*) AS c FROM distill_queue WHERE session_id = ?")
127
+ .get(sessionId).c;
128
+ }
129
+ /** Queue depth (rows pending distillation). */
130
+ function queueDepth(db) {
131
+ return db.prepare("SELECT COUNT(*) AS c FROM distill_queue").get().c;
132
+ }
133
+ /** Age (ms) of the oldest queued item; null when the inbox is empty. */
134
+ function oldestQueueAgeMs(db, now = Date.now()) {
135
+ const row = db
136
+ .prepare("SELECT MIN(arrived_at) AS a FROM distill_queue")
137
+ .get();
138
+ if (!row.a)
139
+ return null;
140
+ const t = new Date(row.a).getTime();
141
+ // Unparseable stamp (clock corruption at write time) reads as age 0 — the
142
+ // honest floor — rather than NaN leaking into status/dashboard output.
143
+ return Number.isFinite(t) ? Math.max(0, now - t) : 0;
144
+ }
145
+ function readQueueStats(db, now = Date.now()) {
146
+ const age = oldestQueueAgeMs(db, now);
147
+ return {
148
+ depth: queueDepth(db),
149
+ oldest_age_hours: age === null ? null : Math.round((age / 3_600_000) * 10) / 10,
150
+ };
151
+ }
152
+ /** Delete one inbox row (its id from `distill_queue`). */
153
+ function deleteQueuedRow(db, id) {
154
+ db.prepare("DELETE FROM distill_queue WHERE id = ?").run(id);
155
+ }
156
+ // ---------------------------------------------------------------------------
157
+ // Queue-mode decision — shared by the /distill handler and its mirror tests
158
+ // ---------------------------------------------------------------------------
159
+ /**
160
+ * #529 bootstrap carve-out: a FRESH brain (empty memories table AND empty
161
+ * inbox) distills synchronously, exactly as pre-#529 — the first-ever
162
+ * delivery must not wait ~12 h for the next scheduled run. Anything else
163
+ * (memories exist OR rows are already queued) is ordinary queue mode.
164
+ */
165
+ function isFreshBrain(db) {
166
+ const mem = db
167
+ .prepare("SELECT EXISTS(SELECT 1 FROM memories LIMIT 1) AS e")
168
+ .get();
169
+ if (mem.e === 1)
170
+ return false;
171
+ const queued = db
172
+ .prepare("SELECT EXISTS(SELECT 1 FROM distill_queue LIMIT 1) AS e")
173
+ .get();
174
+ return queued.e === 0;
175
+ }
176
+ /**
177
+ * The ONE branch condition into the sync flow (owner ruling #529: kill
178
+ * switch and bootstrap carve-out share it — no third path):
179
+ *
180
+ * queueMode = distillQueue enabled && NOT a fresh brain
181
+ *
182
+ * `distillQueueEnabled` is the already-strict-boolean-resolved kill switch
183
+ * (`readStrictBoolean(config, "distillQueue") !== false`, default ON); false
184
+ * lands here always → today's synchronous distill byte-for-byte.
185
+ */
186
+ function resolveQueueMode(db, distillQueueEnabled) {
187
+ return distillQueueEnabled && !isFreshBrain(db);
188
+ }
189
+ /** Chunk-size cache, keyed per endpoint exactly like the handler's. */
190
+ const chunkSizeCache = new Map();
191
+ async function resolveDrainChunkSize(opts) {
192
+ const cfg = opts.llmConfig;
193
+ const detect = opts.detectChunkSize ?? distiller_js_1.detectChunkSize;
194
+ const cacheKey = `${cfg.provider}/${cfg.model}@${cfg.baseUrl}`;
195
+ if (!chunkSizeCache.has(cacheKey)) {
196
+ chunkSizeCache.set(cacheKey, await detect(cfg.provider, cfg.model, cfg.baseUrl, cfg.numCtx));
197
+ }
198
+ return chunkSizeCache.get(cacheKey);
199
+ }
200
+ /** Client key for round-robin: machine + agent (the spec's fairness unit). */
201
+ function clientKey(row) {
202
+ return `${row.source_machine ?? ""}|${row.source_agent ?? ""}`;
203
+ }
204
+ const sleepMs = (ms) => new Promise((r) => setTimeout(r, ms));
205
+ /**
206
+ * Wait out a busy yield signal between items. Returns true to proceed with
207
+ * the item, false when the run deadline fired mid-wait (the caller stops the
208
+ * drain — never starts a fresh LLM call past the deadline). Fail-open by
209
+ * design: unset URL, absent probe, or an unreachable signal all proceed
210
+ * (unreachable logs ONE warn per run). The per-item wait is bounded by
211
+ * DRAIN_YIELD_WAIT_CAP_MS so a stuck "busy" answer cannot pin the run.
212
+ */
213
+ async function waitForYield(opts, warned) {
214
+ if (!opts.yieldUrl || !opts.probeBusy)
215
+ return true;
216
+ const sleep = opts.sleep ?? sleepMs;
217
+ let waited = 0;
218
+ for (;;) {
219
+ let status;
220
+ try {
221
+ status = await opts.probeBusy(opts.yieldUrl);
222
+ }
223
+ catch {
224
+ status = "unreachable"; // a throwing probe is the same fail-open class
225
+ }
226
+ if (status === "busy") {
227
+ if (opts.deadline?.expired())
228
+ return false;
229
+ if (waited >= calibration_js_1.DRAIN_YIELD_WAIT_CAP_MS) {
230
+ console.warn(`[hicortex] drain: yield signal busy for ${Math.round(waited / 60_000)} min ` +
231
+ `(cap ${Math.round(calibration_js_1.DRAIN_YIELD_WAIT_CAP_MS / 60_000)} min) — proceeding fail-open`);
232
+ return true;
233
+ }
234
+ await sleep(calibration_js_1.DRAIN_YIELD_POLL_MS);
235
+ waited += calibration_js_1.DRAIN_YIELD_POLL_MS;
236
+ continue;
237
+ }
238
+ if (status === "unreachable" && !warned.unreachable) {
239
+ warned.unreachable = true; // ONE warn per run, not per item
240
+ console.warn(`[hicortex] drain: yield signal at ${opts.yieldUrl} unreachable — proceeding (fail-open)`);
241
+ }
242
+ return true;
243
+ }
244
+ }
245
+ /**
246
+ * Resolve the memories' `created_at` — TOTAL (never throws) and computed
247
+ * BEFORE distillation starts. An unparseable `session_date` (enqueue stores
248
+ * any non-empty string verbatim) made `new Date(...).toISOString()` throw
249
+ * AFTER the LLM calls had already run: the catch mislabeled it endpoint_down,
250
+ * the row stayed queued, and every scheduled run stopped at it again — a
251
+ * poison row (#530 follow-up). Unparseable → ONE warn naming the session and
252
+ * the bad value, then the row's delivery timestamp (`arrived_at`, always a
253
+ * full ISO written by enqueueDistill). The sync path (mcp-server.ts) has the
254
+ * same latent throw — pre-existing parity; this fix is drain-only by review
255
+ * scope.
256
+ */
257
+ function resolveDrainCreatedAt(item) {
258
+ const stamp = item.session_date ?? new Date().toISOString().slice(0, 10);
259
+ const t = new Date(stamp).getTime();
260
+ if (Number.isFinite(t))
261
+ return new Date(stamp).toISOString();
262
+ console.warn(`[hicortex] drain: unparseable session_date ${JSON.stringify(item.session_date)} on ` +
263
+ `session ${item.session_id ?? "(no session id)"} — using the delivery timestamp instead`);
264
+ return item.arrived_at;
265
+ }
266
+ /**
267
+ * Drain the distill inbox. See the module doc for the ordering + checkpoint
268
+ * contracts. Pure w.r.t. its injected dependencies — no HTTP, no state.json
269
+ * reads (the nightly records the reported usage against the monthly meter).
270
+ */
271
+ async function drainDistillQueue(db, opts) {
272
+ const report = {
273
+ outcome: "empty",
274
+ processed: 0,
275
+ duplicates: 0,
276
+ memories: 0,
277
+ remaining: 0,
278
+ usage: { prompt: 0, completion: 0, total: 0 },
279
+ };
280
+ const rows = db
281
+ .prepare("SELECT * FROM distill_queue ORDER BY rowid ASC")
282
+ .all();
283
+ if (rows.length === 0)
284
+ return report; // the healthy steady state: depth 0
285
+ // No LLM configured: delivery still queued (queue mode needs no LLM), but
286
+ // distillation cannot run. Report + defer — items stay queued; the
287
+ // consolidation block reports its own no_llm right after.
288
+ if (!opts.llm || !opts.llmConfig) {
289
+ console.warn(`[hicortex] Distill inbox holds ${rows.length} item(s) but no LLM is configured — ` +
290
+ `deferring (run npx @gamaze/hicortex init)`);
291
+ report.outcome = "no_llm";
292
+ report.remaining = rows.length;
293
+ return report;
294
+ }
295
+ // Fair order: per-client queues in first-arrival order, each preserving
296
+ // rowid order (= arrival order = per-session ascending), cycled strictly —
297
+ // client A's head, client B's head, A's next, B's next, ...
298
+ const byClient = new Map();
299
+ for (const row of rows) {
300
+ const key = clientKey(row);
301
+ const list = byClient.get(key);
302
+ if (list)
303
+ list.push(row);
304
+ else
305
+ byClient.set(key, [row]);
306
+ }
307
+ const queues = [...byClient.values()];
308
+ console.log(`[hicortex] Distill inbox: ${rows.length} item(s) from ${queues.length} client(s) — draining`);
309
+ const chunkSize = await resolveDrainChunkSize(opts);
310
+ const warned = { unreachable: false };
311
+ let stopped;
312
+ outer: while (queues.some((q) => q.length > 0)) {
313
+ for (const q of queues) {
314
+ if (q.length === 0)
315
+ continue;
316
+ const item = q.shift();
317
+ // #405: the ONE run deadline, BETWEEN items — a safe boundary by
318
+ // construction (each item commits atomically, so a stop here leaves
319
+ // processed items done and the rest queued, dup-over-loss on retry).
320
+ if (opts.deadline?.hit("drain")) {
321
+ stopped = "deferred";
322
+ break outer;
323
+ }
324
+ // #5 token gate, BETWEEN items (and here, before the very first call) —
325
+ // a refusal leaves the rest queued for the next period; consolidation
326
+ // still runs (its own budget accounting is separate).
327
+ if (opts.budgetExceeded?.()) {
328
+ console.warn(`[hicortex] drain: token budget exceeded — ${rows.length - report.processed} item(s) stay queued`);
329
+ stopped = "budget";
330
+ break outer;
331
+ }
332
+ // #529 interactive-yield: wait out a busy signal before the next item.
333
+ if (!(await waitForYield(opts, warned))) {
334
+ opts.deadline?.hit("drain"); // record the stage deferral (idempotent)
335
+ stopped = "deferred";
336
+ break outer;
337
+ }
338
+ // Distill-time dedup re-check (the delivery-time checks ran before
339
+ // enqueue, but the world moved since): a now-duplicate segment skips
340
+ // cleanly — its inbox row is deleted and nothing is re-distilled.
341
+ if (item.session_id) {
342
+ const dup = item.segment_id !== ""
343
+ ? (0, dedup_js_1.countExistingSegment)(db, item.session_id, item.segment_id) > 0
344
+ : (0, dedup_js_1.countExistingSession)(db, item.session_id) > 0;
345
+ if (dup) {
346
+ db.transaction(() => deleteQueuedRow(db, item.id))();
347
+ report.processed++;
348
+ report.duplicates++;
349
+ continue;
350
+ }
351
+ }
352
+ // Distill → embed → ONE transaction (inserts + row delete). Per-item
353
+ // usage accrues even on throw (some chunks' LLM calls already happened)
354
+ // — the caller records it against the monthly meter, same as the sync
355
+ // path's finally.
356
+ const usage = { prompt: 0, completion: 0, total: 0 };
357
+ const label = item.segment_id !== ""
358
+ ? item.segment_id
359
+ : item.session_id ?? undefined;
360
+ // BEFORE distillation — see resolveDrainCreatedAt: a throwing date
361
+ // computation after the LLM calls is the poison row.
362
+ const createdAt = resolveDrainCreatedAt(item);
363
+ try {
364
+ // Cast: DrainLlm is the structural surface distillSession uses.
365
+ const llm = opts.llm;
366
+ const entries = await (0, distiller_js_1.distillSession)(llm, item.text, item.project ?? "unknown", item.session_date ?? new Date().toISOString().slice(0, 10), chunkSize, [], (u) => {
367
+ usage.prompt += u.prompt_tokens ?? 0;
368
+ usage.completion += u.completion_tokens ?? 0;
369
+ usage.total += u.total_tokens ?? 0;
370
+ }, label);
371
+ // Phase 1 — embed every entry up front; ANY failure throws BEFORE the
372
+ // transaction, so nothing is stored and the row stays queued.
373
+ const sourcePrefix = item.session_id
374
+ ? `${item.session_id}${item.segment_id !== "" ? `#${item.segment_id}` : ""}`
375
+ : undefined;
376
+ const toStore = [];
377
+ for (let i = 0; i < entries.length; i++) {
378
+ const entry = entries[i];
379
+ if (typeof entry !== "object" || !entry.content || !entry.content.trim())
380
+ continue;
381
+ toStore.push({
382
+ content: entry.content,
383
+ memoryType: entry.memoryType,
384
+ embedding: await opts.embed(entry.content),
385
+ i,
386
+ });
387
+ }
388
+ // Phase 2 — inserts + the inbox-row delete in ONE transaction: the
389
+ // item is "done" only when both land (per-item checkpointing; a crash
390
+ // between items never sees a half-done item).
391
+ const commit = db.transaction(() => {
392
+ for (const { content, memoryType, embedding, i } of toStore) {
393
+ storage.insertMemory(db, content, embedding, {
394
+ // Same normalizations the sync handler applies at insert time —
395
+ // the row kept the wire fields verbatim.
396
+ sourceAgent: item.source_agent ?? "unknown",
397
+ sourceAgentId: item.source_agent_id,
398
+ sourceDomain: item.source_domain,
399
+ sourceMachine: storage.sanitizeSourceMachine(item.source_machine),
400
+ sourceSession: sourcePrefix ? `${sourcePrefix}#${i}` : undefined,
401
+ project: item.project ?? undefined,
402
+ memoryType,
403
+ privacy: item.privacy,
404
+ createdAt,
405
+ });
406
+ }
407
+ deleteQueuedRow(db, item.id);
408
+ });
409
+ commit();
410
+ report.processed++;
411
+ report.memories += toStore.length;
412
+ report.usage.prompt += usage.prompt;
413
+ report.usage.completion += usage.completion;
414
+ report.usage.total += usage.total;
415
+ console.log(`[hicortex] Drained ${label ?? "segment"}: ${toStore.length} memories`);
416
+ }
417
+ catch (err) {
418
+ // Endpoint failure (ladder + breaker exhausted inside LlmClient) or
419
+ // an embed/insert failure: stop the drain CLEANLY — never retry this
420
+ // item in-run, never a tight loop. Processed items stay done; this
421
+ // and the remaining items stay queued for the next scheduled run.
422
+ report.usage.prompt += usage.prompt;
423
+ report.usage.completion += usage.completion;
424
+ report.usage.total += usage.total;
425
+ console.error(`[hicortex] drain: distilling ${label ?? "segment"} failed — ` +
426
+ `${err instanceof Error ? (err.stack ?? err.message) : String(err)}. ` +
427
+ `Stopping the drain; remaining items stay queued.`);
428
+ stopped = "endpoint_down";
429
+ break outer;
430
+ }
431
+ }
432
+ }
433
+ report.outcome = stopped ?? "completed";
434
+ report.remaining = queueDepth(db);
435
+ if (report.outcome === "completed") {
436
+ console.log(`[hicortex] Distill inbox drained: ${report.processed} item(s), ` +
437
+ `${report.memories} memories, ${report.duplicates} duplicate skip(s)`);
438
+ }
439
+ return report;
440
+ }
package/dist/health.d.ts CHANGED
@@ -11,7 +11,7 @@
11
11
  * bearer-token auth middleware (localhost bypasses as usual) so an
12
12
  * operator running `hicortex status` on the server box, or a co-located
13
13
  * nightly preflight, still gets them — but a remote/anonymous caller does
14
- * not. Spec: `specs/2026-07-27-hosted-service.md` §6, Phase 0a item 5a/b.
14
+ * not. Phase 0a item 5a/b.
15
15
  *
16
16
  * 2. REST `res.status(500).json({error: err.message})` sites were echoing
17
17
  * internal detail (LLM upstream URLs, hostnames, stack frames) to the HTTP
@@ -34,6 +34,10 @@ export declare function publicHealthResponse(): {
34
34
  * standard auth middleware (localhost bypasses auth, so co-located tooling
35
35
  * — `hicortex status`, nightly preflight, `init` detect — sees it without a
36
36
  * token; a remote caller needs the bearer token).
37
+ *
38
+ * `distillQueue` (#529) is the inbox visibility signal: depth + oldest-item
39
+ * age in hours (null = empty inbox). Optional so the helper stays usable
40
+ * without a DB handle (tests) — the route always passes it.
37
41
  */
38
42
  export declare function detailedHealthResponse(opts: {
39
43
  memories: number;
@@ -41,6 +45,10 @@ export declare function detailedHealthResponse(opts: {
41
45
  dbSizeBytes: number;
42
46
  version: string;
43
47
  llmLabel: string;
48
+ distillQueue?: {
49
+ depth: number;
50
+ oldest_age_hours: number | null;
51
+ };
44
52
  }): {
45
53
  status: "ok";
46
54
  version: string;
@@ -48,6 +56,10 @@ export declare function detailedHealthResponse(opts: {
48
56
  links: number;
49
57
  db_size_kb: number;
50
58
  llm: string;
59
+ distill_queue?: {
60
+ depth: number;
61
+ oldest_age_hours: number | null;
62
+ };
51
63
  };
52
64
  /**
53
65
  * Log the full internal error detail server-side and return a generic
package/dist/health.js CHANGED
@@ -12,7 +12,7 @@
12
12
  * bearer-token auth middleware (localhost bypasses as usual) so an
13
13
  * operator running `hicortex status` on the server box, or a co-located
14
14
  * nightly preflight, still gets them — but a remote/anonymous caller does
15
- * not. Spec: `specs/2026-07-27-hosted-service.md` §6, Phase 0a item 5a/b.
15
+ * not. Phase 0a item 5a/b.
16
16
  *
17
17
  * 2. REST `res.status(500).json({error: err.message})` sites were echoing
18
18
  * internal detail (LLM upstream URLs, hostnames, stack frames) to the HTTP
@@ -39,6 +39,10 @@ function publicHealthResponse() {
39
39
  * standard auth middleware (localhost bypasses auth, so co-located tooling
40
40
  * — `hicortex status`, nightly preflight, `init` detect — sees it without a
41
41
  * token; a remote caller needs the bearer token).
42
+ *
43
+ * `distillQueue` (#529) is the inbox visibility signal: depth + oldest-item
44
+ * age in hours (null = empty inbox). Optional so the helper stays usable
45
+ * without a DB handle (tests) — the route always passes it.
42
46
  */
43
47
  function detailedHealthResponse(opts) {
44
48
  return {
@@ -48,6 +52,7 @@ function detailedHealthResponse(opts) {
48
52
  links: opts.links,
49
53
  db_size_kb: Math.round(opts.dbSizeBytes / 1024),
50
54
  llm: opts.llmLabel,
55
+ ...(opts.distillQueue !== undefined ? { distill_queue: opts.distillQueue } : {}),
51
56
  };
52
57
  }
53
58
  /**
@@ -13,7 +13,7 @@
13
13
  * no bypass; a tenant dir provisioned from a restored tar could otherwise
14
14
  * ship with the bypass active).
15
15
  *
16
- * Spec: specs/2026-07-27-hosted-service.md §1-§2 (Phase 0B, issue #271).
16
+ * Phase 0B, issue #271.
17
17
  */
18
18
  export interface HostedBootInput {
19
19
  /** Resolved hostedMode flag from config (absent/false → self-hosted). */
@@ -35,7 +35,7 @@ exports.LOCALHOST_BYPASS_MARKER = ".allow-localhost-bypass";
35
35
  exports.LOCALHOST_BYPASS_MARKER_CONTENT = "# Written by `hicortex init` (self-hosted). Opt-in to the localhost auth\n" +
36
36
  "# bypass. DELETE this file to require the bearer token on localhost too\n" +
37
37
  "# (fail-closed). Hosted-mode (hostedMode:true) refuses to start with this\n" +
38
- "# marker present — see specs/2026-07-27-hosted-service.md §2.\n";
38
+ "# See the hosted-mode boot assertions for the refusal logic.\n";
39
39
  /**
40
40
  * Resolve the marker file path for a given home dir. Defaults to the canonical
41
41
  * Hicortex home (honors HICORTEX_HOME), so callers in tests can point the env
@@ -87,6 +87,7 @@ const seed_lesson_js_1 = require("./seed-lesson.js");
87
87
  const learnings_identity_js_1 = require("./learnings-identity.js");
88
88
  const distiller_js_1 = require("./distiller.js");
89
89
  const dedup_js_1 = require("./dedup.js");
90
+ const distill_queue_js_1 = require("./distill-queue.js");
90
91
  const reconsolidation_js_1 = require("./reconsolidation.js");
91
92
  const redact_js_1 = require("./redact.js");
92
93
  const capture_health_js_1 = require("./capture-health.js");
@@ -659,6 +660,11 @@ async function startServer(options = {}) {
659
660
  // env (provider-set, tenant-immutable) which takes precedence. Initialised here
660
661
  // (after stateDir + savedConfig are known) so the warn-dedup can seed from state.
661
662
  (0, token_budget_js_1.initTokenBudget)(stateDir, savedConfig?.llmTokensPerMonth);
663
+ // #529: the distill-inbox kill switch. Default ON (queue mode) — read once
664
+ // at boot via readStrictBoolean (the llmSingleFlight kill-switch pattern,
665
+ // llm.ts): never coerced ("false" the STRING is invalid + ignored), applied
666
+ // on restart like every other boot-resolved knob.
667
+ const distillQueueEnabled = (0, config_read_js_1.readStrictBoolean)(savedConfig ?? {}, "distillQueue") !== false;
662
668
  // #7: request body-size limit. Env (HICORTEX_DISTILL_BODY_LIMIT_MB) wins;
663
669
  // else the config key; else 5 MB hosted / 25 MB self-hosted (the prior
664
670
  // fixed value → no regression). Guards the OOM vector (the body is fully
@@ -892,6 +898,12 @@ async function startServer(options = {}) {
892
898
  dbSizeBytes: s.db_size_bytes,
893
899
  version: VERSION,
894
900
  llmLabel: llmConfig ? `${llmConfig.provider}/${llmConfig.model}` : "not configured",
901
+ // #529: inbox visibility — queue depth + oldest-item age (null when
902
+ // empty). The 24 h warning threshold itself is applied at render
903
+ // (status/dashboard), not here.
904
+ distillQueue: db
905
+ ? (0, distill_queue_js_1.readQueueStats)(db)
906
+ : { depth: 0, oldest_age_hours: null },
895
907
  }));
896
908
  });
897
909
  // REST /learnings (canonical, #264) + /lessons (alias) — return lessons +
@@ -1303,10 +1315,24 @@ async function startServer(options = {}) {
1303
1315
  res.status(200).json({ skipped: true, paused: true, machine: pause.machine, harness: pause.harness });
1304
1316
  return;
1305
1317
  }
1318
+ // #529 queue mode: the ONE branch condition into the sync flow. Queue
1319
+ // mode = the kill switch is ON (default) AND this is not a fresh brain
1320
+ // (empty memories + empty inbox distills synchronously — the bootstrap
1321
+ // carve-out keeps first-ever memories immediate). Computed once, used at
1322
+ // the no-LLM guard below (queue mode needs no LLM to accept a delivery —
1323
+ // the drain defers when none is configured) and at the queue branch after
1324
+ // the dedup prechecks. `distillQueue: false` lands here as false ALWAYS →
1325
+ // the sync flow below runs byte-identically to pre-#529.
1326
+ const queueMode = (0, distill_queue_js_1.resolveQueueMode)(db, distillQueueEnabled);
1306
1327
  if (!llm || !llmConfig) {
1307
- (0, capture_health_js_1.recordDistillActivity)(db, capEntry(0, "held"));
1308
- res.status(503).json({ error: "No LLM configured — run npx @gamaze/hicortex init. Session will be retried." });
1309
- return;
1328
+ if (!queueMode) {
1329
+ (0, capture_health_js_1.recordDistillActivity)(db, capEntry(0, "held"));
1330
+ res.status(503).json({ error: "No LLM configured — run npx @gamaze/hicortex init. Session will be retried." });
1331
+ return;
1332
+ }
1333
+ // Queue mode with no LLM: fall through — the queue branch below stores
1334
+ // the delivery without any LLM call and the nightly drain defers until
1335
+ // an LLM is configured.
1310
1336
  }
1311
1337
  // Resolve the conversation text from either the pre-denoised string or raw messages array.
1312
1338
  // `fromTextBranch` is captured once so the redaction gate below uses the SAME
@@ -1354,8 +1380,14 @@ async function startServer(options = {}) {
1354
1380
  // marker survives there after the `memories` row is deleted, so a
1355
1381
  // `hicortex dedup --apply` merge can't be undone by a retried/recaptured
1356
1382
  // segment silently re-ingesting the same content.
1383
+ // #529: ALSO consults the distill inbox — a QUEUED-but-undistilled
1384
+ // segment re-POST answers 200 skipped (same client contract as a stored
1385
+ // duplicate: the cursor advances, the segment is never queued twice). In
1386
+ // sync mode the inbox is empty by construction, so behavior there is
1387
+ // unchanged.
1357
1388
  if (session_id && segment_id) {
1358
- const existingCount = (0, dedup_js_1.countExistingSegment)(db, session_id, segment_id);
1389
+ const existingCount = (0, dedup_js_1.countExistingSegment)(db, session_id, segment_id) +
1390
+ (0, distill_queue_js_1.countQueuedSegment)(db, session_id, segment_id);
1359
1391
  if (existingCount > 0) {
1360
1392
  (0, capture_health_js_1.recordDistillActivity)(db, capEntry(conversationText.length, "skipped"));
1361
1393
  res.status(200).json({ skipped: true, existing_count: existingCount });
@@ -1365,15 +1397,64 @@ async function startServer(options = {}) {
1365
1397
  // Session-level dedup: when session_id is present and this is a whole-session
1366
1398
  // POST (no segment_id — legacy ≤0.13.1 clients), skip if any chunk of this
1367
1399
  // session is already stored (memories OR dedup_log — see above).
1400
+ // #529: the inbox is mirrored here too (any queued row of the session).
1368
1401
  // Unchanged: legacy clients keep exact behaviour.
1369
1402
  if (session_id && !segment_id) {
1370
- const existingCount = (0, dedup_js_1.countExistingSession)(db, session_id);
1403
+ const existingCount = (0, dedup_js_1.countExistingSession)(db, session_id) +
1404
+ (0, distill_queue_js_1.countQueuedSession)(db, session_id);
1371
1405
  if (existingCount > 0) {
1372
1406
  (0, capture_health_js_1.recordDistillActivity)(db, capEntry(conversationText.length, "skipped"));
1373
1407
  res.status(200).json({ skipped: true, existing_count: existingCount });
1374
1408
  return;
1375
1409
  }
1376
1410
  }
1411
+ // #529 queue branch: store the redacted segment durably and answer the
1412
+ // SAME 201 shape with zeroed counts + `queued: true` — the client
1413
+ // contract (cursor advances on 201) is untouched; capture.ts only swaps
1414
+ // its log line when `queued` is present. NO LLM call: the probe gate,
1415
+ // chunk-size detection, and the token gate all live in the nightly's
1416
+ // drain now. Redaction already happened above, so the inbox never holds
1417
+ // unredacted text (owner ruling 05.10.2026: same protection as the DB).
1418
+ // The enqueue's UNIQUE(session_id, segment_id) insert is idempotent — a
1419
+ // duplicate (only reachable via a racing identical POST; the prechecks
1420
+ // above already answered the sequential re-POST) returns the same
1421
+ // 200-skip contract.
1422
+ if (queueMode) {
1423
+ const { duplicate } = (0, distill_queue_js_1.enqueueDistill)(db, {
1424
+ text: conversationText,
1425
+ source_agent,
1426
+ source_agent_id,
1427
+ source_domain,
1428
+ source_machine,
1429
+ project,
1430
+ session_id,
1431
+ segment_id,
1432
+ session_date,
1433
+ privacy,
1434
+ });
1435
+ (0, capture_health_js_1.recordDistillActivity)(db, capEntry(conversationText.length, duplicate ? "skipped" : "ok"));
1436
+ if (duplicate) {
1437
+ res.status(200).json({ skipped: true, existing_count: 0 });
1438
+ return;
1439
+ }
1440
+ res.status(201).json({
1441
+ ids: [],
1442
+ distilled: 0,
1443
+ dropped: [],
1444
+ // #287 shape kept (zeros — no LLM ran) so pre-#529 clients parse the
1445
+ // body unchanged; `queued: true` is the new signal.
1446
+ usage: { prompt: 0, completion: 0, total: 0 },
1447
+ queued: true,
1448
+ });
1449
+ return;
1450
+ }
1451
+ // #529: from here on queueMode is false, so the no-LLM guard above has
1452
+ // already answered 503 in every path that reaches this line. Restated so
1453
+ // the sync flow keeps its non-null llm/llmConfig typing unchanged.
1454
+ if (!llm || !llmConfig) {
1455
+ res.status(503).json({ error: "No LLM configured — run npx @gamaze/hicortex init. Session will be retried." });
1456
+ return;
1457
+ }
1377
1458
  // #337: readiness gate — cached minimal generation probe BEFORE
1378
1459
  // detectChunkSize + distillSession, so a dead endpoint never even pays the
1379
1460
  // chunk-size probe. Placed AFTER the dedup short-circuits (a duplicate