@gamaze/hicortex 0.20.6 → 0.20.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +10 -41
  2. package/dist/calibration.d.ts +174 -0
  3. package/dist/calibration.js +231 -0
  4. package/dist/capture.d.ts +15 -3
  5. package/dist/capture.js +10 -1
  6. package/dist/classify-domains.d.ts +6 -0
  7. package/dist/classify-domains.js +7 -1
  8. package/dist/cli.js +2 -3
  9. package/dist/config-read.d.ts +1 -1
  10. package/dist/config-read.js +96 -9
  11. package/dist/consolidate.d.ts +79 -68
  12. package/dist/consolidate.js +218 -174
  13. package/dist/dashboard.d.ts +4 -3
  14. package/dist/dedup.d.ts +34 -26
  15. package/dist/dedup.js +91 -57
  16. package/dist/distiller.js +1 -1
  17. package/dist/domain-classify.d.ts +7 -6
  18. package/dist/domain-classify.js +12 -10
  19. package/dist/eval/decay-eval.d.ts +3 -3
  20. package/dist/eval/decay-eval.js +4 -4
  21. package/dist/eval/planted-eval.d.ts +26 -0
  22. package/dist/eval/planted-eval.js +97 -0
  23. package/dist/eval/planted-fixtures.d.ts +107 -0
  24. package/dist/eval/planted-fixtures.js +283 -0
  25. package/dist/eval/planted-harness.d.ts +176 -0
  26. package/dist/eval/planted-harness.js +649 -0
  27. package/dist/index.js +4 -3
  28. package/dist/init.d.ts +9 -3
  29. package/dist/init.js +52 -9
  30. package/dist/llm.d.ts +43 -58
  31. package/dist/llm.js +87 -101
  32. package/dist/mcp-server.js +29 -29
  33. package/dist/nightly.js +105 -103
  34. package/dist/nofit.d.ts +4 -11
  35. package/dist/nofit.js +6 -23
  36. package/dist/recall-index.d.ts +30 -28
  37. package/dist/recall-index.js +21 -18
  38. package/dist/recall-registry.d.ts +2 -1
  39. package/dist/recall-registry.js +35 -1
  40. package/dist/reconsolidation.d.ts +124 -72
  41. package/dist/reconsolidation.js +390 -169
  42. package/dist/relink.js +3 -4
  43. package/dist/retrieval.d.ts +68 -35
  44. package/dist/retrieval.js +292 -104
  45. package/dist/run-deadline.d.ts +62 -0
  46. package/dist/run-deadline.js +73 -0
  47. package/dist/schema-prototypes.d.ts +3 -3
  48. package/dist/schema-prototypes.js +3 -3
  49. package/dist/state.d.ts +2 -3
  50. package/dist/storage.d.ts +16 -16
  51. package/dist/storage.js +62 -24
  52. package/dist/telemetry.d.ts +8 -7
  53. package/dist/token-budget.js +3 -4
  54. package/dist/type-classify.js +4 -4
  55. package/dist/types.d.ts +95 -155
  56. package/domains.example.json +4 -5
  57. package/hermes-plugin/hicortex/README.md +2 -2
  58. package/openclaw.plugin.json +1 -1
  59. package/package.json +2 -1
  60. package/pi-extension/hicortex/README.md +1 -1
  61. package/server.json +3 -3
package/dist/init.d.ts CHANGED
@@ -69,7 +69,7 @@ export declare function persistLlmConfig(configPath?: string): Promise<void>;
69
69
  * and the writer then OVERWROTE the file — `persistAuthToken` minted a fresh
70
70
  * token (fleet-wide 401), `scaffoldDefaultDomains` re-seeded the generic
71
71
  * vocabulary over the owner list, etc. `authToken` / `licenseKey` /
72
- * `llmApiKey` / `domains` / `weakPrimaryFloor` / `identityClients` (was
72
+ * `llmApiKey` / `domains` / `identityClients` (was
73
73
  * `contextClients`) all gone.
74
74
  * The early-return guards (existing-key checks) did NOT save them: those only
75
75
  * fire on a VALID parse that reads the key, not on a corrupted file.
@@ -174,7 +174,7 @@ export declare function ensureAgentId(config: Record<string, unknown>): {
174
174
  * + unconditional save WIPES config.json when the file exists but is
175
175
  * unparseable (a hand-edit syntax slip) — the catch swallows the parse error,
176
176
  * {} is seeded, and the save overwrites the file with just {"agentId": ...},
177
- * destroying authToken / licenseKey / domains / weakPrimaryFloor, then
177
+ * destroying authToken / licenseKey / domains, then
178
178
  * cascades into scaffoldDefaultDomains re-seeding the generic vocabulary.
179
179
  * This wrapper refuses that path:
180
180
  * - ENOENT (file genuinely absent) → seed {} is correct (new install).
@@ -348,7 +348,8 @@ export declare function runInit(options?: {
348
348
  }): Promise<void>;
349
349
  /**
350
350
  * Resolve the timer-jitter spread (seconds) from config. 0 = disabled. Uses
351
- * readNonNegativeConfig (0 is a valid "off", mirroring ollamaFlushEvery).
351
+ * readNonNegativeConfig (0 is a valid "off", like memorySoftCap's eviction
352
+ * opt-out).
352
353
  */
353
354
  export declare function resolveTimerJitterSeconds(configDir?: string): number;
354
355
  /**
@@ -403,3 +404,8 @@ export declare function formatLaunchdIntervals(hours: number[], minuteOffset?: n
403
404
  * applies to every OnCalendar entry AND to OnUnitActiveSec (systemd semantics).
404
405
  */
405
406
  export declare function formatSystemdTimerBody(isInterval: boolean, intervalSec: number, hours: number[], jitterSec: number): string;
407
+ /**
408
+ * Install the CONSOLIDATION timer (full `nightly`). Reuses the existing
409
+ * `hicortex-nightly` unit name (repurposed from the pre-0.17 single full-nightly).
410
+ */
411
+ export declare function installConsolidationTimer(hours: number[]): void;
package/dist/init.js CHANGED
@@ -52,6 +52,7 @@ exports.resolveNightlyHour = resolveNightlyHour;
52
52
  exports.formatOnCalendarLines = formatOnCalendarLines;
53
53
  exports.formatLaunchdIntervals = formatLaunchdIntervals;
54
54
  exports.formatSystemdTimerBody = formatSystemdTimerBody;
55
+ exports.installConsolidationTimer = installConsolidationTimer;
55
56
  const paths_js_1 = require("./paths.js");
56
57
  const localhost_bypass_js_1 = require("./localhost-bypass.js");
57
58
  const telemetry_js_1 = require("./telemetry.js");
@@ -64,6 +65,7 @@ const node_crypto_1 = require("node:crypto");
64
65
  const claude_md_js_1 = require("./claude-md.js");
65
66
  const claude_desktop_js_1 = require("./claude-desktop.js");
66
67
  const config_read_js_1 = require("./config-read.js");
68
+ const run_deadline_js_1 = require("./run-deadline.js");
67
69
  const identity_store_js_1 = require("./identity-store.js");
68
70
  const HICORTEX_HOME = (0, paths_js_1.hicortexHome)();
69
71
  /** This package's version, for the install lifecycle ping (0.15.2). */
@@ -994,7 +996,7 @@ function saveConfig(configPath, config) {
994
996
  * and the writer then OVERWROTE the file — `persistAuthToken` minted a fresh
995
997
  * token (fleet-wide 401), `scaffoldDefaultDomains` re-seeded the generic
996
998
  * vocabulary over the owner list, etc. `authToken` / `licenseKey` /
997
- * `llmApiKey` / `domains` / `weakPrimaryFloor` / `identityClients` (was
999
+ * `llmApiKey` / `domains` / `identityClients` (was
998
1000
  * `contextClients`) all gone.
999
1001
  * The early-return guards (existing-key checks) did NOT save them: those only
1000
1002
  * fire on a VALID parse that reads the key, not on a corrupted file.
@@ -1099,7 +1101,7 @@ function quarantineMalformedConfig(configPath) {
1099
1101
  console.log(` Keys found in the old file: ${keys.join(", ")}`);
1100
1102
  }
1101
1103
  console.log(` ACTION REQUIRED: copy any of licenseKey / llmApiKey /`);
1102
- console.log(` domains / weakPrimaryFloor back from the backup by hand.`);
1104
+ console.log(` domains back from the backup by hand.`);
1103
1105
  console.log(` A NEW authToken will be generated — every thin client pointing at this`);
1104
1106
  console.log(` server must be updated, or their recall will 401 (silently, fail-soft).`);
1105
1107
  return { quarantined: true, backupPath, keys };
@@ -1167,7 +1169,7 @@ function ensureAgentId(config) {
1167
1169
  * + unconditional save WIPES config.json when the file exists but is
1168
1170
  * unparseable (a hand-edit syntax slip) — the catch swallows the parse error,
1169
1171
  * {} is seeded, and the save overwrites the file with just {"agentId": ...},
1170
- * destroying authToken / licenseKey / domains / weakPrimaryFloor, then
1172
+ * destroying authToken / licenseKey / domains, then
1171
1173
  * cascades into scaffoldDefaultDomains re-seeding the generic vocabulary.
1172
1174
  * This wrapper refuses that path:
1173
1175
  * - ENOENT (file genuinely absent) → seed {} is correct (new install).
@@ -2181,7 +2183,8 @@ const DEFAULT_CONSOLIDATION_HOURS = [10, 22];
2181
2183
  const DEFAULT_TIMER_JITTER_SEC = 3600;
2182
2184
  /**
2183
2185
  * Resolve the timer-jitter spread (seconds) from config. 0 = disabled. Uses
2184
- * readNonNegativeConfig (0 is a valid "off", mirroring ollamaFlushEvery).
2186
+ * readNonNegativeConfig (0 is a valid "off", like memorySoftCap's eviction
2187
+ * opt-out).
2185
2188
  */
2186
2189
  function resolveTimerJitterSeconds(configDir = HICORTEX_HOME) {
2187
2190
  let config = {};
@@ -2471,6 +2474,12 @@ function installCaptureWatchdogTimer() {
2471
2474
  * `hicortex-nightly` unit name (repurposed from the pre-0.17 single full-nightly).
2472
2475
  */
2473
2476
  function installConsolidationTimer(hours) {
2477
+ // #405: DERIVED from the run deadline — nightlyTimeBudgetMinutes + 60 min
2478
+ // slack. The old fixed 360 was prose-coupled to the old call budget (raise
2479
+ // together!); now the unit backstop tracks the deadline it is backing.
2480
+ // Backstop only (NOT an operating limit) — the slack absorbs backup +
2481
+ // telemetry + a slow shutdown after the deadline stops the stages.
2482
+ const timeoutMin = (0, run_deadline_js_1.resolveNightlyTimeBudgetMinutes)(readInstallConfig()) + 60;
2474
2483
  writeScheduleUnit({
2475
2484
  unitBase: "hicortex-nightly",
2476
2485
  plistLabel: "com.gamaze.hicortex-nightly",
@@ -2478,12 +2487,46 @@ function installConsolidationTimer(hours) {
2478
2487
  timerDesc: "Hicortex Consolidation Timer",
2479
2488
  nightlyArgs: ["nightly"],
2480
2489
  hours,
2481
- // Backstop only (NOT an operating limit) — set well above the longest
2482
- // legitimate run so it catches a true hang, never a slow-but-progressing
2483
- // one. ~5000 LLM calls × ~1–3s/call ≈ 1.4–4.2h → 6h clears it with margin.
2484
- // Coupled to the consolidateMaxLlmCalls budget (#241): raise together.
2485
- timeoutMin: 360,
2490
+ // nightlyTimeBudgetMinutes (default 240) + 60 min slack — see above.
2491
+ timeoutMin,
2486
2492
  });
2493
+ // #405 shadow detection: a systemd drop-in can override the derived
2494
+ // TimeoutStartSec silently (a stale pre-#405 drop-in pinning a shorter
2495
+ // timeout would hard-kill legitimate deadline-bounded runs). Detect and
2496
+ // warn — removal is a deploy-time ops step, not an init action.
2497
+ warnSystemdDropInShadow("hicortex-nightly", timeoutMin);
2498
+ }
2499
+ /** Read the install-time config (~/.hicortex/config.json), null-tolerant. */
2500
+ function readInstallConfig() {
2501
+ try {
2502
+ return JSON.parse((0, node_fs_1.readFileSync)((0, node_path_1.join)(HICORTEX_HOME, "config.json"), "utf8"));
2503
+ }
2504
+ catch {
2505
+ return null;
2506
+ }
2507
+ }
2508
+ /**
2509
+ * Warn when `~/.config/systemd/user/<unit>.service.d/*.conf` drop-ins exist —
2510
+ * they can shadow the generated TimeoutStartSec (and anything else the unit
2511
+ * sets). Linux only; the macOS launchd path has no drop-in mechanism.
2512
+ */
2513
+ function warnSystemdDropInShadow(unitBase, derivedTimeoutMin) {
2514
+ if ((0, node_os_1.platform)() !== "linux")
2515
+ return;
2516
+ const dropInDir = (0, node_path_1.join)((0, node_os_1.homedir)(), ".config", "systemd", "user", `${unitBase}.service.d`);
2517
+ let entries = [];
2518
+ try {
2519
+ entries = (0, node_fs_1.readdirSync)(dropInDir).filter((f) => f.endsWith(".conf"));
2520
+ }
2521
+ catch {
2522
+ return; // no drop-in dir — the common case
2523
+ }
2524
+ if (entries.length === 0)
2525
+ return;
2526
+ console.warn(`[hicortex] systemd drop-ins present for ${unitBase}.service (${entries.join(", ")}) — ` +
2527
+ `they can SHADOW the generated unit settings (TimeoutStartSec is now derived: ` +
2528
+ `${derivedTimeoutMin}min = nightlyTimeBudgetMinutes + 60 slack). ` +
2529
+ `Remove stale drop-ins so the derived values take effect.`);
2487
2530
  }
2488
2531
  /**
2489
2532
  * Remove the consolidation timer + service (and the macOS plist). Used on
package/dist/llm.d.ts CHANGED
@@ -23,29 +23,23 @@ export interface LlmConfig {
23
23
  provider: string;
24
24
  /** Max output tokens for all phases (one model). Default 8192. */
25
25
  maxTokens?: number;
26
- /** Max output tokens for the CLASSIFY tier only — the short JSON-verdict
27
- * calls (correction/supersession verdicts, rewrite contracts, type + tag
28
- * classification). Default 1024. See HicortexConfig.classifyMaxTokens (#391). */
29
- classifyMaxTokens?: number;
30
26
  /** Toggle thinking on the openai-compat path for all phases. Absent = no kwarg sent.
31
27
  * LOCAL-endpoint only (ollama / mlx-lm gateway); see HicortexConfig.enableThinking. */
32
28
  enableThinking?: boolean;
33
- /** Context window for ollama (the one model, all phases). Default 8192. */
29
+ /** Context window for ollama (the one model, all phases). Default 8192.
30
+ * #408 diagnostic tier: resolved from HICORTEX_NUM_CTX (env) by
31
+ * applyTierTuningOverlay — never a config key. */
34
32
  numCtx?: number;
35
- /** Flush ollama memory every N ollama calls (0 = off). See HicortexConfig.ollamaFlushEvery. */
33
+ /** Flush ollama memory every N ollama calls (0 = off). #408 diagnostic
34
+ * tier: resolved from HICORTEX_OLLAMA_FLUSH_EVERY by the overlay. */
36
35
  ollamaFlushEvery?: number;
37
- /** Ms to wait after an ollama flush for the runner to release. */
36
+ /** Ms to wait after an ollama flush for the runner to release. #408
37
+ * diagnostic tier: resolved from HICORTEX_OLLAMA_FLUSH_WAIT_MS. */
38
38
  ollamaFlushWaitMs?: number;
39
39
  /** ONE per-call timeout ceiling for every phase (#337). Default 900000 — the
40
- * AbortSignal.timeout value passed by all four phase wrappers (the old
40
+ * AbortSignal.timeout value passed by the single complete() surface (the old
41
41
  * 600000 scoring special-case is gone). See HicortexConfig.llmTimeoutMs. */
42
42
  timeoutMs?: number;
43
- /** Consecutive ladder-exhausted total failures before the circuit breaker
44
- * opens (#337). Default 3; 0 disables. See HicortexConfig.llmBreakerThreshold. */
45
- breakerThreshold?: number;
46
- /** How long an OPEN breaker stays open before the next call becomes a trial
47
- * (#337). Default 600000. See HicortexConfig.llmBreakerCooldownMs. */
48
- breakerCooldownMs?: number;
49
43
  /** Timeout for the readiness probe's single generation attempt (#337).
50
44
  * Default 60000. See HicortexConfig.llmProbeTimeoutMs. */
51
45
  probeTimeoutMs?: number;
@@ -57,10 +51,6 @@ export interface LlmConfig {
57
51
  * single-user model servers (two concurrent large-context calls stall/OOM
58
52
  * the server and the machine under it). See HicortexConfig.llmSingleFlight. */
59
53
  singleFlight?: boolean;
60
- /** How long a queued call waits for the in-flight call before failing as
61
- * endpoint-down (#355). Default 900000 — the same ceiling as llmTimeoutMs.
62
- * See HicortexConfig.llmSingleFlightWaitMs. */
63
- singleFlightWaitMs?: number;
64
54
  }
65
55
  /**
66
56
  * Resolve LLM configuration from explicit config-file overrides or
@@ -85,20 +75,25 @@ export declare function resolveExplicitLlmConfig(overrides?: {
85
75
  */
86
76
  export declare const resolveLlmConfigForCC: typeof resolveExplicitLlmConfig;
87
77
  /**
88
- * Validate + copy the tuning keys (#220: maxTokens + enableThinking + numCtx +
89
- * ollama flush; #391: classifyMaxTokens) from the saved disk config onto a
90
- * runtime LlmConfig. Called by BOTH LlmConfig construction sites — the daemon
91
- * in mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
92
- * runs reflect + classify) — so every process honors the keys, and a future
93
- * site calling this inherits them by construction.
78
+ * Validate + copy the tuning keys (#220: maxTokens + enableThinking) from the
79
+ * saved disk config onto a runtime LlmConfig, and resolve the DIAGNOSTIC
80
+ * TIER (#408: numCtx + the ollama-flush family) from the calibration env
81
+ * resolvers. Called by BOTH LlmConfig construction sites — the daemon in
82
+ * mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
83
+ * runs reflect + classify) — so this is the ONE resolution point per process
84
+ * for every one of these values, and a future site calling this inherits
85
+ * them by construction.
94
86
  *
95
- * All keys are optional; absent = call-site defaults (maxTokens 8192,
96
- * classifyMaxTokens 1024, numCtx 8192, thinking kwarg omitted, flush off).
97
- * Wrong-typed values warn and are dropped (readPositiveConfig /
98
- * readStrictBoolean / readNonNegativeConfig) — notably a JSON slip
87
+ * Config keys are optional; absent = call-site defaults (maxTokens 8192,
88
+ * thinking kwarg omitted). Wrong-typed values warn and are dropped
89
+ * (readPositiveConfig / readStrictBoolean) — notably a JSON slip
99
90
  * `"enableThinking": "false"` (string) is rejected rather than coerced to
100
91
  * truthy thinking-on, which would silently invert the fix this key exists to
101
- * apply.
92
+ * apply. #405: classifyMaxTokens is deleted — maxTokens is the single output
93
+ * ceiling for every call (warned as ignored at the config-read boundary).
94
+ * #408: numCtx / ollamaFlushEvery / ollamaFlushWaitMs are NO LONGER config
95
+ * keys — the env tier always resolves them (env > calibration constant;
96
+ * invalid env warns + falls back), independent of `savedConfig`.
102
97
  */
103
98
  export declare function applyTierTuningOverlay(llmConfig: LlmConfig, savedConfig: Record<string, unknown> | null | undefined): void;
104
99
  /**
@@ -200,48 +195,38 @@ export declare class LlmClient {
200
195
  private resetBreaker;
201
196
  private handleRateLimit;
202
197
  /**
203
- * Fast-tier completion (importance scoring, simple tasks). One model serves
204
- * all phases (#231); numCtx + enableThinking are read from config directly
205
- * inside completeOnce's per-provider dispatch, not threaded here. The periodic
206
- * ollama flush stays (provider-gated) — it is a scoring-call-count cadence and
207
- * scoring is the highest-frequency call, so this is where the flush belongs.
198
+ * The ONE completion surface (#405). Every phase — distill, reflect,
199
+ * classify, scoring — calls this; maxTokens (default 8192) and timeoutMs
200
+ * (default 900 s) resolve from config INSIDE, so no call site can starve a
201
+ * phase with a per-site ceiling (#391's failure direction: hardcoded
202
+ * 16/64/256-token caps starved reasoning models whose internal thinking
203
+ * ate the whole output budget, leaving verdicts empty). numCtx +
204
+ * enableThinking are likewise read inside completeOnce's per-provider
205
+ * dispatch. The periodic ollama flush survives here (provider-gated) — it
206
+ * now counts ALL calls, not just the scoring tier.
208
207
  */
209
- completeFast(prompt: string, maxTokens?: number): Promise<LlmResult>;
210
- /**
211
- * Reflect-tier completion (nightly reflection). One model serves all phases
212
- * (#231) — this is a thin wrapper kept for call-site readability.
213
- */
214
- completeReflect(prompt: string, maxTokens?: number): Promise<LlmResult>;
215
- /**
216
- * Distillation-tier completion (session knowledge extraction). One model
217
- * serves all phases (#231) — thin wrapper kept for call-site readability.
218
- */
219
- completeDistill(prompt: string, maxTokens?: number): Promise<LlmResult>;
220
- /**
221
- * Classification-tier completion (verdicts + tag/type classification). One
222
- * model serves all phases (#231) — thin wrapper kept for call-site
223
- * readability, but the tier keeps its OWN output ceiling (#391).
224
- */
225
- completeClassify(prompt: string, maxTokens?: number): Promise<LlmResult>;
208
+ complete(prompt: string): Promise<LlmResult>;
226
209
  /**
227
210
  * Readiness probe (#337): ONE minimal generation request (max output 1
228
211
  * token) through the normal provider dispatch. Asks the question liveness
229
212
  * checks CANNOT: "can this endpoint GENERATE right now?" — the incident
230
213
  * gateway kept answering /v1/models for hours while every completion hung.
231
214
  * Single attempt: no retry ladder (a dead endpoint must cost one fast
232
- * failure, not a 3.5-min ladder), and it never accrues to the circuit
215
+ * failure, not a ladder), and it never accrues to the circuit
233
216
  * breaker (probing is diagnosis, not traffic). Catch-all → false; the
234
217
  * callers translate that into "endpoint_down" / a 503, never an exception.
218
+ * #405: owned by the DAEMON's /distill gate — the nightly no longer probes.
235
219
  */
236
220
  probe(timeoutMs?: number): Promise<boolean>;
237
- private complete;
221
+ private completeWithRetry;
238
222
  private completeOnce;
239
223
  /** The single-flight wait budget for a call with ceiling `timeoutMs`.
240
- * An explicit `llmSingleFlightWaitMs` wins; otherwise the default is
241
- * max(900 s, llmTimeoutMs) so a waiter never gives up before a legitimate
242
- * in-flight call's own (possibly raised) ceiling expires (2nd-review
243
- * finding 2 — a hardcoded 900 s made a raised-timeout install treat a
244
- * healthy-busy endpoint as down). */
224
+ * #405: always DERIVED — max(900 s, llmTimeoutMs) — so a waiter never
225
+ * gives up before a legitimate in-flight call's own (possibly raised)
226
+ * ceiling expires (2nd-review finding 2 — a hardcoded 900 s made a
227
+ * raised-timeout install treat a healthy-busy endpoint as down). The
228
+ * llmSingleFlightWaitMs knob is deleted (its only documented use was to
229
+ * restore this exact derivation). */
245
230
  private flightWaitMs;
246
231
  /** Resolve the flight guard for this call, or undefined when disabled.
247
232
  * `staleMs` is derived from THIS call's timeout ceiling (≥ the 30-min
package/dist/llm.js CHANGED
@@ -27,6 +27,9 @@ exports.claudeCliConfig = claudeCliConfig;
27
27
  exports.probeOllama = probeOllama;
28
28
  exports.isTotalFailure = isTotalFailure;
29
29
  const config_read_js_1 = require("./config-read.js");
30
+ // #408: the diagnostic tier (numCtx + ollama flush) resolves from the
31
+ // calibration env resolvers — the config keys are gone from the surface.
32
+ const calibration_js_1 = require("./calibration.js");
30
33
  // #355: single-flight guard + canonical home (where the per-endpoint flight
31
34
  // lock files live — the daemon and the nightly share the home, so the lock
32
35
  // serializes them across processes).
@@ -86,57 +89,49 @@ function resolveExplicitLlmConfig(overrides) {
86
89
  */
87
90
  exports.resolveLlmConfigForCC = resolveExplicitLlmConfig;
88
91
  /**
89
- * Validate + copy the tuning keys (#220: maxTokens + enableThinking + numCtx +
90
- * ollama flush; #391: classifyMaxTokens) from the saved disk config onto a
91
- * runtime LlmConfig. Called by BOTH LlmConfig construction sites — the daemon
92
- * in mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
93
- * runs reflect + classify) — so every process honors the keys, and a future
94
- * site calling this inherits them by construction.
92
+ * Validate + copy the tuning keys (#220: maxTokens + enableThinking) from the
93
+ * saved disk config onto a runtime LlmConfig, and resolve the DIAGNOSTIC
94
+ * TIER (#408: numCtx + the ollama-flush family) from the calibration env
95
+ * resolvers. Called by BOTH LlmConfig construction sites — the daemon in
96
+ * mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
97
+ * runs reflect + classify) — so this is the ONE resolution point per process
98
+ * for every one of these values, and a future site calling this inherits
99
+ * them by construction.
95
100
  *
96
- * All keys are optional; absent = call-site defaults (maxTokens 8192,
97
- * classifyMaxTokens 1024, numCtx 8192, thinking kwarg omitted, flush off).
98
- * Wrong-typed values warn and are dropped (readPositiveConfig /
99
- * readStrictBoolean / readNonNegativeConfig) — notably a JSON slip
101
+ * Config keys are optional; absent = call-site defaults (maxTokens 8192,
102
+ * thinking kwarg omitted). Wrong-typed values warn and are dropped
103
+ * (readPositiveConfig / readStrictBoolean) — notably a JSON slip
100
104
  * `"enableThinking": "false"` (string) is rejected rather than coerced to
101
105
  * truthy thinking-on, which would silently invert the fix this key exists to
102
- * apply.
106
+ * apply. #405: classifyMaxTokens is deleted — maxTokens is the single output
107
+ * ceiling for every call (warned as ignored at the config-read boundary).
108
+ * #408: numCtx / ollamaFlushEvery / ollamaFlushWaitMs are NO LONGER config
109
+ * keys — the env tier always resolves them (env > calibration constant;
110
+ * invalid env warns + falls back), independent of `savedConfig`.
103
111
  */
104
112
  function applyTierTuningOverlay(llmConfig, savedConfig) {
113
+ // #408 diagnostic tier — resolved BEFORE the config early-return so the
114
+ // tier applies on every construction site call, config or not.
115
+ llmConfig.numCtx = (0, calibration_js_1.resolveNumCtx)();
116
+ llmConfig.ollamaFlushEvery = (0, calibration_js_1.resolveOllamaFlushEvery)();
117
+ llmConfig.ollamaFlushWaitMs = (0, calibration_js_1.resolveOllamaFlushWaitMs)();
105
118
  if (!savedConfig)
106
119
  return;
107
120
  if (savedConfig.maxTokens !== undefined) {
108
121
  llmConfig.maxTokens = (0, config_read_js_1.readPositiveConfig)(savedConfig, "maxTokens", 8192);
109
122
  }
110
- if (savedConfig.classifyMaxTokens !== undefined) {
111
- llmConfig.classifyMaxTokens = (0, config_read_js_1.readPositiveConfig)(savedConfig, "classifyMaxTokens", 1024);
112
- }
113
123
  const thinking = (0, config_read_js_1.readStrictBoolean)(savedConfig, "enableThinking");
114
124
  if (thinking !== undefined) {
115
125
  llmConfig.enableThinking = thinking;
116
126
  }
117
- if (savedConfig.numCtx !== undefined) {
118
- llmConfig.numCtx = (0, config_read_js_1.readPositiveConfig)(savedConfig, "numCtx", 8192);
119
- }
120
- if (savedConfig.ollamaFlushEvery !== undefined) {
121
- llmConfig.ollamaFlushEvery = (0, config_read_js_1.readNonNegativeConfig)(savedConfig, "ollamaFlushEvery", 0);
122
- }
123
- if (savedConfig.ollamaFlushWaitMs !== undefined) {
124
- llmConfig.ollamaFlushWaitMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "ollamaFlushWaitMs", 180000);
125
- }
126
127
  // #337 resilience knobs. Same boundary discipline as the keys above: absent =
127
- // call-site defaults (timeout 900 s, threshold 3, cooldown 10 min, probe
128
- // timeout 60 s, probe TTL 5 min), wrong-typed values warn and fall back.
129
- // breakerThreshold uses readNonNegativeConfig because 0 is a VALID value
130
- // ("disable the breaker") — the same reason ollamaFlushEvery uses it.
128
+ // call-site defaults (timeout 900 s, probe timeout 60 s, probe TTL 5 min),
129
+ // wrong-typed values warn and fall back. #405: the breaker threshold/
130
+ // cooldown and the single-flight wait are CONSTANTS now (no incident ever
131
+ // required tuning them) — the keys are warned as ignored by config-read.ts.
131
132
  if (savedConfig.llmTimeoutMs !== undefined) {
132
133
  llmConfig.timeoutMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmTimeoutMs", 900000);
133
134
  }
134
- if (savedConfig.llmBreakerThreshold !== undefined) {
135
- llmConfig.breakerThreshold = (0, config_read_js_1.readNonNegativeConfig)(savedConfig, "llmBreakerThreshold", 3);
136
- }
137
- if (savedConfig.llmBreakerCooldownMs !== undefined) {
138
- llmConfig.breakerCooldownMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmBreakerCooldownMs", 600000);
139
- }
140
135
  if (savedConfig.llmProbeTimeoutMs !== undefined) {
141
136
  llmConfig.probeTimeoutMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmProbeTimeoutMs", 60000);
142
137
  }
@@ -144,15 +139,12 @@ function applyTierTuningOverlay(llmConfig, savedConfig) {
144
139
  llmConfig.probeTtlMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmProbeTtlMs", 300000);
145
140
  }
146
141
  // #355 single-flight. Default ON (a correctness property, not a tuning
147
- // option); the wait budget defaults to the timeout ceiling so a queued call
148
- // never gives up before the in-flight call's own ceiling expires.
142
+ // option); the kill switch survives as config. #405: the wait budget is
143
+ // DERIVED (max(900 s, llmTimeoutMs)) — no longer a knob.
149
144
  const singleFlight = (0, config_read_js_1.readStrictBoolean)(savedConfig, "llmSingleFlight");
150
145
  if (singleFlight !== undefined) {
151
146
  llmConfig.singleFlight = singleFlight;
152
147
  }
153
- if (savedConfig.llmSingleFlightWaitMs !== undefined) {
154
- llmConfig.singleFlightWaitMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmSingleFlightWaitMs", 900000);
155
- }
156
148
  }
157
149
  /**
158
150
  * Resolve an LlmConfig from a saved ~/.hicortex/config.json object.
@@ -195,8 +187,9 @@ function resolveSavedLlmConfig(savedConfig, findBinary = findClaudeBinary) {
195
187
  llmModel: savedConfig?.llmModel,
196
188
  });
197
189
  }
198
- // Tuning overlay (#220: maxTokens + enableThinking + numCtx + flush). Applied
199
- // at both construction sites (daemon + nightly) so every phase honors the keys.
190
+ // Tuning overlay (#220: maxTokens + enableThinking; #408: the diagnostic
191
+ // env tier). Applied at both construction sites (daemon + nightly) so every
192
+ // phase honors the same values.
200
193
  if (llmConfig) {
201
194
  applyTierTuningOverlay(llmConfig, savedConfig);
202
195
  }
@@ -312,6 +305,27 @@ exports.RateLimitError = RateLimitError;
312
305
  // (#231), so in practice there is a single client per process today; the
313
306
  // module-level map keeps the state shared correctly if that ever changes.
314
307
  const rateLimitedUntilByEndpoint = new Map();
308
+ // #337: per-endpoint circuit-breaker state, keyed like the rate-limit map.
309
+ // `failures` counts consecutive ladder-exhausted TOTAL failures (the class the
310
+ // retry ladder matches — fetch-failed/ECONNREFUSED/timeout/"Headers Timeout");
311
+ // HTTP errors WITH a response, parse errors, and RateLimitError throw before
312
+ // the ladder can be exhausted and never accrue. Any success resets. At
313
+ // BREAKER_THRESHOLD consecutive failures the breaker opens: calls then
314
+ // throw LlmCircuitOpenError with NO network I/O until the cooldown elapses,
315
+ // after which exactly one call is a trial (a failure re-opens, a success
316
+ // resets). This is what bounds a wedged endpoint to K ladder-exhausted calls
317
+ // instead of the ~200-call nightly retrying into it for hours (incident
318
+ // 2026-08-23/24). `openedAt` stays set (breakerOpen stays true) until a
319
+ // SUCCESS resets it — tripped-but-past-cooldown is still "not proven healthy".
320
+ //
321
+ // #405: threshold + cooldown are CONSTANTS — the config knobs
322
+ // (llmBreakerThreshold / llmBreakerCooldownMs) were deleted after the #403
323
+ // inventory found no incident that ever required tuning them. Worst-case
324
+ // dead-endpoint discovery: 3 logical calls × (1 timeout + 1 retry) ≈ 31 min,
325
+ // once per dead night, bounded by the run deadline (the nightly probe gate
326
+ // that used to catch this in 60 s is also gone — spec-accepted).
327
+ const BREAKER_THRESHOLD = 3;
328
+ const BREAKER_COOLDOWN_MS = 600_000;
315
329
  const breakerByEndpoint = new Map();
316
330
  /** Thrown when the per-endpoint circuit breaker is open (#337) — no network I/O happened. */
317
331
  class LlmCircuitOpenError extends Error {
@@ -364,20 +378,17 @@ class LlmClient {
364
378
  }
365
379
  /** Record one ladder-exhausted TOTAL failure; open the breaker at threshold. */
366
380
  recordBreakerFailure() {
367
- const threshold = this.config.breakerThreshold ?? 3;
368
- if (threshold <= 0)
369
- return; // 0 disables — never open, never fast-fail
370
381
  const st = breakerByEndpoint.get(this.endpointKey) ?? { failures: 0, openedAt: null };
371
382
  st.failures += 1;
372
- if (st.failures >= threshold) {
383
+ if (st.failures >= BREAKER_THRESHOLD) {
373
384
  // (Re)open. Re-opening (a failed trial past cooldown) restarts the
374
385
  // cooldown window from NOW — the endpoint just proved itself still dead.
375
386
  st.openedAt = Date.now();
376
387
  // One structured line per opening — the runbook's grep target. Same
377
388
  // key=value style as event=budget_exhausted (consolidate.ts).
378
389
  console.warn(`[hicortex] event=circuit_open endpoint=${this.endpointKey} ` +
379
- `failures=${st.failures} threshold=${threshold} ` +
380
- `cooldown_ms=${this.config.breakerCooldownMs ?? 600_000}`);
390
+ `failures=${st.failures} threshold=${BREAKER_THRESHOLD} ` +
391
+ `cooldown_ms=${BREAKER_COOLDOWN_MS}`);
381
392
  }
382
393
  breakerByEndpoint.set(this.endpointKey, st);
383
394
  }
@@ -401,19 +412,20 @@ class LlmClient {
401
412
  throw new RateLimitError(retryMs);
402
413
  }
403
414
  /**
404
- * Fast-tier completion (importance scoring, simple tasks). One model serves
405
- * all phases (#231); numCtx + enableThinking are read from config directly
406
- * inside completeOnce's per-provider dispatch, not threaded here. The periodic
407
- * ollama flush stays (provider-gated) — it is a scoring-call-count cadence and
408
- * scoring is the highest-frequency call, so this is where the flush belongs.
415
+ * The ONE completion surface (#405). Every phase — distill, reflect,
416
+ * classify, scoring — calls this; maxTokens (default 8192) and timeoutMs
417
+ * (default 900 s) resolve from config INSIDE, so no call site can starve a
418
+ * phase with a per-site ceiling (#391's failure direction: hardcoded
419
+ * 16/64/256-token caps starved reasoning models whose internal thinking
420
+ * ate the whole output budget, leaving verdicts empty). numCtx +
421
+ * enableThinking are likewise read inside completeOnce's per-provider
422
+ * dispatch. The periodic ollama flush survives here (provider-gated) — it
423
+ * now counts ALL calls, not just the scoring tier.
409
424
  */
410
- async completeFast(prompt, maxTokens) {
411
- const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
412
- // #337: ONE ceiling for every phase (llmTimeoutMs, default 900 s). The old
413
- // 600 s scoring special-case assumed fast-tier calls are short — but the
414
- // ceiling only ever mattered when the endpoint was wedged, and a wedged
415
- // endpoint wedges scoring too. One knob, one place.
416
- const result = await this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
425
+ async complete(prompt) {
426
+ const tokens = this.config.maxTokens ?? 8192;
427
+ // #337: ONE ceiling for every phase (llmTimeoutMs, default 900 s).
428
+ const result = await this.completeWithRetry(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
417
429
  const flushEvery = this.config.ollamaFlushEvery ?? 0;
418
430
  if (this.config.provider === "ollama" && flushEvery > 0) {
419
431
  this.ollamaCallCount++;
@@ -424,46 +436,16 @@ class LlmClient {
424
436
  }
425
437
  return result;
426
438
  }
427
- /**
428
- * Reflect-tier completion (nightly reflection). One model serves all phases
429
- * (#231) — this is a thin wrapper kept for call-site readability.
430
- */
431
- async completeReflect(prompt, maxTokens) {
432
- const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
433
- return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
434
- }
435
- /**
436
- * Distillation-tier completion (session knowledge extraction). One model
437
- * serves all phases (#231) — thin wrapper kept for call-site readability.
438
- */
439
- async completeDistill(prompt, maxTokens) {
440
- const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
441
- return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
442
- }
443
- /**
444
- * Classification-tier completion (verdicts + tag/type classification). One
445
- * model serves all phases (#231) — thin wrapper kept for call-site
446
- * readability, but the tier keeps its OWN output ceiling (#391).
447
- */
448
- async completeClassify(prompt, maxTokens) {
449
- // #391: classify-tier ceiling — the call sites' old hardcoded caps
450
- // (64/32/20, tuned for a local non-reasoning model) starved reasoning
451
- // models whose internal thinking consumed the whole budget, leaving
452
- // verdicts empty. Deliberately NOT this.config.maxTokens: that knob
453
- // governs the heavy phases; this tier has its own (a ceiling, not a
454
- // target — generation still stops at the model's natural end).
455
- const tokens = maxTokens ?? this.config.classifyMaxTokens ?? 1024;
456
- return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
457
- }
458
439
  /**
459
440
  * Readiness probe (#337): ONE minimal generation request (max output 1
460
441
  * token) through the normal provider dispatch. Asks the question liveness
461
442
  * checks CANNOT: "can this endpoint GENERATE right now?" — the incident
462
443
  * gateway kept answering /v1/models for hours while every completion hung.
463
444
  * Single attempt: no retry ladder (a dead endpoint must cost one fast
464
- * failure, not a 3.5-min ladder), and it never accrues to the circuit
445
+ * failure, not a ladder), and it never accrues to the circuit
465
446
  * breaker (probing is diagnosis, not traffic). Catch-all → false; the
466
447
  * callers translate that into "endpoint_down" / a 503, never an exception.
448
+ * #405: owned by the DAEMON's /distill gate — the nightly no longer probes.
467
449
  */
468
450
  async probe(timeoutMs) {
469
451
  try {
@@ -475,22 +457,25 @@ class LlmClient {
475
457
  return false;
476
458
  }
477
459
  }
478
- async complete(model, prompt, maxTokens, timeoutMs) {
460
+ async completeWithRetry(model, prompt, maxTokens, timeoutMs) {
479
461
  // Breaker BEFORE anything else (#337) — an open breaker must cost zero
480
462
  // network I/O and zero ladder time. Past the cooldown we fall through:
481
463
  // this call IS the trial.
482
464
  const breakerSt = breakerByEndpoint.get(this.endpointKey);
483
465
  if (breakerSt !== undefined && breakerSt.openedAt !== null) {
484
466
  const elapsed = Date.now() - breakerSt.openedAt;
485
- const cooldownMs = this.config.breakerCooldownMs ?? 600_000;
486
- if (elapsed < cooldownMs) {
487
- throw new LlmCircuitOpenError(this.endpointKey, cooldownMs - elapsed);
467
+ if (elapsed < BREAKER_COOLDOWN_MS) {
468
+ throw new LlmCircuitOpenError(this.endpointKey, BREAKER_COOLDOWN_MS - elapsed);
488
469
  }
489
470
  }
490
471
  if (this.isRateLimited) {
491
472
  throw new RateLimitError(this.rateLimitedUntil - Date.now());
492
473
  }
493
- const retryDelays = [30_000, 60_000, 120_000]; // 30s, 60s, 120s
474
+ // #405: retry ladder collapsed to ONE 60 s retry (was 30/60/120×3 — the
475
+ // 63.5-min/call worst case from the #403 inventory). Worst-case logical
476
+ // call: 2 × timeoutMs + 60 s ≈ 31 min at the 900 s ceiling. A second
477
+ // consecutive total failure is breaker evidence, not a reason to wait.
478
+ const retryDelays = [60_000];
494
479
  let lastErr;
495
480
  for (let attempt = 0; attempt <= retryDelays.length; attempt++) {
496
481
  try {
@@ -550,13 +535,14 @@ class LlmClient {
550
535
  }
551
536
  }
552
537
  /** The single-flight wait budget for a call with ceiling `timeoutMs`.
553
- * An explicit `llmSingleFlightWaitMs` wins; otherwise the default is
554
- * max(900 s, llmTimeoutMs) so a waiter never gives up before a legitimate
555
- * in-flight call's own (possibly raised) ceiling expires (2nd-review
556
- * finding 2 — a hardcoded 900 s made a raised-timeout install treat a
557
- * healthy-busy endpoint as down). */
538
+ * #405: always DERIVED — max(900 s, llmTimeoutMs) — so a waiter never
539
+ * gives up before a legitimate in-flight call's own (possibly raised)
540
+ * ceiling expires (2nd-review finding 2 — a hardcoded 900 s made a
541
+ * raised-timeout install treat a healthy-busy endpoint as down). The
542
+ * llmSingleFlightWaitMs knob is deleted (its only documented use was to
543
+ * restore this exact derivation). */
558
544
  flightWaitMs(timeoutMs) {
559
- return this.config.singleFlightWaitMs ?? Math.max(900_000, timeoutMs);
545
+ return Math.max(900_000, timeoutMs);
560
546
  }
561
547
  /** Resolve the flight guard for this call, or undefined when disabled.
562
548
  * `staleMs` is derived from THIS call's timeout ceiling (≥ the 30-min