@gamaze/hicortex 0.20.7 → 0.20.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +18 -41
  2. package/assets/dashboard.html +3989 -836
  3. package/dist/calibration.d.ts +293 -0
  4. package/dist/calibration.js +379 -0
  5. package/dist/capture-health.d.ts +87 -0
  6. package/dist/capture-health.js +106 -0
  7. package/dist/capture-pause.d.ts +86 -0
  8. package/dist/capture-pause.js +127 -0
  9. package/dist/capture.d.ts +24 -3
  10. package/dist/capture.js +11 -1
  11. package/dist/classify-domains.d.ts +6 -0
  12. package/dist/classify-domains.js +7 -1
  13. package/dist/cli.js +38 -3
  14. package/dist/config-read.d.ts +1 -1
  15. package/dist/config-read.js +96 -9
  16. package/dist/consolidate.d.ts +114 -68
  17. package/dist/consolidate.js +302 -182
  18. package/dist/dashboard.d.ts +326 -6
  19. package/dist/dashboard.js +592 -7
  20. package/dist/db.js +105 -0
  21. package/dist/dedup.d.ts +34 -26
  22. package/dist/dedup.js +91 -57
  23. package/dist/distiller.js +1 -1
  24. package/dist/domain-classify.d.ts +7 -6
  25. package/dist/domain-classify.js +12 -10
  26. package/dist/eval/decay-eval.d.ts +3 -3
  27. package/dist/eval/decay-eval.js +4 -4
  28. package/dist/eval/importance-eval.d.ts +85 -0
  29. package/dist/eval/importance-eval.js +286 -0
  30. package/dist/eval/planted-eval.d.ts +26 -0
  31. package/dist/eval/planted-eval.js +97 -0
  32. package/dist/eval/planted-fixtures.d.ts +107 -0
  33. package/dist/eval/planted-fixtures.js +283 -0
  34. package/dist/eval/planted-harness.d.ts +176 -0
  35. package/dist/eval/planted-harness.js +649 -0
  36. package/dist/eval/ranking-battery.d.ts +78 -0
  37. package/dist/eval/ranking-battery.js +181 -0
  38. package/dist/eval/ranking-eval.d.ts +41 -0
  39. package/dist/eval/ranking-eval.js +391 -0
  40. package/dist/eval/ranking-fixtures.d.ts +77 -0
  41. package/dist/eval/ranking-fixtures.js +226 -0
  42. package/dist/identity-store.d.ts +21 -0
  43. package/dist/identity-store.js +49 -0
  44. package/dist/index.js +4 -3
  45. package/dist/init.d.ts +23 -3
  46. package/dist/init.js +84 -9
  47. package/dist/llm.d.ts +43 -58
  48. package/dist/llm.js +87 -101
  49. package/dist/mcp-server.d.ts +12 -0
  50. package/dist/mcp-server.js +213 -32
  51. package/dist/nightly.d.ts +9 -1
  52. package/dist/nightly.js +164 -110
  53. package/dist/nofit.d.ts +4 -11
  54. package/dist/nofit.js +6 -23
  55. package/dist/prompts.d.ts +10 -0
  56. package/dist/prompts.js +28 -5
  57. package/dist/recall-index.d.ts +30 -28
  58. package/dist/recall-index.js +21 -18
  59. package/dist/recall-registry.d.ts +2 -1
  60. package/dist/recall-registry.js +35 -1
  61. package/dist/reconsolidation.d.ts +168 -87
  62. package/dist/reconsolidation.js +818 -377
  63. package/dist/relink.js +3 -4
  64. package/dist/rescore-importance.d.ts +80 -0
  65. package/dist/rescore-importance.js +236 -0
  66. package/dist/retrieval.d.ts +80 -35
  67. package/dist/retrieval.js +322 -105
  68. package/dist/run-deadline.d.ts +62 -0
  69. package/dist/run-deadline.js +73 -0
  70. package/dist/schema-prototypes.d.ts +3 -3
  71. package/dist/schema-prototypes.js +3 -3
  72. package/dist/stages.d.ts +37 -0
  73. package/dist/stages.js +51 -0
  74. package/dist/state.d.ts +34 -9
  75. package/dist/storage.d.ts +50 -18
  76. package/dist/storage.js +125 -30
  77. package/dist/telemetry.d.ts +8 -7
  78. package/dist/token-budget.js +3 -4
  79. package/dist/type-classify.js +4 -4
  80. package/dist/types.d.ts +143 -155
  81. package/domains.example.json +4 -5
  82. package/hermes-plugin/hicortex/README.md +2 -2
  83. package/openclaw.plugin.json +1 -1
  84. package/package.json +4 -1
  85. package/pi-extension/hicortex/README.md +1 -1
  86. package/server.json +3 -3
package/dist/llm.d.ts CHANGED
@@ -23,29 +23,23 @@ export interface LlmConfig {
23
23
  provider: string;
24
24
  /** Max output tokens for all phases (one model). Default 8192. */
25
25
  maxTokens?: number;
26
- /** Max output tokens for the CLASSIFY tier only — the short JSON-verdict
27
- * calls (correction/supersession verdicts, rewrite contracts, type + tag
28
- * classification). Default 1024. See HicortexConfig.classifyMaxTokens (#391). */
29
- classifyMaxTokens?: number;
30
26
  /** Toggle thinking on the openai-compat path for all phases. Absent = no kwarg sent.
31
27
  * LOCAL-endpoint only (ollama / mlx-lm gateway); see HicortexConfig.enableThinking. */
32
28
  enableThinking?: boolean;
33
- /** Context window for ollama (the one model, all phases). Default 8192. */
29
+ /** Context window for ollama (the one model, all phases). Default 8192.
30
+ * #408 diagnostic tier: resolved from HICORTEX_NUM_CTX (env) by
31
+ * applyTierTuningOverlay — never a config key. */
34
32
  numCtx?: number;
35
- /** Flush ollama memory every N ollama calls (0 = off). See HicortexConfig.ollamaFlushEvery. */
33
+ /** Flush ollama memory every N ollama calls (0 = off). #408 diagnostic
34
+ * tier: resolved from HICORTEX_OLLAMA_FLUSH_EVERY by the overlay. */
36
35
  ollamaFlushEvery?: number;
37
- /** Ms to wait after an ollama flush for the runner to release. */
36
+ /** Ms to wait after an ollama flush for the runner to release. #408
37
+ * diagnostic tier: resolved from HICORTEX_OLLAMA_FLUSH_WAIT_MS. */
38
38
  ollamaFlushWaitMs?: number;
39
39
  /** ONE per-call timeout ceiling for every phase (#337). Default 900000 — the
40
- * AbortSignal.timeout value passed by all four phase wrappers (the old
40
+ * AbortSignal.timeout value passed by the single complete() surface (the old
41
41
  * 600000 scoring special-case is gone). See HicortexConfig.llmTimeoutMs. */
42
42
  timeoutMs?: number;
43
- /** Consecutive ladder-exhausted total failures before the circuit breaker
44
- * opens (#337). Default 3; 0 disables. See HicortexConfig.llmBreakerThreshold. */
45
- breakerThreshold?: number;
46
- /** How long an OPEN breaker stays open before the next call becomes a trial
47
- * (#337). Default 600000. See HicortexConfig.llmBreakerCooldownMs. */
48
- breakerCooldownMs?: number;
49
43
  /** Timeout for the readiness probe's single generation attempt (#337).
50
44
  * Default 60000. See HicortexConfig.llmProbeTimeoutMs. */
51
45
  probeTimeoutMs?: number;
@@ -57,10 +51,6 @@ export interface LlmConfig {
57
51
  * single-user model servers (two concurrent large-context calls stall/OOM
58
52
  * the server and the machine under it). See HicortexConfig.llmSingleFlight. */
59
53
  singleFlight?: boolean;
60
- /** How long a queued call waits for the in-flight call before failing as
61
- * endpoint-down (#355). Default 900000 — the same ceiling as llmTimeoutMs.
62
- * See HicortexConfig.llmSingleFlightWaitMs. */
63
- singleFlightWaitMs?: number;
64
54
  }
65
55
  /**
66
56
  * Resolve LLM configuration from explicit config-file overrides or
@@ -85,20 +75,25 @@ export declare function resolveExplicitLlmConfig(overrides?: {
85
75
  */
86
76
  export declare const resolveLlmConfigForCC: typeof resolveExplicitLlmConfig;
87
77
  /**
88
- * Validate + copy the tuning keys (#220: maxTokens + enableThinking + numCtx +
89
- * ollama flush; #391: classifyMaxTokens) from the saved disk config onto a
90
- * runtime LlmConfig. Called by BOTH LlmConfig construction sites — the daemon
91
- * in mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
92
- * runs reflect + classify) — so every process honors the keys, and a future
93
- * site calling this inherits them by construction.
78
+ * Validate + copy the tuning keys (#220: maxTokens + enableThinking) from the
79
+ * saved disk config onto a runtime LlmConfig, and resolve the DIAGNOSTIC
80
+ * TIER (#408: numCtx + the ollama-flush family) from the calibration env
81
+ * resolvers. Called by BOTH LlmConfig construction sites — the daemon in
82
+ * mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
83
+ * runs reflect + classify) — so this is the ONE resolution point per process
84
+ * for every one of these values, and a future site calling this inherits
85
+ * them by construction.
94
86
  *
95
- * All keys are optional; absent = call-site defaults (maxTokens 8192,
96
- * classifyMaxTokens 1024, numCtx 8192, thinking kwarg omitted, flush off).
97
- * Wrong-typed values warn and are dropped (readPositiveConfig /
98
- * readStrictBoolean / readNonNegativeConfig) — notably a JSON slip
87
+ * Config keys are optional; absent = call-site defaults (maxTokens 8192,
88
+ * thinking kwarg omitted). Wrong-typed values warn and are dropped
89
+ * (readPositiveConfig / readStrictBoolean) — notably a JSON slip
99
90
  * `"enableThinking": "false"` (string) is rejected rather than coerced to
100
91
  * truthy thinking-on, which would silently invert the fix this key exists to
101
- * apply.
92
+ * apply. #405: classifyMaxTokens is deleted — maxTokens is the single output
93
+ * ceiling for every call (warned as ignored at the config-read boundary).
94
+ * #408: numCtx / ollamaFlushEvery / ollamaFlushWaitMs are NO LONGER config
95
+ * keys — the env tier always resolves them (env > calibration constant;
96
+ * invalid env warns + falls back), independent of `savedConfig`.
102
97
  */
103
98
  export declare function applyTierTuningOverlay(llmConfig: LlmConfig, savedConfig: Record<string, unknown> | null | undefined): void;
104
99
  /**
@@ -200,48 +195,38 @@ export declare class LlmClient {
200
195
  private resetBreaker;
201
196
  private handleRateLimit;
202
197
  /**
203
- * Fast-tier completion (importance scoring, simple tasks). One model serves
204
- * all phases (#231); numCtx + enableThinking are read from config directly
205
- * inside completeOnce's per-provider dispatch, not threaded here. The periodic
206
- * ollama flush stays (provider-gated) — it is a scoring-call-count cadence and
207
- * scoring is the highest-frequency call, so this is where the flush belongs.
198
+ * The ONE completion surface (#405). Every phase — distill, reflect,
199
+ * classify, scoring — calls this; maxTokens (default 8192) and timeoutMs
200
+ * (default 900 s) resolve from config INSIDE, so no call site can starve a
201
+ * phase with a per-site ceiling (#391's failure direction: hardcoded
202
+ * 16/64/256-token caps starved reasoning models whose internal thinking
203
+ * ate the whole output budget, leaving verdicts empty). numCtx +
204
+ * enableThinking are likewise read inside completeOnce's per-provider
205
+ * dispatch. The periodic ollama flush survives here (provider-gated) — it
206
+ * now counts ALL calls, not just the scoring tier.
208
207
  */
209
- completeFast(prompt: string, maxTokens?: number): Promise<LlmResult>;
210
- /**
211
- * Reflect-tier completion (nightly reflection). One model serves all phases
212
- * (#231) — this is a thin wrapper kept for call-site readability.
213
- */
214
- completeReflect(prompt: string, maxTokens?: number): Promise<LlmResult>;
215
- /**
216
- * Distillation-tier completion (session knowledge extraction). One model
217
- * serves all phases (#231) — thin wrapper kept for call-site readability.
218
- */
219
- completeDistill(prompt: string, maxTokens?: number): Promise<LlmResult>;
220
- /**
221
- * Classification-tier completion (verdicts + tag/type classification). One
222
- * model serves all phases (#231) — thin wrapper kept for call-site
223
- * readability, but the tier keeps its OWN output ceiling (#391).
224
- */
225
- completeClassify(prompt: string, maxTokens?: number): Promise<LlmResult>;
208
+ complete(prompt: string): Promise<LlmResult>;
226
209
  /**
227
210
  * Readiness probe (#337): ONE minimal generation request (max output 1
228
211
  * token) through the normal provider dispatch. Asks the question liveness
229
212
  * checks CANNOT: "can this endpoint GENERATE right now?" — the incident
230
213
  * gateway kept answering /v1/models for hours while every completion hung.
231
214
  * Single attempt: no retry ladder (a dead endpoint must cost one fast
232
- * failure, not a 3.5-min ladder), and it never accrues to the circuit
215
+ * failure, not a ladder), and it never accrues to the circuit
233
216
  * breaker (probing is diagnosis, not traffic). Catch-all → false; the
234
217
  * callers translate that into "endpoint_down" / a 503, never an exception.
218
+ * #405: owned by the DAEMON's /distill gate — the nightly no longer probes.
235
219
  */
236
220
  probe(timeoutMs?: number): Promise<boolean>;
237
- private complete;
221
+ private completeWithRetry;
238
222
  private completeOnce;
239
223
  /** The single-flight wait budget for a call with ceiling `timeoutMs`.
240
- * An explicit `llmSingleFlightWaitMs` wins; otherwise the default is
241
- * max(900 s, llmTimeoutMs) so a waiter never gives up before a legitimate
242
- * in-flight call's own (possibly raised) ceiling expires (2nd-review
243
- * finding 2 — a hardcoded 900 s made a raised-timeout install treat a
244
- * healthy-busy endpoint as down). */
224
+ * #405: always DERIVED — max(900 s, llmTimeoutMs) — so a waiter never
225
+ * gives up before a legitimate in-flight call's own (possibly raised)
226
+ * ceiling expires (2nd-review finding 2 — a hardcoded 900 s made a
227
+ * raised-timeout install treat a healthy-busy endpoint as down). The
228
+ * llmSingleFlightWaitMs knob is deleted (its only documented use was to
229
+ * restore this exact derivation). */
245
230
  private flightWaitMs;
246
231
  /** Resolve the flight guard for this call, or undefined when disabled.
247
232
  * `staleMs` is derived from THIS call's timeout ceiling (≥ the 30-min
package/dist/llm.js CHANGED
@@ -27,6 +27,9 @@ exports.claudeCliConfig = claudeCliConfig;
27
27
  exports.probeOllama = probeOllama;
28
28
  exports.isTotalFailure = isTotalFailure;
29
29
  const config_read_js_1 = require("./config-read.js");
30
+ // #408: the diagnostic tier (numCtx + ollama flush) resolves from the
31
+ // calibration env resolvers — the config keys are gone from the surface.
32
+ const calibration_js_1 = require("./calibration.js");
30
33
  // #355: single-flight guard + canonical home (where the per-endpoint flight
31
34
  // lock files live — the daemon and the nightly share the home, so the lock
32
35
  // serializes them across processes).
@@ -86,57 +89,49 @@ function resolveExplicitLlmConfig(overrides) {
86
89
  */
87
90
  exports.resolveLlmConfigForCC = resolveExplicitLlmConfig;
88
91
  /**
89
- * Validate + copy the tuning keys (#220: maxTokens + enableThinking + numCtx +
90
- * ollama flush; #391: classifyMaxTokens) from the saved disk config onto a
91
- * runtime LlmConfig. Called by BOTH LlmConfig construction sites — the daemon
92
- * in mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
93
- * runs reflect + classify) — so every process honors the keys, and a future
94
- * site calling this inherits them by construction.
92
+ * Validate + copy the tuning keys (#220: maxTokens + enableThinking) from the
93
+ * saved disk config onto a runtime LlmConfig, and resolve the DIAGNOSTIC
94
+ * TIER (#408: numCtx + the ollama-flush family) from the calibration env
95
+ * resolvers. Called by BOTH LlmConfig construction sites — the daemon in
96
+ * mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
97
+ * runs reflect + classify) — so this is the ONE resolution point per process
98
+ * for every one of these values, and a future site calling this inherits
99
+ * them by construction.
95
100
  *
96
- * All keys are optional; absent = call-site defaults (maxTokens 8192,
97
- * classifyMaxTokens 1024, numCtx 8192, thinking kwarg omitted, flush off).
98
- * Wrong-typed values warn and are dropped (readPositiveConfig /
99
- * readStrictBoolean / readNonNegativeConfig) — notably a JSON slip
101
+ * Config keys are optional; absent = call-site defaults (maxTokens 8192,
102
+ * thinking kwarg omitted). Wrong-typed values warn and are dropped
103
+ * (readPositiveConfig / readStrictBoolean) — notably a JSON slip
100
104
  * `"enableThinking": "false"` (string) is rejected rather than coerced to
101
105
  * truthy thinking-on, which would silently invert the fix this key exists to
102
- * apply.
106
+ * apply. #405: classifyMaxTokens is deleted — maxTokens is the single output
107
+ * ceiling for every call (warned as ignored at the config-read boundary).
108
+ * #408: numCtx / ollamaFlushEvery / ollamaFlushWaitMs are NO LONGER config
109
+ * keys — the env tier always resolves them (env > calibration constant;
110
+ * invalid env warns + falls back), independent of `savedConfig`.
103
111
  */
104
112
  function applyTierTuningOverlay(llmConfig, savedConfig) {
113
+ // #408 diagnostic tier — resolved BEFORE the config early-return so the
114
+ // tier applies on every construction site call, config or not.
115
+ llmConfig.numCtx = (0, calibration_js_1.resolveNumCtx)();
116
+ llmConfig.ollamaFlushEvery = (0, calibration_js_1.resolveOllamaFlushEvery)();
117
+ llmConfig.ollamaFlushWaitMs = (0, calibration_js_1.resolveOllamaFlushWaitMs)();
105
118
  if (!savedConfig)
106
119
  return;
107
120
  if (savedConfig.maxTokens !== undefined) {
108
121
  llmConfig.maxTokens = (0, config_read_js_1.readPositiveConfig)(savedConfig, "maxTokens", 8192);
109
122
  }
110
- if (savedConfig.classifyMaxTokens !== undefined) {
111
- llmConfig.classifyMaxTokens = (0, config_read_js_1.readPositiveConfig)(savedConfig, "classifyMaxTokens", 1024);
112
- }
113
123
  const thinking = (0, config_read_js_1.readStrictBoolean)(savedConfig, "enableThinking");
114
124
  if (thinking !== undefined) {
115
125
  llmConfig.enableThinking = thinking;
116
126
  }
117
- if (savedConfig.numCtx !== undefined) {
118
- llmConfig.numCtx = (0, config_read_js_1.readPositiveConfig)(savedConfig, "numCtx", 8192);
119
- }
120
- if (savedConfig.ollamaFlushEvery !== undefined) {
121
- llmConfig.ollamaFlushEvery = (0, config_read_js_1.readNonNegativeConfig)(savedConfig, "ollamaFlushEvery", 0);
122
- }
123
- if (savedConfig.ollamaFlushWaitMs !== undefined) {
124
- llmConfig.ollamaFlushWaitMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "ollamaFlushWaitMs", 180000);
125
- }
126
127
  // #337 resilience knobs. Same boundary discipline as the keys above: absent =
127
- // call-site defaults (timeout 900 s, threshold 3, cooldown 10 min, probe
128
- // timeout 60 s, probe TTL 5 min), wrong-typed values warn and fall back.
129
- // breakerThreshold uses readNonNegativeConfig because 0 is a VALID value
130
- // ("disable the breaker") — the same reason ollamaFlushEvery uses it.
128
+ // call-site defaults (timeout 900 s, probe timeout 60 s, probe TTL 5 min),
129
+ // wrong-typed values warn and fall back. #405: the breaker threshold/
130
+ // cooldown and the single-flight wait are CONSTANTS now (no incident ever
131
+ // required tuning them) — the keys are warned as ignored by config-read.ts.
131
132
  if (savedConfig.llmTimeoutMs !== undefined) {
132
133
  llmConfig.timeoutMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmTimeoutMs", 900000);
133
134
  }
134
- if (savedConfig.llmBreakerThreshold !== undefined) {
135
- llmConfig.breakerThreshold = (0, config_read_js_1.readNonNegativeConfig)(savedConfig, "llmBreakerThreshold", 3);
136
- }
137
- if (savedConfig.llmBreakerCooldownMs !== undefined) {
138
- llmConfig.breakerCooldownMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmBreakerCooldownMs", 600000);
139
- }
140
135
  if (savedConfig.llmProbeTimeoutMs !== undefined) {
141
136
  llmConfig.probeTimeoutMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmProbeTimeoutMs", 60000);
142
137
  }
@@ -144,15 +139,12 @@ function applyTierTuningOverlay(llmConfig, savedConfig) {
144
139
  llmConfig.probeTtlMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmProbeTtlMs", 300000);
145
140
  }
146
141
  // #355 single-flight. Default ON (a correctness property, not a tuning
147
- // option); the wait budget defaults to the timeout ceiling so a queued call
148
- // never gives up before the in-flight call's own ceiling expires.
142
+ // option); the kill switch survives as config. #405: the wait budget is
143
+ // DERIVED (max(900 s, llmTimeoutMs)) — no longer a knob.
149
144
  const singleFlight = (0, config_read_js_1.readStrictBoolean)(savedConfig, "llmSingleFlight");
150
145
  if (singleFlight !== undefined) {
151
146
  llmConfig.singleFlight = singleFlight;
152
147
  }
153
- if (savedConfig.llmSingleFlightWaitMs !== undefined) {
154
- llmConfig.singleFlightWaitMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmSingleFlightWaitMs", 900000);
155
- }
156
148
  }
157
149
  /**
158
150
  * Resolve an LlmConfig from a saved ~/.hicortex/config.json object.
@@ -195,8 +187,9 @@ function resolveSavedLlmConfig(savedConfig, findBinary = findClaudeBinary) {
195
187
  llmModel: savedConfig?.llmModel,
196
188
  });
197
189
  }
198
- // Tuning overlay (#220: maxTokens + enableThinking + numCtx + flush). Applied
199
- // at both construction sites (daemon + nightly) so every phase honors the keys.
190
+ // Tuning overlay (#220: maxTokens + enableThinking; #408: the diagnostic
191
+ // env tier). Applied at both construction sites (daemon + nightly) so every
192
+ // phase honors the same values.
200
193
  if (llmConfig) {
201
194
  applyTierTuningOverlay(llmConfig, savedConfig);
202
195
  }
@@ -312,6 +305,27 @@ exports.RateLimitError = RateLimitError;
312
305
  // (#231), so in practice there is a single client per process today; the
313
306
  // module-level map keeps the state shared correctly if that ever changes.
314
307
  const rateLimitedUntilByEndpoint = new Map();
308
+ // #337: per-endpoint circuit-breaker state, keyed like the rate-limit map.
309
+ // `failures` counts consecutive ladder-exhausted TOTAL failures (the class the
310
+ // retry ladder matches — fetch-failed/ECONNREFUSED/timeout/"Headers Timeout");
311
+ // HTTP errors WITH a response, parse errors, and RateLimitError throw before
312
+ // the ladder can be exhausted and never accrue. Any success resets. At
313
+ // BREAKER_THRESHOLD consecutive failures the breaker opens: calls then
314
+ // throw LlmCircuitOpenError with NO network I/O until the cooldown elapses,
315
+ // after which exactly one call is a trial (a failure re-opens, a success
316
+ // resets). This is what bounds a wedged endpoint to K ladder-exhausted calls
317
+ // instead of the ~200-call nightly retrying into it for hours (incident
318
+ // 2026-08-23/24). `openedAt` stays set (breakerOpen stays true) until a
319
+ // SUCCESS resets it — tripped-but-past-cooldown is still "not proven healthy".
320
+ //
321
+ // #405: threshold + cooldown are CONSTANTS — the config knobs
322
+ // (llmBreakerThreshold / llmBreakerCooldownMs) were deleted after the #403
323
+ // inventory found no incident that ever required tuning them. Worst-case
324
+ // dead-endpoint discovery: 3 logical calls × (1 timeout + 1 retry) ≈ 31 min,
325
+ // once per dead night, bounded by the run deadline (the nightly probe gate
326
+ // that used to catch this in 60 s is also gone — spec-accepted).
327
+ const BREAKER_THRESHOLD = 3;
328
+ const BREAKER_COOLDOWN_MS = 600_000;
315
329
  const breakerByEndpoint = new Map();
316
330
  /** Thrown when the per-endpoint circuit breaker is open (#337) — no network I/O happened. */
317
331
  class LlmCircuitOpenError extends Error {
@@ -364,20 +378,17 @@ class LlmClient {
364
378
  }
365
379
  /** Record one ladder-exhausted TOTAL failure; open the breaker at threshold. */
366
380
  recordBreakerFailure() {
367
- const threshold = this.config.breakerThreshold ?? 3;
368
- if (threshold <= 0)
369
- return; // 0 disables — never open, never fast-fail
370
381
  const st = breakerByEndpoint.get(this.endpointKey) ?? { failures: 0, openedAt: null };
371
382
  st.failures += 1;
372
- if (st.failures >= threshold) {
383
+ if (st.failures >= BREAKER_THRESHOLD) {
373
384
  // (Re)open. Re-opening (a failed trial past cooldown) restarts the
374
385
  // cooldown window from NOW — the endpoint just proved itself still dead.
375
386
  st.openedAt = Date.now();
376
387
  // One structured line per opening — the runbook's grep target. Same
377
388
  // key=value style as event=budget_exhausted (consolidate.ts).
378
389
  console.warn(`[hicortex] event=circuit_open endpoint=${this.endpointKey} ` +
379
- `failures=${st.failures} threshold=${threshold} ` +
380
- `cooldown_ms=${this.config.breakerCooldownMs ?? 600_000}`);
390
+ `failures=${st.failures} threshold=${BREAKER_THRESHOLD} ` +
391
+ `cooldown_ms=${BREAKER_COOLDOWN_MS}`);
381
392
  }
382
393
  breakerByEndpoint.set(this.endpointKey, st);
383
394
  }
@@ -401,19 +412,20 @@ class LlmClient {
401
412
  throw new RateLimitError(retryMs);
402
413
  }
403
414
  /**
404
- * Fast-tier completion (importance scoring, simple tasks). One model serves
405
- * all phases (#231); numCtx + enableThinking are read from config directly
406
- * inside completeOnce's per-provider dispatch, not threaded here. The periodic
407
- * ollama flush stays (provider-gated) — it is a scoring-call-count cadence and
408
- * scoring is the highest-frequency call, so this is where the flush belongs.
415
+ * The ONE completion surface (#405). Every phase — distill, reflect,
416
+ * classify, scoring — calls this; maxTokens (default 8192) and timeoutMs
417
+ * (default 900 s) resolve from config INSIDE, so no call site can starve a
418
+ * phase with a per-site ceiling (#391's failure direction: hardcoded
419
+ * 16/64/256-token caps starved reasoning models whose internal thinking
420
+ * ate the whole output budget, leaving verdicts empty). numCtx +
421
+ * enableThinking are likewise read inside completeOnce's per-provider
422
+ * dispatch. The periodic ollama flush survives here (provider-gated) — it
423
+ * now counts ALL calls, not just the scoring tier.
409
424
  */
410
- async completeFast(prompt, maxTokens) {
411
- const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
412
- // #337: ONE ceiling for every phase (llmTimeoutMs, default 900 s). The old
413
- // 600 s scoring special-case assumed fast-tier calls are short — but the
414
- // ceiling only ever mattered when the endpoint was wedged, and a wedged
415
- // endpoint wedges scoring too. One knob, one place.
416
- const result = await this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
425
+ async complete(prompt) {
426
+ const tokens = this.config.maxTokens ?? 8192;
427
+ // #337: ONE ceiling for every phase (llmTimeoutMs, default 900 s).
428
+ const result = await this.completeWithRetry(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
417
429
  const flushEvery = this.config.ollamaFlushEvery ?? 0;
418
430
  if (this.config.provider === "ollama" && flushEvery > 0) {
419
431
  this.ollamaCallCount++;
@@ -424,46 +436,16 @@ class LlmClient {
424
436
  }
425
437
  return result;
426
438
  }
427
- /**
428
- * Reflect-tier completion (nightly reflection). One model serves all phases
429
- * (#231) — this is a thin wrapper kept for call-site readability.
430
- */
431
- async completeReflect(prompt, maxTokens) {
432
- const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
433
- return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
434
- }
435
- /**
436
- * Distillation-tier completion (session knowledge extraction). One model
437
- * serves all phases (#231) — thin wrapper kept for call-site readability.
438
- */
439
- async completeDistill(prompt, maxTokens) {
440
- const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
441
- return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
442
- }
443
- /**
444
- * Classification-tier completion (verdicts + tag/type classification). One
445
- * model serves all phases (#231) — thin wrapper kept for call-site
446
- * readability, but the tier keeps its OWN output ceiling (#391).
447
- */
448
- async completeClassify(prompt, maxTokens) {
449
- // #391: classify-tier ceiling — the call sites' old hardcoded caps
450
- // (64/32/20, tuned for a local non-reasoning model) starved reasoning
451
- // models whose internal thinking consumed the whole budget, leaving
452
- // verdicts empty. Deliberately NOT this.config.maxTokens: that knob
453
- // governs the heavy phases; this tier has its own (a ceiling, not a
454
- // target — generation still stops at the model's natural end).
455
- const tokens = maxTokens ?? this.config.classifyMaxTokens ?? 1024;
456
- return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
457
- }
458
439
  /**
459
440
  * Readiness probe (#337): ONE minimal generation request (max output 1
460
441
  * token) through the normal provider dispatch. Asks the question liveness
461
442
  * checks CANNOT: "can this endpoint GENERATE right now?" — the incident
462
443
  * gateway kept answering /v1/models for hours while every completion hung.
463
444
  * Single attempt: no retry ladder (a dead endpoint must cost one fast
464
- * failure, not a 3.5-min ladder), and it never accrues to the circuit
445
+ * failure, not a ladder), and it never accrues to the circuit
465
446
  * breaker (probing is diagnosis, not traffic). Catch-all → false; the
466
447
  * callers translate that into "endpoint_down" / a 503, never an exception.
448
+ * #405: owned by the DAEMON's /distill gate — the nightly no longer probes.
467
449
  */
468
450
  async probe(timeoutMs) {
469
451
  try {
@@ -475,22 +457,25 @@ class LlmClient {
475
457
  return false;
476
458
  }
477
459
  }
478
- async complete(model, prompt, maxTokens, timeoutMs) {
460
+ async completeWithRetry(model, prompt, maxTokens, timeoutMs) {
479
461
  // Breaker BEFORE anything else (#337) — an open breaker must cost zero
480
462
  // network I/O and zero ladder time. Past the cooldown we fall through:
481
463
  // this call IS the trial.
482
464
  const breakerSt = breakerByEndpoint.get(this.endpointKey);
483
465
  if (breakerSt !== undefined && breakerSt.openedAt !== null) {
484
466
  const elapsed = Date.now() - breakerSt.openedAt;
485
- const cooldownMs = this.config.breakerCooldownMs ?? 600_000;
486
- if (elapsed < cooldownMs) {
487
- throw new LlmCircuitOpenError(this.endpointKey, cooldownMs - elapsed);
467
+ if (elapsed < BREAKER_COOLDOWN_MS) {
468
+ throw new LlmCircuitOpenError(this.endpointKey, BREAKER_COOLDOWN_MS - elapsed);
488
469
  }
489
470
  }
490
471
  if (this.isRateLimited) {
491
472
  throw new RateLimitError(this.rateLimitedUntil - Date.now());
492
473
  }
493
- const retryDelays = [30_000, 60_000, 120_000]; // 30s, 60s, 120s
474
+ // #405: retry ladder collapsed to ONE 60 s retry (was 30/60/120×3 — the
475
+ // 63.5-min/call worst case from the #403 inventory). Worst-case logical
476
+ // call: 2 × timeoutMs + 60 s ≈ 31 min at the 900 s ceiling. A second
477
+ // consecutive total failure is breaker evidence, not a reason to wait.
478
+ const retryDelays = [60_000];
494
479
  let lastErr;
495
480
  for (let attempt = 0; attempt <= retryDelays.length; attempt++) {
496
481
  try {
@@ -550,13 +535,14 @@ class LlmClient {
550
535
  }
551
536
  }
552
537
  /** The single-flight wait budget for a call with ceiling `timeoutMs`.
553
- * An explicit `llmSingleFlightWaitMs` wins; otherwise the default is
554
- * max(900 s, llmTimeoutMs) so a waiter never gives up before a legitimate
555
- * in-flight call's own (possibly raised) ceiling expires (2nd-review
556
- * finding 2 — a hardcoded 900 s made a raised-timeout install treat a
557
- * healthy-busy endpoint as down). */
538
+ * #405: always DERIVED — max(900 s, llmTimeoutMs) — so a waiter never
539
+ * gives up before a legitimate in-flight call's own (possibly raised)
540
+ * ceiling expires (2nd-review finding 2 — a hardcoded 900 s made a
541
+ * raised-timeout install treat a healthy-busy endpoint as down). The
542
+ * llmSingleFlightWaitMs knob is deleted (its only documented use was to
543
+ * restore this exact derivation). */
558
544
  flightWaitMs(timeoutMs) {
559
- return this.config.singleFlightWaitMs ?? Math.max(900_000, timeoutMs);
545
+ return Math.max(900_000, timeoutMs);
560
546
  }
561
547
  /** Resolve the flight guard for this call, or undefined when disabled.
562
548
  * `staleMs` is derived from THIS call's timeout ceiling (≥ the 30-min
@@ -42,6 +42,18 @@ export declare function createMcpServer(): McpServer;
42
42
  * at each step; invalid/absent falls through.
43
43
  */
44
44
  export declare function resolveBodyLimitMb(configVal: unknown, hostedMode: boolean): number;
45
+ /**
46
+ * Resolve the OPTIONAL /search relevance floor (`minSimilarity` query param,
47
+ * #409 console polish). Pure — exported for tests. Absent/blank/invalid →
48
+ * undefined = NO gate (byte-identical to every pre-existing caller: agents,
49
+ * plugins, MCP tools never send the param). A finite number in [0, 1] → that
50
+ * floor, clamped into range so a hostile `?minSimilarity=42` cannot widen or
51
+ * invert the gate. Applied AFTER retrieve() with the exported recall gate
52
+ * (passesRelevanceGate: FTS/`both` hits pass regardless — a token match is
53
+ * real evidence; vector-only hits must clear the floor) — the same post-hoc
54
+ * shape /recall-index uses, so the two recall surfaces gate identically.
55
+ */
56
+ export declare function resolveSearchSimilarityFloor(raw: unknown): number | undefined;
45
57
  /**
46
58
  * Express error middleware (#7): translate express.json's default HTML 413
47
59
  * (entity.too.large) into a consistent JSON response. Catches body-parser