@gamaze/hicortex 0.20.7 → 0.20.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -41
- package/assets/dashboard.html +3989 -836
- package/dist/calibration.d.ts +293 -0
- package/dist/calibration.js +379 -0
- package/dist/capture-health.d.ts +87 -0
- package/dist/capture-health.js +106 -0
- package/dist/capture-pause.d.ts +86 -0
- package/dist/capture-pause.js +127 -0
- package/dist/capture.d.ts +24 -3
- package/dist/capture.js +11 -1
- package/dist/classify-domains.d.ts +6 -0
- package/dist/classify-domains.js +7 -1
- package/dist/cli.js +38 -3
- package/dist/config-read.d.ts +1 -1
- package/dist/config-read.js +96 -9
- package/dist/consolidate.d.ts +114 -68
- package/dist/consolidate.js +302 -182
- package/dist/dashboard.d.ts +326 -6
- package/dist/dashboard.js +592 -7
- package/dist/db.js +105 -0
- package/dist/dedup.d.ts +34 -26
- package/dist/dedup.js +91 -57
- package/dist/distiller.js +1 -1
- package/dist/domain-classify.d.ts +7 -6
- package/dist/domain-classify.js +12 -10
- package/dist/eval/decay-eval.d.ts +3 -3
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/importance-eval.d.ts +85 -0
- package/dist/eval/importance-eval.js +286 -0
- package/dist/eval/planted-eval.d.ts +26 -0
- package/dist/eval/planted-eval.js +97 -0
- package/dist/eval/planted-fixtures.d.ts +107 -0
- package/dist/eval/planted-fixtures.js +283 -0
- package/dist/eval/planted-harness.d.ts +176 -0
- package/dist/eval/planted-harness.js +649 -0
- package/dist/eval/ranking-battery.d.ts +78 -0
- package/dist/eval/ranking-battery.js +181 -0
- package/dist/eval/ranking-eval.d.ts +41 -0
- package/dist/eval/ranking-eval.js +391 -0
- package/dist/eval/ranking-fixtures.d.ts +77 -0
- package/dist/eval/ranking-fixtures.js +226 -0
- package/dist/identity-store.d.ts +21 -0
- package/dist/identity-store.js +49 -0
- package/dist/index.js +4 -3
- package/dist/init.d.ts +23 -3
- package/dist/init.js +84 -9
- package/dist/llm.d.ts +43 -58
- package/dist/llm.js +87 -101
- package/dist/mcp-server.d.ts +12 -0
- package/dist/mcp-server.js +213 -32
- package/dist/nightly.d.ts +9 -1
- package/dist/nightly.js +164 -110
- package/dist/nofit.d.ts +4 -11
- package/dist/nofit.js +6 -23
- package/dist/prompts.d.ts +10 -0
- package/dist/prompts.js +28 -5
- package/dist/recall-index.d.ts +30 -28
- package/dist/recall-index.js +21 -18
- package/dist/recall-registry.d.ts +2 -1
- package/dist/recall-registry.js +35 -1
- package/dist/reconsolidation.d.ts +168 -87
- package/dist/reconsolidation.js +818 -377
- package/dist/relink.js +3 -4
- package/dist/rescore-importance.d.ts +80 -0
- package/dist/rescore-importance.js +236 -0
- package/dist/retrieval.d.ts +80 -35
- package/dist/retrieval.js +322 -105
- package/dist/run-deadline.d.ts +62 -0
- package/dist/run-deadline.js +73 -0
- package/dist/schema-prototypes.d.ts +3 -3
- package/dist/schema-prototypes.js +3 -3
- package/dist/stages.d.ts +37 -0
- package/dist/stages.js +51 -0
- package/dist/state.d.ts +34 -9
- package/dist/storage.d.ts +50 -18
- package/dist/storage.js +125 -30
- package/dist/telemetry.d.ts +8 -7
- package/dist/token-budget.js +3 -4
- package/dist/type-classify.js +4 -4
- package/dist/types.d.ts +143 -155
- package/domains.example.json +4 -5
- package/hermes-plugin/hicortex/README.md +2 -2
- package/openclaw.plugin.json +1 -1
- package/package.json +4 -1
- package/pi-extension/hicortex/README.md +1 -1
- package/server.json +3 -3
package/dist/llm.d.ts
CHANGED
|
@@ -23,29 +23,23 @@ export interface LlmConfig {
|
|
|
23
23
|
provider: string;
|
|
24
24
|
/** Max output tokens for all phases (one model). Default 8192. */
|
|
25
25
|
maxTokens?: number;
|
|
26
|
-
/** Max output tokens for the CLASSIFY tier only — the short JSON-verdict
|
|
27
|
-
* calls (correction/supersession verdicts, rewrite contracts, type + tag
|
|
28
|
-
* classification). Default 1024. See HicortexConfig.classifyMaxTokens (#391). */
|
|
29
|
-
classifyMaxTokens?: number;
|
|
30
26
|
/** Toggle thinking on the openai-compat path for all phases. Absent = no kwarg sent.
|
|
31
27
|
* LOCAL-endpoint only (ollama / mlx-lm gateway); see HicortexConfig.enableThinking. */
|
|
32
28
|
enableThinking?: boolean;
|
|
33
|
-
/** Context window for ollama (the one model, all phases). Default 8192.
|
|
29
|
+
/** Context window for ollama (the one model, all phases). Default 8192.
|
|
30
|
+
* #408 diagnostic tier: resolved from HICORTEX_NUM_CTX (env) by
|
|
31
|
+
* applyTierTuningOverlay — never a config key. */
|
|
34
32
|
numCtx?: number;
|
|
35
|
-
/** Flush ollama memory every N ollama calls (0 = off).
|
|
33
|
+
/** Flush ollama memory every N ollama calls (0 = off). #408 diagnostic
|
|
34
|
+
* tier: resolved from HICORTEX_OLLAMA_FLUSH_EVERY by the overlay. */
|
|
36
35
|
ollamaFlushEvery?: number;
|
|
37
|
-
/** Ms to wait after an ollama flush for the runner to release.
|
|
36
|
+
/** Ms to wait after an ollama flush for the runner to release. #408
|
|
37
|
+
* diagnostic tier: resolved from HICORTEX_OLLAMA_FLUSH_WAIT_MS. */
|
|
38
38
|
ollamaFlushWaitMs?: number;
|
|
39
39
|
/** ONE per-call timeout ceiling for every phase (#337). Default 900000 — the
|
|
40
|
-
* AbortSignal.timeout value passed by
|
|
40
|
+
* AbortSignal.timeout value passed by the single complete() surface (the old
|
|
41
41
|
* 600000 scoring special-case is gone). See HicortexConfig.llmTimeoutMs. */
|
|
42
42
|
timeoutMs?: number;
|
|
43
|
-
/** Consecutive ladder-exhausted total failures before the circuit breaker
|
|
44
|
-
* opens (#337). Default 3; 0 disables. See HicortexConfig.llmBreakerThreshold. */
|
|
45
|
-
breakerThreshold?: number;
|
|
46
|
-
/** How long an OPEN breaker stays open before the next call becomes a trial
|
|
47
|
-
* (#337). Default 600000. See HicortexConfig.llmBreakerCooldownMs. */
|
|
48
|
-
breakerCooldownMs?: number;
|
|
49
43
|
/** Timeout for the readiness probe's single generation attempt (#337).
|
|
50
44
|
* Default 60000. See HicortexConfig.llmProbeTimeoutMs. */
|
|
51
45
|
probeTimeoutMs?: number;
|
|
@@ -57,10 +51,6 @@ export interface LlmConfig {
|
|
|
57
51
|
* single-user model servers (two concurrent large-context calls stall/OOM
|
|
58
52
|
* the server and the machine under it). See HicortexConfig.llmSingleFlight. */
|
|
59
53
|
singleFlight?: boolean;
|
|
60
|
-
/** How long a queued call waits for the in-flight call before failing as
|
|
61
|
-
* endpoint-down (#355). Default 900000 — the same ceiling as llmTimeoutMs.
|
|
62
|
-
* See HicortexConfig.llmSingleFlightWaitMs. */
|
|
63
|
-
singleFlightWaitMs?: number;
|
|
64
54
|
}
|
|
65
55
|
/**
|
|
66
56
|
* Resolve LLM configuration from explicit config-file overrides or
|
|
@@ -85,20 +75,25 @@ export declare function resolveExplicitLlmConfig(overrides?: {
|
|
|
85
75
|
*/
|
|
86
76
|
export declare const resolveLlmConfigForCC: typeof resolveExplicitLlmConfig;
|
|
87
77
|
/**
|
|
88
|
-
* Validate + copy the tuning keys (#220: maxTokens + enableThinking
|
|
89
|
-
*
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
* runs
|
|
93
|
-
*
|
|
78
|
+
* Validate + copy the tuning keys (#220: maxTokens + enableThinking) from the
|
|
79
|
+
* saved disk config onto a runtime LlmConfig, and resolve the DIAGNOSTIC
|
|
80
|
+
* TIER (#408: numCtx + the ollama-flush family) from the calibration env
|
|
81
|
+
* resolvers. Called by BOTH LlmConfig construction sites — the daemon in
|
|
82
|
+
* mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
|
|
83
|
+
* runs reflect + classify) — so this is the ONE resolution point per process
|
|
84
|
+
* for every one of these values, and a future site calling this inherits
|
|
85
|
+
* them by construction.
|
|
94
86
|
*
|
|
95
|
-
*
|
|
96
|
-
*
|
|
97
|
-
*
|
|
98
|
-
* readStrictBoolean / readNonNegativeConfig) — notably a JSON slip
|
|
87
|
+
* Config keys are optional; absent = call-site defaults (maxTokens 8192,
|
|
88
|
+
* thinking kwarg omitted). Wrong-typed values warn and are dropped
|
|
89
|
+
* (readPositiveConfig / readStrictBoolean) — notably a JSON slip
|
|
99
90
|
* `"enableThinking": "false"` (string) is rejected rather than coerced to
|
|
100
91
|
* truthy thinking-on, which would silently invert the fix this key exists to
|
|
101
|
-
* apply.
|
|
92
|
+
* apply. #405: classifyMaxTokens is deleted — maxTokens is the single output
|
|
93
|
+
* ceiling for every call (warned as ignored at the config-read boundary).
|
|
94
|
+
* #408: numCtx / ollamaFlushEvery / ollamaFlushWaitMs are NO LONGER config
|
|
95
|
+
* keys — the env tier always resolves them (env > calibration constant;
|
|
96
|
+
* invalid env warns + falls back), independent of `savedConfig`.
|
|
102
97
|
*/
|
|
103
98
|
export declare function applyTierTuningOverlay(llmConfig: LlmConfig, savedConfig: Record<string, unknown> | null | undefined): void;
|
|
104
99
|
/**
|
|
@@ -200,48 +195,38 @@ export declare class LlmClient {
|
|
|
200
195
|
private resetBreaker;
|
|
201
196
|
private handleRateLimit;
|
|
202
197
|
/**
|
|
203
|
-
*
|
|
204
|
-
*
|
|
205
|
-
*
|
|
206
|
-
*
|
|
207
|
-
*
|
|
198
|
+
* The ONE completion surface (#405). Every phase — distill, reflect,
|
|
199
|
+
* classify, scoring — calls this; maxTokens (default 8192) and timeoutMs
|
|
200
|
+
* (default 900 s) resolve from config INSIDE, so no call site can starve a
|
|
201
|
+
* phase with a per-site ceiling (#391's failure direction: hardcoded
|
|
202
|
+
* 16/64/256-token caps starved reasoning models whose internal thinking
|
|
203
|
+
* ate the whole output budget, leaving verdicts empty). numCtx +
|
|
204
|
+
* enableThinking are likewise read inside completeOnce's per-provider
|
|
205
|
+
* dispatch. The periodic ollama flush survives here (provider-gated) — it
|
|
206
|
+
* now counts ALL calls, not just the scoring tier.
|
|
208
207
|
*/
|
|
209
|
-
|
|
210
|
-
/**
|
|
211
|
-
* Reflect-tier completion (nightly reflection). One model serves all phases
|
|
212
|
-
* (#231) — this is a thin wrapper kept for call-site readability.
|
|
213
|
-
*/
|
|
214
|
-
completeReflect(prompt: string, maxTokens?: number): Promise<LlmResult>;
|
|
215
|
-
/**
|
|
216
|
-
* Distillation-tier completion (session knowledge extraction). One model
|
|
217
|
-
* serves all phases (#231) — thin wrapper kept for call-site readability.
|
|
218
|
-
*/
|
|
219
|
-
completeDistill(prompt: string, maxTokens?: number): Promise<LlmResult>;
|
|
220
|
-
/**
|
|
221
|
-
* Classification-tier completion (verdicts + tag/type classification). One
|
|
222
|
-
* model serves all phases (#231) — thin wrapper kept for call-site
|
|
223
|
-
* readability, but the tier keeps its OWN output ceiling (#391).
|
|
224
|
-
*/
|
|
225
|
-
completeClassify(prompt: string, maxTokens?: number): Promise<LlmResult>;
|
|
208
|
+
complete(prompt: string): Promise<LlmResult>;
|
|
226
209
|
/**
|
|
227
210
|
* Readiness probe (#337): ONE minimal generation request (max output 1
|
|
228
211
|
* token) through the normal provider dispatch. Asks the question liveness
|
|
229
212
|
* checks CANNOT: "can this endpoint GENERATE right now?" — the incident
|
|
230
213
|
* gateway kept answering /v1/models for hours while every completion hung.
|
|
231
214
|
* Single attempt: no retry ladder (a dead endpoint must cost one fast
|
|
232
|
-
* failure, not a
|
|
215
|
+
* failure, not a ladder), and it never accrues to the circuit
|
|
233
216
|
* breaker (probing is diagnosis, not traffic). Catch-all → false; the
|
|
234
217
|
* callers translate that into "endpoint_down" / a 503, never an exception.
|
|
218
|
+
* #405: owned by the DAEMON's /distill gate — the nightly no longer probes.
|
|
235
219
|
*/
|
|
236
220
|
probe(timeoutMs?: number): Promise<boolean>;
|
|
237
|
-
private
|
|
221
|
+
private completeWithRetry;
|
|
238
222
|
private completeOnce;
|
|
239
223
|
/** The single-flight wait budget for a call with ceiling `timeoutMs`.
|
|
240
|
-
*
|
|
241
|
-
*
|
|
242
|
-
*
|
|
243
|
-
*
|
|
244
|
-
*
|
|
224
|
+
* #405: always DERIVED — max(900 s, llmTimeoutMs) — so a waiter never
|
|
225
|
+
* gives up before a legitimate in-flight call's own (possibly raised)
|
|
226
|
+
* ceiling expires (2nd-review finding 2 — a hardcoded 900 s made a
|
|
227
|
+
* raised-timeout install treat a healthy-busy endpoint as down). The
|
|
228
|
+
* llmSingleFlightWaitMs knob is deleted (its only documented use was to
|
|
229
|
+
* restore this exact derivation). */
|
|
245
230
|
private flightWaitMs;
|
|
246
231
|
/** Resolve the flight guard for this call, or undefined when disabled.
|
|
247
232
|
* `staleMs` is derived from THIS call's timeout ceiling (≥ the 30-min
|
package/dist/llm.js
CHANGED
|
@@ -27,6 +27,9 @@ exports.claudeCliConfig = claudeCliConfig;
|
|
|
27
27
|
exports.probeOllama = probeOllama;
|
|
28
28
|
exports.isTotalFailure = isTotalFailure;
|
|
29
29
|
const config_read_js_1 = require("./config-read.js");
|
|
30
|
+
// #408: the diagnostic tier (numCtx + ollama flush) resolves from the
|
|
31
|
+
// calibration env resolvers — the config keys are gone from the surface.
|
|
32
|
+
const calibration_js_1 = require("./calibration.js");
|
|
30
33
|
// #355: single-flight guard + canonical home (where the per-endpoint flight
|
|
31
34
|
// lock files live — the daemon and the nightly share the home, so the lock
|
|
32
35
|
// serializes them across processes).
|
|
@@ -86,57 +89,49 @@ function resolveExplicitLlmConfig(overrides) {
|
|
|
86
89
|
*/
|
|
87
90
|
exports.resolveLlmConfigForCC = resolveExplicitLlmConfig;
|
|
88
91
|
/**
|
|
89
|
-
* Validate + copy the tuning keys (#220: maxTokens + enableThinking
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
* runs
|
|
94
|
-
*
|
|
92
|
+
* Validate + copy the tuning keys (#220: maxTokens + enableThinking) from the
|
|
93
|
+
* saved disk config onto a runtime LlmConfig, and resolve the DIAGNOSTIC
|
|
94
|
+
* TIER (#408: numCtx + the ollama-flush family) from the calibration env
|
|
95
|
+
* resolvers. Called by BOTH LlmConfig construction sites — the daemon in
|
|
96
|
+
* mcp-server.ts (runs distill) AND resolveSavedLlmConfig below (the nightly
|
|
97
|
+
* runs reflect + classify) — so this is the ONE resolution point per process
|
|
98
|
+
* for every one of these values, and a future site calling this inherits
|
|
99
|
+
* them by construction.
|
|
95
100
|
*
|
|
96
|
-
*
|
|
97
|
-
*
|
|
98
|
-
*
|
|
99
|
-
* readStrictBoolean / readNonNegativeConfig) — notably a JSON slip
|
|
101
|
+
* Config keys are optional; absent = call-site defaults (maxTokens 8192,
|
|
102
|
+
* thinking kwarg omitted). Wrong-typed values warn and are dropped
|
|
103
|
+
* (readPositiveConfig / readStrictBoolean) — notably a JSON slip
|
|
100
104
|
* `"enableThinking": "false"` (string) is rejected rather than coerced to
|
|
101
105
|
* truthy thinking-on, which would silently invert the fix this key exists to
|
|
102
|
-
* apply.
|
|
106
|
+
* apply. #405: classifyMaxTokens is deleted — maxTokens is the single output
|
|
107
|
+
* ceiling for every call (warned as ignored at the config-read boundary).
|
|
108
|
+
* #408: numCtx / ollamaFlushEvery / ollamaFlushWaitMs are NO LONGER config
|
|
109
|
+
* keys — the env tier always resolves them (env > calibration constant;
|
|
110
|
+
* invalid env warns + falls back), independent of `savedConfig`.
|
|
103
111
|
*/
|
|
104
112
|
function applyTierTuningOverlay(llmConfig, savedConfig) {
|
|
113
|
+
// #408 diagnostic tier — resolved BEFORE the config early-return so the
|
|
114
|
+
// tier applies on every construction site call, config or not.
|
|
115
|
+
llmConfig.numCtx = (0, calibration_js_1.resolveNumCtx)();
|
|
116
|
+
llmConfig.ollamaFlushEvery = (0, calibration_js_1.resolveOllamaFlushEvery)();
|
|
117
|
+
llmConfig.ollamaFlushWaitMs = (0, calibration_js_1.resolveOllamaFlushWaitMs)();
|
|
105
118
|
if (!savedConfig)
|
|
106
119
|
return;
|
|
107
120
|
if (savedConfig.maxTokens !== undefined) {
|
|
108
121
|
llmConfig.maxTokens = (0, config_read_js_1.readPositiveConfig)(savedConfig, "maxTokens", 8192);
|
|
109
122
|
}
|
|
110
|
-
if (savedConfig.classifyMaxTokens !== undefined) {
|
|
111
|
-
llmConfig.classifyMaxTokens = (0, config_read_js_1.readPositiveConfig)(savedConfig, "classifyMaxTokens", 1024);
|
|
112
|
-
}
|
|
113
123
|
const thinking = (0, config_read_js_1.readStrictBoolean)(savedConfig, "enableThinking");
|
|
114
124
|
if (thinking !== undefined) {
|
|
115
125
|
llmConfig.enableThinking = thinking;
|
|
116
126
|
}
|
|
117
|
-
if (savedConfig.numCtx !== undefined) {
|
|
118
|
-
llmConfig.numCtx = (0, config_read_js_1.readPositiveConfig)(savedConfig, "numCtx", 8192);
|
|
119
|
-
}
|
|
120
|
-
if (savedConfig.ollamaFlushEvery !== undefined) {
|
|
121
|
-
llmConfig.ollamaFlushEvery = (0, config_read_js_1.readNonNegativeConfig)(savedConfig, "ollamaFlushEvery", 0);
|
|
122
|
-
}
|
|
123
|
-
if (savedConfig.ollamaFlushWaitMs !== undefined) {
|
|
124
|
-
llmConfig.ollamaFlushWaitMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "ollamaFlushWaitMs", 180000);
|
|
125
|
-
}
|
|
126
127
|
// #337 resilience knobs. Same boundary discipline as the keys above: absent =
|
|
127
|
-
// call-site defaults (timeout 900 s,
|
|
128
|
-
//
|
|
129
|
-
//
|
|
130
|
-
//
|
|
128
|
+
// call-site defaults (timeout 900 s, probe timeout 60 s, probe TTL 5 min),
|
|
129
|
+
// wrong-typed values warn and fall back. #405: the breaker threshold/
|
|
130
|
+
// cooldown and the single-flight wait are CONSTANTS now (no incident ever
|
|
131
|
+
// required tuning them) — the keys are warned as ignored by config-read.ts.
|
|
131
132
|
if (savedConfig.llmTimeoutMs !== undefined) {
|
|
132
133
|
llmConfig.timeoutMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmTimeoutMs", 900000);
|
|
133
134
|
}
|
|
134
|
-
if (savedConfig.llmBreakerThreshold !== undefined) {
|
|
135
|
-
llmConfig.breakerThreshold = (0, config_read_js_1.readNonNegativeConfig)(savedConfig, "llmBreakerThreshold", 3);
|
|
136
|
-
}
|
|
137
|
-
if (savedConfig.llmBreakerCooldownMs !== undefined) {
|
|
138
|
-
llmConfig.breakerCooldownMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmBreakerCooldownMs", 600000);
|
|
139
|
-
}
|
|
140
135
|
if (savedConfig.llmProbeTimeoutMs !== undefined) {
|
|
141
136
|
llmConfig.probeTimeoutMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmProbeTimeoutMs", 60000);
|
|
142
137
|
}
|
|
@@ -144,15 +139,12 @@ function applyTierTuningOverlay(llmConfig, savedConfig) {
|
|
|
144
139
|
llmConfig.probeTtlMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmProbeTtlMs", 300000);
|
|
145
140
|
}
|
|
146
141
|
// #355 single-flight. Default ON (a correctness property, not a tuning
|
|
147
|
-
// option); the
|
|
148
|
-
//
|
|
142
|
+
// option); the kill switch survives as config. #405: the wait budget is
|
|
143
|
+
// DERIVED (max(900 s, llmTimeoutMs)) — no longer a knob.
|
|
149
144
|
const singleFlight = (0, config_read_js_1.readStrictBoolean)(savedConfig, "llmSingleFlight");
|
|
150
145
|
if (singleFlight !== undefined) {
|
|
151
146
|
llmConfig.singleFlight = singleFlight;
|
|
152
147
|
}
|
|
153
|
-
if (savedConfig.llmSingleFlightWaitMs !== undefined) {
|
|
154
|
-
llmConfig.singleFlightWaitMs = (0, config_read_js_1.readPositiveConfig)(savedConfig, "llmSingleFlightWaitMs", 900000);
|
|
155
|
-
}
|
|
156
148
|
}
|
|
157
149
|
/**
|
|
158
150
|
* Resolve an LlmConfig from a saved ~/.hicortex/config.json object.
|
|
@@ -195,8 +187,9 @@ function resolveSavedLlmConfig(savedConfig, findBinary = findClaudeBinary) {
|
|
|
195
187
|
llmModel: savedConfig?.llmModel,
|
|
196
188
|
});
|
|
197
189
|
}
|
|
198
|
-
// Tuning overlay (#220: maxTokens + enableThinking
|
|
199
|
-
// at both construction sites (daemon + nightly) so every
|
|
190
|
+
// Tuning overlay (#220: maxTokens + enableThinking; #408: the diagnostic
|
|
191
|
+
// env tier). Applied at both construction sites (daemon + nightly) so every
|
|
192
|
+
// phase honors the same values.
|
|
200
193
|
if (llmConfig) {
|
|
201
194
|
applyTierTuningOverlay(llmConfig, savedConfig);
|
|
202
195
|
}
|
|
@@ -312,6 +305,27 @@ exports.RateLimitError = RateLimitError;
|
|
|
312
305
|
// (#231), so in practice there is a single client per process today; the
|
|
313
306
|
// module-level map keeps the state shared correctly if that ever changes.
|
|
314
307
|
const rateLimitedUntilByEndpoint = new Map();
|
|
308
|
+
// #337: per-endpoint circuit-breaker state, keyed like the rate-limit map.
|
|
309
|
+
// `failures` counts consecutive ladder-exhausted TOTAL failures (the class the
|
|
310
|
+
// retry ladder matches — fetch-failed/ECONNREFUSED/timeout/"Headers Timeout");
|
|
311
|
+
// HTTP errors WITH a response, parse errors, and RateLimitError throw before
|
|
312
|
+
// the ladder can be exhausted and never accrue. Any success resets. At
|
|
313
|
+
// BREAKER_THRESHOLD consecutive failures the breaker opens: calls then
|
|
314
|
+
// throw LlmCircuitOpenError with NO network I/O until the cooldown elapses,
|
|
315
|
+
// after which exactly one call is a trial (a failure re-opens, a success
|
|
316
|
+
// resets). This is what bounds a wedged endpoint to K ladder-exhausted calls
|
|
317
|
+
// instead of the ~200-call nightly retrying into it for hours (incident
|
|
318
|
+
// 2026-08-23/24). `openedAt` stays set (breakerOpen stays true) until a
|
|
319
|
+
// SUCCESS resets it — tripped-but-past-cooldown is still "not proven healthy".
|
|
320
|
+
//
|
|
321
|
+
// #405: threshold + cooldown are CONSTANTS — the config knobs
|
|
322
|
+
// (llmBreakerThreshold / llmBreakerCooldownMs) were deleted after the #403
|
|
323
|
+
// inventory found no incident that ever required tuning them. Worst-case
|
|
324
|
+
// dead-endpoint discovery: 3 logical calls × (1 timeout + 1 retry) ≈ 31 min,
|
|
325
|
+
// once per dead night, bounded by the run deadline (the nightly probe gate
|
|
326
|
+
// that used to catch this in 60 s is also gone — spec-accepted).
|
|
327
|
+
const BREAKER_THRESHOLD = 3;
|
|
328
|
+
const BREAKER_COOLDOWN_MS = 600_000;
|
|
315
329
|
const breakerByEndpoint = new Map();
|
|
316
330
|
/** Thrown when the per-endpoint circuit breaker is open (#337) — no network I/O happened. */
|
|
317
331
|
class LlmCircuitOpenError extends Error {
|
|
@@ -364,20 +378,17 @@ class LlmClient {
|
|
|
364
378
|
}
|
|
365
379
|
/** Record one ladder-exhausted TOTAL failure; open the breaker at threshold. */
|
|
366
380
|
recordBreakerFailure() {
|
|
367
|
-
const threshold = this.config.breakerThreshold ?? 3;
|
|
368
|
-
if (threshold <= 0)
|
|
369
|
-
return; // 0 disables — never open, never fast-fail
|
|
370
381
|
const st = breakerByEndpoint.get(this.endpointKey) ?? { failures: 0, openedAt: null };
|
|
371
382
|
st.failures += 1;
|
|
372
|
-
if (st.failures >=
|
|
383
|
+
if (st.failures >= BREAKER_THRESHOLD) {
|
|
373
384
|
// (Re)open. Re-opening (a failed trial past cooldown) restarts the
|
|
374
385
|
// cooldown window from NOW — the endpoint just proved itself still dead.
|
|
375
386
|
st.openedAt = Date.now();
|
|
376
387
|
// One structured line per opening — the runbook's grep target. Same
|
|
377
388
|
// key=value style as event=budget_exhausted (consolidate.ts).
|
|
378
389
|
console.warn(`[hicortex] event=circuit_open endpoint=${this.endpointKey} ` +
|
|
379
|
-
`failures=${st.failures} threshold=${
|
|
380
|
-
`cooldown_ms=${
|
|
390
|
+
`failures=${st.failures} threshold=${BREAKER_THRESHOLD} ` +
|
|
391
|
+
`cooldown_ms=${BREAKER_COOLDOWN_MS}`);
|
|
381
392
|
}
|
|
382
393
|
breakerByEndpoint.set(this.endpointKey, st);
|
|
383
394
|
}
|
|
@@ -401,19 +412,20 @@ class LlmClient {
|
|
|
401
412
|
throw new RateLimitError(retryMs);
|
|
402
413
|
}
|
|
403
414
|
/**
|
|
404
|
-
*
|
|
405
|
-
*
|
|
406
|
-
*
|
|
407
|
-
*
|
|
408
|
-
*
|
|
415
|
+
* The ONE completion surface (#405). Every phase — distill, reflect,
|
|
416
|
+
* classify, scoring — calls this; maxTokens (default 8192) and timeoutMs
|
|
417
|
+
* (default 900 s) resolve from config INSIDE, so no call site can starve a
|
|
418
|
+
* phase with a per-site ceiling (#391's failure direction: hardcoded
|
|
419
|
+
* 16/64/256-token caps starved reasoning models whose internal thinking
|
|
420
|
+
* ate the whole output budget, leaving verdicts empty). numCtx +
|
|
421
|
+
* enableThinking are likewise read inside completeOnce's per-provider
|
|
422
|
+
* dispatch. The periodic ollama flush survives here (provider-gated) — it
|
|
423
|
+
* now counts ALL calls, not just the scoring tier.
|
|
409
424
|
*/
|
|
410
|
-
async
|
|
411
|
-
const tokens =
|
|
412
|
-
// #337: ONE ceiling for every phase (llmTimeoutMs, default 900 s).
|
|
413
|
-
|
|
414
|
-
// ceiling only ever mattered when the endpoint was wedged, and a wedged
|
|
415
|
-
// endpoint wedges scoring too. One knob, one place.
|
|
416
|
-
const result = await this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
|
|
425
|
+
async complete(prompt) {
|
|
426
|
+
const tokens = this.config.maxTokens ?? 8192;
|
|
427
|
+
// #337: ONE ceiling for every phase (llmTimeoutMs, default 900 s).
|
|
428
|
+
const result = await this.completeWithRetry(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
|
|
417
429
|
const flushEvery = this.config.ollamaFlushEvery ?? 0;
|
|
418
430
|
if (this.config.provider === "ollama" && flushEvery > 0) {
|
|
419
431
|
this.ollamaCallCount++;
|
|
@@ -424,46 +436,16 @@ class LlmClient {
|
|
|
424
436
|
}
|
|
425
437
|
return result;
|
|
426
438
|
}
|
|
427
|
-
/**
|
|
428
|
-
* Reflect-tier completion (nightly reflection). One model serves all phases
|
|
429
|
-
* (#231) — this is a thin wrapper kept for call-site readability.
|
|
430
|
-
*/
|
|
431
|
-
async completeReflect(prompt, maxTokens) {
|
|
432
|
-
const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
|
|
433
|
-
return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
|
|
434
|
-
}
|
|
435
|
-
/**
|
|
436
|
-
* Distillation-tier completion (session knowledge extraction). One model
|
|
437
|
-
* serves all phases (#231) — thin wrapper kept for call-site readability.
|
|
438
|
-
*/
|
|
439
|
-
async completeDistill(prompt, maxTokens) {
|
|
440
|
-
const tokens = maxTokens ?? this.config.maxTokens ?? 8192;
|
|
441
|
-
return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
|
|
442
|
-
}
|
|
443
|
-
/**
|
|
444
|
-
* Classification-tier completion (verdicts + tag/type classification). One
|
|
445
|
-
* model serves all phases (#231) — thin wrapper kept for call-site
|
|
446
|
-
* readability, but the tier keeps its OWN output ceiling (#391).
|
|
447
|
-
*/
|
|
448
|
-
async completeClassify(prompt, maxTokens) {
|
|
449
|
-
// #391: classify-tier ceiling — the call sites' old hardcoded caps
|
|
450
|
-
// (64/32/20, tuned for a local non-reasoning model) starved reasoning
|
|
451
|
-
// models whose internal thinking consumed the whole budget, leaving
|
|
452
|
-
// verdicts empty. Deliberately NOT this.config.maxTokens: that knob
|
|
453
|
-
// governs the heavy phases; this tier has its own (a ceiling, not a
|
|
454
|
-
// target — generation still stops at the model's natural end).
|
|
455
|
-
const tokens = maxTokens ?? this.config.classifyMaxTokens ?? 1024;
|
|
456
|
-
return this.complete(this.config.model, prompt, tokens, this.config.timeoutMs ?? 900_000);
|
|
457
|
-
}
|
|
458
439
|
/**
|
|
459
440
|
* Readiness probe (#337): ONE minimal generation request (max output 1
|
|
460
441
|
* token) through the normal provider dispatch. Asks the question liveness
|
|
461
442
|
* checks CANNOT: "can this endpoint GENERATE right now?" — the incident
|
|
462
443
|
* gateway kept answering /v1/models for hours while every completion hung.
|
|
463
444
|
* Single attempt: no retry ladder (a dead endpoint must cost one fast
|
|
464
|
-
* failure, not a
|
|
445
|
+
* failure, not a ladder), and it never accrues to the circuit
|
|
465
446
|
* breaker (probing is diagnosis, not traffic). Catch-all → false; the
|
|
466
447
|
* callers translate that into "endpoint_down" / a 503, never an exception.
|
|
448
|
+
* #405: owned by the DAEMON's /distill gate — the nightly no longer probes.
|
|
467
449
|
*/
|
|
468
450
|
async probe(timeoutMs) {
|
|
469
451
|
try {
|
|
@@ -475,22 +457,25 @@ class LlmClient {
|
|
|
475
457
|
return false;
|
|
476
458
|
}
|
|
477
459
|
}
|
|
478
|
-
async
|
|
460
|
+
async completeWithRetry(model, prompt, maxTokens, timeoutMs) {
|
|
479
461
|
// Breaker BEFORE anything else (#337) — an open breaker must cost zero
|
|
480
462
|
// network I/O and zero ladder time. Past the cooldown we fall through:
|
|
481
463
|
// this call IS the trial.
|
|
482
464
|
const breakerSt = breakerByEndpoint.get(this.endpointKey);
|
|
483
465
|
if (breakerSt !== undefined && breakerSt.openedAt !== null) {
|
|
484
466
|
const elapsed = Date.now() - breakerSt.openedAt;
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
throw new LlmCircuitOpenError(this.endpointKey, cooldownMs - elapsed);
|
|
467
|
+
if (elapsed < BREAKER_COOLDOWN_MS) {
|
|
468
|
+
throw new LlmCircuitOpenError(this.endpointKey, BREAKER_COOLDOWN_MS - elapsed);
|
|
488
469
|
}
|
|
489
470
|
}
|
|
490
471
|
if (this.isRateLimited) {
|
|
491
472
|
throw new RateLimitError(this.rateLimitedUntil - Date.now());
|
|
492
473
|
}
|
|
493
|
-
|
|
474
|
+
// #405: retry ladder collapsed to ONE 60 s retry (was 30/60/120×3 — the
|
|
475
|
+
// 63.5-min/call worst case from the #403 inventory). Worst-case logical
|
|
476
|
+
// call: 2 × timeoutMs + 60 s ≈ 31 min at the 900 s ceiling. A second
|
|
477
|
+
// consecutive total failure is breaker evidence, not a reason to wait.
|
|
478
|
+
const retryDelays = [60_000];
|
|
494
479
|
let lastErr;
|
|
495
480
|
for (let attempt = 0; attempt <= retryDelays.length; attempt++) {
|
|
496
481
|
try {
|
|
@@ -550,13 +535,14 @@ class LlmClient {
|
|
|
550
535
|
}
|
|
551
536
|
}
|
|
552
537
|
/** The single-flight wait budget for a call with ceiling `timeoutMs`.
|
|
553
|
-
*
|
|
554
|
-
*
|
|
555
|
-
*
|
|
556
|
-
*
|
|
557
|
-
*
|
|
538
|
+
* #405: always DERIVED — max(900 s, llmTimeoutMs) — so a waiter never
|
|
539
|
+
* gives up before a legitimate in-flight call's own (possibly raised)
|
|
540
|
+
* ceiling expires (2nd-review finding 2 — a hardcoded 900 s made a
|
|
541
|
+
* raised-timeout install treat a healthy-busy endpoint as down). The
|
|
542
|
+
* llmSingleFlightWaitMs knob is deleted (its only documented use was to
|
|
543
|
+
* restore this exact derivation). */
|
|
558
544
|
flightWaitMs(timeoutMs) {
|
|
559
|
-
return
|
|
545
|
+
return Math.max(900_000, timeoutMs);
|
|
560
546
|
}
|
|
561
547
|
/** Resolve the flight guard for this call, or undefined when disabled.
|
|
562
548
|
* `staleMs` is derived from THIS call's timeout ceiling (≥ the 30-min
|
package/dist/mcp-server.d.ts
CHANGED
|
@@ -42,6 +42,18 @@ export declare function createMcpServer(): McpServer;
|
|
|
42
42
|
* at each step; invalid/absent falls through.
|
|
43
43
|
*/
|
|
44
44
|
export declare function resolveBodyLimitMb(configVal: unknown, hostedMode: boolean): number;
|
|
45
|
+
/**
|
|
46
|
+
* Resolve the OPTIONAL /search relevance floor (`minSimilarity` query param,
|
|
47
|
+
* #409 console polish). Pure — exported for tests. Absent/blank/invalid →
|
|
48
|
+
* undefined = NO gate (byte-identical to every pre-existing caller: agents,
|
|
49
|
+
* plugins, MCP tools never send the param). A finite number in [0, 1] → that
|
|
50
|
+
* floor, clamped into range so a hostile `?minSimilarity=42` cannot widen or
|
|
51
|
+
* invert the gate. Applied AFTER retrieve() with the exported recall gate
|
|
52
|
+
* (passesRelevanceGate: FTS/`both` hits pass regardless — a token match is
|
|
53
|
+
* real evidence; vector-only hits must clear the floor) — the same post-hoc
|
|
54
|
+
* shape /recall-index uses, so the two recall surfaces gate identically.
|
|
55
|
+
*/
|
|
56
|
+
export declare function resolveSearchSimilarityFloor(raw: unknown): number | undefined;
|
|
45
57
|
/**
|
|
46
58
|
* Express error middleware (#7): translate express.json's default HTML 413
|
|
47
59
|
* (entity.too.large) into a consistent JSON response. Catches body-parser
|