@gamaze/hicortex 0.20.7 → 0.20.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +10 -41
  2. package/dist/calibration.d.ts +174 -0
  3. package/dist/calibration.js +231 -0
  4. package/dist/capture.d.ts +15 -3
  5. package/dist/capture.js +10 -1
  6. package/dist/classify-domains.d.ts +6 -0
  7. package/dist/classify-domains.js +7 -1
  8. package/dist/cli.js +2 -3
  9. package/dist/config-read.d.ts +1 -1
  10. package/dist/config-read.js +96 -9
  11. package/dist/consolidate.d.ts +79 -68
  12. package/dist/consolidate.js +218 -174
  13. package/dist/dashboard.d.ts +4 -3
  14. package/dist/dedup.d.ts +34 -26
  15. package/dist/dedup.js +91 -57
  16. package/dist/distiller.js +1 -1
  17. package/dist/domain-classify.d.ts +7 -6
  18. package/dist/domain-classify.js +12 -10
  19. package/dist/eval/decay-eval.d.ts +3 -3
  20. package/dist/eval/decay-eval.js +4 -4
  21. package/dist/eval/planted-eval.d.ts +26 -0
  22. package/dist/eval/planted-eval.js +97 -0
  23. package/dist/eval/planted-fixtures.d.ts +107 -0
  24. package/dist/eval/planted-fixtures.js +283 -0
  25. package/dist/eval/planted-harness.d.ts +176 -0
  26. package/dist/eval/planted-harness.js +649 -0
  27. package/dist/index.js +4 -3
  28. package/dist/init.d.ts +9 -3
  29. package/dist/init.js +52 -9
  30. package/dist/llm.d.ts +43 -58
  31. package/dist/llm.js +87 -101
  32. package/dist/mcp-server.js +29 -29
  33. package/dist/nightly.js +105 -103
  34. package/dist/nofit.d.ts +4 -11
  35. package/dist/nofit.js +6 -23
  36. package/dist/recall-index.d.ts +30 -28
  37. package/dist/recall-index.js +21 -18
  38. package/dist/recall-registry.d.ts +2 -1
  39. package/dist/recall-registry.js +35 -1
  40. package/dist/reconsolidation.d.ts +124 -72
  41. package/dist/reconsolidation.js +359 -148
  42. package/dist/relink.js +3 -4
  43. package/dist/retrieval.d.ts +68 -35
  44. package/dist/retrieval.js +292 -104
  45. package/dist/run-deadline.d.ts +62 -0
  46. package/dist/run-deadline.js +73 -0
  47. package/dist/schema-prototypes.d.ts +3 -3
  48. package/dist/schema-prototypes.js +3 -3
  49. package/dist/state.d.ts +2 -3
  50. package/dist/storage.d.ts +16 -16
  51. package/dist/storage.js +62 -24
  52. package/dist/telemetry.d.ts +8 -7
  53. package/dist/token-budget.js +3 -4
  54. package/dist/type-classify.js +4 -4
  55. package/dist/types.d.ts +95 -155
  56. package/domains.example.json +4 -5
  57. package/hermes-plugin/hicortex/README.md +2 -2
  58. package/openclaw.plugin.json +1 -1
  59. package/package.json +2 -1
  60. package/pi-extension/hicortex/README.md +1 -1
  61. package/server.json +3 -3
package/README.md CHANGED
@@ -116,11 +116,11 @@ Edit `~/.hicortex/config.json` on the server machine:
116
116
  }
117
117
  ```
118
118
 
119
- A richer example — including a `compartment: true` work/life firewall and a custom `weakPrimaryFloor` — ships as `domains.example.json` in the package.
119
+ A richer power-user example (a wider life-sphere set) ships as `domains.example.json` in the package.
120
120
 
121
121
  **How classification works:** the LLM decides only *which* of your domains apply to a memory — never weights or rankings. The weight of each tag is derived from your own data: each domain builds a prototype from the memories already in it, and a tag's weight is how strongly the memory's embedding matches that prototype. The primary domain is picked deterministically from those weights, and everything is recomputed each nightly, so your categories drift with your data instead of going stale. Memories that genuinely fit nothing get a weak association when they are close enough to some domain — and otherwise fade away over time. No junk drawer, no "Unsorted" pile.
122
122
 
123
- `weakPrimaryFloor` (config, default 0.45) sets how close a no-fit memory must be to its nearest domain to earn that weak association instead of fading.
123
+ How close a no-fit memory must be to its nearest domain to earn that weak association (instead of fading) is a release-managed calibration constant — it ships with each release and changes only with published eval evidence, not a config key.
124
124
 
125
125
  **Backfill an existing corpus** (server mode, needs `domains` in config):
126
126
 
@@ -250,19 +250,12 @@ Config at `~/.hicortex/config.json`. Created by `init`. Key options:
250
250
  | `mode` | `"server"` (default) or `"client"` |
251
251
  | `serverUrl` | Remote server URL (client mode) |
252
252
  | `llmModel` | The one model used by all phases (distill, score, classify, reflect). Set via `init`. |
253
- | `numCtx` | Context window for ollama (default 8192, one value for all phases). Scoring uses ~850 tokens, so 2048 is ample; distill/reflect/classify need more for `detectChunkSize`'s chunk sizing. |
254
253
  | `enableThinking` | Toggle the model's internal reasoning ("thinking") stream for OpenAI-compatible endpoints (default false). Only meaningful for local chat-template-aware servers (ollama, mlx-lm); leave unset for cloud OpenAI/OpenRouter/Groq endpoints (they 400 on the unknown `chat_template_kwargs` field). |
255
254
  | `maxTokens` | Max output tokens for all phases (default 8192). A ceiling, not a target — the model stops early when done. |
256
- | `classifyMaxTokens` | Max output tokens for the classify tier — the short JSON-verdict calls (correction/supersession verdicts, rewrite contracts, type and domain tag classification). Default 1024. A ceiling, not a target: raise it when a reasoning-style model spends the budget on internal reasoning and returns empty verdicts; `maxTokens` keeps governing the heavy phases (distill, reflect). |
257
- | `ollamaFlushEvery` | Flush ollama's accumulated memory every N scoring calls. **Off by default (0)** — opt-in only for an **ollama** install whose runner RSS growth (~171 MB/call) swap-thrashes long consolidations on a RAM-constrained box; N=15 caps a cycle at ~2.5 GB. Gated on the provider being ollama (local **or** remote) — no effect for non-ollama providers. Only you can judge whether your ollama endpoint actually suffers the growth (a managed/cloud ollama host may not), so it stays off until you set it. |
258
- | `ollamaFlushWaitMs` | Milliseconds to wait after an ollama flush for the runner to exit + release memory (default 180000 = 3 min). |
259
255
  | `llmTimeoutMs` | The ONE timeout ceiling on every LLM call in every phase (default 900000 = 15 min). The LLM request paths disable the HTTP client's hidden 5-minute response-header timer, so this knob is the only bound — one place to tune when the endpoint is slow, no per-phase special cases. |
260
- | `llmBreakerThreshold` | Consecutive fully-failed LLM calls (after their built-in retry ladder) that open the per-endpoint circuit breaker (default 3; **0 disables the breaker**). While open, calls fail fast with no network I/O; an HTTP error with a response body, a malformed-reply parse error, or a rate-limit 429 never counts — only the endpoint being unreachable/hung does. |
261
- | `llmBreakerCooldownMs` | How long the breaker stays open before one half-open trial call goes out (default 600000 = 10 min). A failing trial re-opens it; a succeeding one resets the counter. |
262
- | `llmProbeTimeoutMs` | Patience of the readiness probe — one minimal 1-token generation request the nightly sends before consolidating and the daemon sends before distilling (default 60000 = 1 min). Catches a gateway that answers health/model-list queries while generation is dead; a failed probe skips consolidation (`endpoint_down`, retried next run) and answers `/distill` with a 503 so capture holds its cursor. |
256
+ | `llmProbeTimeoutMs` | Patience of the readiness probe — one minimal 1-token generation request the daemon sends before distilling (default 60000 = 1 min). Catches a gateway that answers health/model-list queries while generation is dead; a failed probe answers `/distill` with a 503 so capture holds its cursor. The nightly no longer probes — its dead-endpoint signal is the circuit breaker (`endpoint_down`, retried next run) |
263
257
  | `llmProbeTtlMs` | How long the daemon caches a `/distill` probe outcome (default 300000 = 5 min). A healthy capture cadence pays at most one probe per window; a dead endpoint turns into fast cached 503s instead of every request paying the probe timeout. |
264
258
  | `llmSingleFlight` | **Serialized LLM calls — default `true`.** At most ONE request in flight per endpoint at any moment, across every process (the daemon distilling concurrent captures, the nightly consolidating, CLI backfills take turns via a per-endpoint lock file). One Hicortex server is several callers at once, and local single-user model servers (a Mac mini or laptop serving one big-context model) can stall or crash — taking the machine with them — under two concurrent large requests. Queued calls wait their turn; batches run back-to-back. **Set `false` if your endpoint is a beefy multi-tenant service that parallelizes well and you want faster consolidation** — you opt into responsibility for the endpoint's concurrency safety. |
265
- | `llmSingleFlightWaitMs` | How long a queued call waits for the in-flight call before failing as endpoint-down and retrying per the normal ladder/breaker rules (default: `max(900000, llmTimeoutMs)` — a waiter never gives up before a legitimate in-flight call's own, possibly raised, ceiling expires). Only meaningful with serialization on. Note under contention: the readiness probe queues too, but its wait is capped by its own probe budget — during a long consolidation batch, `/distill` answers 503 quickly (captures cursor-hold and retry, lossless) rather than holding the socket for the full queue budget. |
266
259
  | `authToken` | Bearer token for endpoint auth. Generated on first `init` in server mode. Find the active token with `hicortex status` or in `~/.hicortex/config.json`. |
267
260
  | `corsAllowedOrigins` | Browser origins allowed to read cross-origin responses, e.g. `["https://ui.example.com"]`. **Empty by default** — the server sends no `Access-Control-Allow-Origin` and never `Allow-Credentials`, so no external web page can read its data. The bundled `/viz` and `/identity/ui` pages are same-origin and need no entry. |
268
261
  | `licenseKey` | Commercial license key (optional; for display in `hicortex status`) |
@@ -270,8 +263,6 @@ Config at `~/.hicortex/config.json`. Created by `init`. Key options:
270
263
  | _env_ `HICORTEX_DISTILL_BODY_LIMIT_MB` | Environment override for the `/distill` body limit — **wins over the `distillBodyLimitMb` config key in every mode** (that is the point: a deployment operator pins it so tenant-writable config cannot raise it). Unset = config/default applies. |
271
264
  | _env_ `HICORTEX_MEMORY_CAP` | Environment override for the memory soft cap — a **positive** value wins over the `memorySoftCap` config key in every mode; `0`/negative/malformed fall through to the config key, then the 10000 default (an env can pin a cap, never disable one — config `memorySoftCap: 0` still disables when the env is unset). Mode-agnostic operator knob: nightly eviction, the dashboard-snapshot capacity stamp, and the live dashboard gauge all resolve through the same resolver, so the enforced and displayed caps can never disagree. Drives `nightly --evict-only` the same way. |
272
265
  | `domains` | Your memory domain list (`[{name, description}]`). Scaffolded by `init`; edit freely — see [Memory Domains & Tags](#memory-domains--tags) |
273
- | `weakPrimaryFloor` | Minimum similarity for a no-fit memory to keep a weak domain association (default: 0.45) |
274
- | `moduleIndexTokenBudget` | Max tokens for domain index in lessons context (default: 500) |
275
266
  | `lessonsLimit` | Max lessons injected into an agent's session-start context (default: 10). Lessons are ranked per session by project/domain affinity + recency + strength + access, so each session sees its most-relevant slice. Lower = leaner system prompts. |
276
267
  | `identityClients` | Which harnesses inject the [identity layer](#identity-layer) at session start (default `["cc"]`; `"all"` or any subset of `cc`/`hermes`/`oc`/`pi`/`opencode`) |
277
268
  | `identityAgents` | Per-agent identity modes (0.13): `{ "<id>": "override" \| "global" \| "off" }`. Absent + no `agents/<id>/` dir → every agent gets the global set. Boot-time (restart to apply) — see [Per-agent identity](#per-agent-identity-013) |
@@ -279,39 +270,17 @@ Config at `~/.hicortex/config.json`. Created by `init`. Key options:
279
270
  | `captureCooldownHours` | Success-cooldown (hours) for the **capture watchdog** (0.17). The capture timer polls every ~20 min; the watchdog captures only if more than this has elapsed since the last *successful* capture (`state.lastNightly`). Default `6` (≈4 captures/day). A failed preflight retries on the next poll (~20 min) — so a transient fire-instant network miss costs minutes, not a day (#239) |
280
271
  | `consolidationHours` | Hours (0–23, local) for the **consolidation** timer — the full nightly (capture + distill + score + reflect + link). Installed for **server/co-located only** (clients have no local DB). Default `[10, 22]`: the 22:00 evening slot runs after the day's capture waves (same-day results); the 10:00 morning slot runs *after* the morning capture so wake-up pushes are caught. Omitted on clients |
281
272
  | `timerJitterSeconds` | Max random delay (seconds) added to **generated** consolidation timers (#256), so a fleet doesn't all fire on the same minute (thundering-herd → LLM-backend contention). systemd: a single `RandomizedDelaySec=<n>`; launchd has no native equivalent so a per-install randomized `Minute` offset is baked into every `StartCalendarInterval` dict (sub-60s values no-op on launchd). Default `3600` (≈±30 min spread on the 2-slot/day cadence); `0` disables. Affects timers on the next `init` (re-init rewrites the unit files; installs that don't re-init keep their existing timers) |
282
- | `consolidateMaxLlmCalls` | Ceiling on total LLM calls across all classify-tier consolidation stages (content-domain, link discovery, supersession) per run. A runaway **backstop**, not a throughput throttle — on a free local model the binding constraint is the nightly unit's wall-clock timeout, not call count. Default `5000` (was a hard-coded 200 that starved link/supersession during a classification backlog) |
273
+ | `nightlyTimeBudgetMinutes` | The ONE wall-clock budget (minutes) for a nightly run: capture and every consolidation stage share one cooperative deadline, checked at safe boundaries (capture segments, stage boundaries, item loops, the merge zone). A run whose deadline fires reports consolidation `deferred` and resumes from its cursors next run — no work lost, none redone. Default `240`; `0`/invalid → default (a deadline always exists — there is no "off"). The systemd unit's `TimeoutStartSec` is derived from this (+60 min slack) at `init` |
274
+ | `nightlyLlmCallBudget` | The ONE per-run ceiling on LLM calls across the whole pipeline. Consumed in run order — a stage that exhausts it defers its remainder via its cursor. Bounds money/load independent of latency: a fast metered or capacity-limited endpoint permits thousands of calls inside the wall-clock budget, so time alone cannot protect it. Default `5000`; `0`/invalid → default. `consolidateMaxLlmCalls` is a **deprecated alias** (honored one release when the new key is absent — rename it) |
283
275
  | `memorySoftCap` | Soft cap on the memory corpus (default 10000). When the corpus exceeds this, the nightly's capacity-eviction stage removes the lowest-`effectiveStrength` memories (ties broken by oldest access) until under the cap — the active forgetting mechanism that bounds DB size, vector-index RAM, and consolidation workload. `0` disables eviction (indefinite growth — the pre-#245 behaviour). The evicted tail is cold by construction (effectiveStrength is the same decay-weighted score the recall ranker uses, so these were not surfacing in the top-k anyway). At 10K memories the load + JS sort is <100 ms |
284
276
  | `updateChannel` | Release channel pinned into the generated daemon/timer ExecStart for **npx-thin** installs (global-binary installs use the absolute binary and are unaffected). A dist-tag (`"rc"`, `"next"`) or an exact version (`"0.17.1"`). E.g. `"rc"` → the timer runs `npx -y @gamaze/hicortex@rc nightly`, so the host tracks the rc dist-tag (an internal fleet can ride rc through a pre-promotion soak). Validated as `[\w.\-]+` (rejects anything that'd break the unit/plist templates). Absent → auto-detect (bare on `latest`, else `@next`). (0.17.1) |
285
277
  | `nightlyHour` | **Deprecated (0.17) single-slot fallback.** Local hour (0–23) honoured only when `consolidationHours` is absent — yields one daily consolidation slot at that hour (preserves the pre-0.17 "one daily job" intent). New installs should use `consolidationHours` |
286
- | `preflightTimeoutMs` | **Client mode only.** Per-attempt timeout for the nightly's server-reachability check before it starts capturing (default: 20000 ms, bumped from 15000 in 0.17 to absorb a slow link re-establishing after the client wakes) |
287
- | `preflightAttempts` | **Client mode only.** Reachability-check retries before the nightly aborts (default: 3; floored at 1). `1` = single try, no retry |
288
- | `preflightRetryGapMs` | **Client mode only.** Delay between reachability retries (default: 60000 ms). Note: timers don't advance while the machine is asleep, so on a sleeping laptop this gap counts awake-time, not wall-clock |
289
- | `scoreSimilarityWeight` | Weight of semantic similarity in the ranking score (default: 0.50) |
290
- | `scoreStrengthWeight` | Weight of effective strength — importance/use/recency of access (default: 0.20) |
291
- | `scoreConnectionsWeight` | Weight of graph centrality (default: 0.15) |
292
- | `scoreRecencyWeight` | Weight of the slow recency curve (default: 0.15) |
293
- | `freshnessBoostDays` | Fresh-memory window: new memories rank higher for this many days (default: 7) |
294
- | `freshnessBoostWeight` | Size of the fresh-memory bonus at age 0, fading linearly to 0 at the window edge (default: 0.15; set 0 to disable) |
295
- | `supersededDemotion` | Score multiplier for a memory a later decision reversed (default: 0.50) |
296
- | `decayHalfLifeDays` | Memory decay half-life in days at reference importance (default: 365). Larger = slower forgetting; importance, access, and links slow it further |
297
- | `searchLimit` / `recentLimit` | Default result counts for search (8) and recent (12) |
298
- | `recentWindowDays` | Candidate window for recent recall (default: 180) |
299
- | `coldExposureSlots` | Top-k slots reservable for never-accessed memories so the long tail gets exposure (default: 2) |
300
- | `recallMaxItems` | Max lines in the pushed recall index (default: 5) |
301
- | `noveltyFloorSlots` | Slots of `recallMaxItems` guaranteed to the top passing hit(s) of the pure-prompt (unblended) search — the novelty floor. Keeps a session whose earlier turns set a strong intent from burying a topic-switching prompt's best matches: the floor's picks render first, turn-based re-show suppression still applies, and the total never exceeds `recallMaxItems` (default: 2; set 0 to disable) |
302
- | `recallMinSimilarity` | Relevance floor for index entries (default: 0.62; text-search matches always pass) |
303
- | `recallReshowTurns` | Turns before an already-shown memory may reappear in the same session (default: 30) |
304
- | `recallMinPromptChars` | Prompts shorter than this skip the recall index (default: 20) |
305
- | `recallTitleChars` | Chars of each memory's first line shown in an index entry (default: 100, range 40–400). Reverted from 150 on 2026-08-03: a full-corpus relevance eval found 100 and 150 statistically identical while 100 saves ~13% of the block's tokens |
306
- | `sessionIntentWeight` | Blend weight of the session-intent rolling centroid in the recall search vector: `query = (1-w)·prompt + w·centroid` (default: 0.33; set 0 to disable — pure-prompt recall, the kill-switch). The first turn of a session searches with pure prompt and seeds the centroid; subsequent turns blend so recall follows the session's intent instead of being query-literal. The EMA rate (0.4) is a shipped constant, not configurable |
307
- | `dedupAutoMergeThreshold` | The deterministic merge ceiling of the unified resolution pass: memory pairs at/above this cosine merge automatically (zero LLM) via the dedup core's clustering; pairs between `correctionMinSimilarity` and this value get the one merge/corrects/supersedes/none verdict. Also the default threshold for `hicortex dedup` (default: 0.92) |
308
- | `dedupMergeThreshold` | Legacy alias for `dedupAutoMergeThreshold`, still honored when the newer key is absent |
309
- | `dedupNightlyMaxMerges` | Pacing cap on merge operations per nightly run — deterministic-zone clusters plus verdict-confirmed pair merges count against one cap, so a large duplicate backlog drains over a few nights (default: 250; `0` disables the merge machinery) |
310
- | `supersessionMinSimilarity` | Minimum cosine similarity for a nightly supersession candidate pair (default: 0.80) |
311
- | `supersessionMaxCalls` | Max classify-tier LLM calls the nightly's supersession stage spends per run (default: 30) |
312
- | `supersessionPenalty` | Multiplier applied to a superseded memory's `base_strength` (default: 0.5) |
313
278
  | `telemetry` | Anonymous usage telemetry. **On by default and not written into config by `init`** — add `"telemetry": false` yourself (or set `HICORTEX_TELEMETRY=off`) to opt out. Inspect exactly what is sent with `hicortex telemetry` |
314
279
 
280
+ **Calibration is release-managed.** The ~35 tuning keys earlier releases exposed (recall breadth, relevance floors, ranking weights, decay speed, dedup/supersession/correction thresholds, the weak-primary floor) are no longer config: they are constants that ship with each release and change only in releases, with the eval evidence linked in the changelog. Config values for them are ignored — the server prints a one-time boot warning naming each ignored key. Your config file now describes your *install* (mode, model, schedules, identity, budgets), not the brain's tuning.
281
+
282
+ **Diagnostic tier (environment).** Three niche, ollama-only operational values moved from config to environment variables: `HICORTEX_NUM_CTX` (context window for ollama, default 8192), `HICORTEX_OLLAMA_FLUSH_EVERY` (flush ollama's accumulated memory every N LLM calls; default 0 = off), and `HICORTEX_OLLAMA_FLUSH_WAIT_MS` (post-flush wait, default 180000). Pin them in a service unit's environment when needed; the old config keys are ignored with a boot warning naming the replacement.
283
+
315
284
  Full docs: [hicortex.gamaze.com/docs/configuration.html](https://hicortex.gamaze.com/docs/configuration.html)
316
285
 
317
286
  ## REST API
@@ -355,7 +324,7 @@ Optional config (add to plugin entry in `~/.openclaw/openclaw.json`):
355
324
  | `serverUrl` | `http://127.0.0.1:8787` | Hicortex server URL. Change for remote servers. |
356
325
  | `authToken` | _(none)_ | Bearer token. Localhost bypasses auth; required for remote servers. Get the token from `hicortex status` on the server. |
357
326
  | `defaultProject` | _(none)_ | Project name sent on recall, search, recent, and ingest whenever the gateway supplies no project (Hermes `default_project` parity). |
358
- | `recallLimit` | `8` | Max memories per recall on the pre-0.14 `/search` fallback. The pushed recall index is sized by SERVER config (`recallMaxItems`) — the server accepts no client limit. |
327
+ | `recallLimit` | `8` | Max memories per recall on the pre-0.14 `/search` fallback. The pushed recall index is sized by the server's release-managed calibration — the server accepts no client limit. |
359
328
  | `scaffoldDeadMan` | `true` | Auto-scaffold the dead-man identity-guard line into the agent workspace bootstrap (`BOOTSTRAP.md`) at startup. Set `false` to disable the write and any file creation entirely. |
360
329
 
361
330
  If `serverUrl`/`authToken` are absent from the config, the `HICORTEX_URL` and `HICORTEX_AUTH_TOKEN` environment variables are used as fallbacks (config always wins).
@@ -0,0 +1,174 @@
1
+ /**
2
+ * Release-managed calibration constants (#408) — the single home of every
3
+ * tuning value the product ships. One value, one definition, one provenance
4
+ * comment. Nothing in here is read from ~/.hicortex/config.json anymore: the
5
+ * ~35 tuning keys that 0.15–0.20 exposed as config are now CONSTANTS that
6
+ * move only in releases. The user config surface shrinks to the keys that
7
+ * describe an install (mode, model, schedules, identity, budgets), not the
8
+ * ones that tune the brain.
9
+ *
10
+ * EVOLUTION CONTRACT: these values change ONLY in releases, with the
11
+ * eval/band-stats evidence linked in the changelog line that moves them
12
+ * (eval harness = `npm run eval` + the resolution band stats in the nightly
13
+ * report). Never in a patch to quiet one corpus, never behind a new config
14
+ * key. The seams for EXPERIMENTS are the configure*() functions in
15
+ * retrieval.ts / storage.ts (and the stage Options fields) — the eval and
16
+ * the tests sweep values through them; production never passes anything, so
17
+ * every process scores with exactly these constants.
18
+ *
19
+ * Every value below equals the default the code shipped the day this module
20
+ * was introduced (verified by tests/calibration.test.ts) — an install that
21
+ * never set the old config keys sees byte-identical behavior. The old keys
22
+ * are warned as RELEASE-MANAGED at the config boundary (config-read.ts) so
23
+ * the removal is never silent.
24
+ */
25
+ /**
26
+ * Memory-decay half-life (days) at the reference importance 0.5. #192 recall/
27
+ * decay alignment: was ~115 days — aggressive enough to bury the long tail in
28
+ * ranking. Long-term remembering is the product; time preference stays mild.
29
+ */
30
+ export declare const DECAY_HALF_LIFE_DAYS = 365;
31
+ /** Default k for retrieve() (/search without an explicit limit). #192. */
32
+ export declare const SEARCH_LIMIT = 8;
33
+ /** Default k for searchRecent() (/recent without an explicit limit). #192. */
34
+ export declare const RECENT_LIMIT = 12;
35
+ /** searchRecent() candidate window, days. #192. */
36
+ export declare const RECENT_WINDOW_DAYS = 180;
37
+ /** Top-k slots reservable for never-accessed memories (cold exposure). #192:
38
+ * recall was too passive (88% of memories never accessed) — the long tail
39
+ * gets guaranteed slots instead of waiting for the strength clock. */
40
+ export declare const COLD_EXPOSURE_SLOTS = 2;
41
+ /** Blend weight of the session-intent centroid in the recall search vector
42
+ * (#192, 0.15.3): query = (1-w)·prompt + w·centroid. The kill-switch is the
43
+ * configureSessionIntent(0) seam (eval-only); production always ships 0.33. */
44
+ export declare const SESSION_INTENT_WEIGHT = 0.33;
45
+ /** Relevance-gate floor for vector-only /recall-index candidates. 0.62
46
+ * (raised from 0.55 on 2026-08-03 per a 0.01-step floor sweep on the
47
+ * rewritten corpus): steady ~3:1 noise:signal removal with no knee; sits
48
+ * below the 0.63 local pessimum. FTS-matched candidates pass regardless. */
49
+ export declare const RECALL_MIN_SIMILARITY = 0.62;
50
+ /** Max lines in the pushed recall index. 5 (lowered from 6 on 2026-08-03):
51
+ * per-slot decomposition at floor 0.62 showed slot 6 gives NO prompt its
52
+ * first relevant memory. The K-sweep is monotone toward 4, but the 4-vs-5
53
+ * distinction rests on 5 of 98 prompts — 5 hedges with coverage. */
54
+ export declare const RECALL_MAX_ITEMS = 5;
55
+ /** Prompts shorter than this skip the recall index (continuations, "yes"). */
56
+ export declare const RECALL_MIN_PROMPT_CHARS = 20;
57
+ /** Chars of a memory's first line shown in an index entry. 100 (reverted
58
+ * from 150 on 2026-08-03): the full-corpus relevance eval found 100 vs 150
59
+ * statistically identical (full CI overlap at N=40); 100 saves ~13% tokens. */
60
+ export declare const RECALL_TITLE_CHARS = 100;
61
+ /** Slots of RECALL_MAX_ITEMS guaranteed to the pure-prompt (unblended)
62
+ * search's top passing hit(s) — the #324 novelty floor. 2 mirrors
63
+ * COLD_EXPOSURE_SLOTS sizing: a floor, never a takeover. */
64
+ export declare const NOVELTY_FLOOR_SLOTS = 2;
65
+ /** Turns an already-shown memory stays suppressed in the same session before
66
+ * it may reappear in the pushed index (#192 turn-based dedup). */
67
+ export declare const RECALL_RESHOW_TURNS = 30;
68
+ /** Semantic-similarity share of the composite score. 0.50 (raised from 0.40
69
+ * in the 0.15.2 rebalance, #191 Phase B): on the production corpus effective
70
+ * strength (0.30) outweighed what similarity could recover — hardened old
71
+ * memories beat exact matches for their own topic. Similarity now leads;
72
+ * strength breaks ties and rewards real use. */
73
+ export declare const SCORE_SIMILARITY_WEIGHT = 0.5;
74
+ /** Effective-strength share of the composite score (was 0.30; see above). */
75
+ export declare const SCORE_STRENGTH_WEIGHT = 0.2;
76
+ /** Graph-centrality share of the composite score (was 0.20; see above). */
77
+ export declare const SCORE_CONNECTIONS_WEIGHT = 0.15;
78
+ /** Slow recency curve share of the composite score (was 0.10; see above). */
79
+ export declare const SCORE_RECENCY_WEIGHT = 0.15;
80
+ /** Fresh-memory window: the additive bonus fades linearly to 0 over this
81
+ * many days. 7 — nightly capture means 1 day is the floor of "fresh"
82
+ * (#191 Phase B). */
83
+ export declare const FRESHNESS_BOOST_DAYS = 7;
84
+ /** Fresh-memory bonus size at age 0 (#191 Phase B; 0 = disabled via seam). */
85
+ export declare const FRESHNESS_BOOST_WEIGHT = 0.15;
86
+ /** Score multiplier for a memory a later decision superseded (0.15.2; the
87
+ * belief walk (#393 D) is the primary mechanism — this is the safety net
88
+ * for rows the walk does not reach). */
89
+ export declare const SUPERSEDED_DEMOTION = 0.5;
90
+ /** #203 soft boost on exact project match. ADDITIVE, zero-boost neutral,
91
+ * never a penalty — a foreign memory ranks equal, not lower. */
92
+ export declare const PROJECT_AFFINITY_WEIGHT = 0.15;
93
+ /** #203 soft boost multiplier on max overlapping domain-tag weight. */
94
+ export declare const DOMAIN_AFFINITY_WEIGHT = 0.15;
95
+ /** #205 RRF k parameter (1/(k+rank+1)) — matches the pre-#205 hardcoded 60
96
+ * so the no-config path was byte-identical to 0.15.3. */
97
+ export declare const RRF_K = 60;
98
+ /** #205 composite-score share of the final blend (RRF gets the remainder);
99
+ * pre-#205 hardcoded value carried forward. */
100
+ export declare const RRF_COMPOSITE_WEIGHT = 0.8;
101
+ /** #205 per-list RRF weight for the FTS list. 0.5 is the bisection point
102
+ * where BM25F + composite-affinity flip the token-exact marine body match
103
+ * below the same-scope hardware field (Q4 contamination 0.20 → 0.00) while
104
+ * pure-keyword queries keep recall@5 = 1.0. 0.7 was measured too timid. */
105
+ export declare const RRF_FTS_WEIGHT = 0.5;
106
+ /** #205 per-list RRF weight for the vector list (vec stays at 1.0 — the
107
+ * conservative nudge is on the FTS side only). */
108
+ export declare const RRF_VECTOR_WEIGHT = 1;
109
+ export declare const BM25_WEIGHT_BODY = 1;
110
+ export declare const BM25_WEIGHT_PROJECT = 2;
111
+ export declare const BM25_WEIGHT_DOMAIN = 2;
112
+ /** Deterministic merge ceiling of the unified resolution pass (#392): pairs
113
+ * at/above this cosine merge LLM-free; [CORRECTION_MIN_SIMILARITY, this)
114
+ * get the one verdict call. 0.92 — measured on the #191 mechanical audit
115
+ * corpus (89 clusters / 110 excess rows; data/audit-20260729). */
116
+ export declare const DEDUP_AUTO_MERGE_THRESHOLD = 0.92;
117
+ /** Minimum cosine for a nightly supersession candidate pair (#100 stage,
118
+ * 0.15.0): one classify-tier call per pair above the bar. */
119
+ export declare const SUPERSESSION_MIN_SIMILARITY = 0.8;
120
+ /** Minimum cosine for a reconsolidation correction pair (#384). Deliberately
121
+ * wider than supersession's 0.80: a retraction often rides inside an
122
+ * otherwise unrelated memory; the verdict + confidence gate carry the
123
+ * precision. */
124
+ export declare const CORRECTION_MIN_SIMILARITY = 0.75;
125
+ /** Minimum verdict confidence for the REWRITE (and #392 merge-apply) fork
126
+ * (#384): below it a `corrects` degrades to mark-only — a weak mark is
127
+ * recoverable, a weak rewrite is corruption. */
128
+ export declare const CORRECTION_REWRITE_MIN_CONFIDENCE = 0.8;
129
+ /** Minimum cosine(memory embedding, best domain prototype) for a no-fit
130
+ * memory to earn a WEAK primary instead of decaying (owner amendment
131
+ * 07.07). Starting point for bge-small-en-v1.5. */
132
+ export declare const WEAK_PRIMARY_FLOOR = 0.45;
133
+ /**
134
+ * Context window for ollama (one value, all phases; #220/#228). 8192 is
135
+ * where context stops being the binding constraint for a sub-8B model on
136
+ * ollama. Also drives `detectChunkSize` (chunkChars ≤ numCtx × 0.6 × 4
137
+ * chars) so the chunker and the request agree by construction.
138
+ */
139
+ export declare const NUM_CTX = 8192;
140
+ /** Flush ollama's accumulated memory every N LLM calls (0 = off; #220).
141
+ * Opt-in operational workaround for ollama runner RSS growth — never a
142
+ * default-on behavior. */
143
+ export declare const OLLAMA_FLUSH_EVERY = 0;
144
+ /** Ms to wait after an ollama flush (`keep_alive:0`) for the runner to exit
145
+ * + release memory. The runner takes >90 s to exit; 3 min allows margin. */
146
+ export declare const OLLAMA_FLUSH_WAIT_MS = 180000;
147
+ /** The env-tier table (release surface: names + defaults are frozen —
148
+ * adding a knob here is a release decision, not a runtime one). */
149
+ export declare const DIAGNOSTIC_ENV_TIER: Readonly<{
150
+ numCtx: Readonly<{
151
+ env: string;
152
+ default: number;
153
+ }>;
154
+ ollamaFlushEvery: Readonly<{
155
+ env: string;
156
+ default: number;
157
+ }>;
158
+ ollamaFlushWaitMs: Readonly<{
159
+ env: string;
160
+ default: number;
161
+ }>;
162
+ }>;
163
+ /** Resolve the effective ollama context window: a positive finite
164
+ * HICORTEX_NUM_CTX wins; anything else (absent/blank/invalid) keeps NUM_CTX
165
+ * with a warn on the invalid case. */
166
+ export declare function resolveNumCtx(): number;
167
+ /** Resolve the flush cadence: a non-negative finite
168
+ * HICORTEX_OLLAMA_FLUSH_EVERY wins (0 = the valid off value); invalid warns
169
+ * and keeps OLLAMA_FLUSH_EVERY. */
170
+ export declare function resolveOllamaFlushEvery(): number;
171
+ /** Resolve the post-flush wait: a positive finite
172
+ * HICORTEX_OLLAMA_FLUSH_WAIT_MS wins; invalid warns and keeps
173
+ * OLLAMA_FLUSH_WAIT_MS. */
174
+ export declare function resolveOllamaFlushWaitMs(): number;
@@ -0,0 +1,231 @@
1
+ "use strict";
2
+ /**
3
+ * Release-managed calibration constants (#408) — the single home of every
4
+ * tuning value the product ships. One value, one definition, one provenance
5
+ * comment. Nothing in here is read from ~/.hicortex/config.json anymore: the
6
+ * ~35 tuning keys that 0.15–0.20 exposed as config are now CONSTANTS that
7
+ * move only in releases. The user config surface shrinks to the keys that
8
+ * describe an install (mode, model, schedules, identity, budgets), not the
9
+ * ones that tune the brain.
10
+ *
11
+ * EVOLUTION CONTRACT: these values change ONLY in releases, with the
12
+ * eval/band-stats evidence linked in the changelog line that moves them
13
+ * (eval harness = `npm run eval` + the resolution band stats in the nightly
14
+ * report). Never in a patch to quiet one corpus, never behind a new config
15
+ * key. The seams for EXPERIMENTS are the configure*() functions in
16
+ * retrieval.ts / storage.ts (and the stage Options fields) — the eval and
17
+ * the tests sweep values through them; production never passes anything, so
18
+ * every process scores with exactly these constants.
19
+ *
20
+ * Every value below equals the default the code shipped the day this module
21
+ * was introduced (verified by tests/calibration.test.ts) — an install that
22
+ * never set the old config keys sees byte-identical behavior. The old keys
23
+ * are warned as RELEASE-MANAGED at the config boundary (config-read.ts) so
24
+ * the removal is never silent.
25
+ */
26
+ Object.defineProperty(exports, "__esModule", { value: true });
27
+ exports.DIAGNOSTIC_ENV_TIER = exports.OLLAMA_FLUSH_WAIT_MS = exports.OLLAMA_FLUSH_EVERY = exports.NUM_CTX = exports.WEAK_PRIMARY_FLOOR = exports.CORRECTION_REWRITE_MIN_CONFIDENCE = exports.CORRECTION_MIN_SIMILARITY = exports.SUPERSESSION_MIN_SIMILARITY = exports.DEDUP_AUTO_MERGE_THRESHOLD = exports.BM25_WEIGHT_DOMAIN = exports.BM25_WEIGHT_PROJECT = exports.BM25_WEIGHT_BODY = exports.RRF_VECTOR_WEIGHT = exports.RRF_FTS_WEIGHT = exports.RRF_COMPOSITE_WEIGHT = exports.RRF_K = exports.DOMAIN_AFFINITY_WEIGHT = exports.PROJECT_AFFINITY_WEIGHT = exports.SUPERSEDED_DEMOTION = exports.FRESHNESS_BOOST_WEIGHT = exports.FRESHNESS_BOOST_DAYS = exports.SCORE_RECENCY_WEIGHT = exports.SCORE_CONNECTIONS_WEIGHT = exports.SCORE_STRENGTH_WEIGHT = exports.SCORE_SIMILARITY_WEIGHT = exports.RECALL_RESHOW_TURNS = exports.NOVELTY_FLOOR_SLOTS = exports.RECALL_TITLE_CHARS = exports.RECALL_MIN_PROMPT_CHARS = exports.RECALL_MAX_ITEMS = exports.RECALL_MIN_SIMILARITY = exports.SESSION_INTENT_WEIGHT = exports.COLD_EXPOSURE_SLOTS = exports.RECENT_WINDOW_DAYS = exports.RECENT_LIMIT = exports.SEARCH_LIMIT = exports.DECAY_HALF_LIFE_DAYS = void 0;
28
+ exports.resolveNumCtx = resolveNumCtx;
29
+ exports.resolveOllamaFlushEvery = resolveOllamaFlushEvery;
30
+ exports.resolveOllamaFlushWaitMs = resolveOllamaFlushWaitMs;
31
+ // ---------------------------------------------------------------------------
32
+ // Recall / decay family (was: decayHalfLifeDays, searchLimit, recentLimit,
33
+ // recentWindowDays, coldExposureSlots, sessionIntentWeight, recall*,
34
+ // noveltyFloorSlots, recallReshowTurns)
35
+ // ---------------------------------------------------------------------------
36
+ /**
37
+ * Memory-decay half-life (days) at the reference importance 0.5. #192 recall/
38
+ * decay alignment: was ~115 days — aggressive enough to bury the long tail in
39
+ * ranking. Long-term remembering is the product; time preference stays mild.
40
+ */
41
+ exports.DECAY_HALF_LIFE_DAYS = 365;
42
+ /** Default k for retrieve() (/search without an explicit limit). #192. */
43
+ exports.SEARCH_LIMIT = 8;
44
+ /** Default k for searchRecent() (/recent without an explicit limit). #192. */
45
+ exports.RECENT_LIMIT = 12;
46
+ /** searchRecent() candidate window, days. #192. */
47
+ exports.RECENT_WINDOW_DAYS = 180;
48
+ /** Top-k slots reservable for never-accessed memories (cold exposure). #192:
49
+ * recall was too passive (88% of memories never accessed) — the long tail
50
+ * gets guaranteed slots instead of waiting for the strength clock. */
51
+ exports.COLD_EXPOSURE_SLOTS = 2;
52
+ /** Blend weight of the session-intent centroid in the recall search vector
53
+ * (#192, 0.15.3): query = (1-w)·prompt + w·centroid. The kill-switch is the
54
+ * configureSessionIntent(0) seam (eval-only); production always ships 0.33. */
55
+ exports.SESSION_INTENT_WEIGHT = 0.33;
56
+ /** Relevance-gate floor for vector-only /recall-index candidates. 0.62
57
+ * (raised from 0.55 on 2026-08-03 per a 0.01-step floor sweep on the
58
+ * rewritten corpus): steady ~3:1 noise:signal removal with no knee; sits
59
+ * below the 0.63 local pessimum. FTS-matched candidates pass regardless. */
60
+ exports.RECALL_MIN_SIMILARITY = 0.62;
61
+ /** Max lines in the pushed recall index. 5 (lowered from 6 on 2026-08-03):
62
+ * per-slot decomposition at floor 0.62 showed slot 6 gives NO prompt its
63
+ * first relevant memory. The K-sweep is monotone toward 4, but the 4-vs-5
64
+ * distinction rests on 5 of 98 prompts — 5 hedges with coverage. */
65
+ exports.RECALL_MAX_ITEMS = 5;
66
+ /** Prompts shorter than this skip the recall index (continuations, "yes"). */
67
+ exports.RECALL_MIN_PROMPT_CHARS = 20;
68
+ /** Chars of a memory's first line shown in an index entry. 100 (reverted
69
+ * from 150 on 2026-08-03): the full-corpus relevance eval found 100 vs 150
70
+ * statistically identical (full CI overlap at N=40); 100 saves ~13% tokens. */
71
+ exports.RECALL_TITLE_CHARS = 100;
72
+ /** Slots of RECALL_MAX_ITEMS guaranteed to the pure-prompt (unblended)
73
+ * search's top passing hit(s) — the #324 novelty floor. 2 mirrors
74
+ * COLD_EXPOSURE_SLOTS sizing: a floor, never a takeover. */
75
+ exports.NOVELTY_FLOOR_SLOTS = 2;
76
+ /** Turns an already-shown memory stays suppressed in the same session before
77
+ * it may reappear in the pushed index (#192 turn-based dedup). */
78
+ exports.RECALL_RESHOW_TURNS = 30;
79
+ // ---------------------------------------------------------------------------
80
+ // Composite ranking weights (was: score*Weight, freshnessBoost*,
81
+ // supersededDemotion, *AffinityWeight, rrf*)
82
+ // ---------------------------------------------------------------------------
83
+ /** Semantic-similarity share of the composite score. 0.50 (raised from 0.40
84
+ * in the 0.15.2 rebalance, #191 Phase B): on the production corpus effective
85
+ * strength (0.30) outweighed what similarity could recover — hardened old
86
+ * memories beat exact matches for their own topic. Similarity now leads;
87
+ * strength breaks ties and rewards real use. */
88
+ exports.SCORE_SIMILARITY_WEIGHT = 0.50;
89
+ /** Effective-strength share of the composite score (was 0.30; see above). */
90
+ exports.SCORE_STRENGTH_WEIGHT = 0.20;
91
+ /** Graph-centrality share of the composite score (was 0.20; see above). */
92
+ exports.SCORE_CONNECTIONS_WEIGHT = 0.15;
93
+ /** Slow recency curve share of the composite score (was 0.10; see above). */
94
+ exports.SCORE_RECENCY_WEIGHT = 0.15;
95
+ /** Fresh-memory window: the additive bonus fades linearly to 0 over this
96
+ * many days. 7 — nightly capture means 1 day is the floor of "fresh"
97
+ * (#191 Phase B). */
98
+ exports.FRESHNESS_BOOST_DAYS = 7;
99
+ /** Fresh-memory bonus size at age 0 (#191 Phase B; 0 = disabled via seam). */
100
+ exports.FRESHNESS_BOOST_WEIGHT = 0.15;
101
+ /** Score multiplier for a memory a later decision superseded (0.15.2; the
102
+ * belief walk (#393 D) is the primary mechanism — this is the safety net
103
+ * for rows the walk does not reach). */
104
+ exports.SUPERSEDED_DEMOTION = 0.50;
105
+ /** #203 soft boost on exact project match. ADDITIVE, zero-boost neutral,
106
+ * never a penalty — a foreign memory ranks equal, not lower. */
107
+ exports.PROJECT_AFFINITY_WEIGHT = 0.15;
108
+ /** #203 soft boost multiplier on max overlapping domain-tag weight. */
109
+ exports.DOMAIN_AFFINITY_WEIGHT = 0.15;
110
+ /** #205 RRF k parameter (1/(k+rank+1)) — matches the pre-#205 hardcoded 60
111
+ * so the no-config path was byte-identical to 0.15.3. */
112
+ exports.RRF_K = 60;
113
+ /** #205 composite-score share of the final blend (RRF gets the remainder);
114
+ * pre-#205 hardcoded value carried forward. */
115
+ exports.RRF_COMPOSITE_WEIGHT = 0.8;
116
+ /** #205 per-list RRF weight for the FTS list. 0.5 is the bisection point
117
+ * where BM25F + composite-affinity flip the token-exact marine body match
118
+ * below the same-scope hardware field (Q4 contamination 0.20 → 0.00) while
119
+ * pure-keyword queries keep recall@5 = 1.0. 0.7 was measured too timid. */
120
+ exports.RRF_FTS_WEIGHT = 0.5;
121
+ /** #205 per-list RRF weight for the vector list (vec stays at 1.0 — the
122
+ * conservative nudge is on the FTS side only). */
123
+ exports.RRF_VECTOR_WEIGHT = 1.0;
124
+ // ---------------------------------------------------------------------------
125
+ // BM25F field weights (was: bm25WeightBody/Project/Domain) — #205. Body is
126
+ // down-weighted relative to the scope fields so a project/domain token match
127
+ // outranks a token-exact body collision from a foreign scope.
128
+ // ---------------------------------------------------------------------------
129
+ exports.BM25_WEIGHT_BODY = 1.0;
130
+ exports.BM25_WEIGHT_PROJECT = 2.0;
131
+ exports.BM25_WEIGHT_DOMAIN = 2.0;
132
+ // ---------------------------------------------------------------------------
133
+ // Resolution / dedup family (was: dedupAutoMergeThreshold [legacy
134
+ // dedupMergeThreshold], supersessionMinSimilarity, correctionMinSimilarity,
135
+ // correctionRewriteMinConfidence, weakPrimaryFloor)
136
+ // ---------------------------------------------------------------------------
137
+ /** Deterministic merge ceiling of the unified resolution pass (#392): pairs
138
+ * at/above this cosine merge LLM-free; [CORRECTION_MIN_SIMILARITY, this)
139
+ * get the one verdict call. 0.92 — measured on the #191 mechanical audit
140
+ * corpus (89 clusters / 110 excess rows; data/audit-20260729). */
141
+ exports.DEDUP_AUTO_MERGE_THRESHOLD = 0.92;
142
+ /** Minimum cosine for a nightly supersession candidate pair (#100 stage,
143
+ * 0.15.0): one classify-tier call per pair above the bar. */
144
+ exports.SUPERSESSION_MIN_SIMILARITY = 0.80;
145
+ /** Minimum cosine for a reconsolidation correction pair (#384). Deliberately
146
+ * wider than supersession's 0.80: a retraction often rides inside an
147
+ * otherwise unrelated memory; the verdict + confidence gate carry the
148
+ * precision. */
149
+ exports.CORRECTION_MIN_SIMILARITY = 0.75;
150
+ /** Minimum verdict confidence for the REWRITE (and #392 merge-apply) fork
151
+ * (#384): below it a `corrects` degrades to mark-only — a weak mark is
152
+ * recoverable, a weak rewrite is corruption. */
153
+ exports.CORRECTION_REWRITE_MIN_CONFIDENCE = 0.80;
154
+ /** Minimum cosine(memory embedding, best domain prototype) for a no-fit
155
+ * memory to earn a WEAK primary instead of decaying (owner amendment
156
+ * 07.07). Starting point for bge-small-en-v1.5. */
157
+ exports.WEAK_PRIMARY_FLOOR = 0.45;
158
+ // ---------------------------------------------------------------------------
159
+ // Diagnostic tier (env-overridable — #408). The ollama-operational family is
160
+ // NOT user tuning: it exists so an operator of a constrained box can pin the
161
+ // three values into a service unit's environment without a config-file
162
+ // round-trip. Precedence: env > the constant below. An invalid env value
163
+ // warns and falls back to the constant (the resolveMemorySoftCap boundary
164
+ // posture, applied to the env half).
165
+ // ---------------------------------------------------------------------------
166
+ /**
167
+ * Context window for ollama (one value, all phases; #220/#228). 8192 is
168
+ * where context stops being the binding constraint for a sub-8B model on
169
+ * ollama. Also drives `detectChunkSize` (chunkChars ≤ numCtx × 0.6 × 4
170
+ * chars) so the chunker and the request agree by construction.
171
+ */
172
+ exports.NUM_CTX = 8192;
173
+ /** Flush ollama's accumulated memory every N LLM calls (0 = off; #220).
174
+ * Opt-in operational workaround for ollama runner RSS growth — never a
175
+ * default-on behavior. */
176
+ exports.OLLAMA_FLUSH_EVERY = 0;
177
+ /** Ms to wait after an ollama flush (`keep_alive:0`) for the runner to exit
178
+ * + release memory. The runner takes >90 s to exit; 3 min allows margin. */
179
+ exports.OLLAMA_FLUSH_WAIT_MS = 180000;
180
+ /** The env-tier table (release surface: names + defaults are frozen —
181
+ * adding a knob here is a release decision, not a runtime one). */
182
+ exports.DIAGNOSTIC_ENV_TIER = Object.freeze({
183
+ numCtx: Object.freeze({ env: "HICORTEX_NUM_CTX", default: exports.NUM_CTX }),
184
+ ollamaFlushEvery: Object.freeze({
185
+ env: "HICORTEX_OLLAMA_FLUSH_EVERY",
186
+ default: exports.OLLAMA_FLUSH_EVERY,
187
+ }),
188
+ ollamaFlushWaitMs: Object.freeze({
189
+ env: "HICORTEX_OLLAMA_FLUSH_WAIT_MS",
190
+ default: exports.OLLAMA_FLUSH_WAIT_MS,
191
+ }),
192
+ });
193
+ /** Resolve the effective ollama context window: a positive finite
194
+ * HICORTEX_NUM_CTX wins; anything else (absent/blank/invalid) keeps NUM_CTX
195
+ * with a warn on the invalid case. */
196
+ function resolveNumCtx() {
197
+ const raw = process.env[exports.DIAGNOSTIC_ENV_TIER.numCtx.env];
198
+ if (raw === undefined || raw === "")
199
+ return exports.NUM_CTX;
200
+ const v = Number(raw);
201
+ if (Number.isFinite(v) && v > 0)
202
+ return v;
203
+ console.warn(`[hicortex] env HICORTEX_NUM_CTX=${JSON.stringify(raw)} is not a positive finite number — using default ${exports.NUM_CTX}.`);
204
+ return exports.NUM_CTX;
205
+ }
206
+ /** Resolve the flush cadence: a non-negative finite
207
+ * HICORTEX_OLLAMA_FLUSH_EVERY wins (0 = the valid off value); invalid warns
208
+ * and keeps OLLAMA_FLUSH_EVERY. */
209
+ function resolveOllamaFlushEvery() {
210
+ const raw = process.env[exports.DIAGNOSTIC_ENV_TIER.ollamaFlushEvery.env];
211
+ if (raw === undefined || raw === "")
212
+ return exports.OLLAMA_FLUSH_EVERY;
213
+ const v = Number(raw);
214
+ if (Number.isFinite(v) && v >= 0)
215
+ return Math.floor(v);
216
+ console.warn(`[hicortex] env HICORTEX_OLLAMA_FLUSH_EVERY=${JSON.stringify(raw)} is not a non-negative finite number — using default ${exports.OLLAMA_FLUSH_EVERY}.`);
217
+ return exports.OLLAMA_FLUSH_EVERY;
218
+ }
219
+ /** Resolve the post-flush wait: a positive finite
220
+ * HICORTEX_OLLAMA_FLUSH_WAIT_MS wins; invalid warns and keeps
221
+ * OLLAMA_FLUSH_WAIT_MS. */
222
+ function resolveOllamaFlushWaitMs() {
223
+ const raw = process.env[exports.DIAGNOSTIC_ENV_TIER.ollamaFlushWaitMs.env];
224
+ if (raw === undefined || raw === "")
225
+ return exports.OLLAMA_FLUSH_WAIT_MS;
226
+ const v = Number(raw);
227
+ if (Number.isFinite(v) && v > 0)
228
+ return v;
229
+ console.warn(`[hicortex] env HICORTEX_OLLAMA_FLUSH_WAIT_MS=${JSON.stringify(raw)} is not a positive finite number — using default ${exports.OLLAMA_FLUSH_WAIT_MS}.`);
230
+ return exports.OLLAMA_FLUSH_WAIT_MS;
231
+ }
package/dist/capture.d.ts CHANGED
@@ -14,6 +14,7 @@
14
14
  */
15
15
  import type { TranscriptBatch } from "./transcript-reader.js";
16
16
  import type { CursorStore } from "./capture-cursors.js";
17
+ import type { RunDeadline } from "./run-deadline.js";
17
18
  /**
18
19
  * Max denoised chars per segment. Kept below the server's 80K distill cap
19
20
  * (distiller.ts MAX_TRANSCRIPT_CHARS) with ~20K headroom so NO capture path can
@@ -105,6 +106,15 @@ export interface CaptureOptions {
105
106
  * `source_domain` provenance. Null when undeclared.
106
107
  */
107
108
  sourceDomain?: string | null;
109
+ /**
110
+ * The run-wide pipeline deadline (#405), checked BETWEEN segment POSTs —
111
+ * a boundary the per-session cursor discipline already guarantees is safe
112
+ * (the cursor only advances past server-confirmed segments, so a deadline
113
+ * stop holds every unconfirmed segment for the next run; dup-over-loss).
114
+ * Full and consolidate-only nightlies pass it; capture-only/watchdog runs
115
+ * keep their 30-min unit backstop instead.
116
+ */
117
+ deadline?: RunDeadline;
108
118
  }
109
119
  export interface CaptureResult {
110
120
  memoriesIngested: number;
@@ -113,10 +123,12 @@ export interface CaptureResult {
113
123
  /**
114
124
  * Set when the loop stopped early on a terminal server response: "limit"
115
125
  * (token-budget 429, mcp-server.ts's `"token budget exceeded"` gate) or
116
- * "auth" (401). A rate-limit 429 never sets this — it is transient (#327).
117
- * The caller decides watermark handling.
126
+ * "auth" (401); "deadline" (#405) when the run-wide pipeline deadline fired
127
+ * between segments (transient — the watermark holds, everything unconfirmed
128
+ * retries next run). A rate-limit 429 never sets this — it is transient
129
+ * (#327). The caller decides watermark handling.
118
130
  */
119
- stopped?: "limit" | "auth";
131
+ stopped?: "limit" | "auth" | "deadline";
120
132
  /**
121
133
  * Run-global rate-429 latch (#327 CR): true when at least one session
122
134
  * SURRENDERED to a rate-limit 429 (the transient kind — postWithRateRetry
package/dist/capture.js CHANGED
@@ -204,7 +204,7 @@ async function postWithRateRetry(post, body) {
204
204
  * re-paying the Retry-After ladder (#327).
205
205
  */
206
206
  async function captureBatches(batches, opts) {
207
- const { post, cursorStore, dryRun = false, segmentMaxChars = exports.SEGMENT_MAX_CHARS, sourceAgentId, sourceDomain } = opts;
207
+ const { post, cursorStore, dryRun = false, segmentMaxChars = exports.SEGMENT_MAX_CHARS, sourceAgentId, sourceDomain, deadline } = opts;
208
208
  let memoriesIngested = 0;
209
209
  let sessionsSent = 0;
210
210
  let hadTransientFailure = false;
@@ -249,6 +249,15 @@ async function captureBatches(batches, opts) {
249
249
  let sessionPosted = false;
250
250
  for (let s = 0; s < segments.length; s++) {
251
251
  const seg = segments[s];
252
+ // #405: stop BETWEEN segments — a safe boundary by construction (the
253
+ // cursor below only advances past server-confirmed segments). The whole
254
+ // session loop breaks on `stopped` at the bottom; unconfirmed segments
255
+ // hold and retry next run.
256
+ if (!dryRun && deadline?.hit("capture")) {
257
+ console.warn(`[hicortex] Run deadline reached — capture stops after the last confirmed segment`);
258
+ stopped = "deadline";
259
+ break;
260
+ }
252
261
  // A segment advances the cursor to its segEnd only when it is the LAST
253
262
  // segment ending at that boundary. Hard-split pieces (.p0,.p1,…) of one
254
263
  // entry share the same segEnd; confirming an earlier piece must NOT move
@@ -50,6 +50,12 @@ export interface ClassifyDomainsOptions {
50
50
  llm?: LlmClient;
51
51
  /** Config override (tests). Defaults to reading stateDir/config.json. */
52
52
  config?: Record<string, unknown> | null;
53
+ /**
54
+ * Weak-primary floor (#408): release-managed default (calibration.ts via
55
+ * nofit's DEFAULT_WEAK_PRIMARY_FLOOR); this field is the eval/test seam —
56
+ * the config key is gone from the surface. Invalid → default.
57
+ */
58
+ weakPrimaryFloor?: number;
53
59
  /**
54
60
  * Embedder override (tests). Used only for domain-description prototype
55
61
  * seeds; defaults to the local ONNX embedder, loaded lazily on first need
@@ -118,7 +118,13 @@ async function runClassifyDomains(options = {}) {
118
118
  '{ "name": "Boating", "description": "..." }] } and re-run. ' +
119
119
  "No fallback bucket is needed — no-fit memories are handled automatically.");
120
120
  }
121
- const weakPrimaryFloor = (0, nofit_js_1.resolveWeakPrimaryFloor)(config);
121
+ // #408: the floor is a release-managed calibration constant; the Options
122
+ // field is the eval/test seam (invalid values keep the default, the stage
123
+ // knob-validation style — silent fallback, no warn).
124
+ const floorRaw = Number(options.weakPrimaryFloor);
125
+ const weakPrimaryFloor = Number.isFinite(floorRaw) && floorRaw > 0 && floorRaw < 1
126
+ ? floorRaw
127
+ : nofit_js_1.DEFAULT_WEAK_PRIMARY_FLOOR;
122
128
  // Resolve the LLM (one model serves all phases — #231).
123
129
  let llm;
124
130
  if (options.llm) {