@gamaze/hicortex 0.20.7 → 0.20.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -41
- package/dist/calibration.d.ts +174 -0
- package/dist/calibration.js +231 -0
- package/dist/capture.d.ts +15 -3
- package/dist/capture.js +10 -1
- package/dist/classify-domains.d.ts +6 -0
- package/dist/classify-domains.js +7 -1
- package/dist/cli.js +2 -3
- package/dist/config-read.d.ts +1 -1
- package/dist/config-read.js +96 -9
- package/dist/consolidate.d.ts +79 -68
- package/dist/consolidate.js +218 -174
- package/dist/dashboard.d.ts +4 -3
- package/dist/dedup.d.ts +34 -26
- package/dist/dedup.js +91 -57
- package/dist/distiller.js +1 -1
- package/dist/domain-classify.d.ts +7 -6
- package/dist/domain-classify.js +12 -10
- package/dist/eval/decay-eval.d.ts +3 -3
- package/dist/eval/decay-eval.js +4 -4
- package/dist/eval/planted-eval.d.ts +26 -0
- package/dist/eval/planted-eval.js +97 -0
- package/dist/eval/planted-fixtures.d.ts +107 -0
- package/dist/eval/planted-fixtures.js +283 -0
- package/dist/eval/planted-harness.d.ts +176 -0
- package/dist/eval/planted-harness.js +649 -0
- package/dist/index.js +4 -3
- package/dist/init.d.ts +9 -3
- package/dist/init.js +52 -9
- package/dist/llm.d.ts +43 -58
- package/dist/llm.js +87 -101
- package/dist/mcp-server.js +29 -29
- package/dist/nightly.js +105 -103
- package/dist/nofit.d.ts +4 -11
- package/dist/nofit.js +6 -23
- package/dist/recall-index.d.ts +30 -28
- package/dist/recall-index.js +21 -18
- package/dist/recall-registry.d.ts +2 -1
- package/dist/recall-registry.js +35 -1
- package/dist/reconsolidation.d.ts +124 -72
- package/dist/reconsolidation.js +359 -148
- package/dist/relink.js +3 -4
- package/dist/retrieval.d.ts +68 -35
- package/dist/retrieval.js +292 -104
- package/dist/run-deadline.d.ts +62 -0
- package/dist/run-deadline.js +73 -0
- package/dist/schema-prototypes.d.ts +3 -3
- package/dist/schema-prototypes.js +3 -3
- package/dist/state.d.ts +2 -3
- package/dist/storage.d.ts +16 -16
- package/dist/storage.js +62 -24
- package/dist/telemetry.d.ts +8 -7
- package/dist/token-budget.js +3 -4
- package/dist/type-classify.js +4 -4
- package/dist/types.d.ts +95 -155
- package/domains.example.json +4 -5
- package/hermes-plugin/hicortex/README.md +2 -2
- package/openclaw.plugin.json +1 -1
- package/package.json +2 -1
- package/pi-extension/hicortex/README.md +1 -1
- package/server.json +3 -3
package/README.md
CHANGED
|
@@ -116,11 +116,11 @@ Edit `~/.hicortex/config.json` on the server machine:
|
|
|
116
116
|
}
|
|
117
117
|
```
|
|
118
118
|
|
|
119
|
-
A richer example
|
|
119
|
+
A richer power-user example (a wider life-sphere set) ships as `domains.example.json` in the package.
|
|
120
120
|
|
|
121
121
|
**How classification works:** the LLM decides only *which* of your domains apply to a memory — never weights or rankings. The weight of each tag is derived from your own data: each domain builds a prototype from the memories already in it, and a tag's weight is how strongly the memory's embedding matches that prototype. The primary domain is picked deterministically from those weights, and everything is recomputed each nightly, so your categories drift with your data instead of going stale. Memories that genuinely fit nothing get a weak association when they are close enough to some domain — and otherwise fade away over time. No junk drawer, no "Unsorted" pile.
|
|
122
122
|
|
|
123
|
-
|
|
123
|
+
How close a no-fit memory must be to its nearest domain to earn that weak association (instead of fading) is a release-managed calibration constant — it ships with each release and changes only with published eval evidence, not a config key.
|
|
124
124
|
|
|
125
125
|
**Backfill an existing corpus** (server mode, needs `domains` in config):
|
|
126
126
|
|
|
@@ -250,19 +250,12 @@ Config at `~/.hicortex/config.json`. Created by `init`. Key options:
|
|
|
250
250
|
| `mode` | `"server"` (default) or `"client"` |
|
|
251
251
|
| `serverUrl` | Remote server URL (client mode) |
|
|
252
252
|
| `llmModel` | The one model used by all phases (distill, score, classify, reflect). Set via `init`. |
|
|
253
|
-
| `numCtx` | Context window for ollama (default 8192, one value for all phases). Scoring uses ~850 tokens, so 2048 is ample; distill/reflect/classify need more for `detectChunkSize`'s chunk sizing. |
|
|
254
253
|
| `enableThinking` | Toggle the model's internal reasoning ("thinking") stream for OpenAI-compatible endpoints (default false). Only meaningful for local chat-template-aware servers (ollama, mlx-lm); leave unset for cloud OpenAI/OpenRouter/Groq endpoints (they 400 on the unknown `chat_template_kwargs` field). |
|
|
255
254
|
| `maxTokens` | Max output tokens for all phases (default 8192). A ceiling, not a target — the model stops early when done. |
|
|
256
|
-
| `classifyMaxTokens` | Max output tokens for the classify tier — the short JSON-verdict calls (correction/supersession verdicts, rewrite contracts, type and domain tag classification). Default 1024. A ceiling, not a target: raise it when a reasoning-style model spends the budget on internal reasoning and returns empty verdicts; `maxTokens` keeps governing the heavy phases (distill, reflect). |
|
|
257
|
-
| `ollamaFlushEvery` | Flush ollama's accumulated memory every N scoring calls. **Off by default (0)** — opt-in only for an **ollama** install whose runner RSS growth (~171 MB/call) swap-thrashes long consolidations on a RAM-constrained box; N=15 caps a cycle at ~2.5 GB. Gated on the provider being ollama (local **or** remote) — no effect for non-ollama providers. Only you can judge whether your ollama endpoint actually suffers the growth (a managed/cloud ollama host may not), so it stays off until you set it. |
|
|
258
|
-
| `ollamaFlushWaitMs` | Milliseconds to wait after an ollama flush for the runner to exit + release memory (default 180000 = 3 min). |
|
|
259
255
|
| `llmTimeoutMs` | The ONE timeout ceiling on every LLM call in every phase (default 900000 = 15 min). The LLM request paths disable the HTTP client's hidden 5-minute response-header timer, so this knob is the only bound — one place to tune when the endpoint is slow, no per-phase special cases. |
|
|
260
|
-
| `
|
|
261
|
-
| `llmBreakerCooldownMs` | How long the breaker stays open before one half-open trial call goes out (default 600000 = 10 min). A failing trial re-opens it; a succeeding one resets the counter. |
|
|
262
|
-
| `llmProbeTimeoutMs` | Patience of the readiness probe — one minimal 1-token generation request the nightly sends before consolidating and the daemon sends before distilling (default 60000 = 1 min). Catches a gateway that answers health/model-list queries while generation is dead; a failed probe skips consolidation (`endpoint_down`, retried next run) and answers `/distill` with a 503 so capture holds its cursor. |
|
|
256
|
+
| `llmProbeTimeoutMs` | Patience of the readiness probe — one minimal 1-token generation request the daemon sends before distilling (default 60000 = 1 min). Catches a gateway that answers health/model-list queries while generation is dead; a failed probe answers `/distill` with a 503 so capture holds its cursor. The nightly no longer probes — its dead-endpoint signal is the circuit breaker (`endpoint_down`, retried next run) |
|
|
263
257
|
| `llmProbeTtlMs` | How long the daemon caches a `/distill` probe outcome (default 300000 = 5 min). A healthy capture cadence pays at most one probe per window; a dead endpoint turns into fast cached 503s instead of every request paying the probe timeout. |
|
|
264
258
|
| `llmSingleFlight` | **Serialized LLM calls — default `true`.** At most ONE request in flight per endpoint at any moment, across every process (the daemon distilling concurrent captures, the nightly consolidating, CLI backfills take turns via a per-endpoint lock file). One Hicortex server is several callers at once, and local single-user model servers (a Mac mini or laptop serving one big-context model) can stall or crash — taking the machine with them — under two concurrent large requests. Queued calls wait their turn; batches run back-to-back. **Set `false` if your endpoint is a beefy multi-tenant service that parallelizes well and you want faster consolidation** — you opt into responsibility for the endpoint's concurrency safety. |
|
|
265
|
-
| `llmSingleFlightWaitMs` | How long a queued call waits for the in-flight call before failing as endpoint-down and retrying per the normal ladder/breaker rules (default: `max(900000, llmTimeoutMs)` — a waiter never gives up before a legitimate in-flight call's own, possibly raised, ceiling expires). Only meaningful with serialization on. Note under contention: the readiness probe queues too, but its wait is capped by its own probe budget — during a long consolidation batch, `/distill` answers 503 quickly (captures cursor-hold and retry, lossless) rather than holding the socket for the full queue budget. |
|
|
266
259
|
| `authToken` | Bearer token for endpoint auth. Generated on first `init` in server mode. Find the active token with `hicortex status` or in `~/.hicortex/config.json`. |
|
|
267
260
|
| `corsAllowedOrigins` | Browser origins allowed to read cross-origin responses, e.g. `["https://ui.example.com"]`. **Empty by default** — the server sends no `Access-Control-Allow-Origin` and never `Allow-Credentials`, so no external web page can read its data. The bundled `/viz` and `/identity/ui` pages are same-origin and need no entry. |
|
|
268
261
|
| `licenseKey` | Commercial license key (optional; for display in `hicortex status`) |
|
|
@@ -270,8 +263,6 @@ Config at `~/.hicortex/config.json`. Created by `init`. Key options:
|
|
|
270
263
|
| _env_ `HICORTEX_DISTILL_BODY_LIMIT_MB` | Environment override for the `/distill` body limit — **wins over the `distillBodyLimitMb` config key in every mode** (that is the point: a deployment operator pins it so tenant-writable config cannot raise it). Unset = config/default applies. |
|
|
271
264
|
| _env_ `HICORTEX_MEMORY_CAP` | Environment override for the memory soft cap — a **positive** value wins over the `memorySoftCap` config key in every mode; `0`/negative/malformed fall through to the config key, then the 10000 default (an env can pin a cap, never disable one — config `memorySoftCap: 0` still disables when the env is unset). Mode-agnostic operator knob: nightly eviction, the dashboard-snapshot capacity stamp, and the live dashboard gauge all resolve through the same resolver, so the enforced and displayed caps can never disagree. Drives `nightly --evict-only` the same way. |
|
|
272
265
|
| `domains` | Your memory domain list (`[{name, description}]`). Scaffolded by `init`; edit freely — see [Memory Domains & Tags](#memory-domains--tags) |
|
|
273
|
-
| `weakPrimaryFloor` | Minimum similarity for a no-fit memory to keep a weak domain association (default: 0.45) |
|
|
274
|
-
| `moduleIndexTokenBudget` | Max tokens for domain index in lessons context (default: 500) |
|
|
275
266
|
| `lessonsLimit` | Max lessons injected into an agent's session-start context (default: 10). Lessons are ranked per session by project/domain affinity + recency + strength + access, so each session sees its most-relevant slice. Lower = leaner system prompts. |
|
|
276
267
|
| `identityClients` | Which harnesses inject the [identity layer](#identity-layer) at session start (default `["cc"]`; `"all"` or any subset of `cc`/`hermes`/`oc`/`pi`/`opencode`) |
|
|
277
268
|
| `identityAgents` | Per-agent identity modes (0.13): `{ "<id>": "override" \| "global" \| "off" }`. Absent + no `agents/<id>/` dir → every agent gets the global set. Boot-time (restart to apply) — see [Per-agent identity](#per-agent-identity-013) |
|
|
@@ -279,39 +270,17 @@ Config at `~/.hicortex/config.json`. Created by `init`. Key options:
|
|
|
279
270
|
| `captureCooldownHours` | Success-cooldown (hours) for the **capture watchdog** (0.17). The capture timer polls every ~20 min; the watchdog captures only if more than this has elapsed since the last *successful* capture (`state.lastNightly`). Default `6` (≈4 captures/day). A failed preflight retries on the next poll (~20 min) — so a transient fire-instant network miss costs minutes, not a day (#239) |
|
|
280
271
|
| `consolidationHours` | Hours (0–23, local) for the **consolidation** timer — the full nightly (capture + distill + score + reflect + link). Installed for **server/co-located only** (clients have no local DB). Default `[10, 22]`: the 22:00 evening slot runs after the day's capture waves (same-day results); the 10:00 morning slot runs *after* the morning capture so wake-up pushes are caught. Omitted on clients |
|
|
281
272
|
| `timerJitterSeconds` | Max random delay (seconds) added to **generated** consolidation timers (#256), so a fleet doesn't all fire on the same minute (thundering-herd → LLM-backend contention). systemd: a single `RandomizedDelaySec=<n>`; launchd has no native equivalent so a per-install randomized `Minute` offset is baked into every `StartCalendarInterval` dict (sub-60s values no-op on launchd). Default `3600` (≈±30 min spread on the 2-slot/day cadence); `0` disables. Affects timers on the next `init` (re-init rewrites the unit files; installs that don't re-init keep their existing timers) |
|
|
282
|
-
| `
|
|
273
|
+
| `nightlyTimeBudgetMinutes` | The ONE wall-clock budget (minutes) for a nightly run: capture and every consolidation stage share one cooperative deadline, checked at safe boundaries (capture segments, stage boundaries, item loops, the merge zone). A run whose deadline fires reports consolidation `deferred` and resumes from its cursors next run — no work lost, none redone. Default `240`; `0`/invalid → default (a deadline always exists — there is no "off"). The systemd unit's `TimeoutStartSec` is derived from this (+60 min slack) at `init` |
|
|
274
|
+
| `nightlyLlmCallBudget` | The ONE per-run ceiling on LLM calls across the whole pipeline. Consumed in run order — a stage that exhausts it defers its remainder via its cursor. Bounds money/load independent of latency: a fast metered or capacity-limited endpoint permits thousands of calls inside the wall-clock budget, so time alone cannot protect it. Default `5000`; `0`/invalid → default. `consolidateMaxLlmCalls` is a **deprecated alias** (honored one release when the new key is absent — rename it) |
|
|
283
275
|
| `memorySoftCap` | Soft cap on the memory corpus (default 10000). When the corpus exceeds this, the nightly's capacity-eviction stage removes the lowest-`effectiveStrength` memories (ties broken by oldest access) until under the cap — the active forgetting mechanism that bounds DB size, vector-index RAM, and consolidation workload. `0` disables eviction (indefinite growth — the pre-#245 behaviour). The evicted tail is cold by construction (effectiveStrength is the same decay-weighted score the recall ranker uses, so these were not surfacing in the top-k anyway). At 10K memories the load + JS sort is <100 ms |
|
|
284
276
|
| `updateChannel` | Release channel pinned into the generated daemon/timer ExecStart for **npx-thin** installs (global-binary installs use the absolute binary and are unaffected). A dist-tag (`"rc"`, `"next"`) or an exact version (`"0.17.1"`). E.g. `"rc"` → the timer runs `npx -y @gamaze/hicortex@rc nightly`, so the host tracks the rc dist-tag (an internal fleet can ride rc through a pre-promotion soak). Validated as `[\w.\-]+` (rejects anything that'd break the unit/plist templates). Absent → auto-detect (bare on `latest`, else `@next`). (0.17.1) |
|
|
285
277
|
| `nightlyHour` | **Deprecated (0.17) single-slot fallback.** Local hour (0–23) honoured only when `consolidationHours` is absent — yields one daily consolidation slot at that hour (preserves the pre-0.17 "one daily job" intent). New installs should use `consolidationHours` |
|
|
286
|
-
| `preflightTimeoutMs` | **Client mode only.** Per-attempt timeout for the nightly's server-reachability check before it starts capturing (default: 20000 ms, bumped from 15000 in 0.17 to absorb a slow link re-establishing after the client wakes) |
|
|
287
|
-
| `preflightAttempts` | **Client mode only.** Reachability-check retries before the nightly aborts (default: 3; floored at 1). `1` = single try, no retry |
|
|
288
|
-
| `preflightRetryGapMs` | **Client mode only.** Delay between reachability retries (default: 60000 ms). Note: timers don't advance while the machine is asleep, so on a sleeping laptop this gap counts awake-time, not wall-clock |
|
|
289
|
-
| `scoreSimilarityWeight` | Weight of semantic similarity in the ranking score (default: 0.50) |
|
|
290
|
-
| `scoreStrengthWeight` | Weight of effective strength — importance/use/recency of access (default: 0.20) |
|
|
291
|
-
| `scoreConnectionsWeight` | Weight of graph centrality (default: 0.15) |
|
|
292
|
-
| `scoreRecencyWeight` | Weight of the slow recency curve (default: 0.15) |
|
|
293
|
-
| `freshnessBoostDays` | Fresh-memory window: new memories rank higher for this many days (default: 7) |
|
|
294
|
-
| `freshnessBoostWeight` | Size of the fresh-memory bonus at age 0, fading linearly to 0 at the window edge (default: 0.15; set 0 to disable) |
|
|
295
|
-
| `supersededDemotion` | Score multiplier for a memory a later decision reversed (default: 0.50) |
|
|
296
|
-
| `decayHalfLifeDays` | Memory decay half-life in days at reference importance (default: 365). Larger = slower forgetting; importance, access, and links slow it further |
|
|
297
|
-
| `searchLimit` / `recentLimit` | Default result counts for search (8) and recent (12) |
|
|
298
|
-
| `recentWindowDays` | Candidate window for recent recall (default: 180) |
|
|
299
|
-
| `coldExposureSlots` | Top-k slots reservable for never-accessed memories so the long tail gets exposure (default: 2) |
|
|
300
|
-
| `recallMaxItems` | Max lines in the pushed recall index (default: 5) |
|
|
301
|
-
| `noveltyFloorSlots` | Slots of `recallMaxItems` guaranteed to the top passing hit(s) of the pure-prompt (unblended) search — the novelty floor. Keeps a session whose earlier turns set a strong intent from burying a topic-switching prompt's best matches: the floor's picks render first, turn-based re-show suppression still applies, and the total never exceeds `recallMaxItems` (default: 2; set 0 to disable) |
|
|
302
|
-
| `recallMinSimilarity` | Relevance floor for index entries (default: 0.62; text-search matches always pass) |
|
|
303
|
-
| `recallReshowTurns` | Turns before an already-shown memory may reappear in the same session (default: 30) |
|
|
304
|
-
| `recallMinPromptChars` | Prompts shorter than this skip the recall index (default: 20) |
|
|
305
|
-
| `recallTitleChars` | Chars of each memory's first line shown in an index entry (default: 100, range 40–400). Reverted from 150 on 2026-08-03: a full-corpus relevance eval found 100 and 150 statistically identical while 100 saves ~13% of the block's tokens |
|
|
306
|
-
| `sessionIntentWeight` | Blend weight of the session-intent rolling centroid in the recall search vector: `query = (1-w)·prompt + w·centroid` (default: 0.33; set 0 to disable — pure-prompt recall, the kill-switch). The first turn of a session searches with pure prompt and seeds the centroid; subsequent turns blend so recall follows the session's intent instead of being query-literal. The EMA rate (0.4) is a shipped constant, not configurable |
|
|
307
|
-
| `dedupAutoMergeThreshold` | The deterministic merge ceiling of the unified resolution pass: memory pairs at/above this cosine merge automatically (zero LLM) via the dedup core's clustering; pairs between `correctionMinSimilarity` and this value get the one merge/corrects/supersedes/none verdict. Also the default threshold for `hicortex dedup` (default: 0.92) |
|
|
308
|
-
| `dedupMergeThreshold` | Legacy alias for `dedupAutoMergeThreshold`, still honored when the newer key is absent |
|
|
309
|
-
| `dedupNightlyMaxMerges` | Pacing cap on merge operations per nightly run — deterministic-zone clusters plus verdict-confirmed pair merges count against one cap, so a large duplicate backlog drains over a few nights (default: 250; `0` disables the merge machinery) |
|
|
310
|
-
| `supersessionMinSimilarity` | Minimum cosine similarity for a nightly supersession candidate pair (default: 0.80) |
|
|
311
|
-
| `supersessionMaxCalls` | Max classify-tier LLM calls the nightly's supersession stage spends per run (default: 30) |
|
|
312
|
-
| `supersessionPenalty` | Multiplier applied to a superseded memory's `base_strength` (default: 0.5) |
|
|
313
278
|
| `telemetry` | Anonymous usage telemetry. **On by default and not written into config by `init`** — add `"telemetry": false` yourself (or set `HICORTEX_TELEMETRY=off`) to opt out. Inspect exactly what is sent with `hicortex telemetry` |
|
|
314
279
|
|
|
280
|
+
**Calibration is release-managed.** The ~35 tuning keys earlier releases exposed (recall breadth, relevance floors, ranking weights, decay speed, dedup/supersession/correction thresholds, the weak-primary floor) are no longer config: they are constants that ship with each release and change only in releases, with the eval evidence linked in the changelog. Config values for them are ignored — the server prints a one-time boot warning naming each ignored key. Your config file now describes your *install* (mode, model, schedules, identity, budgets), not the brain's tuning.
|
|
281
|
+
|
|
282
|
+
**Diagnostic tier (environment).** Three niche, ollama-only operational values moved from config to environment variables: `HICORTEX_NUM_CTX` (context window for ollama, default 8192), `HICORTEX_OLLAMA_FLUSH_EVERY` (flush ollama's accumulated memory every N LLM calls; default 0 = off), and `HICORTEX_OLLAMA_FLUSH_WAIT_MS` (post-flush wait, default 180000). Pin them in a service unit's environment when needed; the old config keys are ignored with a boot warning naming the replacement.
|
|
283
|
+
|
|
315
284
|
Full docs: [hicortex.gamaze.com/docs/configuration.html](https://hicortex.gamaze.com/docs/configuration.html)
|
|
316
285
|
|
|
317
286
|
## REST API
|
|
@@ -355,7 +324,7 @@ Optional config (add to plugin entry in `~/.openclaw/openclaw.json`):
|
|
|
355
324
|
| `serverUrl` | `http://127.0.0.1:8787` | Hicortex server URL. Change for remote servers. |
|
|
356
325
|
| `authToken` | _(none)_ | Bearer token. Localhost bypasses auth; required for remote servers. Get the token from `hicortex status` on the server. |
|
|
357
326
|
| `defaultProject` | _(none)_ | Project name sent on recall, search, recent, and ingest whenever the gateway supplies no project (Hermes `default_project` parity). |
|
|
358
|
-
| `recallLimit` | `8` | Max memories per recall on the pre-0.14 `/search` fallback. The pushed recall index is sized by
|
|
327
|
+
| `recallLimit` | `8` | Max memories per recall on the pre-0.14 `/search` fallback. The pushed recall index is sized by the server's release-managed calibration — the server accepts no client limit. |
|
|
359
328
|
| `scaffoldDeadMan` | `true` | Auto-scaffold the dead-man identity-guard line into the agent workspace bootstrap (`BOOTSTRAP.md`) at startup. Set `false` to disable the write and any file creation entirely. |
|
|
360
329
|
|
|
361
330
|
If `serverUrl`/`authToken` are absent from the config, the `HICORTEX_URL` and `HICORTEX_AUTH_TOKEN` environment variables are used as fallbacks (config always wins).
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Release-managed calibration constants (#408) — the single home of every
|
|
3
|
+
* tuning value the product ships. One value, one definition, one provenance
|
|
4
|
+
* comment. Nothing in here is read from ~/.hicortex/config.json anymore: the
|
|
5
|
+
* ~35 tuning keys that 0.15–0.20 exposed as config are now CONSTANTS that
|
|
6
|
+
* move only in releases. The user config surface shrinks to the keys that
|
|
7
|
+
* describe an install (mode, model, schedules, identity, budgets), not the
|
|
8
|
+
* ones that tune the brain.
|
|
9
|
+
*
|
|
10
|
+
* EVOLUTION CONTRACT: these values change ONLY in releases, with the
|
|
11
|
+
* eval/band-stats evidence linked in the changelog line that moves them
|
|
12
|
+
* (eval harness = `npm run eval` + the resolution band stats in the nightly
|
|
13
|
+
* report). Never in a patch to quiet one corpus, never behind a new config
|
|
14
|
+
* key. The seams for EXPERIMENTS are the configure*() functions in
|
|
15
|
+
* retrieval.ts / storage.ts (and the stage Options fields) — the eval and
|
|
16
|
+
* the tests sweep values through them; production never passes anything, so
|
|
17
|
+
* every process scores with exactly these constants.
|
|
18
|
+
*
|
|
19
|
+
* Every value below equals the default the code shipped the day this module
|
|
20
|
+
* was introduced (verified by tests/calibration.test.ts) — an install that
|
|
21
|
+
* never set the old config keys sees byte-identical behavior. The old keys
|
|
22
|
+
* are warned as RELEASE-MANAGED at the config boundary (config-read.ts) so
|
|
23
|
+
* the removal is never silent.
|
|
24
|
+
*/
|
|
25
|
+
/**
|
|
26
|
+
* Memory-decay half-life (days) at the reference importance 0.5. #192 recall/
|
|
27
|
+
* decay alignment: was ~115 days — aggressive enough to bury the long tail in
|
|
28
|
+
* ranking. Long-term remembering is the product; time preference stays mild.
|
|
29
|
+
*/
|
|
30
|
+
export declare const DECAY_HALF_LIFE_DAYS = 365;
|
|
31
|
+
/** Default k for retrieve() (/search without an explicit limit). #192. */
|
|
32
|
+
export declare const SEARCH_LIMIT = 8;
|
|
33
|
+
/** Default k for searchRecent() (/recent without an explicit limit). #192. */
|
|
34
|
+
export declare const RECENT_LIMIT = 12;
|
|
35
|
+
/** searchRecent() candidate window, days. #192. */
|
|
36
|
+
export declare const RECENT_WINDOW_DAYS = 180;
|
|
37
|
+
/** Top-k slots reservable for never-accessed memories (cold exposure). #192:
|
|
38
|
+
* recall was too passive (88% of memories never accessed) — the long tail
|
|
39
|
+
* gets guaranteed slots instead of waiting for the strength clock. */
|
|
40
|
+
export declare const COLD_EXPOSURE_SLOTS = 2;
|
|
41
|
+
/** Blend weight of the session-intent centroid in the recall search vector
|
|
42
|
+
* (#192, 0.15.3): query = (1-w)·prompt + w·centroid. The kill-switch is the
|
|
43
|
+
* configureSessionIntent(0) seam (eval-only); production always ships 0.33. */
|
|
44
|
+
export declare const SESSION_INTENT_WEIGHT = 0.33;
|
|
45
|
+
/** Relevance-gate floor for vector-only /recall-index candidates. 0.62
|
|
46
|
+
* (raised from 0.55 on 2026-08-03 per a 0.01-step floor sweep on the
|
|
47
|
+
* rewritten corpus): steady ~3:1 noise:signal removal with no knee; sits
|
|
48
|
+
* below the 0.63 local pessimum. FTS-matched candidates pass regardless. */
|
|
49
|
+
export declare const RECALL_MIN_SIMILARITY = 0.62;
|
|
50
|
+
/** Max lines in the pushed recall index. 5 (lowered from 6 on 2026-08-03):
|
|
51
|
+
* per-slot decomposition at floor 0.62 showed slot 6 gives NO prompt its
|
|
52
|
+
* first relevant memory. The K-sweep is monotone toward 4, but the 4-vs-5
|
|
53
|
+
* distinction rests on 5 of 98 prompts — 5 hedges with coverage. */
|
|
54
|
+
export declare const RECALL_MAX_ITEMS = 5;
|
|
55
|
+
/** Prompts shorter than this skip the recall index (continuations, "yes"). */
|
|
56
|
+
export declare const RECALL_MIN_PROMPT_CHARS = 20;
|
|
57
|
+
/** Chars of a memory's first line shown in an index entry. 100 (reverted
|
|
58
|
+
* from 150 on 2026-08-03): the full-corpus relevance eval found 100 vs 150
|
|
59
|
+
* statistically identical (full CI overlap at N=40); 100 saves ~13% tokens. */
|
|
60
|
+
export declare const RECALL_TITLE_CHARS = 100;
|
|
61
|
+
/** Slots of RECALL_MAX_ITEMS guaranteed to the pure-prompt (unblended)
|
|
62
|
+
* search's top passing hit(s) — the #324 novelty floor. 2 mirrors
|
|
63
|
+
* COLD_EXPOSURE_SLOTS sizing: a floor, never a takeover. */
|
|
64
|
+
export declare const NOVELTY_FLOOR_SLOTS = 2;
|
|
65
|
+
/** Turns an already-shown memory stays suppressed in the same session before
|
|
66
|
+
* it may reappear in the pushed index (#192 turn-based dedup). */
|
|
67
|
+
export declare const RECALL_RESHOW_TURNS = 30;
|
|
68
|
+
/** Semantic-similarity share of the composite score. 0.50 (raised from 0.40
|
|
69
|
+
* in the 0.15.2 rebalance, #191 Phase B): on the production corpus effective
|
|
70
|
+
* strength (0.30) outweighed what similarity could recover — hardened old
|
|
71
|
+
* memories beat exact matches for their own topic. Similarity now leads;
|
|
72
|
+
* strength breaks ties and rewards real use. */
|
|
73
|
+
export declare const SCORE_SIMILARITY_WEIGHT = 0.5;
|
|
74
|
+
/** Effective-strength share of the composite score (was 0.30; see above). */
|
|
75
|
+
export declare const SCORE_STRENGTH_WEIGHT = 0.2;
|
|
76
|
+
/** Graph-centrality share of the composite score (was 0.20; see above). */
|
|
77
|
+
export declare const SCORE_CONNECTIONS_WEIGHT = 0.15;
|
|
78
|
+
/** Slow recency curve share of the composite score (was 0.10; see above). */
|
|
79
|
+
export declare const SCORE_RECENCY_WEIGHT = 0.15;
|
|
80
|
+
/** Fresh-memory window: the additive bonus fades linearly to 0 over this
|
|
81
|
+
* many days. 7 — nightly capture means 1 day is the floor of "fresh"
|
|
82
|
+
* (#191 Phase B). */
|
|
83
|
+
export declare const FRESHNESS_BOOST_DAYS = 7;
|
|
84
|
+
/** Fresh-memory bonus size at age 0 (#191 Phase B; 0 = disabled via seam). */
|
|
85
|
+
export declare const FRESHNESS_BOOST_WEIGHT = 0.15;
|
|
86
|
+
/** Score multiplier for a memory a later decision superseded (0.15.2; the
|
|
87
|
+
* belief walk (#393 D) is the primary mechanism — this is the safety net
|
|
88
|
+
* for rows the walk does not reach). */
|
|
89
|
+
export declare const SUPERSEDED_DEMOTION = 0.5;
|
|
90
|
+
/** #203 soft boost on exact project match. ADDITIVE, zero-boost neutral,
|
|
91
|
+
* never a penalty — a foreign memory ranks equal, not lower. */
|
|
92
|
+
export declare const PROJECT_AFFINITY_WEIGHT = 0.15;
|
|
93
|
+
/** #203 soft boost multiplier on max overlapping domain-tag weight. */
|
|
94
|
+
export declare const DOMAIN_AFFINITY_WEIGHT = 0.15;
|
|
95
|
+
/** #205 RRF k parameter (1/(k+rank+1)) — matches the pre-#205 hardcoded 60
|
|
96
|
+
* so the no-config path was byte-identical to 0.15.3. */
|
|
97
|
+
export declare const RRF_K = 60;
|
|
98
|
+
/** #205 composite-score share of the final blend (RRF gets the remainder);
|
|
99
|
+
* pre-#205 hardcoded value carried forward. */
|
|
100
|
+
export declare const RRF_COMPOSITE_WEIGHT = 0.8;
|
|
101
|
+
/** #205 per-list RRF weight for the FTS list. 0.5 is the bisection point
|
|
102
|
+
* where BM25F + composite-affinity flip the token-exact marine body match
|
|
103
|
+
* below the same-scope hardware field (Q4 contamination 0.20 → 0.00) while
|
|
104
|
+
* pure-keyword queries keep recall@5 = 1.0. 0.7 was measured too timid. */
|
|
105
|
+
export declare const RRF_FTS_WEIGHT = 0.5;
|
|
106
|
+
/** #205 per-list RRF weight for the vector list (vec stays at 1.0 — the
|
|
107
|
+
* conservative nudge is on the FTS side only). */
|
|
108
|
+
export declare const RRF_VECTOR_WEIGHT = 1;
|
|
109
|
+
export declare const BM25_WEIGHT_BODY = 1;
|
|
110
|
+
export declare const BM25_WEIGHT_PROJECT = 2;
|
|
111
|
+
export declare const BM25_WEIGHT_DOMAIN = 2;
|
|
112
|
+
/** Deterministic merge ceiling of the unified resolution pass (#392): pairs
|
|
113
|
+
* at/above this cosine merge LLM-free; [CORRECTION_MIN_SIMILARITY, this)
|
|
114
|
+
* get the one verdict call. 0.92 — measured on the #191 mechanical audit
|
|
115
|
+
* corpus (89 clusters / 110 excess rows; data/audit-20260729). */
|
|
116
|
+
export declare const DEDUP_AUTO_MERGE_THRESHOLD = 0.92;
|
|
117
|
+
/** Minimum cosine for a nightly supersession candidate pair (#100 stage,
|
|
118
|
+
* 0.15.0): one classify-tier call per pair above the bar. */
|
|
119
|
+
export declare const SUPERSESSION_MIN_SIMILARITY = 0.8;
|
|
120
|
+
/** Minimum cosine for a reconsolidation correction pair (#384). Deliberately
|
|
121
|
+
* wider than supersession's 0.80: a retraction often rides inside an
|
|
122
|
+
* otherwise unrelated memory; the verdict + confidence gate carry the
|
|
123
|
+
* precision. */
|
|
124
|
+
export declare const CORRECTION_MIN_SIMILARITY = 0.75;
|
|
125
|
+
/** Minimum verdict confidence for the REWRITE (and #392 merge-apply) fork
|
|
126
|
+
* (#384): below it a `corrects` degrades to mark-only — a weak mark is
|
|
127
|
+
* recoverable, a weak rewrite is corruption. */
|
|
128
|
+
export declare const CORRECTION_REWRITE_MIN_CONFIDENCE = 0.8;
|
|
129
|
+
/** Minimum cosine(memory embedding, best domain prototype) for a no-fit
|
|
130
|
+
* memory to earn a WEAK primary instead of decaying (owner amendment
|
|
131
|
+
* 07.07). Starting point for bge-small-en-v1.5. */
|
|
132
|
+
export declare const WEAK_PRIMARY_FLOOR = 0.45;
|
|
133
|
+
/**
|
|
134
|
+
* Context window for ollama (one value, all phases; #220/#228). 8192 is
|
|
135
|
+
* where context stops being the binding constraint for a sub-8B model on
|
|
136
|
+
* ollama. Also drives `detectChunkSize` (chunkChars ≤ numCtx × 0.6 × 4
|
|
137
|
+
* chars) so the chunker and the request agree by construction.
|
|
138
|
+
*/
|
|
139
|
+
export declare const NUM_CTX = 8192;
|
|
140
|
+
/** Flush ollama's accumulated memory every N LLM calls (0 = off; #220).
|
|
141
|
+
* Opt-in operational workaround for ollama runner RSS growth — never a
|
|
142
|
+
* default-on behavior. */
|
|
143
|
+
export declare const OLLAMA_FLUSH_EVERY = 0;
|
|
144
|
+
/** Ms to wait after an ollama flush (`keep_alive:0`) for the runner to exit
|
|
145
|
+
* + release memory. The runner takes >90 s to exit; 3 min allows margin. */
|
|
146
|
+
export declare const OLLAMA_FLUSH_WAIT_MS = 180000;
|
|
147
|
+
/** The env-tier table (release surface: names + defaults are frozen —
|
|
148
|
+
* adding a knob here is a release decision, not a runtime one). */
|
|
149
|
+
export declare const DIAGNOSTIC_ENV_TIER: Readonly<{
|
|
150
|
+
numCtx: Readonly<{
|
|
151
|
+
env: string;
|
|
152
|
+
default: number;
|
|
153
|
+
}>;
|
|
154
|
+
ollamaFlushEvery: Readonly<{
|
|
155
|
+
env: string;
|
|
156
|
+
default: number;
|
|
157
|
+
}>;
|
|
158
|
+
ollamaFlushWaitMs: Readonly<{
|
|
159
|
+
env: string;
|
|
160
|
+
default: number;
|
|
161
|
+
}>;
|
|
162
|
+
}>;
|
|
163
|
+
/** Resolve the effective ollama context window: a positive finite
|
|
164
|
+
* HICORTEX_NUM_CTX wins; anything else (absent/blank/invalid) keeps NUM_CTX
|
|
165
|
+
* with a warn on the invalid case. */
|
|
166
|
+
export declare function resolveNumCtx(): number;
|
|
167
|
+
/** Resolve the flush cadence: a non-negative finite
|
|
168
|
+
* HICORTEX_OLLAMA_FLUSH_EVERY wins (0 = the valid off value); invalid warns
|
|
169
|
+
* and keeps OLLAMA_FLUSH_EVERY. */
|
|
170
|
+
export declare function resolveOllamaFlushEvery(): number;
|
|
171
|
+
/** Resolve the post-flush wait: a positive finite
|
|
172
|
+
* HICORTEX_OLLAMA_FLUSH_WAIT_MS wins; invalid warns and keeps
|
|
173
|
+
* OLLAMA_FLUSH_WAIT_MS. */
|
|
174
|
+
export declare function resolveOllamaFlushWaitMs(): number;
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Release-managed calibration constants (#408) — the single home of every
|
|
4
|
+
* tuning value the product ships. One value, one definition, one provenance
|
|
5
|
+
* comment. Nothing in here is read from ~/.hicortex/config.json anymore: the
|
|
6
|
+
* ~35 tuning keys that 0.15–0.20 exposed as config are now CONSTANTS that
|
|
7
|
+
* move only in releases. The user config surface shrinks to the keys that
|
|
8
|
+
* describe an install (mode, model, schedules, identity, budgets), not the
|
|
9
|
+
* ones that tune the brain.
|
|
10
|
+
*
|
|
11
|
+
* EVOLUTION CONTRACT: these values change ONLY in releases, with the
|
|
12
|
+
* eval/band-stats evidence linked in the changelog line that moves them
|
|
13
|
+
* (eval harness = `npm run eval` + the resolution band stats in the nightly
|
|
14
|
+
* report). Never in a patch to quiet one corpus, never behind a new config
|
|
15
|
+
* key. The seams for EXPERIMENTS are the configure*() functions in
|
|
16
|
+
* retrieval.ts / storage.ts (and the stage Options fields) — the eval and
|
|
17
|
+
* the tests sweep values through them; production never passes anything, so
|
|
18
|
+
* every process scores with exactly these constants.
|
|
19
|
+
*
|
|
20
|
+
* Every value below equals the default the code shipped the day this module
|
|
21
|
+
* was introduced (verified by tests/calibration.test.ts) — an install that
|
|
22
|
+
* never set the old config keys sees byte-identical behavior. The old keys
|
|
23
|
+
* are warned as RELEASE-MANAGED at the config boundary (config-read.ts) so
|
|
24
|
+
* the removal is never silent.
|
|
25
|
+
*/
|
|
26
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
27
|
+
exports.DIAGNOSTIC_ENV_TIER = exports.OLLAMA_FLUSH_WAIT_MS = exports.OLLAMA_FLUSH_EVERY = exports.NUM_CTX = exports.WEAK_PRIMARY_FLOOR = exports.CORRECTION_REWRITE_MIN_CONFIDENCE = exports.CORRECTION_MIN_SIMILARITY = exports.SUPERSESSION_MIN_SIMILARITY = exports.DEDUP_AUTO_MERGE_THRESHOLD = exports.BM25_WEIGHT_DOMAIN = exports.BM25_WEIGHT_PROJECT = exports.BM25_WEIGHT_BODY = exports.RRF_VECTOR_WEIGHT = exports.RRF_FTS_WEIGHT = exports.RRF_COMPOSITE_WEIGHT = exports.RRF_K = exports.DOMAIN_AFFINITY_WEIGHT = exports.PROJECT_AFFINITY_WEIGHT = exports.SUPERSEDED_DEMOTION = exports.FRESHNESS_BOOST_WEIGHT = exports.FRESHNESS_BOOST_DAYS = exports.SCORE_RECENCY_WEIGHT = exports.SCORE_CONNECTIONS_WEIGHT = exports.SCORE_STRENGTH_WEIGHT = exports.SCORE_SIMILARITY_WEIGHT = exports.RECALL_RESHOW_TURNS = exports.NOVELTY_FLOOR_SLOTS = exports.RECALL_TITLE_CHARS = exports.RECALL_MIN_PROMPT_CHARS = exports.RECALL_MAX_ITEMS = exports.RECALL_MIN_SIMILARITY = exports.SESSION_INTENT_WEIGHT = exports.COLD_EXPOSURE_SLOTS = exports.RECENT_WINDOW_DAYS = exports.RECENT_LIMIT = exports.SEARCH_LIMIT = exports.DECAY_HALF_LIFE_DAYS = void 0;
|
|
28
|
+
exports.resolveNumCtx = resolveNumCtx;
|
|
29
|
+
exports.resolveOllamaFlushEvery = resolveOllamaFlushEvery;
|
|
30
|
+
exports.resolveOllamaFlushWaitMs = resolveOllamaFlushWaitMs;
|
|
31
|
+
// ---------------------------------------------------------------------------
|
|
32
|
+
// Recall / decay family (was: decayHalfLifeDays, searchLimit, recentLimit,
|
|
33
|
+
// recentWindowDays, coldExposureSlots, sessionIntentWeight, recall*,
|
|
34
|
+
// noveltyFloorSlots, recallReshowTurns)
|
|
35
|
+
// ---------------------------------------------------------------------------
|
|
36
|
+
/**
|
|
37
|
+
* Memory-decay half-life (days) at the reference importance 0.5. #192 recall/
|
|
38
|
+
* decay alignment: was ~115 days — aggressive enough to bury the long tail in
|
|
39
|
+
* ranking. Long-term remembering is the product; time preference stays mild.
|
|
40
|
+
*/
|
|
41
|
+
exports.DECAY_HALF_LIFE_DAYS = 365;
|
|
42
|
+
/** Default k for retrieve() (/search without an explicit limit). #192. */
|
|
43
|
+
exports.SEARCH_LIMIT = 8;
|
|
44
|
+
/** Default k for searchRecent() (/recent without an explicit limit). #192. */
|
|
45
|
+
exports.RECENT_LIMIT = 12;
|
|
46
|
+
/** searchRecent() candidate window, days. #192. */
|
|
47
|
+
exports.RECENT_WINDOW_DAYS = 180;
|
|
48
|
+
/** Top-k slots reservable for never-accessed memories (cold exposure). #192:
|
|
49
|
+
* recall was too passive (88% of memories never accessed) — the long tail
|
|
50
|
+
* gets guaranteed slots instead of waiting for the strength clock. */
|
|
51
|
+
exports.COLD_EXPOSURE_SLOTS = 2;
|
|
52
|
+
/** Blend weight of the session-intent centroid in the recall search vector
|
|
53
|
+
* (#192, 0.15.3): query = (1-w)·prompt + w·centroid. The kill-switch is the
|
|
54
|
+
* configureSessionIntent(0) seam (eval-only); production always ships 0.33. */
|
|
55
|
+
exports.SESSION_INTENT_WEIGHT = 0.33;
|
|
56
|
+
/** Relevance-gate floor for vector-only /recall-index candidates. 0.62
|
|
57
|
+
* (raised from 0.55 on 2026-08-03 per a 0.01-step floor sweep on the
|
|
58
|
+
* rewritten corpus): steady ~3:1 noise:signal removal with no knee; sits
|
|
59
|
+
* below the 0.63 local pessimum. FTS-matched candidates pass regardless. */
|
|
60
|
+
exports.RECALL_MIN_SIMILARITY = 0.62;
|
|
61
|
+
/** Max lines in the pushed recall index. 5 (lowered from 6 on 2026-08-03):
|
|
62
|
+
* per-slot decomposition at floor 0.62 showed slot 6 gives NO prompt its
|
|
63
|
+
* first relevant memory. The K-sweep is monotone toward 4, but the 4-vs-5
|
|
64
|
+
* distinction rests on 5 of 98 prompts — 5 hedges with coverage. */
|
|
65
|
+
exports.RECALL_MAX_ITEMS = 5;
|
|
66
|
+
/** Prompts shorter than this skip the recall index (continuations, "yes"). */
|
|
67
|
+
exports.RECALL_MIN_PROMPT_CHARS = 20;
|
|
68
|
+
/** Chars of a memory's first line shown in an index entry. 100 (reverted
|
|
69
|
+
* from 150 on 2026-08-03): the full-corpus relevance eval found 100 vs 150
|
|
70
|
+
* statistically identical (full CI overlap at N=40); 100 saves ~13% tokens. */
|
|
71
|
+
exports.RECALL_TITLE_CHARS = 100;
|
|
72
|
+
/** Slots of RECALL_MAX_ITEMS guaranteed to the pure-prompt (unblended)
|
|
73
|
+
* search's top passing hit(s) — the #324 novelty floor. 2 mirrors
|
|
74
|
+
* COLD_EXPOSURE_SLOTS sizing: a floor, never a takeover. */
|
|
75
|
+
exports.NOVELTY_FLOOR_SLOTS = 2;
|
|
76
|
+
/** Turns an already-shown memory stays suppressed in the same session before
|
|
77
|
+
* it may reappear in the pushed index (#192 turn-based dedup). */
|
|
78
|
+
exports.RECALL_RESHOW_TURNS = 30;
|
|
79
|
+
// ---------------------------------------------------------------------------
|
|
80
|
+
// Composite ranking weights (was: score*Weight, freshnessBoost*,
|
|
81
|
+
// supersededDemotion, *AffinityWeight, rrf*)
|
|
82
|
+
// ---------------------------------------------------------------------------
|
|
83
|
+
/** Semantic-similarity share of the composite score. 0.50 (raised from 0.40
|
|
84
|
+
* in the 0.15.2 rebalance, #191 Phase B): on the production corpus effective
|
|
85
|
+
* strength (0.30) outweighed what similarity could recover — hardened old
|
|
86
|
+
* memories beat exact matches for their own topic. Similarity now leads;
|
|
87
|
+
* strength breaks ties and rewards real use. */
|
|
88
|
+
exports.SCORE_SIMILARITY_WEIGHT = 0.50;
|
|
89
|
+
/** Effective-strength share of the composite score (was 0.30; see above). */
|
|
90
|
+
exports.SCORE_STRENGTH_WEIGHT = 0.20;
|
|
91
|
+
/** Graph-centrality share of the composite score (was 0.20; see above). */
|
|
92
|
+
exports.SCORE_CONNECTIONS_WEIGHT = 0.15;
|
|
93
|
+
/** Slow recency curve share of the composite score (was 0.10; see above). */
|
|
94
|
+
exports.SCORE_RECENCY_WEIGHT = 0.15;
|
|
95
|
+
/** Fresh-memory window: the additive bonus fades linearly to 0 over this
|
|
96
|
+
* many days. 7 — nightly capture means 1 day is the floor of "fresh"
|
|
97
|
+
* (#191 Phase B). */
|
|
98
|
+
exports.FRESHNESS_BOOST_DAYS = 7;
|
|
99
|
+
/** Fresh-memory bonus size at age 0 (#191 Phase B; 0 = disabled via seam). */
|
|
100
|
+
exports.FRESHNESS_BOOST_WEIGHT = 0.15;
|
|
101
|
+
/** Score multiplier for a memory a later decision superseded (0.15.2; the
|
|
102
|
+
* belief walk (#393 D) is the primary mechanism — this is the safety net
|
|
103
|
+
* for rows the walk does not reach). */
|
|
104
|
+
exports.SUPERSEDED_DEMOTION = 0.50;
|
|
105
|
+
/** #203 soft boost on exact project match. ADDITIVE, zero-boost neutral,
|
|
106
|
+
* never a penalty — a foreign memory ranks equal, not lower. */
|
|
107
|
+
exports.PROJECT_AFFINITY_WEIGHT = 0.15;
|
|
108
|
+
/** #203 soft boost multiplier on max overlapping domain-tag weight. */
|
|
109
|
+
exports.DOMAIN_AFFINITY_WEIGHT = 0.15;
|
|
110
|
+
/** #205 RRF k parameter (1/(k+rank+1)) — matches the pre-#205 hardcoded 60
|
|
111
|
+
* so the no-config path was byte-identical to 0.15.3. */
|
|
112
|
+
exports.RRF_K = 60;
|
|
113
|
+
/** #205 composite-score share of the final blend (RRF gets the remainder);
|
|
114
|
+
* pre-#205 hardcoded value carried forward. */
|
|
115
|
+
exports.RRF_COMPOSITE_WEIGHT = 0.8;
|
|
116
|
+
/** #205 per-list RRF weight for the FTS list. 0.5 is the bisection point
|
|
117
|
+
* where BM25F + composite-affinity flip the token-exact marine body match
|
|
118
|
+
* below the same-scope hardware field (Q4 contamination 0.20 → 0.00) while
|
|
119
|
+
* pure-keyword queries keep recall@5 = 1.0. 0.7 was measured too timid. */
|
|
120
|
+
exports.RRF_FTS_WEIGHT = 0.5;
|
|
121
|
+
/** #205 per-list RRF weight for the vector list (vec stays at 1.0 — the
|
|
122
|
+
* conservative nudge is on the FTS side only). */
|
|
123
|
+
exports.RRF_VECTOR_WEIGHT = 1.0;
|
|
124
|
+
// ---------------------------------------------------------------------------
|
|
125
|
+
// BM25F field weights (was: bm25WeightBody/Project/Domain) — #205. Body is
|
|
126
|
+
// down-weighted relative to the scope fields so a project/domain token match
|
|
127
|
+
// outranks a token-exact body collision from a foreign scope.
|
|
128
|
+
// ---------------------------------------------------------------------------
|
|
129
|
+
exports.BM25_WEIGHT_BODY = 1.0;
|
|
130
|
+
exports.BM25_WEIGHT_PROJECT = 2.0;
|
|
131
|
+
exports.BM25_WEIGHT_DOMAIN = 2.0;
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// Resolution / dedup family (was: dedupAutoMergeThreshold [legacy
|
|
134
|
+
// dedupMergeThreshold], supersessionMinSimilarity, correctionMinSimilarity,
|
|
135
|
+
// correctionRewriteMinConfidence, weakPrimaryFloor)
|
|
136
|
+
// ---------------------------------------------------------------------------
|
|
137
|
+
/** Deterministic merge ceiling of the unified resolution pass (#392): pairs
|
|
138
|
+
* at/above this cosine merge LLM-free; [CORRECTION_MIN_SIMILARITY, this)
|
|
139
|
+
* get the one verdict call. 0.92 — measured on the #191 mechanical audit
|
|
140
|
+
* corpus (89 clusters / 110 excess rows; data/audit-20260729). */
|
|
141
|
+
exports.DEDUP_AUTO_MERGE_THRESHOLD = 0.92;
|
|
142
|
+
/** Minimum cosine for a nightly supersession candidate pair (#100 stage,
|
|
143
|
+
* 0.15.0): one classify-tier call per pair above the bar. */
|
|
144
|
+
exports.SUPERSESSION_MIN_SIMILARITY = 0.80;
|
|
145
|
+
/** Minimum cosine for a reconsolidation correction pair (#384). Deliberately
|
|
146
|
+
* wider than supersession's 0.80: a retraction often rides inside an
|
|
147
|
+
* otherwise unrelated memory; the verdict + confidence gate carry the
|
|
148
|
+
* precision. */
|
|
149
|
+
exports.CORRECTION_MIN_SIMILARITY = 0.75;
|
|
150
|
+
/** Minimum verdict confidence for the REWRITE (and #392 merge-apply) fork
|
|
151
|
+
* (#384): below it a `corrects` degrades to mark-only — a weak mark is
|
|
152
|
+
* recoverable, a weak rewrite is corruption. */
|
|
153
|
+
exports.CORRECTION_REWRITE_MIN_CONFIDENCE = 0.80;
|
|
154
|
+
/** Minimum cosine(memory embedding, best domain prototype) for a no-fit
|
|
155
|
+
* memory to earn a WEAK primary instead of decaying (owner amendment
|
|
156
|
+
* 07.07). Starting point for bge-small-en-v1.5. */
|
|
157
|
+
exports.WEAK_PRIMARY_FLOOR = 0.45;
|
|
158
|
+
// ---------------------------------------------------------------------------
|
|
159
|
+
// Diagnostic tier (env-overridable — #408). The ollama-operational family is
|
|
160
|
+
// NOT user tuning: it exists so an operator of a constrained box can pin the
|
|
161
|
+
// three values into a service unit's environment without a config-file
|
|
162
|
+
// round-trip. Precedence: env > the constant below. An invalid env value
|
|
163
|
+
// warns and falls back to the constant (the resolveMemorySoftCap boundary
|
|
164
|
+
// posture, applied to the env half).
|
|
165
|
+
// ---------------------------------------------------------------------------
|
|
166
|
+
/**
|
|
167
|
+
* Context window for ollama (one value, all phases; #220/#228). 8192 is
|
|
168
|
+
* where context stops being the binding constraint for a sub-8B model on
|
|
169
|
+
* ollama. Also drives `detectChunkSize` (chunkChars ≤ numCtx × 0.6 × 4
|
|
170
|
+
* chars) so the chunker and the request agree by construction.
|
|
171
|
+
*/
|
|
172
|
+
exports.NUM_CTX = 8192;
|
|
173
|
+
/** Flush ollama's accumulated memory every N LLM calls (0 = off; #220).
|
|
174
|
+
* Opt-in operational workaround for ollama runner RSS growth — never a
|
|
175
|
+
* default-on behavior. */
|
|
176
|
+
exports.OLLAMA_FLUSH_EVERY = 0;
|
|
177
|
+
/** Ms to wait after an ollama flush (`keep_alive:0`) for the runner to exit
|
|
178
|
+
* + release memory. The runner takes >90 s to exit; 3 min allows margin. */
|
|
179
|
+
exports.OLLAMA_FLUSH_WAIT_MS = 180000;
|
|
180
|
+
/** The env-tier table (release surface: names + defaults are frozen —
|
|
181
|
+
* adding a knob here is a release decision, not a runtime one). */
|
|
182
|
+
exports.DIAGNOSTIC_ENV_TIER = Object.freeze({
|
|
183
|
+
numCtx: Object.freeze({ env: "HICORTEX_NUM_CTX", default: exports.NUM_CTX }),
|
|
184
|
+
ollamaFlushEvery: Object.freeze({
|
|
185
|
+
env: "HICORTEX_OLLAMA_FLUSH_EVERY",
|
|
186
|
+
default: exports.OLLAMA_FLUSH_EVERY,
|
|
187
|
+
}),
|
|
188
|
+
ollamaFlushWaitMs: Object.freeze({
|
|
189
|
+
env: "HICORTEX_OLLAMA_FLUSH_WAIT_MS",
|
|
190
|
+
default: exports.OLLAMA_FLUSH_WAIT_MS,
|
|
191
|
+
}),
|
|
192
|
+
});
|
|
193
|
+
/** Resolve the effective ollama context window: a positive finite
|
|
194
|
+
* HICORTEX_NUM_CTX wins; anything else (absent/blank/invalid) keeps NUM_CTX
|
|
195
|
+
* with a warn on the invalid case. */
|
|
196
|
+
function resolveNumCtx() {
|
|
197
|
+
const raw = process.env[exports.DIAGNOSTIC_ENV_TIER.numCtx.env];
|
|
198
|
+
if (raw === undefined || raw === "")
|
|
199
|
+
return exports.NUM_CTX;
|
|
200
|
+
const v = Number(raw);
|
|
201
|
+
if (Number.isFinite(v) && v > 0)
|
|
202
|
+
return v;
|
|
203
|
+
console.warn(`[hicortex] env HICORTEX_NUM_CTX=${JSON.stringify(raw)} is not a positive finite number — using default ${exports.NUM_CTX}.`);
|
|
204
|
+
return exports.NUM_CTX;
|
|
205
|
+
}
|
|
206
|
+
/** Resolve the flush cadence: a non-negative finite
|
|
207
|
+
* HICORTEX_OLLAMA_FLUSH_EVERY wins (0 = the valid off value); invalid warns
|
|
208
|
+
* and keeps OLLAMA_FLUSH_EVERY. */
|
|
209
|
+
function resolveOllamaFlushEvery() {
|
|
210
|
+
const raw = process.env[exports.DIAGNOSTIC_ENV_TIER.ollamaFlushEvery.env];
|
|
211
|
+
if (raw === undefined || raw === "")
|
|
212
|
+
return exports.OLLAMA_FLUSH_EVERY;
|
|
213
|
+
const v = Number(raw);
|
|
214
|
+
if (Number.isFinite(v) && v >= 0)
|
|
215
|
+
return Math.floor(v);
|
|
216
|
+
console.warn(`[hicortex] env HICORTEX_OLLAMA_FLUSH_EVERY=${JSON.stringify(raw)} is not a non-negative finite number — using default ${exports.OLLAMA_FLUSH_EVERY}.`);
|
|
217
|
+
return exports.OLLAMA_FLUSH_EVERY;
|
|
218
|
+
}
|
|
219
|
+
/** Resolve the post-flush wait: a positive finite
|
|
220
|
+
* HICORTEX_OLLAMA_FLUSH_WAIT_MS wins; invalid warns and keeps
|
|
221
|
+
* OLLAMA_FLUSH_WAIT_MS. */
|
|
222
|
+
function resolveOllamaFlushWaitMs() {
|
|
223
|
+
const raw = process.env[exports.DIAGNOSTIC_ENV_TIER.ollamaFlushWaitMs.env];
|
|
224
|
+
if (raw === undefined || raw === "")
|
|
225
|
+
return exports.OLLAMA_FLUSH_WAIT_MS;
|
|
226
|
+
const v = Number(raw);
|
|
227
|
+
if (Number.isFinite(v) && v > 0)
|
|
228
|
+
return v;
|
|
229
|
+
console.warn(`[hicortex] env HICORTEX_OLLAMA_FLUSH_WAIT_MS=${JSON.stringify(raw)} is not a positive finite number — using default ${exports.OLLAMA_FLUSH_WAIT_MS}.`);
|
|
230
|
+
return exports.OLLAMA_FLUSH_WAIT_MS;
|
|
231
|
+
}
|
package/dist/capture.d.ts
CHANGED
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
*/
|
|
15
15
|
import type { TranscriptBatch } from "./transcript-reader.js";
|
|
16
16
|
import type { CursorStore } from "./capture-cursors.js";
|
|
17
|
+
import type { RunDeadline } from "./run-deadline.js";
|
|
17
18
|
/**
|
|
18
19
|
* Max denoised chars per segment. Kept below the server's 80K distill cap
|
|
19
20
|
* (distiller.ts MAX_TRANSCRIPT_CHARS) with ~20K headroom so NO capture path can
|
|
@@ -105,6 +106,15 @@ export interface CaptureOptions {
|
|
|
105
106
|
* `source_domain` provenance. Null when undeclared.
|
|
106
107
|
*/
|
|
107
108
|
sourceDomain?: string | null;
|
|
109
|
+
/**
|
|
110
|
+
* The run-wide pipeline deadline (#405), checked BETWEEN segment POSTs —
|
|
111
|
+
* a boundary the per-session cursor discipline already guarantees is safe
|
|
112
|
+
* (the cursor only advances past server-confirmed segments, so a deadline
|
|
113
|
+
* stop holds every unconfirmed segment for the next run; dup-over-loss).
|
|
114
|
+
* Full and consolidate-only nightlies pass it; capture-only/watchdog runs
|
|
115
|
+
* keep their 30-min unit backstop instead.
|
|
116
|
+
*/
|
|
117
|
+
deadline?: RunDeadline;
|
|
108
118
|
}
|
|
109
119
|
export interface CaptureResult {
|
|
110
120
|
memoriesIngested: number;
|
|
@@ -113,10 +123,12 @@ export interface CaptureResult {
|
|
|
113
123
|
/**
|
|
114
124
|
* Set when the loop stopped early on a terminal server response: "limit"
|
|
115
125
|
* (token-budget 429, mcp-server.ts's `"token budget exceeded"` gate) or
|
|
116
|
-
* "auth" (401)
|
|
117
|
-
*
|
|
126
|
+
* "auth" (401); "deadline" (#405) when the run-wide pipeline deadline fired
|
|
127
|
+
* between segments (transient — the watermark holds, everything unconfirmed
|
|
128
|
+
* retries next run). A rate-limit 429 never sets this — it is transient
|
|
129
|
+
* (#327). The caller decides watermark handling.
|
|
118
130
|
*/
|
|
119
|
-
stopped?: "limit" | "auth";
|
|
131
|
+
stopped?: "limit" | "auth" | "deadline";
|
|
120
132
|
/**
|
|
121
133
|
* Run-global rate-429 latch (#327 CR): true when at least one session
|
|
122
134
|
* SURRENDERED to a rate-limit 429 (the transient kind — postWithRateRetry
|
package/dist/capture.js
CHANGED
|
@@ -204,7 +204,7 @@ async function postWithRateRetry(post, body) {
|
|
|
204
204
|
* re-paying the Retry-After ladder (#327).
|
|
205
205
|
*/
|
|
206
206
|
async function captureBatches(batches, opts) {
|
|
207
|
-
const { post, cursorStore, dryRun = false, segmentMaxChars = exports.SEGMENT_MAX_CHARS, sourceAgentId, sourceDomain } = opts;
|
|
207
|
+
const { post, cursorStore, dryRun = false, segmentMaxChars = exports.SEGMENT_MAX_CHARS, sourceAgentId, sourceDomain, deadline } = opts;
|
|
208
208
|
let memoriesIngested = 0;
|
|
209
209
|
let sessionsSent = 0;
|
|
210
210
|
let hadTransientFailure = false;
|
|
@@ -249,6 +249,15 @@ async function captureBatches(batches, opts) {
|
|
|
249
249
|
let sessionPosted = false;
|
|
250
250
|
for (let s = 0; s < segments.length; s++) {
|
|
251
251
|
const seg = segments[s];
|
|
252
|
+
// #405: stop BETWEEN segments — a safe boundary by construction (the
|
|
253
|
+
// cursor below only advances past server-confirmed segments). The whole
|
|
254
|
+
// session loop breaks on `stopped` at the bottom; unconfirmed segments
|
|
255
|
+
// hold and retry next run.
|
|
256
|
+
if (!dryRun && deadline?.hit("capture")) {
|
|
257
|
+
console.warn(`[hicortex] Run deadline reached — capture stops after the last confirmed segment`);
|
|
258
|
+
stopped = "deadline";
|
|
259
|
+
break;
|
|
260
|
+
}
|
|
252
261
|
// A segment advances the cursor to its segEnd only when it is the LAST
|
|
253
262
|
// segment ending at that boundary. Hard-split pieces (.p0,.p1,…) of one
|
|
254
263
|
// entry share the same segEnd; confirming an earlier piece must NOT move
|
|
@@ -50,6 +50,12 @@ export interface ClassifyDomainsOptions {
|
|
|
50
50
|
llm?: LlmClient;
|
|
51
51
|
/** Config override (tests). Defaults to reading stateDir/config.json. */
|
|
52
52
|
config?: Record<string, unknown> | null;
|
|
53
|
+
/**
|
|
54
|
+
* Weak-primary floor (#408): release-managed default (calibration.ts via
|
|
55
|
+
* nofit's DEFAULT_WEAK_PRIMARY_FLOOR); this field is the eval/test seam —
|
|
56
|
+
* the config key is gone from the surface. Invalid → default.
|
|
57
|
+
*/
|
|
58
|
+
weakPrimaryFloor?: number;
|
|
53
59
|
/**
|
|
54
60
|
* Embedder override (tests). Used only for domain-description prototype
|
|
55
61
|
* seeds; defaults to the local ONNX embedder, loaded lazily on first need
|
package/dist/classify-domains.js
CHANGED
|
@@ -118,7 +118,13 @@ async function runClassifyDomains(options = {}) {
|
|
|
118
118
|
'{ "name": "Boating", "description": "..." }] } and re-run. ' +
|
|
119
119
|
"No fallback bucket is needed — no-fit memories are handled automatically.");
|
|
120
120
|
}
|
|
121
|
-
|
|
121
|
+
// #408: the floor is a release-managed calibration constant; the Options
|
|
122
|
+
// field is the eval/test seam (invalid values keep the default, the stage
|
|
123
|
+
// knob-validation style — silent fallback, no warn).
|
|
124
|
+
const floorRaw = Number(options.weakPrimaryFloor);
|
|
125
|
+
const weakPrimaryFloor = Number.isFinite(floorRaw) && floorRaw > 0 && floorRaw < 1
|
|
126
|
+
? floorRaw
|
|
127
|
+
: nofit_js_1.DEFAULT_WEAK_PRIMARY_FLOOR;
|
|
122
128
|
// Resolve the LLM (one model serves all phases — #231).
|
|
123
129
|
let llm;
|
|
124
130
|
if (options.llm) {
|