@gamaze/hicortex 0.20.7 → 0.20.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/README.md +18 -41
  2. package/assets/dashboard.html +3989 -836
  3. package/dist/calibration.d.ts +293 -0
  4. package/dist/calibration.js +379 -0
  5. package/dist/capture-health.d.ts +87 -0
  6. package/dist/capture-health.js +106 -0
  7. package/dist/capture-pause.d.ts +86 -0
  8. package/dist/capture-pause.js +127 -0
  9. package/dist/capture.d.ts +24 -3
  10. package/dist/capture.js +11 -1
  11. package/dist/classify-domains.d.ts +6 -0
  12. package/dist/classify-domains.js +7 -1
  13. package/dist/cli.js +38 -3
  14. package/dist/config-read.d.ts +1 -1
  15. package/dist/config-read.js +96 -9
  16. package/dist/consolidate.d.ts +114 -68
  17. package/dist/consolidate.js +302 -182
  18. package/dist/dashboard.d.ts +326 -6
  19. package/dist/dashboard.js +592 -7
  20. package/dist/db.js +105 -0
  21. package/dist/dedup.d.ts +34 -26
  22. package/dist/dedup.js +91 -57
  23. package/dist/distiller.js +1 -1
  24. package/dist/domain-classify.d.ts +7 -6
  25. package/dist/domain-classify.js +12 -10
  26. package/dist/eval/decay-eval.d.ts +3 -3
  27. package/dist/eval/decay-eval.js +4 -4
  28. package/dist/eval/importance-eval.d.ts +85 -0
  29. package/dist/eval/importance-eval.js +286 -0
  30. package/dist/eval/planted-eval.d.ts +26 -0
  31. package/dist/eval/planted-eval.js +97 -0
  32. package/dist/eval/planted-fixtures.d.ts +107 -0
  33. package/dist/eval/planted-fixtures.js +283 -0
  34. package/dist/eval/planted-harness.d.ts +176 -0
  35. package/dist/eval/planted-harness.js +649 -0
  36. package/dist/eval/ranking-battery.d.ts +78 -0
  37. package/dist/eval/ranking-battery.js +181 -0
  38. package/dist/eval/ranking-eval.d.ts +41 -0
  39. package/dist/eval/ranking-eval.js +391 -0
  40. package/dist/eval/ranking-fixtures.d.ts +77 -0
  41. package/dist/eval/ranking-fixtures.js +226 -0
  42. package/dist/identity-store.d.ts +21 -0
  43. package/dist/identity-store.js +49 -0
  44. package/dist/index.js +4 -3
  45. package/dist/init.d.ts +23 -3
  46. package/dist/init.js +84 -9
  47. package/dist/llm.d.ts +43 -58
  48. package/dist/llm.js +87 -101
  49. package/dist/mcp-server.d.ts +12 -0
  50. package/dist/mcp-server.js +213 -32
  51. package/dist/nightly.d.ts +9 -1
  52. package/dist/nightly.js +164 -110
  53. package/dist/nofit.d.ts +4 -11
  54. package/dist/nofit.js +6 -23
  55. package/dist/prompts.d.ts +10 -0
  56. package/dist/prompts.js +28 -5
  57. package/dist/recall-index.d.ts +30 -28
  58. package/dist/recall-index.js +21 -18
  59. package/dist/recall-registry.d.ts +2 -1
  60. package/dist/recall-registry.js +35 -1
  61. package/dist/reconsolidation.d.ts +168 -87
  62. package/dist/reconsolidation.js +818 -377
  63. package/dist/relink.js +3 -4
  64. package/dist/rescore-importance.d.ts +80 -0
  65. package/dist/rescore-importance.js +236 -0
  66. package/dist/retrieval.d.ts +80 -35
  67. package/dist/retrieval.js +322 -105
  68. package/dist/run-deadline.d.ts +62 -0
  69. package/dist/run-deadline.js +73 -0
  70. package/dist/schema-prototypes.d.ts +3 -3
  71. package/dist/schema-prototypes.js +3 -3
  72. package/dist/stages.d.ts +37 -0
  73. package/dist/stages.js +51 -0
  74. package/dist/state.d.ts +34 -9
  75. package/dist/storage.d.ts +50 -18
  76. package/dist/storage.js +125 -30
  77. package/dist/telemetry.d.ts +8 -7
  78. package/dist/token-budget.js +3 -4
  79. package/dist/type-classify.js +4 -4
  80. package/dist/types.d.ts +143 -155
  81. package/domains.example.json +4 -5
  82. package/hermes-plugin/hicortex/README.md +2 -2
  83. package/openclaw.plugin.json +1 -1
  84. package/package.json +4 -1
  85. package/pi-extension/hicortex/README.md +1 -1
  86. package/server.json +3 -3
@@ -9,18 +9,27 @@ import type { LlmClient } from "./llm.js";
9
9
  import type { EmbedFn } from "./retrieval.js";
10
10
  import { type DomainDef } from "./domain-classify.js";
11
11
  import { type ReconsolidationOptions } from "./reconsolidation.js";
12
+ import type { RunDeadline } from "./run-deadline.js";
12
13
  /**
13
- * Default ceiling on total LLM calls across all classify-tier consolidation
14
- * stages (content-domain, link discovery, supersession) per run. This is a
15
- * runaway BACKSTOP, not a throughput throttle — on a free local model there is
16
- * no per-call cost to defend against; the binding constraint is the nightly
17
- * unit's wall-clock timeout (TimeoutStartSec), not call count. 5000 clears a
18
- * one-time classification backlog (a ~2000-memory batch drains in ~1-2 runs
19
- * instead of ~11 nights at the old 200) with margin for link/supersession, and
20
- * ~5000 calls x ~1-3s/call ≈ 1.4-4.2h fits the 6h consolidation backstop.
21
- * Config-overridable as `consolidateMaxLlmCalls` (#241).
14
+ * Default ceiling on LLM calls across the WHOLE nightly pipeline (#405; the
15
+ * #241 consolidateMaxLlmCalls mechanism, renamed and widened). A runaway
16
+ * BACKSTOP that bounds money/load INDEPENDENT OF LATENCY — a fast metered or
17
+ * capacity-limited endpoint permits thousands of calls inside the wall-clock
18
+ * budget, so time alone cannot protect it (owner ruling 2026-09-12). Consumed
19
+ * in run order: a stage that exhausts it defers its remainder via its cursor.
20
+ * 5000 clears a one-time classification backlog (a ~2000-memory batch drains
21
+ * in ~1-2 runs) with margin. Config: `nightlyLlmCallBudget` (#405); the old
22
+ * `consolidateMaxLlmCalls` key is a deprecated alias honored one release.
22
23
  */
23
- export declare const CONSOLIDATE_MAX_LLM_CALLS = 5000;
24
+ export declare const DEFAULT_NIGHTLY_LLM_CALL_BUDGET = 5000;
25
+ /**
26
+ * Resolve the per-run LLM call budget from config (#405):
27
+ * - `nightlyLlmCallBudget` present (positive finite) → it wins;
28
+ * - else `consolidateMaxLlmCalls` present → used as a DEPRECATED ALIAS with
29
+ * a warn naming the replacement (honored one release);
30
+ * - absent/invalid → the 5000 default.
31
+ */
32
+ export declare function resolveNightlyLlmCallBudget(config: Record<string, unknown> | null | undefined): number;
24
33
  /**
25
34
  * Minimum COSINE similarity for a link candidate.
26
35
  *
@@ -66,14 +75,12 @@ export declare class BudgetTracker {
66
75
  callsByStage: Record<string, number>;
67
76
  /**
68
77
  * Per-stage count of LLM-call REQUESTS refused because the budget was
69
- * exhausted (#255). Keys are the same stage labels passed to `use()`. The
70
- * value is the SUM of the `count` args passed to each refused `use()` call
71
- * in that stage (in production every `use()` call passes count=1, so each
72
- * refused call adds 1 — but the API accepts a batch count, so a single
73
- * refused batch request accrues its full count). Stages break on the first
74
- * refusal, so a stage's value is the count of the one request that crossed
75
- * the boundary. For item-level skip counts (how many memories or pairs were
76
- * left unprocessed), see the per-stage reports — e.g.
78
+ * exhausted (#255). Keys are the same stage labels passed to `use()`; each
79
+ * refused call adds 1 (the dead batch `count` param is gone — #405 —
80
+ * production always passed 1 anyway). Stages break on the first refusal,
81
+ * so a stage's value is the count of requests that crossed the boundary.
82
+ * For item-level skip counts (how many memories or pairs were left
83
+ * unprocessed), see the per-stage reports — e.g.
77
84
  * `stages.importance.skipped_budget` — which count MEMORIES, not call
78
85
  * requests. Surfaced in summary() and ConsolidationReport as
79
86
  * `deferred_by_stage`.
@@ -99,7 +106,7 @@ export declare class BudgetTracker {
99
106
  constructor(maxCalls: number);
100
107
  get exhausted(): boolean;
101
108
  get remaining(): number;
102
- use(stage: string, count?: number): boolean;
109
+ use(stage: string): boolean;
103
110
  /**
104
111
  * Record token usage from one LLM call (#246). Called by the consolidation
105
112
  * stages after each metered completion. `undefined` usage (claude-cli path,
@@ -113,6 +120,26 @@ export declare class BudgetTracker {
113
120
  } | undefined): void;
114
121
  summary(): NonNullable<ConsolidationReport["budget"]>;
115
122
  }
123
+ /**
124
+ * #427 observability: warn when a consolidation run made LLM CALLS but
125
+ * metered ZERO tokens — the endpoint returned no usage objects on its
126
+ * completions (recordUsage skips undefined by design, never fabricates a
127
+ * zero). Such a run still spends budget calls but its snapshot carries token
128
+ * nulls, which read as a mystery on the dashboard. The warn is a structured
129
+ * event in the same journald-greppable style as `event=budget_exhausted`
130
+ * (grep `event=tokens_unmetered`), so the blind spot is visible instead of
131
+ * silent. Returns true when it warned (for tests); no fabrication either
132
+ * way — the numbers stay exactly what the endpoint reported.
133
+ */
134
+ export declare function warnUnmeteredTokensRun(budget: NonNullable<ConsolidationReport["budget"]>): boolean;
135
+ /**
136
+ * True when a token-period start stamp is ABSENT or sits in a previous UTC
137
+ * calendar month than `now` — the monthly-reset staleness check. #405: ONE
138
+ * shared helper — the check was triplicated (the nightly's throttle branch,
139
+ * the nightly's accrual write, token-budget.ts recordDistillUsage) and each
140
+ * copy re-derived the year+month comparison by hand.
141
+ */
142
+ export declare function isStaleTokenPeriod(periodStart: string | undefined, now?: Date): boolean;
116
143
  /**
117
144
  * Decide whether consolidation should be throttled this run based on the
118
145
  * `llmTokensPerMonth` fair-use cap. Pure (no I/O) so it can be unit-tested
@@ -125,8 +152,8 @@ export declare class BudgetTracker {
125
152
  *
126
153
  * `cap = 0` (the self-hosted default) → never throttle (unlimited).
127
154
  * `periodStart` in a previous calendar month → period resets to 0 first
128
- * (mirrors the reset logic in nightly.ts; both sides agree because both read
129
- * the same state + clock).
155
+ * (isStaleTokenPeriod — the same helper every monthly-reset site uses, so
156
+ * the sides agree because they read the same state + clock).
130
157
  */
131
158
  export declare function shouldThrottleTokens(cap: number, period: {
132
159
  total: number;
@@ -140,6 +167,29 @@ export declare function shouldThrottleTokens(cap: number, period: {
140
167
  * Parse JSON from LLM output, tolerating markdown fences and indexed formats.
141
168
  */
142
169
  export declare function parseJsonLenient<T>(text: string, fallback: T): T;
170
+ /**
171
+ * The shared importance-scoring loop (#425 extraction): one LLM call per
172
+ * 10-memory batch through the production `importanceScoring` prompt, each
173
+ * written score clamped at IMPORTANCE_CEILING and stamped with the
174
+ * importance_scored_at watermark. Used by the nightly's stageImportance AND
175
+ * `hicortex rescore-importance` — there is exactly one scoring code path
176
+ * (no forked backfill logic; cap + watermark write identically everywhere).
177
+ *
178
+ * Failure semantics: a batch whose LLM call THROWS writes nothing (counted
179
+ * in `failed` — retried naturally later); a batch whose reply parses to a
180
+ * non-array falls back to 0.5 per memory (written + watermarked — the
181
+ * endpoint answered, the answer was unusable).
182
+ */
183
+ export declare function scoreMemoriesImportance(db: Database.Database, memories: Memory[], llm: LlmClient, opts?: {
184
+ budget?: BudgetTracker;
185
+ deadline?: RunDeadline;
186
+ dryRun?: boolean;
187
+ onBatch?: (written: number, failed: number) => void;
188
+ }): Promise<{
189
+ scored: number;
190
+ failed: number;
191
+ skipped_budget: number;
192
+ }>;
143
193
  /**
144
194
  * Rebuild moduleIndex from the configured domain set + live DB counts, and
145
195
  * persist it. Shared by the nightly stage and `hicortex classify-domains`.
@@ -176,19 +226,16 @@ export declare function discoverLinkCandidates(db: Database.Database, mem: Memor
176
226
  * 672-link audit (see the Stage 3 header) found the LLM-classified UPPERCASE
177
227
  * types near-useless (CONTRADICTS 4% acceptable). Every candidate now takes its
178
228
  * pre-computed `heuristicType` (only `extends` or `relates_to` — see
179
- * classifyRelationship). No LLM call is made.
180
- *
181
- * Signature stability: `llm` and `budget` are RETAINED but intentionally
182
- * ignored so the callers (nightly `stageLinks`, `hicortex relink`) and the
183
- * tests that import this need no change to their call sites. The return shape
184
- * is unchanged; `llmClassified` is always 0 now and `heuristicFallback` counts
185
- * every candidate. Do NOT re-add an LLM path here without a classifier that
186
- * passes the audit harness at >= 70% acceptable.
229
+ * classifyRelationship). No LLM call is made. #405: the ignored `llm`/`budget`
230
+ * params are deleted — the signature now tells the truth.
231
+ * The return shape is unchanged; `llmClassified` is always 0 and
232
+ * `heuristicFallback` counts every candidate. Do NOT re-add an LLM path here
233
+ * without a classifier that passes the audit harness at >= 70% acceptable.
187
234
  *
188
235
  * Shared between the nightly `stageLinks` and `hicortex relink`.
189
236
  * Returns one relationship type per candidate (same order as input).
190
237
  */
191
- export declare function classifyLinkCandidates(candidates: LinkCandidate[], _llm: LlmClient | null, _budget: BudgetTracker): Promise<{
238
+ export declare function classifyLinkCandidates(candidates: LinkCandidate[]): Promise<{
192
239
  types: string[];
193
240
  llmClassified: number;
194
241
  heuristicFallback: number;
@@ -210,22 +257,17 @@ export declare function classifyLinkCandidates(candidates: LinkCandidate[], _llm
210
257
  * boundary from the l2ToCosine calibration is preserved.
211
258
  */
212
259
  export declare function classifyRelationship(source: Memory, target: Memory, similarity: number): string;
213
- /** Default minimum COSINE similarity for a supersession candidate pair. */
260
+ /** Default minimum COSINE similarity for a supersession candidate pair —
261
+ * RELEASE-MANAGED since #408 (calibration.ts SUPERSESSION_MIN_SIMILARITY). */
214
262
  export declare const DEFAULT_SUPERSESSION_MIN_SIMILARITY = 0.8;
215
- /**
216
- * Default max classify-tier LLM calls (pairs evaluated) spent per nightly run.
217
- * 0 = no separate cap — supersession shares the consolidation budget
218
- * (CONSOLIDATE_MAX_LLM_CALLS, default 5000) like every other stage. The old
219
- * default of 30 was set when the corpus had 14 decisions; with the distiller
220
- * now classifying types correctly (#216), decisions are common and the cap
221
- * was throttling supersession to a crawl. On a local free model there is no
222
- * per-call cost to defend against — the binding constraint is the wall-clock
223
- * timeout (TimeoutStartSec), not call count.
224
- */
225
- export declare const DEFAULT_SUPERSESSION_MAX_CALLS = 0;
226
263
  export interface SupersessionOptions {
264
+ /** Candidate-pair cosine floor. Release-managed default (calibration.ts);
265
+ * this field is the eval/test seam. Invalid → default. */
227
266
  minSimilarity?: number;
228
- maxCalls?: number;
267
+ /** The run-wide pipeline deadline (#405) — checked at each candidate
268
+ * boundary; on expiry the scan stops and the cursor holds at the last
269
+ * fully-considered candidate (resumed next run). */
270
+ deadline?: RunDeadline;
229
271
  }
230
272
  export interface SupersessionStageResult {
231
273
  scanned: number;
@@ -265,6 +307,13 @@ export declare function parseSupersessionReply(reply: string): boolean | null;
265
307
  * candidacy. It only stops SHORT of a candidate when the budget is already
266
308
  * exhausted before that candidate starts, so the cursor never skips a
267
309
  * candidate that was never looked at.
310
+ *
311
+ * #405: the cursor persists after EVERY fully-considered candidate (the
312
+ * post-#404 reconsolidation pattern), not at stage end — a run killed or
313
+ * deadline-deferred mid-stage loses at most the candidate in flight. No
314
+ * orphan clamp is needed (unlike reconsolidation): supersession applies each
315
+ * verdict's link immediately, so `cursor = candidate.__rowid` always sits
316
+ * after all of that candidate's writes.
268
317
  */
269
318
  export declare function stageSupersession(db: Database.Database, llm: LlmClient, budget: BudgetTracker, embedFn: EmbedFn, dryRun: boolean, stateDir: string | undefined, options?: SupersessionOptions): Promise<SupersessionStageResult>;
270
319
  /**
@@ -320,7 +369,7 @@ export declare function stageMemoryCapEviction(db: Database.Database, dryRun: bo
320
369
  * When `domains` is a non-empty list, the pipeline uses content-based
321
370
  * classification (config-owned) INSTEAD of project grouping. The single
322
371
  * model serves all phases; if it's unavailable, `complete()` retries
323
- * internally (30s/60s/120s) and the phase fails soft on persistence —
372
+ * internally (one 60 s retry, #405) and the phase fails soft on persistence —
324
373
  * the nightly retries on the next run. No pre-flight health checks; the
325
374
  * phase either answers or is skipped until the next scheduled run. When
326
375
  * `domains` is absent/empty, the legacy project-grouping curation runs
@@ -330,38 +379,35 @@ export interface DomainStageOptions {
330
379
  domains?: DomainDef[] | null;
331
380
  contentDomainsReady?: boolean;
332
381
  /**
333
- * Weak-primary floor for the no-fit path (see nofit.ts). Resolved by the
334
- * caller from config (`weakPrimaryFloor`); defaults to
335
- * DEFAULT_WEAK_PRIMARY_FLOOR when absent.
382
+ * Weak-primary floor for the no-fit path (see nofit.ts). Release-managed
383
+ * default (#408 — calibration.ts WEAK_PRIMARY_FLOOR via nofit's
384
+ * DEFAULT_WEAK_PRIMARY_FLOOR); this field stays as the eval/test seam.
336
385
  */
337
386
  weakPrimaryFloor?: number;
338
387
  }
339
388
  export declare function runConsolidation(db: Database.Database, llm: LlmClient, embedFn: EmbedFn, dryRun?: boolean, skipReflection?: boolean, stateDir?: string, domainOptions?: DomainStageOptions, supersessionOptions?: SupersessionOptions,
340
- /** Total LLM-call ceiling across classify-tier stages (#241). The caller
341
- * reads `consolidateMaxLlmCalls` from config and passes it; unset → the
342
- * exported `CONSOLIDATE_MAX_LLM_CALLS` default (5000). */
389
+ /** The ONE per-run LLM-call ceiling (#405/#241). The caller resolves
390
+ * `nightlyLlmCallBudget` from config (consolidateMaxLlmCalls is a
391
+ * deprecated alias — resolveNightlyLlmCallBudget) and passes it; unset →
392
+ * the exported DEFAULT_NIGHTLY_LLM_CALL_BUDGET (5000). */
343
393
  budgetMaxCalls?: number,
344
394
  /** Soft cap on the corpus (#245). Nightly.ts reads `memorySoftCap` from
345
395
  * config and passes it; unset → `DEFAULT_MEMORY_SOFT_CAP` (10000). `0`
346
396
  * disables eviction (indefinite growth). */
347
397
  memorySoftCap?: number,
348
- /** Reconsolidation-stage knobs (#384), threaded from config by nightly.ts
349
- * (correctionMinSimilarity / correctionRewriteMinConfidence) exactly like
350
- * supersessionOptions above; unset fields → the stage's defaults.
351
- * Appended AFTER the pre-#384 params so every existing positional caller
352
- * (tests, hosted nightly) keeps its argument meaning. */
353
- reconsolidationOptions?: ReconsolidationOptions): Promise<ConsolidationReport>;
354
- /**
355
- * Calculate milliseconds until the next occurrence of a given hour (local time).
356
- */
357
- export declare function msUntilHour(hour: number): number;
398
+ /** Reconsolidation-stage knobs (#384) — eval/test seams since #408 (the
399
+ * values are release-managed calibration constants; nightly.ts threads
400
+ * NOTHING), exactly like supersessionOptions above; unset fields → the
401
+ * stage's calibration defaults. Appended AFTER the pre-#384 params so
402
+ * every existing positional caller (tests, hosted nightly) keeps its
403
+ * argument meaning. */
404
+ reconsolidationOptions?: ReconsolidationOptions,
358
405
  /**
359
- * Schedule the consolidation pipeline to run nightly.
360
- * Returns a cleanup function to cancel the timer.
361
- *
362
- * NOTE: currently unused (nightly.ts drives consolidation directly). Any future
363
- * caller MUST read config.domains and thread `domainOptions` into runConsolidation
364
- * when content domains are configured — otherwise it silently falls back to the
365
- * legacy project-grouping path even when a domain list is set.
406
+ * The run-wide pipeline deadline (#405), created at nightly start and
407
+ * shared by capture + every consolidation stage. Absent = no deadline
408
+ * (tests, evict-only paths, pre-#405 callers). When it fires, every
409
+ * not-yet-run stage defers (logs event=deadline_deferred stage=<name>) and
410
+ * the report status becomes "deferred" — which keeps lastConsolidated
411
+ * un-advanced so the next run re-finds the pending work.
366
412
  */
367
- export declare function scheduleConsolidation(db: Database.Database, llm: LlmClient, embedFn: EmbedFn, hour?: number): () => void;
413
+ deadline?: RunDeadline): Promise<ConsolidationReport>;