@gamaze/hicortex 0.23.1 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/assets/dashboard.html +64 -13
  2. package/dist/calibration.d.ts +31 -0
  3. package/dist/calibration.js +38 -1
  4. package/dist/capture.d.ts +7 -0
  5. package/dist/capture.js +10 -1
  6. package/dist/dashboard.d.ts +32 -7
  7. package/dist/dashboard.js +50 -10
  8. package/dist/db.js +45 -0
  9. package/dist/dedup.js +2 -2
  10. package/dist/distill-queue.d.ts +203 -0
  11. package/dist/distill-queue.js +440 -0
  12. package/dist/health.d.ts +13 -1
  13. package/dist/health.js +6 -1
  14. package/dist/hosted-boot.d.ts +1 -1
  15. package/dist/index.js +17 -2
  16. package/dist/learnings-identity.js +20 -1
  17. package/dist/localhost-bypass.js +1 -1
  18. package/dist/mcp-server.js +92 -7
  19. package/dist/nightly.js +88 -1
  20. package/dist/nofit.d.ts +1 -1
  21. package/dist/nofit.js +1 -1
  22. package/dist/schema-prototypes.d.ts +1 -1
  23. package/dist/schema-prototypes.js +1 -1
  24. package/dist/status.d.ts +11 -0
  25. package/dist/status.js +25 -0
  26. package/dist/types.d.ts +20 -0
  27. package/hermes-plugin/hicortex/README.md +13 -6
  28. package/hermes-plugin/hicortex/__init__.py +7 -0
  29. package/hermes-plugin/hicortex/client.py +64 -2
  30. package/hermes-plugin/hicortex/plugin.yaml +1 -1
  31. package/hermes-plugin/hicortex/provider.py +24 -1
  32. package/opencode-plugin/hicortex/index.ts +24 -1
  33. package/package.json +2 -1
  34. package/pi-extension/hicortex/index.ts +24 -1
  35. package/server.json +2 -2
  36. package/dist/eval/decay-eval.d.ts +0 -111
  37. package/dist/eval/decay-eval.js +0 -214
  38. package/dist/eval/dups.d.ts +0 -100
  39. package/dist/eval/dups.js +0 -174
  40. package/dist/eval/eval-clock.d.ts +0 -32
  41. package/dist/eval/eval-clock.js +0 -47
  42. package/dist/eval/eval-db.d.ts +0 -25
  43. package/dist/eval/eval-db.js +0 -67
  44. package/dist/eval/graph-eval.d.ts +0 -89
  45. package/dist/eval/graph-eval.js +0 -246
  46. package/dist/eval/importance-eval.d.ts +0 -85
  47. package/dist/eval/importance-eval.js +0 -286
  48. package/dist/eval/planted-eval.d.ts +0 -30
  49. package/dist/eval/planted-eval.js +0 -122
  50. package/dist/eval/planted-fixtures.d.ts +0 -107
  51. package/dist/eval/planted-fixtures.js +0 -283
  52. package/dist/eval/planted-harness.d.ts +0 -183
  53. package/dist/eval/planted-harness.js +0 -651
  54. package/dist/eval/ranking-battery.d.ts +0 -125
  55. package/dist/eval/ranking-battery.js +0 -289
  56. package/dist/eval/ranking-eval.d.ts +0 -61
  57. package/dist/eval/ranking-eval.js +0 -554
  58. package/dist/eval/ranking-fixtures.d.ts +0 -117
  59. package/dist/eval/ranking-fixtures.js +0 -485
  60. package/dist/eval/recall-sweep.d.ts +0 -87
  61. package/dist/eval/recall-sweep.js +0 -1030
  62. package/dist/eval/reflection-census.d.ts +0 -19
  63. package/dist/eval/reflection-census.js +0 -25
  64. package/dist/eval/relevance-eval.d.ts +0 -178
  65. package/dist/eval/relevance-eval.js +0 -2240
  66. package/dist/eval/run-eval.d.ts +0 -20
  67. package/dist/eval/run-eval.js +0 -299
@@ -1,19 +0,0 @@
1
- /**
2
- * Reflection census (#191 mechanical baseline) — lesson output volume over
3
- * time and the lesson/episode yield ratio. This is NOT a quality read
4
- * (sampling lessons for actionable-vs-noise is Phase-A grading work,
5
- * deferred per the issue's owner comment until recall is active) — just the
6
- * mechanical count.
7
- */
8
- import type Database from "better-sqlite3";
9
- export interface ReflectionCensus {
10
- lessonsByDate: Array<{
11
- date: string;
12
- count: number;
13
- }>;
14
- totalLessons: number;
15
- totalEpisodes: number;
16
- /** lessons / episodes — null when there are no episodes to divide by. */
17
- yieldRatio: number | null;
18
- }
19
- export declare function runReflectionCensus(db: Database.Database): ReflectionCensus;
@@ -1,25 +0,0 @@
1
- "use strict";
2
- /**
3
- * Reflection census (#191 mechanical baseline) — lesson output volume over
4
- * time and the lesson/episode yield ratio. This is NOT a quality read
5
- * (sampling lessons for actionable-vs-noise is Phase-A grading work,
6
- * deferred per the issue's owner comment until recall is active) — just the
7
- * mechanical count.
8
- */
9
- Object.defineProperty(exports, "__esModule", { value: true });
10
- exports.runReflectionCensus = runReflectionCensus;
11
- function runReflectionCensus(db) {
12
- const lessonsByDate = db
13
- .prepare(`SELECT date(created_at) AS date, COUNT(*) AS count
14
- FROM memories WHERE memory_type = 'learnings'
15
- GROUP BY date ORDER BY date`)
16
- .all();
17
- const totalLessons = db.prepare("SELECT COUNT(*) AS c FROM memories WHERE memory_type = 'learnings'").get().c;
18
- const totalEpisodes = db.prepare("SELECT COUNT(*) AS c FROM memories WHERE memory_type = 'experience'").get().c;
19
- return {
20
- lessonsByDate,
21
- totalLessons,
22
- totalEpisodes,
23
- yieldRatio: totalEpisodes > 0 ? totalLessons / totalEpisodes : null,
24
- };
25
- }
@@ -1,178 +0,0 @@
1
- #!/usr/bin/env node
2
- /**
3
- * Real-query relevance + SNIPPET eval — recall QUALITY on real agent prompts.
4
- *
5
- * v2 (spec `specs/2026-08-02-relevance-eval.md`) extends the v1 selection-only
6
- * eval with the SNIPPET layer: v1 asked "did retrieve() surface the right
7
- * memories?" (judge sees up to 2000 chars). Production shows the agent only a
8
- * ~100-char one-liner (`recall-index.ts#memoryTitle`), so a memory can be
9
- * genuinely relevant while its rendered line is useless — v1 scored that as a
10
- * win. v2 grades BOTH: `full_verdict` (selection quality, judge sees full
11
- * content) and `line_verdict` (snippet quality, judge sees ONLY the rendered
12
- * production one-liner — imported from `recall-index.ts`, never reimplemented).
13
- *
14
- * v2 additions (spec §4, §5, §6, §6b) layered onto the v1 base (prompt
15
- * sampling, readonly snapshot handling, embed-once + neverCalledEmbed,
16
- * lenient JSON parse, distribution/CI reporting):
17
- * 1. Dual verdict per surfaced memory — TWO separate, blind judge calls.
18
- * 2. Snippet-length sweep (100/200/300/title+1st-sentence) on a fixed
19
- * 40-prompt subset (8 per source).
20
- * 3. Similarity-floor + retrieval-source analysis (near-free — logged, not
21
- * re-judged).
22
- * 4. Token-cost estimate (char/4) per K and per snippet-length variant.
23
- * 5. ~20 rendered ACTUAL production blocks dumped into the report.
24
- * 6. Redundancy — one set-level judge call per (prompt × mode) over the
25
- * production 6.
26
- * 7. Rate limiting + resumability (§6b, MANDATORY): serial calls,
27
- * `--judge-delay-ms` (default 2000), exponential backoff with jitter on
28
- * 429/5xx/timeout (5→10→20→40→80s, max 5 retries, respects
29
- * `Retry-After`), checkpoint-per-call to a `.jsonl` sidecar, `--resume`,
30
- * progress logging, 10%-error-rate abort, `--max-calls` budget guard
31
- * (default 900).
32
- *
33
- * Prompt corpus (spec §2, owner decision §11.1): EVEN split, 20 prompts per
34
- * source × 5 sources — Hermes (lenny, raider, nano) + CC (the DevOps
35
- * `infrastructure` project, the `aironic-marine` project). Saved to
36
- * `data/prompts.json`, stable/reused verbatim once a valid v2 set exists.
37
- *
38
- * Judge: GLM-5.2 via z.ai — the INSTRUMENT only. It never picks candidates;
39
- * retrieve() (LLM-free) does. A dedicated raw HTTP caller (NOT `LlmClient`) is
40
- * used here on purpose: `LlmClient.completeReflect` bakes in a
41
- * nightly-tolerant retry policy (30s/60s/120s, unlimited rate-limit patience)
42
- * that conflicts with §6b's specific real-time batch policy (5/10/20/40/80s +
43
- * jitter, 5 retries, a hard call budget). Implemented directly here rather
44
- * than adding a second retry mode to `llm.ts` (out of scope for this eval,
45
- * and another agent is concurrently working elsewhere in this repo).
46
- *
47
- * Honesty invariants (non-negotiable — mirror recall-sweep.ts + spec §7):
48
- * - Snapshot opened READONLY via openSnapshot — never initDb.
49
- * - noStrengthen: true on every retrieve() call.
50
- * - Real bge-small-en-v1.5 embedder, embed-once + queryEmbedding reuse.
51
- * - neverCalledEmbed self-check ABORTS the run if retrieve() ignores
52
- * queryEmbedding (would invalidate every measured number).
53
- * - GLM-5.2 is the JUDGE only, real prompts, no synthetic queries.
54
- * - Production renderer (`formatIndexLine`/`memoryTitle`) imported from
55
- * `recall-index.ts`, never reimplemented.
56
- * - judge_error batches/units excluded from every denominator, reported
57
- * separately.
58
- *
59
- * Run:
60
- * npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
61
- * [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
62
- * [--verdicts-json=path] [--verdicts-jsonl=path] \
63
- * [--now=<ISO>] [--runs=N]
64
- *
65
- * #458 clock pin + judge variance protocol:
66
- * - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
67
- * retrieve() `now` seam) — before/after runs become wall-clock-independent
68
- * (default: live clock, the pre-#458 behavior).
69
- * - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
70
- * selections (one retrieve sweep) and reports the run-to-run noise floor
71
- * (median + spread + selection-identical flip counts, §15b). Phases 3+4
72
- * stay single-shot; the shared --max-calls budget spans all runs and the
73
- * default scales ×N unless --max-calls was passed explicitly.
74
- */
75
- type Verdict = "relevant" | "noise" | "stale";
76
- interface JudgeVerdict {
77
- /** Q1..Q8 label — mapped back to the real memory id via the caller's
78
- * position-ordered topIds list. */
79
- id: string;
80
- verdict: Verdict;
81
- reason: string;
82
- }
83
- type VerdictsOrError = JudgeVerdict[] | {
84
- judgeError: string;
85
- };
86
- interface CliArgs {
87
- positionals: string[];
88
- judgeDelayMs: number;
89
- maxCalls: number;
90
- /** True when --max-calls was passed explicitly (#458) — an explicit budget
91
- * was sized by the caller for the whole invocation and must NOT be scaled
92
- * by --runs (see scaleMaxCalls). */
93
- maxCallsExplicit: boolean;
94
- resume: boolean;
95
- /** Override the raw-verdicts JSON path (default VERDICTS_JSON_PATH) — added
96
- * 2026-08-02 so a re-run against a NEW snapshot can never silently
97
- * overwrite a baseline's raw verdicts. */
98
- verdictsJsonPath: string;
99
- /** Override the checkpoint sidecar / raw-verdicts JSONL path (default
100
- * VERDICTS_JSONL_PATH) — same reason. */
101
- verdictsJsonlPath: string;
102
- /** #458: raw --now=<ISO> value — undefined = live clock. Validated in main
103
- * via parsePinnedNow (invalid → usage-style error, exit 1). */
104
- nowIso: string | undefined;
105
- /** #458: judge runs over the fixed selections (phases 1+2 only). Default 1
106
- * = the pre-#458 single-run behavior. */
107
- runs: number;
108
- }
109
- /** Parsed and exported for tests (#458). Throws on an invalid --runs value
110
- * (NaN or <1) — fail-explicit, never a silent default. */
111
- export declare function parseArgs(argv: string[]): CliArgs;
112
- /**
113
- * #458 — resolve the effective --max-calls budget. The ONE shared budget
114
- * spans ALL judge runs, so when --runs>1 and the caller did NOT pass
115
- * --max-calls explicitly, the default scales ×N (each run re-judges every
116
- * unit). An explicit budget always wins — the caller sized it for the whole
117
- * invocation. Pure, exported for tests.
118
- */
119
- export declare function scaleMaxCalls(runs: number, explicitMaxCalls: number | undefined, defaultMaxCalls: number): number;
120
- /** Median of a numeric sample (even count → mean of the two middle values).
121
- * Empty input → NaN, the mean() convention above. Does not mutate input. */
122
- export declare function median(xs: number[]): number;
123
- /** One judged unit's verdicts for variance accounting (#458): `key` is the
124
- * stable (promptIdx|mode) unit identity, identical across runs because the
125
- * retrieve sweep runs ONCE. */
126
- export interface VarianceUnitVerdicts {
127
- key: string;
128
- verdicts: VerdictsOrError;
129
- }
130
- /** A headline metric measured once per run (e.g. precision@5 OFF). Values are
131
- * in raw fractions; null = not computable that run (excluded from
132
- * median/spread). */
133
- export interface VarianceMetricSeries {
134
- label: string;
135
- values: Array<number | null>;
136
- }
137
- export interface VarianceMetricStats extends VarianceMetricSeries {
138
- median: number | null;
139
- /** (max − min) × 100, in points — the measured run-to-run noise floor. */
140
- spreadPts: number | null;
141
- }
142
- /** Pairwise flip counts between runs i and j (1-based run numbers). */
143
- export interface PairwiseFlips {
144
- runA: number;
145
- runB: number;
146
- /** Verdict ROWS compared: (unit, Q-label) pairs present as parsed verdict
147
- * arrays in BOTH runs. judge_error units and labels the judge omitted in
148
- * one run are not comparable — excluded, never counted as flips. */
149
- rowsCompared: number;
150
- rowsFlipped: number;
151
- unitsCompared: number;
152
- /** A UNIT flips if any of its comparable rows differ. */
153
- unitsFlipped: number;
154
- }
155
- export interface JudgeVariance {
156
- runs: number;
157
- metrics: VarianceMetricStats[];
158
- flips: PairwiseFlips[];
159
- maxRowsFlipped: number;
160
- maxUnitsFlipped: number;
161
- meanRowsFlipped: number;
162
- meanUnitsFlipped: number;
163
- /** Comparable rows of the first pair — selections are identical across
164
- * runs, so only judge errors shrink this (reported per run below). */
165
- totalComparableRows: number;
166
- /** judge_error UNITS per run (a failed unit yields no parsed rows; it is
167
- * excluded from every flip denominator). */
168
- judgeErrorUnitsPerRun: number[];
169
- }
170
- /**
171
- * Compute the #458 judge-variance account over N runs of the same selections:
172
- * per-metric median + spread (the noise floor), and pairwise verdict flip
173
- * counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
174
- * if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
175
- * order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
176
- */
177
- export declare function computeJudgeVariance(perRunVerdicts: VarianceUnitVerdicts[][], metricSeries: VarianceMetricSeries[]): JudgeVariance;
178
- export {};