@gamaze/hicortex 0.23.0 → 0.23.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +455 -153
- package/dist/dashboard.d.ts +18 -7
- package/dist/dashboard.js +42 -10
- package/dist/dedup.js +2 -2
- package/dist/index.js +17 -2
- package/dist/learnings-identity.js +20 -1
- package/dist/llm.d.ts +12 -1
- package/dist/llm.js +14 -3
- package/dist/mcp-server.js +6 -2
- package/dist/mcp-stdio.d.ts +61 -3
- package/dist/mcp-stdio.js +272 -51
- package/dist/nightly.js +1 -1
- package/hermes-plugin/hicortex/provider.py +23 -0
- package/opencode-plugin/hicortex/index.ts +24 -1
- package/package.json +2 -1
- package/pi-extension/hicortex/index.ts +24 -1
- package/server.json +2 -2
- package/dist/eval/decay-eval.d.ts +0 -111
- package/dist/eval/decay-eval.js +0 -214
- package/dist/eval/dups.d.ts +0 -100
- package/dist/eval/dups.js +0 -174
- package/dist/eval/eval-clock.d.ts +0 -32
- package/dist/eval/eval-clock.js +0 -47
- package/dist/eval/eval-db.d.ts +0 -25
- package/dist/eval/eval-db.js +0 -67
- package/dist/eval/graph-eval.d.ts +0 -89
- package/dist/eval/graph-eval.js +0 -246
- package/dist/eval/importance-eval.d.ts +0 -85
- package/dist/eval/importance-eval.js +0 -286
- package/dist/eval/planted-eval.d.ts +0 -30
- package/dist/eval/planted-eval.js +0 -122
- package/dist/eval/planted-fixtures.d.ts +0 -107
- package/dist/eval/planted-fixtures.js +0 -283
- package/dist/eval/planted-harness.d.ts +0 -183
- package/dist/eval/planted-harness.js +0 -651
- package/dist/eval/ranking-battery.d.ts +0 -125
- package/dist/eval/ranking-battery.js +0 -289
- package/dist/eval/ranking-eval.d.ts +0 -61
- package/dist/eval/ranking-eval.js +0 -554
- package/dist/eval/ranking-fixtures.d.ts +0 -117
- package/dist/eval/ranking-fixtures.js +0 -485
- package/dist/eval/recall-sweep.d.ts +0 -87
- package/dist/eval/recall-sweep.js +0 -1030
- package/dist/eval/reflection-census.d.ts +0 -19
- package/dist/eval/reflection-census.js +0 -25
- package/dist/eval/relevance-eval.d.ts +0 -178
- package/dist/eval/relevance-eval.js +0 -2240
- package/dist/eval/run-eval.d.ts +0 -20
- package/dist/eval/run-eval.js +0 -299
|
@@ -1,19 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Reflection census (#191 mechanical baseline) — lesson output volume over
|
|
3
|
-
* time and the lesson/episode yield ratio. This is NOT a quality read
|
|
4
|
-
* (sampling lessons for actionable-vs-noise is Phase-A grading work,
|
|
5
|
-
* deferred per the issue's owner comment until recall is active) — just the
|
|
6
|
-
* mechanical count.
|
|
7
|
-
*/
|
|
8
|
-
import type Database from "better-sqlite3";
|
|
9
|
-
export interface ReflectionCensus {
|
|
10
|
-
lessonsByDate: Array<{
|
|
11
|
-
date: string;
|
|
12
|
-
count: number;
|
|
13
|
-
}>;
|
|
14
|
-
totalLessons: number;
|
|
15
|
-
totalEpisodes: number;
|
|
16
|
-
/** lessons / episodes — null when there are no episodes to divide by. */
|
|
17
|
-
yieldRatio: number | null;
|
|
18
|
-
}
|
|
19
|
-
export declare function runReflectionCensus(db: Database.Database): ReflectionCensus;
|
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
/**
|
|
3
|
-
* Reflection census (#191 mechanical baseline) — lesson output volume over
|
|
4
|
-
* time and the lesson/episode yield ratio. This is NOT a quality read
|
|
5
|
-
* (sampling lessons for actionable-vs-noise is Phase-A grading work,
|
|
6
|
-
* deferred per the issue's owner comment until recall is active) — just the
|
|
7
|
-
* mechanical count.
|
|
8
|
-
*/
|
|
9
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
10
|
-
exports.runReflectionCensus = runReflectionCensus;
|
|
11
|
-
function runReflectionCensus(db) {
|
|
12
|
-
const lessonsByDate = db
|
|
13
|
-
.prepare(`SELECT date(created_at) AS date, COUNT(*) AS count
|
|
14
|
-
FROM memories WHERE memory_type = 'learnings'
|
|
15
|
-
GROUP BY date ORDER BY date`)
|
|
16
|
-
.all();
|
|
17
|
-
const totalLessons = db.prepare("SELECT COUNT(*) AS c FROM memories WHERE memory_type = 'learnings'").get().c;
|
|
18
|
-
const totalEpisodes = db.prepare("SELECT COUNT(*) AS c FROM memories WHERE memory_type = 'experience'").get().c;
|
|
19
|
-
return {
|
|
20
|
-
lessonsByDate,
|
|
21
|
-
totalLessons,
|
|
22
|
-
totalEpisodes,
|
|
23
|
-
yieldRatio: totalEpisodes > 0 ? totalLessons / totalEpisodes : null,
|
|
24
|
-
};
|
|
25
|
-
}
|
|
@@ -1,178 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
/**
|
|
3
|
-
* Real-query relevance + SNIPPET eval — recall QUALITY on real agent prompts.
|
|
4
|
-
*
|
|
5
|
-
* v2 (spec `specs/2026-08-02-relevance-eval.md`) extends the v1 selection-only
|
|
6
|
-
* eval with the SNIPPET layer: v1 asked "did retrieve() surface the right
|
|
7
|
-
* memories?" (judge sees up to 2000 chars). Production shows the agent only a
|
|
8
|
-
* ~100-char one-liner (`recall-index.ts#memoryTitle`), so a memory can be
|
|
9
|
-
* genuinely relevant while its rendered line is useless — v1 scored that as a
|
|
10
|
-
* win. v2 grades BOTH: `full_verdict` (selection quality, judge sees full
|
|
11
|
-
* content) and `line_verdict` (snippet quality, judge sees ONLY the rendered
|
|
12
|
-
* production one-liner — imported from `recall-index.ts`, never reimplemented).
|
|
13
|
-
*
|
|
14
|
-
* v2 additions (spec §4, §5, §6, §6b) layered onto the v1 base (prompt
|
|
15
|
-
* sampling, readonly snapshot handling, embed-once + neverCalledEmbed,
|
|
16
|
-
* lenient JSON parse, distribution/CI reporting):
|
|
17
|
-
* 1. Dual verdict per surfaced memory — TWO separate, blind judge calls.
|
|
18
|
-
* 2. Snippet-length sweep (100/200/300/title+1st-sentence) on a fixed
|
|
19
|
-
* 40-prompt subset (8 per source).
|
|
20
|
-
* 3. Similarity-floor + retrieval-source analysis (near-free — logged, not
|
|
21
|
-
* re-judged).
|
|
22
|
-
* 4. Token-cost estimate (char/4) per K and per snippet-length variant.
|
|
23
|
-
* 5. ~20 rendered ACTUAL production blocks dumped into the report.
|
|
24
|
-
* 6. Redundancy — one set-level judge call per (prompt × mode) over the
|
|
25
|
-
* production 6.
|
|
26
|
-
* 7. Rate limiting + resumability (§6b, MANDATORY): serial calls,
|
|
27
|
-
* `--judge-delay-ms` (default 2000), exponential backoff with jitter on
|
|
28
|
-
* 429/5xx/timeout (5→10→20→40→80s, max 5 retries, respects
|
|
29
|
-
* `Retry-After`), checkpoint-per-call to a `.jsonl` sidecar, `--resume`,
|
|
30
|
-
* progress logging, 10%-error-rate abort, `--max-calls` budget guard
|
|
31
|
-
* (default 900).
|
|
32
|
-
*
|
|
33
|
-
* Prompt corpus (spec §2, owner decision §11.1): EVEN split, 20 prompts per
|
|
34
|
-
* source × 5 sources — Hermes (lenny, raider, nano) + CC (the DevOps
|
|
35
|
-
* `infrastructure` project, the `aironic-marine` project). Saved to
|
|
36
|
-
* `data/prompts.json`, stable/reused verbatim once a valid v2 set exists.
|
|
37
|
-
*
|
|
38
|
-
* Judge: GLM-5.2 via z.ai — the INSTRUMENT only. It never picks candidates;
|
|
39
|
-
* retrieve() (LLM-free) does. A dedicated raw HTTP caller (NOT `LlmClient`) is
|
|
40
|
-
* used here on purpose: `LlmClient.completeReflect` bakes in a
|
|
41
|
-
* nightly-tolerant retry policy (30s/60s/120s, unlimited rate-limit patience)
|
|
42
|
-
* that conflicts with §6b's specific real-time batch policy (5/10/20/40/80s +
|
|
43
|
-
* jitter, 5 retries, a hard call budget). Implemented directly here rather
|
|
44
|
-
* than adding a second retry mode to `llm.ts` (out of scope for this eval,
|
|
45
|
-
* and another agent is concurrently working elsewhere in this repo).
|
|
46
|
-
*
|
|
47
|
-
* Honesty invariants (non-negotiable — mirror recall-sweep.ts + spec §7):
|
|
48
|
-
* - Snapshot opened READONLY via openSnapshot — never initDb.
|
|
49
|
-
* - noStrengthen: true on every retrieve() call.
|
|
50
|
-
* - Real bge-small-en-v1.5 embedder, embed-once + queryEmbedding reuse.
|
|
51
|
-
* - neverCalledEmbed self-check ABORTS the run if retrieve() ignores
|
|
52
|
-
* queryEmbedding (would invalidate every measured number).
|
|
53
|
-
* - GLM-5.2 is the JUDGE only, real prompts, no synthetic queries.
|
|
54
|
-
* - Production renderer (`formatIndexLine`/`memoryTitle`) imported from
|
|
55
|
-
* `recall-index.ts`, never reimplemented.
|
|
56
|
-
* - judge_error batches/units excluded from every denominator, reported
|
|
57
|
-
* separately.
|
|
58
|
-
*
|
|
59
|
-
* Run:
|
|
60
|
-
* npm run eval:relevance -- <snapshot.db> [prompts.json] [report.md] \
|
|
61
|
-
* [--judge-delay-ms=2000] [--max-calls=900] [--resume] \
|
|
62
|
-
* [--verdicts-json=path] [--verdicts-jsonl=path] \
|
|
63
|
-
* [--now=<ISO>] [--runs=N]
|
|
64
|
-
*
|
|
65
|
-
* #458 clock pin + judge variance protocol:
|
|
66
|
-
* - `--now=<ISO>` pins the clock the retrieve sweep scores against (the
|
|
67
|
-
* retrieve() `now` seam) — before/after runs become wall-clock-independent
|
|
68
|
-
* (default: live clock, the pre-#458 behavior).
|
|
69
|
-
* - `--runs=N` (default 1) runs judge phases 1+2 N times over the SAME
|
|
70
|
-
* selections (one retrieve sweep) and reports the run-to-run noise floor
|
|
71
|
-
* (median + spread + selection-identical flip counts, §15b). Phases 3+4
|
|
72
|
-
* stay single-shot; the shared --max-calls budget spans all runs and the
|
|
73
|
-
* default scales ×N unless --max-calls was passed explicitly.
|
|
74
|
-
*/
|
|
75
|
-
type Verdict = "relevant" | "noise" | "stale";
|
|
76
|
-
interface JudgeVerdict {
|
|
77
|
-
/** Q1..Q8 label — mapped back to the real memory id via the caller's
|
|
78
|
-
* position-ordered topIds list. */
|
|
79
|
-
id: string;
|
|
80
|
-
verdict: Verdict;
|
|
81
|
-
reason: string;
|
|
82
|
-
}
|
|
83
|
-
type VerdictsOrError = JudgeVerdict[] | {
|
|
84
|
-
judgeError: string;
|
|
85
|
-
};
|
|
86
|
-
interface CliArgs {
|
|
87
|
-
positionals: string[];
|
|
88
|
-
judgeDelayMs: number;
|
|
89
|
-
maxCalls: number;
|
|
90
|
-
/** True when --max-calls was passed explicitly (#458) — an explicit budget
|
|
91
|
-
* was sized by the caller for the whole invocation and must NOT be scaled
|
|
92
|
-
* by --runs (see scaleMaxCalls). */
|
|
93
|
-
maxCallsExplicit: boolean;
|
|
94
|
-
resume: boolean;
|
|
95
|
-
/** Override the raw-verdicts JSON path (default VERDICTS_JSON_PATH) — added
|
|
96
|
-
* 2026-08-02 so a re-run against a NEW snapshot can never silently
|
|
97
|
-
* overwrite a baseline's raw verdicts. */
|
|
98
|
-
verdictsJsonPath: string;
|
|
99
|
-
/** Override the checkpoint sidecar / raw-verdicts JSONL path (default
|
|
100
|
-
* VERDICTS_JSONL_PATH) — same reason. */
|
|
101
|
-
verdictsJsonlPath: string;
|
|
102
|
-
/** #458: raw --now=<ISO> value — undefined = live clock. Validated in main
|
|
103
|
-
* via parsePinnedNow (invalid → usage-style error, exit 1). */
|
|
104
|
-
nowIso: string | undefined;
|
|
105
|
-
/** #458: judge runs over the fixed selections (phases 1+2 only). Default 1
|
|
106
|
-
* = the pre-#458 single-run behavior. */
|
|
107
|
-
runs: number;
|
|
108
|
-
}
|
|
109
|
-
/** Parsed and exported for tests (#458). Throws on an invalid --runs value
|
|
110
|
-
* (NaN or <1) — fail-explicit, never a silent default. */
|
|
111
|
-
export declare function parseArgs(argv: string[]): CliArgs;
|
|
112
|
-
/**
|
|
113
|
-
* #458 — resolve the effective --max-calls budget. The ONE shared budget
|
|
114
|
-
* spans ALL judge runs, so when --runs>1 and the caller did NOT pass
|
|
115
|
-
* --max-calls explicitly, the default scales ×N (each run re-judges every
|
|
116
|
-
* unit). An explicit budget always wins — the caller sized it for the whole
|
|
117
|
-
* invocation. Pure, exported for tests.
|
|
118
|
-
*/
|
|
119
|
-
export declare function scaleMaxCalls(runs: number, explicitMaxCalls: number | undefined, defaultMaxCalls: number): number;
|
|
120
|
-
/** Median of a numeric sample (even count → mean of the two middle values).
|
|
121
|
-
* Empty input → NaN, the mean() convention above. Does not mutate input. */
|
|
122
|
-
export declare function median(xs: number[]): number;
|
|
123
|
-
/** One judged unit's verdicts for variance accounting (#458): `key` is the
|
|
124
|
-
* stable (promptIdx|mode) unit identity, identical across runs because the
|
|
125
|
-
* retrieve sweep runs ONCE. */
|
|
126
|
-
export interface VarianceUnitVerdicts {
|
|
127
|
-
key: string;
|
|
128
|
-
verdicts: VerdictsOrError;
|
|
129
|
-
}
|
|
130
|
-
/** A headline metric measured once per run (e.g. precision@5 OFF). Values are
|
|
131
|
-
* in raw fractions; null = not computable that run (excluded from
|
|
132
|
-
* median/spread). */
|
|
133
|
-
export interface VarianceMetricSeries {
|
|
134
|
-
label: string;
|
|
135
|
-
values: Array<number | null>;
|
|
136
|
-
}
|
|
137
|
-
export interface VarianceMetricStats extends VarianceMetricSeries {
|
|
138
|
-
median: number | null;
|
|
139
|
-
/** (max − min) × 100, in points — the measured run-to-run noise floor. */
|
|
140
|
-
spreadPts: number | null;
|
|
141
|
-
}
|
|
142
|
-
/** Pairwise flip counts between runs i and j (1-based run numbers). */
|
|
143
|
-
export interface PairwiseFlips {
|
|
144
|
-
runA: number;
|
|
145
|
-
runB: number;
|
|
146
|
-
/** Verdict ROWS compared: (unit, Q-label) pairs present as parsed verdict
|
|
147
|
-
* arrays in BOTH runs. judge_error units and labels the judge omitted in
|
|
148
|
-
* one run are not comparable — excluded, never counted as flips. */
|
|
149
|
-
rowsCompared: number;
|
|
150
|
-
rowsFlipped: number;
|
|
151
|
-
unitsCompared: number;
|
|
152
|
-
/** A UNIT flips if any of its comparable rows differ. */
|
|
153
|
-
unitsFlipped: number;
|
|
154
|
-
}
|
|
155
|
-
export interface JudgeVariance {
|
|
156
|
-
runs: number;
|
|
157
|
-
metrics: VarianceMetricStats[];
|
|
158
|
-
flips: PairwiseFlips[];
|
|
159
|
-
maxRowsFlipped: number;
|
|
160
|
-
maxUnitsFlipped: number;
|
|
161
|
-
meanRowsFlipped: number;
|
|
162
|
-
meanUnitsFlipped: number;
|
|
163
|
-
/** Comparable rows of the first pair — selections are identical across
|
|
164
|
-
* runs, so only judge errors shrink this (reported per run below). */
|
|
165
|
-
totalComparableRows: number;
|
|
166
|
-
/** judge_error UNITS per run (a failed unit yields no parsed rows; it is
|
|
167
|
-
* excluded from every flip denominator). */
|
|
168
|
-
judgeErrorUnitsPerRun: number[];
|
|
169
|
-
}
|
|
170
|
-
/**
|
|
171
|
-
* Compute the #458 judge-variance account over N runs of the same selections:
|
|
172
|
-
* per-metric median + spread (the noise floor), and pairwise verdict flip
|
|
173
|
-
* counts (rows = one Q-label judgment on one (prompt,mode) unit; units flip
|
|
174
|
-
* if any row does). `perRunVerdicts[r]` is run r+1's units in a stable
|
|
175
|
-
* order/identity; `metricSeries[s].values[r]` is run r+1's headline value.
|
|
176
|
-
*/
|
|
177
|
-
export declare function computeJudgeVariance(perRunVerdicts: VarianceUnitVerdicts[][], metricSeries: VarianceMetricSeries[]): JudgeVariance;
|
|
178
|
-
export {};
|