@gamaze/hicortex 0.23.1 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/dashboard.html +64 -13
- package/dist/calibration.d.ts +31 -0
- package/dist/calibration.js +38 -1
- package/dist/capture.d.ts +7 -0
- package/dist/capture.js +10 -1
- package/dist/dashboard.d.ts +32 -7
- package/dist/dashboard.js +50 -10
- package/dist/db.js +45 -0
- package/dist/dedup.js +2 -2
- package/dist/distill-queue.d.ts +203 -0
- package/dist/distill-queue.js +440 -0
- package/dist/health.d.ts +13 -1
- package/dist/health.js +6 -1
- package/dist/hosted-boot.d.ts +1 -1
- package/dist/index.js +17 -2
- package/dist/learnings-identity.js +20 -1
- package/dist/localhost-bypass.js +1 -1
- package/dist/mcp-server.js +92 -7
- package/dist/nightly.js +88 -1
- package/dist/nofit.d.ts +1 -1
- package/dist/nofit.js +1 -1
- package/dist/schema-prototypes.d.ts +1 -1
- package/dist/schema-prototypes.js +1 -1
- package/dist/status.d.ts +11 -0
- package/dist/status.js +25 -0
- package/dist/types.d.ts +20 -0
- package/hermes-plugin/hicortex/README.md +13 -6
- package/hermes-plugin/hicortex/__init__.py +7 -0
- package/hermes-plugin/hicortex/client.py +64 -2
- package/hermes-plugin/hicortex/plugin.yaml +1 -1
- package/hermes-plugin/hicortex/provider.py +24 -1
- package/opencode-plugin/hicortex/index.ts +24 -1
- package/package.json +2 -1
- package/pi-extension/hicortex/index.ts +24 -1
- package/server.json +2 -2
- package/dist/eval/decay-eval.d.ts +0 -111
- package/dist/eval/decay-eval.js +0 -214
- package/dist/eval/dups.d.ts +0 -100
- package/dist/eval/dups.js +0 -174
- package/dist/eval/eval-clock.d.ts +0 -32
- package/dist/eval/eval-clock.js +0 -47
- package/dist/eval/eval-db.d.ts +0 -25
- package/dist/eval/eval-db.js +0 -67
- package/dist/eval/graph-eval.d.ts +0 -89
- package/dist/eval/graph-eval.js +0 -246
- package/dist/eval/importance-eval.d.ts +0 -85
- package/dist/eval/importance-eval.js +0 -286
- package/dist/eval/planted-eval.d.ts +0 -30
- package/dist/eval/planted-eval.js +0 -122
- package/dist/eval/planted-fixtures.d.ts +0 -107
- package/dist/eval/planted-fixtures.js +0 -283
- package/dist/eval/planted-harness.d.ts +0 -183
- package/dist/eval/planted-harness.js +0 -651
- package/dist/eval/ranking-battery.d.ts +0 -125
- package/dist/eval/ranking-battery.js +0 -289
- package/dist/eval/ranking-eval.d.ts +0 -61
- package/dist/eval/ranking-eval.js +0 -554
- package/dist/eval/ranking-fixtures.d.ts +0 -117
- package/dist/eval/ranking-fixtures.js +0 -485
- package/dist/eval/recall-sweep.d.ts +0 -87
- package/dist/eval/recall-sweep.js +0 -1030
- package/dist/eval/reflection-census.d.ts +0 -19
- package/dist/eval/reflection-census.js +0 -25
- package/dist/eval/relevance-eval.d.ts +0 -178
- package/dist/eval/relevance-eval.js +0 -2240
- package/dist/eval/run-eval.d.ts +0 -20
- package/dist/eval/run-eval.js +0 -299
|
@@ -1,183 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The planted-pairs acceptance gate (#393 increment A) — machinery shared by
|
|
3
|
-
* the CLI runner (planted-eval.ts, REAL embedder + real-text corpus) and the
|
|
4
|
-
* vitest suite (tests/planted-pairs.test.ts, controlled synthetic vectors).
|
|
5
|
-
*
|
|
6
|
-
* What it measures, per planted pair, before/after running the REAL
|
|
7
|
-
* production resolution stage (`stageReconsolidation`, which internally runs
|
|
8
|
-
* the deterministic merge zone LAST — guard-C: judgment outranks the sweep):
|
|
9
|
-
*
|
|
10
|
-
* DETECTED — the pair entered the resolution machinery as a candidate: a
|
|
11
|
-
* verdict call was made on it (judged band, logged by the ground-truth
|
|
12
|
-
* judge) OR it co-clustered in the pre-stage `planDedup` discovery at the
|
|
13
|
-
* zone ceiling. Tagged by source ("judge" = KNN/similarity, "scout" = the
|
|
14
|
-
* #393 B reference-extraction source — attribution rule: a verdict call
|
|
15
|
-
* on a below-floor pair can only be scout-sourced, "zone" = the
|
|
16
|
-
* deterministic ceiling).
|
|
17
|
-
* BOUND — a resolution structure reflecting the ground truth now connects
|
|
18
|
-
* the pair: `corrected_by` / `superseded_by` / `conflicts` link, or a
|
|
19
|
-
* merge (`dedup_log` row). Guard-C added the `conflicts` bind (the judge
|
|
20
|
-
* flags a genuine conflict instead of leaving it inexpressible).
|
|
21
|
-
* RESOLVED — the class-specific desired END STATE holds in the store:
|
|
22
|
-
* corrects = correction effective (older demoted/absorbed/rewritten) and
|
|
23
|
-
* the correction information live;
|
|
24
|
-
* supersedes= older demoted-or-absorbed and newer live in recall;
|
|
25
|
-
* merge = exactly one live row remains, the other absorbed via dedup;
|
|
26
|
-
* conflict = both live AND not blended AND the conflicts link exists
|
|
27
|
-
* (guard-C's end state — the consumer sees both truths);
|
|
28
|
-
* none = both live, not blended, no resolution link between them.
|
|
29
|
-
*
|
|
30
|
-
* The version_chain CLASS end-state is measured on the RECALL surface, not
|
|
31
|
-
* the store (#393 D): a retrieval-only change cannot move a store probe, so
|
|
32
|
-
* post-stage the harness runs the production retrieve() (read-only: limit 3,
|
|
33
|
-
* below the cold-exposure threshold, noStrengthen) with the OLDEST chain
|
|
34
|
-
* member's content as query and requires the belief walk's guarantee — no
|
|
35
|
-
* superseded ancestor surfaces as a competing truth, and the linked chain's
|
|
36
|
-
* terminal surfaces. Store facts (terminal live, non-terminals gone) remain
|
|
37
|
-
* in the class note as corroboration.
|
|
38
|
-
*
|
|
39
|
-
* The judge is a GROUND-TRUTH-STUBBED LlmClient (the refine addendum's Q3
|
|
40
|
-
* recommendation): it answers each planted pair's declared relation and
|
|
41
|
-
* defaults unknown/collider pairs to `none`. This isolates the MACHINERY
|
|
42
|
-
* (detection sources, binding actions, merge guard, walk) from judge quality —
|
|
43
|
-
* the harness answers "would the pipeline resolve a KNOWN correction if the
|
|
44
|
-
* judge were perfect?". Live-LLM verdict quality is the real-corpus soak run.
|
|
45
|
-
*
|
|
46
|
-
* Read/write discipline: the gate NEVER touches ~/.hicortex — the fixture DB
|
|
47
|
-
* and the stage's stateDir are caller-supplied (temp) paths. A real snapshot
|
|
48
|
-
* is only ever PLANTED INTO A COPY (planted-eval.ts --snapshot).
|
|
49
|
-
*/
|
|
50
|
-
import type Database from "better-sqlite3";
|
|
51
|
-
import type { LlmClient } from "../llm.js";
|
|
52
|
-
import { type EmbedFn } from "../retrieval.js";
|
|
53
|
-
import type { acquireCaptureLock } from "../capture.js";
|
|
54
|
-
import { type ReconsolidationStageResult } from "../reconsolidation.js";
|
|
55
|
-
import { type PlanDedupResult } from "../dedup.js";
|
|
56
|
-
import { type PlantedCorpus, type FixtureClass, type GroundTruthRelation } from "./planted-fixtures.js";
|
|
57
|
-
/** Insert the corpus into a writable DB (fresh, or a copy of a snapshot). */
|
|
58
|
-
export declare function plantCorpus(db: Database.Database, corpus: PlantedCorpus, embedFn: EmbedFn): Promise<Map<string, string>>;
|
|
59
|
-
export interface VerdictCall {
|
|
60
|
-
olderKey: string;
|
|
61
|
-
newerKey: string;
|
|
62
|
-
/** True when the pair matched a declared ground-truth pair; false = default-none fallback. */
|
|
63
|
-
grounded: boolean;
|
|
64
|
-
}
|
|
65
|
-
export interface RewriteCall {
|
|
66
|
-
targetKey: string;
|
|
67
|
-
triggerKeys: string[];
|
|
68
|
-
}
|
|
69
|
-
/** A scout correction-shape call (#393 B) on one memory, ground-truth answered. */
|
|
70
|
-
export interface ScoutCall {
|
|
71
|
-
key: string;
|
|
72
|
-
correction: boolean;
|
|
73
|
-
}
|
|
74
|
-
export interface GroundTruthJudge {
|
|
75
|
-
llm: LlmClient;
|
|
76
|
-
verdictCalls: VerdictCall[];
|
|
77
|
-
rewriteCalls: RewriteCall[];
|
|
78
|
-
scoutCalls: ScoutCall[];
|
|
79
|
-
}
|
|
80
|
-
/**
|
|
81
|
-
* Deterministic, ground-truth-keyed judge. Parses each prompt back into its
|
|
82
|
-
* (older, newer) contents — the exact texts `buildCorrectionVerdictPrompt` /
|
|
83
|
-
* `buildRewritePrompt` embed — and answers from the corpus's declared
|
|
84
|
-
* relations:
|
|
85
|
-
* corrects -> {"action":"corrects","confidence":0.95} (above the gate)
|
|
86
|
-
* supersedes-> {"action":"supersedes","confidence":0.9}
|
|
87
|
-
* merge -> {"action":"merge","confidence":0.95} (above the gate)
|
|
88
|
-
* conflict -> {"action":"conflicts","confidence":0.9} (guard-C: flag,
|
|
89
|
-
* both live, never blended)
|
|
90
|
-
* none -> {"action":"none","confidence":0.9}
|
|
91
|
-
* Unknown pairs (colliders, filler cross-pairs) default to none.
|
|
92
|
-
*
|
|
93
|
-
* #393 B: the judge also answers the scout's per-memory shape calls — a
|
|
94
|
-
* memory is correction-shaped iff it is the newer side of a declared
|
|
95
|
-
* corrects/supersedes/conflict pair (declared `references` returned verbatim,
|
|
96
|
-
* throwing if a resolution-shaped pair never declared them; guard-C added
|
|
97
|
-
* `conflict` to the shaped set — two records disagreeing on one quantity
|
|
98
|
-
* share their wording); everything else answers correction=false. Shape calls
|
|
99
|
-
* are logged in scoutCalls.
|
|
100
|
-
*/
|
|
101
|
-
export declare function makeGroundTruthJudge(corpus: PlantedCorpus, idByKey: Map<string, string>): GroundTruthJudge;
|
|
102
|
-
export interface PlantedPairResult {
|
|
103
|
-
class: FixtureClass;
|
|
104
|
-
olderKey: string;
|
|
105
|
-
newerKey: string;
|
|
106
|
-
relation: GroundTruthRelation;
|
|
107
|
-
cosine: number;
|
|
108
|
-
inBand: boolean;
|
|
109
|
-
detected: boolean;
|
|
110
|
-
/**
|
|
111
|
-
* Which source(s) made the pair a candidate: "judge" (verdict call via the
|
|
112
|
-
* KNN similarity source), "scout" (verdict call on a BELOW-floor pair — only
|
|
113
|
-
* the scout can source one; #393 B), "zone" (co-clustered at the ceiling).
|
|
114
|
-
* In-band pairs the KNN source could have found are attributed "judge" even
|
|
115
|
-
* when the scout also found them — the attribution answers "what made this
|
|
116
|
-
* pair reachable", and for in-band pairs that is the similarity floor.
|
|
117
|
-
*/
|
|
118
|
-
detectionSource: "" | "judge" | "zone" | "judge+zone" | "scout" | "scout+zone";
|
|
119
|
-
bound: boolean | null;
|
|
120
|
-
boundHow: string;
|
|
121
|
-
resolved: boolean;
|
|
122
|
-
notes: string;
|
|
123
|
-
}
|
|
124
|
-
export interface PlantedClassResult {
|
|
125
|
-
class: FixtureClass;
|
|
126
|
-
pairs: number;
|
|
127
|
-
detected: number;
|
|
128
|
-
bound: number;
|
|
129
|
-
resolved: number;
|
|
130
|
-
/** Class-level end state (version_chain: terminal live + non-terminals gone). */
|
|
131
|
-
classResolved: boolean;
|
|
132
|
-
notes: string;
|
|
133
|
-
}
|
|
134
|
-
export interface ColliderRow {
|
|
135
|
-
aKey: string;
|
|
136
|
-
bKey: string;
|
|
137
|
-
cosine: number;
|
|
138
|
-
verdictCalled: boolean;
|
|
139
|
-
merged: boolean;
|
|
140
|
-
bothLive: boolean;
|
|
141
|
-
}
|
|
142
|
-
export interface PlantedGateResult {
|
|
143
|
-
pairResults: PlantedPairResult[];
|
|
144
|
-
classResults: PlantedClassResult[];
|
|
145
|
-
colliders: ColliderRow[];
|
|
146
|
-
stageReport: ReconsolidationStageResult;
|
|
147
|
-
zonePlanBefore: PlanDedupResult;
|
|
148
|
-
/** Pre-run zone-blend predictions: ground-truth pair -> surviving (canonical) key. */
|
|
149
|
-
blendPredictions: Array<{
|
|
150
|
-
pair: string;
|
|
151
|
-
canonicalKey: string;
|
|
152
|
-
loserKey: string;
|
|
153
|
-
}>;
|
|
154
|
-
}
|
|
155
|
-
export interface PlantedGateOptions {
|
|
156
|
-
/** Writable fixture DB path — created fresh, or a snapshot COPY to plant into. */
|
|
157
|
-
dbPath: string;
|
|
158
|
-
/** Temp state dir for the stage (state.json cursor, backups, band stats). */
|
|
159
|
-
stateDir: string;
|
|
160
|
-
embedFn: EmbedFn;
|
|
161
|
-
/** Lock acquirer for the zone/merge windows. Default: always-acquire (hermetic). */
|
|
162
|
-
acquireLock?: typeof acquireCaptureLock;
|
|
163
|
-
/** Zone ceiling. Default: the production default (0.92). */
|
|
164
|
-
threshold?: number;
|
|
165
|
-
/** #458: the pinned clock for the recall-surface retrieve() (undefined =
|
|
166
|
-
* live). Fixtures keep their absolute createdAt strings — only the
|
|
167
|
-
* scoring instant is pinned. */
|
|
168
|
-
now?: Date;
|
|
169
|
-
}
|
|
170
|
-
/**
|
|
171
|
-
* Build (or plant into) the DB, run the production stage, measure the gate.
|
|
172
|
-
* This is the ONE orchestration shared by the CLI runner and the vitest suite.
|
|
173
|
-
*/
|
|
174
|
-
export declare function runPlantedGate(corpus: PlantedCorpus, opts: PlantedGateOptions): Promise<PlantedGateResult>;
|
|
175
|
-
export declare function renderPlantedReport(args: {
|
|
176
|
-
corpusLabel: string;
|
|
177
|
-
generatedAt: string;
|
|
178
|
-
embedderLabel: string;
|
|
179
|
-
/** #458: clock-mode label ("pinned <ISO>" | "live") — the instant the
|
|
180
|
-
* recall-surface retrieve() scored against. */
|
|
181
|
-
clock: string;
|
|
182
|
-
result: PlantedGateResult;
|
|
183
|
-
}): string;
|