hippo-memory 1.55.0 → 1.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/README.md +11 -0
  2. package/dist/api.d.ts +19 -9
  3. package/dist/api.js +112 -35
  4. package/dist/card-detail.d.ts +1 -1
  5. package/dist/card-detail.js +1 -1
  6. package/dist/cli/shared.d.ts +137 -0
  7. package/dist/cli/shared.js +830 -0
  8. package/dist/cli/sleep.d.ts +10 -0
  9. package/dist/cli/sleep.js +171 -0
  10. package/dist/cli.d.ts +0 -7
  11. package/dist/cli.js +313 -1806
  12. package/dist/config.d.ts +5 -0
  13. package/dist/config.js +21 -0
  14. package/dist/connectors/github/webhook.d.ts +19 -0
  15. package/dist/connectors/github/webhook.js +313 -0
  16. package/dist/connectors/slack/webhook.d.ts +22 -0
  17. package/dist/connectors/slack/webhook.js +203 -0
  18. package/dist/consolidate.js +3 -2
  19. package/dist/context-auto.d.ts +3 -0
  20. package/dist/context-auto.js +34 -0
  21. package/dist/customer-notes.js +2 -1
  22. package/dist/dashboard.js +2 -1
  23. package/dist/db.js +67 -1
  24. package/dist/decisions.js +2 -1
  25. package/dist/delivery-recorder.d.ts +127 -0
  26. package/dist/delivery-recorder.js +218 -0
  27. package/dist/eval-stats.d.ts +58 -0
  28. package/dist/eval-stats.js +111 -0
  29. package/dist/goals.d.ts +49 -25
  30. package/dist/goals.js +39 -22
  31. package/dist/graph-extract.js +1 -1
  32. package/dist/graph-recall.d.ts +1 -1
  33. package/dist/graph-recall.js +1 -1
  34. package/dist/graph.js +1 -1
  35. package/dist/hooks.d.ts +1 -3
  36. package/dist/hooks.js +2 -4
  37. package/dist/http-util.d.ts +31 -0
  38. package/dist/http-util.js +46 -0
  39. package/dist/incidents.js +2 -1
  40. package/dist/index.d.ts +5 -2
  41. package/dist/index.js +5 -2
  42. package/dist/mcp/server.js +173 -285
  43. package/dist/memory.d.ts +19 -0
  44. package/dist/memory.js +38 -0
  45. package/dist/policies.js +2 -1
  46. package/dist/predictions.js +2 -1
  47. package/dist/processes.js +2 -1
  48. package/dist/project-briefs.js +3 -1
  49. package/dist/prompt-recall.js +1 -1
  50. package/dist/recall-history.d.ts +5 -0
  51. package/dist/recall-history.js +9 -0
  52. package/dist/recall-pipeline.d.ts +101 -0
  53. package/dist/recall-pipeline.js +313 -0
  54. package/dist/recall-scope.d.ts +22 -0
  55. package/dist/recall-scope.js +27 -1
  56. package/dist/recall-trace.d.ts +69 -0
  57. package/dist/recall-trace.js +136 -0
  58. package/dist/search.d.ts +0 -20
  59. package/dist/search.js +2 -49
  60. package/dist/server.js +1901 -2384
  61. package/dist/skills.js +2 -1
  62. package/dist/store-cards.d.ts +53 -0
  63. package/dist/store-cards.js +512 -0
  64. package/dist/store.d.ts +2 -89
  65. package/dist/store.js +6 -562
  66. package/dist/tenant.d.ts +22 -0
  67. package/dist/tenant.js +26 -0
  68. package/dist/token-ledger.d.ts +2 -0
  69. package/dist/token-ledger.js +5 -0
  70. package/dist/tokenize.d.ts +2 -0
  71. package/dist/tokenize.js +8 -0
  72. package/dist/version.d.ts +1 -1
  73. package/dist/version.js +1 -1
  74. package/extensions/openclaw-plugin/openclaw.plugin.json +1 -1
  75. package/extensions/openclaw-plugin/package.json +1 -1
  76. package/openclaw.plugin.json +1 -1
  77. package/package.json +2 -1
package/dist/dashboard.js CHANGED
@@ -9,7 +9,8 @@ import * as os from 'os';
9
9
  import * as path from 'path';
10
10
  import * as fs from 'fs';
11
11
  import { extractPathTags } from './path-context.js';
12
- import { loadAllEntries, listCards, listMemoryConflicts, readEntry, writeEntry } from './store.js';
12
+ import { loadAllEntries, listMemoryConflicts, readEntry, writeEntry } from './store.js';
13
+ import { listCards } from './store-cards.js';
13
14
  import { calculateStrength, confidenceFacets } from './memory.js';
14
15
  import { loadConfig } from './config.js';
15
16
  import { listPeers } from './shared.js';
package/dist/db.js CHANGED
@@ -11,7 +11,7 @@ const require = createRequire(import.meta.url);
11
11
  // runtime (Node's built-in synchronous SQLite module); there are no bundled
12
12
  // types for it here, so this require + cast is the module's documented boundary.
13
13
  const { DatabaseSync } = require('node:sqlite');
14
- const CURRENT_SCHEMA_VERSION = 49;
14
+ const CURRENT_SCHEMA_VERSION = 50;
15
15
  const MIGRATIONS = [
16
16
  {
17
17
  version: 1,
@@ -2501,6 +2501,72 @@ const MIGRATIONS = [
2501
2501
  `);
2502
2502
  },
2503
2503
  },
2504
+ {
2505
+ version: 50,
2506
+ up: (db) => {
2507
+ // Per-turn delivery events (src/recall-trace.ts). Additive only: no min_compatible_binary bump; rollback drops both tables
2508
+ // and sets schema_version back to 49. No CHECK on enum columns since SQLite cannot alter one; delivery-recorder.ts unions are the allowlist.
2509
+ db.exec(`
2510
+ CREATE TABLE IF NOT EXISTS delivery_events (
2511
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
2512
+ ts TEXT NOT NULL,
2513
+ ledger_version INTEGER NOT NULL,
2514
+ tenant_id TEXT NOT NULL DEFAULT 'default',
2515
+ runtime TEXT NOT NULL,
2516
+ event_type TEXT NOT NULL,
2517
+ surface TEXT NOT NULL,
2518
+ store_hash TEXT NOT NULL,
2519
+ write_store TEXT NOT NULL,
2520
+ project_hash TEXT,
2521
+ session_id TEXT,
2522
+ session_state TEXT NOT NULL,
2523
+ host_turn_id TEXT,
2524
+ turn_seq INTEGER,
2525
+ duplicate_of INTEGER,
2526
+ prompt_hash TEXT,
2527
+ prompt_length INTEGER NOT NULL DEFAULT 0,
2528
+ query_hash TEXT,
2529
+ recall_trace_id INTEGER,
2530
+ block_state TEXT NOT NULL,
2531
+ prompt_recall INTEGER NOT NULL DEFAULT 0,
2532
+ considered_count INTEGER NOT NULL DEFAULT 0,
2533
+ filtered_count INTEGER NOT NULL DEFAULT 0,
2534
+ selected_count INTEGER NOT NULL DEFAULT 0,
2535
+ emitted_count INTEGER NOT NULL DEFAULT 0,
2536
+ rejected_count INTEGER NOT NULL DEFAULT 0,
2537
+ rejected_unlisted INTEGER NOT NULL DEFAULT 0,
2538
+ sections_shown INTEGER NOT NULL DEFAULT 0,
2539
+ sections_dropped INTEGER NOT NULL DEFAULT 0,
2540
+ budget_tokens INTEGER NOT NULL DEFAULT 0,
2541
+ selected_tokens INTEGER NOT NULL DEFAULT 0,
2542
+ injected_tokens INTEGER NOT NULL DEFAULT 0,
2543
+ static_hash TEXT,
2544
+ recall_hash TEXT,
2545
+ emitted_hash TEXT,
2546
+ elapsed_ms INTEGER NOT NULL DEFAULT 0
2547
+ );
2548
+ CREATE INDEX IF NOT EXISTS idx_delivery_events_session ON delivery_events(tenant_id, session_id, id);
2549
+ CREATE INDEX IF NOT EXISTS idx_delivery_events_ts ON delivery_events(ts);
2550
+ CREATE UNIQUE INDEX IF NOT EXISTS idx_delivery_events_turn
2551
+ ON delivery_events(tenant_id, session_id, event_type, turn_seq) WHERE turn_seq IS NOT NULL;
2552
+
2553
+ CREATE TABLE IF NOT EXISTS delivery_candidates (
2554
+ event_id INTEGER NOT NULL REFERENCES delivery_events(id) ON DELETE CASCADE,
2555
+ tenant_id TEXT NOT NULL DEFAULT 'default',
2556
+ memory_id TEXT NOT NULL, -- no FK: events outlive forgotten memories, as recall traces do
2557
+ source_store TEXT NOT NULL,
2558
+ pool TEXT NOT NULL,
2559
+ stage TEXT NOT NULL,
2560
+ outcome TEXT NOT NULL,
2561
+ reason TEXT,
2562
+ cand_rank INTEGER,
2563
+ score REAL,
2564
+ tokens INTEGER,
2565
+ PRIMARY KEY (event_id, memory_id)
2566
+ ) WITHOUT ROWID;
2567
+ `);
2568
+ },
2569
+ },
2504
2570
  ];
2505
2571
  function tableHasColumn(db, tableName, columnName) {
2506
2572
  if (!/^[a-z_]+$/i.test(tableName))
package/dist/decisions.js CHANGED
@@ -24,7 +24,8 @@
24
24
  * step rolls all of them back. Pattern matches savePrediction (predictions.ts).
25
25
  */
26
26
  import { openHippoDb, closeHippoDb } from './db.js';
27
- import { writeEntry, assertTenantId } from './store.js';
27
+ import { writeEntry } from './store.js';
28
+ import { assertTenantId } from './tenant.js';
28
29
  import { markGraphDirty, removeGraphEntitiesForObject } from './graph.js';
29
30
  import { createMemory, Layer } from './memory.js';
30
31
  import { appendAuditEvent } from './audit.js';
@@ -0,0 +1,127 @@
1
+ import type { MemoryEntry } from './memory.js';
2
+ import { type PromptRecallGate } from './prompt-recall.js';
3
+ export type DeliveryRuntime = 'claude-code' | 'codex' | 'unknown';
4
+ export type DeliveryEventType = 'prompt-submit' | 'pinned-manual';
5
+ export type DeliverySurface = 'hook' | 'context';
6
+ export type DeliveryWriteStore = 'local' | 'global';
7
+ export type DeliverySessionState = 'payload' | 'env' | 'missing' | 'subagent';
8
+ export type DeliveryBlockState = 'sent' | 'reused' | 'reused-recall-sent' | 'empty' | 'disabled';
9
+ export type DeliveryPool = 'pin' | 'recent' | 'prompt-recall' | 'strength' | 'search';
10
+ export type DeliveryStage = 'load' | 'eligible' | 'gate' | 'budget' | 'limit' | 'final';
11
+ export type DeliveryOutcome = 'emitted' | 'reused' | 'rejected';
12
+ export type DeliveryRejectReason = 'budget' | 'gate-below-threshold' | 'gate-max-items' | 'duplicate' | 'scope' | 'quality' | 'limit';
13
+ /** Row format version written to `delivery_events.ledger_version`. */
14
+ export declare const DELIVERY_LEDGER_VERSION = 1;
15
+ /** Rejected candidate rows kept per event; the rest only add to `rejected_unlisted`. */
16
+ export declare const DELIVERY_REJECTED_ROW_CAP = 16;
17
+ export interface DeliveryCandidateInput {
18
+ memoryId: string;
19
+ sourceStore: DeliveryWriteStore;
20
+ pool: DeliveryPool;
21
+ stage: DeliveryStage;
22
+ outcome: DeliveryOutcome;
23
+ reason: DeliveryRejectReason | null;
24
+ rank: number | null;
25
+ score: number | null;
26
+ tokens: number | null;
27
+ }
28
+ /** One event as the writer stores it; the writer adds `turn_seq`, `duplicate_of` and the format version. */
29
+ export interface DeliveryEventInput {
30
+ ts: string;
31
+ tenantId: string;
32
+ runtime: DeliveryRuntime;
33
+ eventType: DeliveryEventType;
34
+ surface: DeliverySurface;
35
+ storeHash: string;
36
+ writeStore: DeliveryWriteStore;
37
+ projectHash: string | null;
38
+ sessionId: string | null;
39
+ sessionState: DeliverySessionState;
40
+ hostTurnId: string | null;
41
+ promptHash: string | null;
42
+ promptLength: number;
43
+ queryHash: string | null;
44
+ recallTraceId: number | null;
45
+ blockState: DeliveryBlockState;
46
+ promptRecall: boolean;
47
+ consideredCount: number;
48
+ filteredCount: number;
49
+ selectedCount: number;
50
+ emittedCount: number;
51
+ rejectedCount: number;
52
+ rejectedUnlisted: number;
53
+ sectionsShown: number;
54
+ sectionsDropped: number;
55
+ budgetTokens: number;
56
+ selectedTokens: number;
57
+ injectedTokens: number;
58
+ staticHash: string | null;
59
+ recallHash: string | null;
60
+ emittedHash: string | null;
61
+ elapsedMs: number;
62
+ candidates: readonly DeliveryCandidateInput[];
63
+ }
64
+ export interface DeliveryFacts {
65
+ projectName: string;
66
+ budgetTokens: number;
67
+ promptRecall: boolean;
68
+ }
69
+ /** The shape of a returned context entry the observer reads. */
70
+ export interface DeliverySelected {
71
+ entry: MemoryEntry;
72
+ score: number;
73
+ tokens: number;
74
+ isGlobal?: boolean;
75
+ promptRecall?: boolean;
76
+ }
77
+ /** What getContext reports while it selects. Every method only reads; none changes what is selected. */
78
+ export interface DeliveryObserver {
79
+ facts(facts: DeliveryFacts): void;
80
+ sections(shown: number, dropped: number): void;
81
+ /** Returns `admit`'s own answer unchanged and lets its throws through. */
82
+ watchAdmit(admit: (e: MemoryEntry) => boolean): (e: MemoryEntry) => boolean;
83
+ /** The loader's quality floor dropped a row admit let through; with prompt recall on, eligibility reports it instead. */
84
+ qualityDropped(entry: MemoryEntry, isGlobal: boolean): void;
85
+ disabled(): void;
86
+ /** No `pool` means pin or recent by the entry's own flag. */
87
+ offer(entries: readonly MemoryEntry[], isGlobal: boolean, pool?: DeliveryPool): void;
88
+ reject(entry: MemoryEntry, stage: DeliveryStage, reason: DeliveryRejectReason, score?: number, tokens?: number): void;
89
+ dropMissing(before: readonly MemoryEntry[], after: readonly MemoryEntry[], stage: DeliveryStage, reason: DeliveryRejectReason): void;
90
+ gated(prompt: ReadonlySet<string>, candidates: readonly {
91
+ id: string;
92
+ tokens: ReadonlySet<string>;
93
+ }[], gate: PromptRecallGate, kept: readonly {
94
+ item: {
95
+ id: string;
96
+ };
97
+ }[]): void;
98
+ selected(items: readonly DeliverySelected[]): void;
99
+ }
100
+ /** What the renderer sent, reported once at its exit. */
101
+ export interface DeliveryOutcomeInput {
102
+ state: DeliveryBlockState;
103
+ staticHash?: string | null;
104
+ recallHash?: string | null;
105
+ /** The exact text the agent receives: the hook's additionalContext, or every stdout byte, newline included. */
106
+ emittedText?: string | null;
107
+ /** The static block was skipped as unchanged, so its entries are reused, not sent. */
108
+ staticReused?: boolean;
109
+ }
110
+ export interface DeliveryRecorder extends DeliveryObserver {
111
+ readonly root: string;
112
+ delivered(outcome: DeliveryOutcomeInput): void;
113
+ /** Builds the event and hands it to `write` once per call; throws on an injected fault, and writes nothing once broken. */
114
+ flush(write: (input: DeliveryEventInput) => number | null): void;
115
+ }
116
+ export interface DeliveryRecorderInit {
117
+ /** The store the event is written to. */
118
+ root: string;
119
+ storeHash: string;
120
+ writeStore: DeliveryWriteStore;
121
+ tenantId: string;
122
+ stdinText?: string;
123
+ envSessionId?: string;
124
+ }
125
+ /** A recorder for one call; every observer method is guarded, and a throw marks it broken instead of escaping. */
126
+ export declare function createDeliveryRecorder(init: DeliveryRecorderInit): DeliveryRecorder;
127
+ //# sourceMappingURL=delivery-recorder.d.ts.map
@@ -0,0 +1,218 @@
1
+ import { evalNow } from './ablation.js';
2
+ import { scoreOverlap } from './prompt-recall.js';
3
+ import { blockHash, estimateTokens, hookPayloadSessionId, hookPayloadString, isSubagentPayload } from './token-ledger.js';
4
+ /** Row format version written to `delivery_events.ledger_version`. */
5
+ export const DELIVERY_LEDGER_VERSION = 1;
6
+ /** Rejected candidate rows kept per event; the rest only add to `rejected_unlisted`. */
7
+ export const DELIVERY_REJECTED_ROW_CAP = 16;
8
+ // Deeper stages were closer to being sent, so the row cap keeps them first.
9
+ const STAGE_DEPTH = new Map([
10
+ ['load', 0], ['eligible', 1], ['gate', 2], ['budget', 3], ['limit', 4], ['final', 5],
11
+ ]);
12
+ function storeOf(isGlobal) {
13
+ return isGlobal === true ? 'global' : 'local';
14
+ }
15
+ function byDepthThenScore(a, b) {
16
+ const depth = (STAGE_DEPTH.get(b.stage ?? 'load') ?? 0) - (STAGE_DEPTH.get(a.stage ?? 'load') ?? 0);
17
+ if (depth !== 0)
18
+ return depth;
19
+ const score = (b.score ?? Number.NEGATIVE_INFINITY) - (a.score ?? Number.NEGATIVE_INFINITY);
20
+ if (score !== 0 && !Number.isNaN(score))
21
+ return score;
22
+ return a.id < b.id ? -1 : a.id > b.id ? 1 : 0;
23
+ }
24
+ /** A recorder for one call; every observer method is guarded, and a throw marks it broken instead of escaping. */
25
+ export function createDeliveryRecorder(init) {
26
+ const startedMs = Date.now();
27
+ const ts = evalNow().toISOString();
28
+ // Test-only fault injection, as HIPPO_FAKE_NOW is for time.
29
+ const fault = process.env.HIPPO_TEST_DELIVERY_FAULT ?? '';
30
+ const payloadSession = hookPayloadSessionId(init.stdinText);
31
+ const subagent = isSubagentPayload(init.stdinText);
32
+ const envSession = init.envSessionId !== undefined && init.envSessionId !== '' ? init.envSessionId : null;
33
+ const prompt = hookPayloadString(init.stdinText, 'prompt');
34
+ const rawTurnId = hookPayloadString(init.stdinText, 'turn_id');
35
+ const hostTurnId = rawTurnId !== null && rawTurnId.trim() !== '' ? rawTurnId : null;
36
+ const hookEvent = hookPayloadString(init.stdinText, 'hook_event_name');
37
+ const sessionState = subagent
38
+ ? 'subagent'
39
+ : payloadSession !== null ? 'payload' : envSession !== null ? 'env' : 'missing';
40
+ const candidates = new Map();
41
+ const picked = new Map();
42
+ const filtered = new Set();
43
+ let facts = null;
44
+ let shown = 0;
45
+ let dropped = 0;
46
+ let disabledSeen = false;
47
+ let outcome = { state: 'empty' };
48
+ let broken = null;
49
+ let flushed = false;
50
+ const guard = (fn) => {
51
+ if (broken !== null)
52
+ return;
53
+ try {
54
+ if (fault === 'observe')
55
+ throw new Error('injected observe fault');
56
+ fn();
57
+ }
58
+ catch (error) {
59
+ broken = error instanceof Error ? error.message : String(error);
60
+ }
61
+ };
62
+ const rejectId = (id, stage, reason, score, tokens) => {
63
+ const held = candidates.get(id);
64
+ // First rejection wins; an id never offered is not a candidate of this call.
65
+ if (!held || held.reason !== null)
66
+ return;
67
+ candidates.set(id, { ...held, stage, reason, score, tokens });
68
+ };
69
+ const build = () => {
70
+ if (fault === 'build')
71
+ throw new Error('injected build fault');
72
+ const staticReused = outcome.staticReused === true;
73
+ const rows = [];
74
+ for (const p of picked.values()) {
75
+ const held = candidates.get(p.entry.id);
76
+ rows.push({
77
+ memoryId: p.entry.id,
78
+ sourceStore: p.sourceStore,
79
+ pool: p.promptRecall ? 'prompt-recall' : held?.pool === 'pin' || p.entry.pinned ? 'pin' : 'recent',
80
+ stage: 'final',
81
+ outcome: staticReused && !p.promptRecall ? 'reused' : 'emitted',
82
+ reason: null,
83
+ rank: p.rank,
84
+ score: p.score,
85
+ tokens: p.tokens,
86
+ });
87
+ }
88
+ const rejected = [];
89
+ let undecided = 0;
90
+ for (const c of candidates.values()) {
91
+ if (picked.has(c.id))
92
+ continue;
93
+ if (c.reason === null)
94
+ undecided += 1;
95
+ else
96
+ rejected.push(c);
97
+ }
98
+ rejected.sort(byDepthThenScore);
99
+ for (const c of rejected.slice(0, DELIVERY_REJECTED_ROW_CAP)) {
100
+ rows.push({
101
+ memoryId: c.id, sourceStore: c.sourceStore, pool: c.pool, stage: c.stage ?? 'load', outcome: 'rejected',
102
+ reason: c.reason, rank: null, score: c.score, tokens: c.tokens,
103
+ });
104
+ }
105
+ const overflow = Math.max(0, rejected.length - DELIVERY_REJECTED_ROW_CAP);
106
+ const emitted = outcome.emittedText ?? null;
107
+ return {
108
+ ts,
109
+ tenantId: init.tenantId,
110
+ runtime: hostTurnId !== null ? 'codex' : hookEvent !== null ? 'claude-code' : 'unknown',
111
+ eventType: hookEvent === 'UserPromptSubmit' ? 'prompt-submit' : 'pinned-manual',
112
+ surface: 'hook',
113
+ storeHash: init.storeHash,
114
+ writeStore: init.writeStore,
115
+ projectHash: facts !== null && facts.projectName !== '' ? blockHash(facts.projectName) : null,
116
+ sessionId: payloadSession ?? envSession,
117
+ sessionState,
118
+ hostTurnId,
119
+ promptHash: prompt !== null ? blockHash(prompt) : null,
120
+ promptLength: prompt?.length ?? 0,
121
+ queryHash: null,
122
+ recallTraceId: null,
123
+ blockState: disabledSeen ? 'disabled' : outcome.state,
124
+ promptRecall: facts?.promptRecall === true,
125
+ consideredCount: new Set([...candidates.keys(), ...picked.keys()]).size,
126
+ filteredCount: filtered.size,
127
+ selectedCount: picked.size,
128
+ emittedCount: rows.filter((r) => r.outcome === 'emitted').length,
129
+ rejectedCount: rejected.length + undecided,
130
+ rejectedUnlisted: overflow + undecided,
131
+ sectionsShown: shown,
132
+ sectionsDropped: dropped,
133
+ budgetTokens: facts?.budgetTokens ?? 0,
134
+ selectedTokens: [...picked.values()].reduce((sum, p) => sum + p.tokens, 0),
135
+ injectedTokens: emitted !== null ? estimateTokens(emitted) : 0,
136
+ staticHash: outcome.staticHash ?? null,
137
+ recallHash: outcome.recallHash ?? null,
138
+ emittedHash: emitted !== null ? blockHash(emitted) : null,
139
+ elapsedMs: Math.max(0, Date.now() - startedMs),
140
+ candidates: rows,
141
+ };
142
+ };
143
+ return {
144
+ root: init.root,
145
+ facts: (f) => guard(() => { facts = { ...f }; }),
146
+ sections: (s, d) => guard(() => { shown = s; dropped = d; }),
147
+ watchAdmit: (admit) => (e) => {
148
+ const ok = admit(e);
149
+ if (!ok)
150
+ guard(() => { filtered.add(e.id); });
151
+ return ok;
152
+ },
153
+ qualityDropped: (e, isGlobal) => guard(() => {
154
+ filtered.add(e.id);
155
+ if (!candidates.has(e.id)) {
156
+ candidates.set(e.id, {
157
+ id: e.id, sourceStore: storeOf(isGlobal), pool: 'recent', stage: null, reason: null, score: null, tokens: null,
158
+ });
159
+ }
160
+ rejectId(e.id, 'load', 'quality', null, null);
161
+ }),
162
+ disabled: () => guard(() => { disabledSeen = true; }),
163
+ offer: (entries, isGlobal, pool) => guard(() => {
164
+ for (const e of entries) {
165
+ const held = candidates.get(e.id);
166
+ const wanted = pool ?? (e.pinned ? 'pin' : 'recent');
167
+ // A loaded recent row the prompt-recall gate then judges belongs to that pool.
168
+ const relabel = held !== undefined && held.reason === null && held.pool === 'recent' && wanted === 'prompt-recall';
169
+ if (held !== undefined && !relabel)
170
+ continue;
171
+ candidates.set(e.id, {
172
+ id: e.id, sourceStore: storeOf(isGlobal), pool: wanted, stage: null, reason: null, score: null, tokens: null,
173
+ });
174
+ }
175
+ }),
176
+ reject: (e, stage, reason, score, tokens) => guard(() => rejectId(e.id, stage, reason, score ?? null, tokens ?? null)),
177
+ dropMissing: (before, after, stage, reason) => guard(() => {
178
+ const kept = new Set(after.map((e) => e.id));
179
+ for (const e of before)
180
+ if (!kept.has(e.id))
181
+ rejectId(e.id, stage, reason, null, null);
182
+ }),
183
+ gated: (prompt, items, gate, kept) => guard(() => {
184
+ const keptIds = new Set(kept.map((g) => g.item.id));
185
+ for (const c of items) {
186
+ if (keptIds.has(c.id))
187
+ continue;
188
+ const { score, shared } = scoreOverlap(prompt, c.tokens, gate.metric);
189
+ const cleared = score >= gate.threshold && shared >= gate.minShared;
190
+ rejectId(c.id, 'gate', cleared ? 'gate-max-items' : 'gate-below-threshold', score, null);
191
+ }
192
+ }),
193
+ selected: (items) => guard(() => {
194
+ picked.clear();
195
+ items.forEach((r, i) => {
196
+ picked.set(r.entry.id, {
197
+ entry: r.entry, rank: i + 1, score: r.score, tokens: r.tokens,
198
+ sourceStore: storeOf(r.isGlobal), promptRecall: r.promptRecall === true,
199
+ });
200
+ });
201
+ }),
202
+ delivered: (o) => guard(() => { outcome = { ...o }; }),
203
+ flush: (write) => {
204
+ if (flushed)
205
+ return;
206
+ flushed = true;
207
+ if (broken !== null) {
208
+ console.error(`[hippo] delivery ledger skipped: recorder failed: ${broken}`);
209
+ return;
210
+ }
211
+ const input = build();
212
+ if (fault === 'flush')
213
+ throw new Error('injected flush fault');
214
+ write(input);
215
+ },
216
+ };
217
+ }
218
+ //# sourceMappingURL=delivery-recorder.js.map
@@ -120,4 +120,62 @@ export declare function passAtK(runsByTask: boolean[][], k: number): number;
120
120
  * NaN when no task has k runs.
121
121
  */
122
122
  export declare function passHatK(runsByTask: boolean[][], k: number): number;
123
+ /** All units of one family (tasks by seeds), resampled as one block so seeds stay together. */
124
+ export type Family<T> = readonly T[];
125
+ /** One repository's families; a task with no lesson is a family of one. */
126
+ export type Repository<T> = readonly Family<T>[];
127
+ /** An {@link Estimate} with a two-sided p-value from the same resamples. */
128
+ export interface TestedEstimate extends Estimate {
129
+ /** 2 x min(share <= null, share >= null), capped at 1; NaN with fewer than two repositories. */
130
+ readonly p: number;
131
+ /** Non-finite resamples, dropped; `iterations + dropped` is the number requested. */
132
+ readonly dropped: number;
133
+ /** The null the p-value tested against; {@link verdict} reads its sides from here. */
134
+ readonly nullValue: number;
135
+ }
136
+ /** Options for {@link twoLevelBootstrap}. */
137
+ export interface TwoLevelOpts extends BootstrapOpts {
138
+ /** Resamples. Default 10,000. */
139
+ readonly iterations?: number;
140
+ /** Value the p-value tests against: 0 for differences, 1 for ratios. Default 0. */
141
+ readonly nullValue?: number;
142
+ }
143
+ /** Resamples repositories, then families inside each; only the repository draw carries a shared-store fault.
144
+ * The statistic never uses the PRNG, so a second call on one seed draws the same resamples. */
145
+ export declare function twoLevelBootstrap<T>(repos: readonly Repository<T>[], statistic: (units: readonly T[]) => number, opts?: TwoLevelOpts): TestedEstimate;
146
+ /** Holm step-down in input order; the family size is `ps.length`, as preregistered.
147
+ * A NaN stays NaN but ranks as 1, so it never loosens the others; a finite p outside [0, 1] throws. */
148
+ export declare function holmAdjust(ps: readonly number[]): number[];
149
+ /** One of the four mutually exclusive outcomes the preregistration allows per hypothesis. */
150
+ export type Verdict = 'loss' | 'win' | 'tie' | 'inconclusive';
151
+ /** What a hypothesis needs to be read; see {@link verdict}. */
152
+ export interface VerdictSpec {
153
+ /** Direction that favours the treatment arm. */
154
+ readonly helpful: 'lower' | 'higher';
155
+ /** Inclusive band the interval must sit inside for a tie. */
156
+ readonly tieBand: readonly [number, number];
157
+ /** An estimate on this value or beyond it, on the helpful side, reaches the minimum effect. */
158
+ readonly minimumEffectAt: number;
159
+ /** Default 0.05. */
160
+ readonly alpha?: number;
161
+ }
162
+ /** A verdict, plus whether a win reaches the minimum effect (a win below it is a small win). */
163
+ export interface VerdictResult {
164
+ readonly verdict: Verdict;
165
+ readonly reachesMinimum: boolean;
166
+ }
167
+ /** Checks in the preregistered order; the null comes from `e.nullValue`, so p and sides cannot disagree.
168
+ * A NaN estimate, p or bound is inconclusive, since a win or loss cannot be ruled out. */
169
+ export declare function verdict(e: TestedEstimate, adjustedP: number, spec: VerdictSpec): VerdictResult;
170
+ /** A win must hold under both codings of not-applicable, while a loss under either is reported. */
171
+ export declare function combineCodings(a: VerdictResult, b: VerdictResult): VerdictResult;
172
+ /** Outcome of the harm gate; see {@link harmGate}. */
173
+ export interface HarmGate {
174
+ readonly pass: boolean;
175
+ readonly costOk: boolean;
176
+ readonly resolveOk: boolean;
177
+ }
178
+ /** Cost ratio's upper bound below 1.10, resolve difference's lower bound (a fraction) above -0.05.
179
+ * Both estimates must use alpha 0.05, the conservative reading of "upper 95% bound"; a NaN bound fails. */
180
+ export declare function harmGate(costRatio: Estimate, resolveDiff: Estimate): HarmGate;
123
181
  //# sourceMappingURL=eval-stats.d.ts.map
@@ -184,4 +184,115 @@ export function passHatK(runsByTask, k) {
184
184
  return Number.NaN;
185
185
  return eligible.filter((r) => r.slice(0, k).every(Boolean)).length / eligible.length;
186
186
  }
187
+ function notANumber(nullValue, dropped) {
188
+ const nan = Number.NaN;
189
+ return { estimate: nan, low: nan, high: nan, p: nan, iterations: 0, dropped, nullValue };
190
+ }
191
+ function resampleUnits(repos, rand) {
192
+ const units = [];
193
+ for (let i = 0; i < repos.length; i++) {
194
+ const repo = repos[Math.floor(rand() * repos.length)];
195
+ for (let j = 0; j < repo.length; j++) {
196
+ for (const unit of repo[Math.floor(rand() * repo.length)])
197
+ units.push(unit);
198
+ }
199
+ }
200
+ return units;
201
+ }
202
+ // Inclusive on both sides, as the p-value is worded; float noise around the null breaks a tie.
203
+ function twoSidedP(samples, nullValue) {
204
+ let below = 0;
205
+ let above = 0;
206
+ for (const s of samples) {
207
+ if (s <= nullValue)
208
+ below++;
209
+ if (s >= nullValue)
210
+ above++;
211
+ }
212
+ return Math.min(1, (2 * Math.min(below, above)) / samples.length);
213
+ }
214
+ /** Resamples repositories, then families inside each; only the repository draw carries a shared-store fault.
215
+ * The statistic never uses the PRNG, so a second call on one seed draws the same resamples. */
216
+ export function twoLevelBootstrap(repos, statistic, opts = {}) {
217
+ const requested = opts.iterations ?? 10_000;
218
+ if (!Number.isInteger(requested) || requested <= 0) {
219
+ throw new RangeError(`iterations must be a positive integer, got ${requested}`);
220
+ }
221
+ const nullValue = opts.nullValue ?? 0;
222
+ const kept = repos.map((r) => r.filter((f) => f.length > 0)).filter((r) => r.length > 0);
223
+ if (kept.length === 0)
224
+ return notANumber(nullValue, 0);
225
+ const rand = seededRandom(opts.seed ?? 1);
226
+ const samples = [];
227
+ for (let b = 0; b < requested; b++) {
228
+ const value = statistic(resampleUnits(kept, rand));
229
+ if (Number.isFinite(value))
230
+ samples.push(value);
231
+ }
232
+ const dropped = requested - samples.length;
233
+ if (samples.length === 0)
234
+ return notANumber(nullValue, dropped);
235
+ const estimate = statistic(kept.flatMap((r) => r.flat()));
236
+ const p = kept.length < 2 ? Number.NaN : twoSidedP(samples, nullValue);
237
+ const interval = percentileInterval(samples, opts.alpha ?? 0.05);
238
+ return { estimate, ...interval, p, iterations: samples.length, dropped, nullValue };
239
+ }
240
+ /** Holm step-down in input order; the family size is `ps.length`, as preregistered.
241
+ * A NaN stays NaN but ranks as 1, so it never loosens the others; a finite p outside [0, 1] throws. */
242
+ export function holmAdjust(ps) {
243
+ for (const p of ps) {
244
+ if (!Number.isNaN(p) && !(p >= 0 && p <= 1))
245
+ throw new RangeError(`p-value out of range: ${p}`);
246
+ }
247
+ const ranked = ps
248
+ .map((p, index) => ({ index, key: Number.isNaN(p) ? 1 : p }))
249
+ .sort((a, b) => a.key - b.key || a.index - b.index);
250
+ const adjusted = Array.from({ length: ps.length }, () => Number.NaN);
251
+ let running = 0;
252
+ ranked.forEach(({ index, key }, rank) => {
253
+ running = Math.max(running, (ps.length - rank) * key);
254
+ if (!Number.isNaN(ps[index]))
255
+ adjusted[index] = Math.min(1, running);
256
+ });
257
+ return adjusted;
258
+ }
259
+ /** Checks in the preregistered order; the null comes from `e.nullValue`, so p and sides cannot disagree.
260
+ * A NaN estimate, p or bound is inconclusive, since a win or loss cannot be ruled out. */
261
+ export function verdict(e, adjustedP, spec) {
262
+ const inconclusive = { verdict: 'inconclusive', reachesMinimum: false };
263
+ if ([adjustedP, e.estimate, e.low, e.high].some(Number.isNaN))
264
+ return inconclusive;
265
+ const nullValue = e.nullValue;
266
+ const lowerIsHelpful = spec.helpful === 'lower';
267
+ if (adjustedP < (spec.alpha ?? 0.05)) {
268
+ const helpfulSide = lowerIsHelpful ? e.estimate < nullValue : e.estimate > nullValue;
269
+ const harmfulSide = lowerIsHelpful ? e.estimate > nullValue : e.estimate < nullValue;
270
+ if (harmfulSide)
271
+ return { verdict: 'loss', reachesMinimum: false };
272
+ if (helpfulSide) {
273
+ const reaches = lowerIsHelpful ? e.estimate <= spec.minimumEffectAt : e.estimate >= spec.minimumEffectAt;
274
+ return { verdict: 'win', reachesMinimum: reaches };
275
+ }
276
+ }
277
+ const insideBand = e.low >= spec.tieBand[0] && e.high <= spec.tieBand[1];
278
+ return insideBand ? { verdict: 'tie', reachesMinimum: false } : inconclusive;
279
+ }
280
+ /** A win must hold under both codings of not-applicable, while a loss under either is reported. */
281
+ export function combineCodings(a, b) {
282
+ if (a.verdict === 'loss' || b.verdict === 'loss')
283
+ return { verdict: 'loss', reachesMinimum: false };
284
+ if (a.verdict === 'win' && b.verdict === 'win') {
285
+ return { verdict: 'win', reachesMinimum: a.reachesMinimum && b.reachesMinimum };
286
+ }
287
+ if (a.verdict === 'tie' && b.verdict === 'tie')
288
+ return { verdict: 'tie', reachesMinimum: false };
289
+ return { verdict: 'inconclusive', reachesMinimum: false };
290
+ }
291
+ /** Cost ratio's upper bound below 1.10, resolve difference's lower bound (a fraction) above -0.05.
292
+ * Both estimates must use alpha 0.05, the conservative reading of "upper 95% bound"; a NaN bound fails. */
293
+ export function harmGate(costRatio, resolveDiff) {
294
+ const costOk = costRatio.high < 1.1;
295
+ const resolveOk = resolveDiff.low > -0.05;
296
+ return { pass: costOk && resolveOk, costOk, resolveOk };
297
+ }
187
298
  //# sourceMappingURL=eval-stats.js.map