hippo-memory 1.45.0 → 1.47.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +58 -15
  2. package/bin/hippo.js +0 -0
  3. package/dist/ablation.d.ts +10 -1
  4. package/dist/ablation.js +17 -1
  5. package/dist/api.d.ts +66 -1
  6. package/dist/api.js +202 -7
  7. package/dist/audit.d.ts +1 -1
  8. package/dist/capture-error.d.ts +26 -0
  9. package/dist/capture-error.js +110 -0
  10. package/dist/capture.d.ts +25 -8
  11. package/dist/capture.js +100 -5
  12. package/dist/cli.d.ts +6 -1
  13. package/dist/cli.js +432 -48
  14. package/dist/config.d.ts +20 -0
  15. package/dist/config.js +35 -0
  16. package/dist/consolidate.d.ts +6 -0
  17. package/dist/consolidate.js +98 -13
  18. package/dist/db.js +81 -1
  19. package/dist/doctor.d.ts +34 -0
  20. package/dist/doctor.js +183 -0
  21. package/dist/dormant.d.ts +91 -0
  22. package/dist/dormant.js +121 -0
  23. package/dist/eval-stats.d.ts +123 -0
  24. package/dist/eval-stats.js +187 -0
  25. package/dist/failure-log.d.ts +49 -0
  26. package/dist/failure-log.js +58 -0
  27. package/dist/half-life-migration.d.ts +55 -0
  28. package/dist/half-life-migration.js +111 -0
  29. package/dist/hooks.d.ts +4 -0
  30. package/dist/hooks.js +47 -0
  31. package/dist/mcp/server.d.ts +6 -0
  32. package/dist/mcp/server.js +70 -13
  33. package/dist/memory.d.ts +16 -2
  34. package/dist/memory.js +27 -5
  35. package/dist/physics-config.js +5 -1
  36. package/dist/recall-scope.d.ts +24 -0
  37. package/dist/recall-scope.js +41 -0
  38. package/dist/reject-flow.d.ts +3 -3
  39. package/dist/reject-flow.js +10 -3
  40. package/dist/search.d.ts +4 -4
  41. package/dist/search.js +23 -18
  42. package/dist/server.js +11 -1
  43. package/dist/store.d.ts +12 -1
  44. package/dist/store.js +58 -12
  45. package/dist/token-ledger.d.ts +119 -0
  46. package/dist/token-ledger.js +181 -0
  47. package/dist/version.d.ts +1 -1
  48. package/dist/version.js +1 -1
  49. package/extensions/openclaw-plugin/openclaw.plugin.json +1 -1
  50. package/extensions/openclaw-plugin/package.json +1 -1
  51. package/openclaw.plugin.json +1 -1
  52. package/package.json +2 -1
@@ -0,0 +1,91 @@
1
+ /**
2
+ * Dormant memories: what sleep does with a faded memory instead of deleting
3
+ * it (config `dormant.enabled`, on by default; `retentionDays` bounds how
4
+ * long one is kept).
5
+ *
6
+ * A dormant memory keeps its full content in `dormant_memories` (schema v44)
7
+ * but is no longer a `memories` row, so recall, context, every sleep pass and
8
+ * every other reader of `memories` stop seeing it exactly as if it had been
9
+ * deleted. `hippo dormant` lists and searches them, `hippo dormant restore`
10
+ * brings one back, `hippo dormant forget` deletes one for good.
11
+ *
12
+ * Not to be confused with the raw archive (`raw_archive`, `archiveRawMemory`,
13
+ * `hippo forget --archive`): that path removes a raw receipt's content and
14
+ * keeps only its metadata. A dormant memory keeps its content.
15
+ *
16
+ * DB-only helpers: the caller owns the handle and any transaction. The
17
+ * tenant-scoped entry points are api.listDormant / restoreDormant /
18
+ * forgetDormant.
19
+ */
20
+ import type { DatabaseSyncLike } from './db.js';
21
+ import type { MemoryEntry } from './memory.js';
22
+ /** Why sleep made a memory dormant. Only the decay pass does today. */
23
+ export type DormantReason = 'decay';
24
+ /** One memory that sleep is moving out of active memory into the dormant store. */
25
+ export interface DormantMove {
26
+ /** The memory as it stood when it faded; restored verbatim apart from its recall clock. */
27
+ entry: MemoryEntry;
28
+ /** Live strength at the moment it went dormant (below the decay threshold). */
29
+ strength: number;
30
+ reason: DormantReason;
31
+ /** ISO time of the sleep that made it dormant. */
32
+ dormantAt: string;
33
+ }
34
+ /** A dormant memory as listed to a user. */
35
+ export interface DormantMemory {
36
+ id: string;
37
+ tenantId: string;
38
+ content: string;
39
+ tags: string[];
40
+ /** Live strength when it went dormant. */
41
+ strength: number;
42
+ /** Why it went dormant (`decay`). */
43
+ reason: string;
44
+ /** ISO time it went dormant. */
45
+ dormantAt: string;
46
+ }
47
+ /** Options for {@link listDormantRows}. */
48
+ export interface ListDormantOpts {
49
+ /** Whitespace-separated terms; a row must contain every term (case-insensitive substring). */
50
+ query?: string;
51
+ /** Maximum rows returned, newest first. Default 20. */
52
+ limit?: number;
53
+ }
54
+ /**
55
+ * Insert (or refresh) the dormant snapshot for `move.entry`. Does not touch
56
+ * the `memories` row: the caller deletes it in the same transaction, so the
57
+ * memory is never in both places or in neither.
58
+ */
59
+ export declare function insertDormantRow(db: DatabaseSyncLike, move: DormantMove): void;
60
+ /** A tenant's dormant memories, newest first, optionally filtered by search terms. */
61
+ export declare function listDormantRows(db: DatabaseSyncLike, tenantId: string, opts?: ListDormantOpts): DormantMemory[];
62
+ /** A dormant memory's stored snapshot plus when and why it went dormant. */
63
+ export interface DormantSnapshot {
64
+ entry: MemoryEntry;
65
+ reason: string;
66
+ strength: number;
67
+ dormantAt: string;
68
+ }
69
+ /**
70
+ * The stored snapshot for a tenant's dormant memory, or null when the tenant
71
+ * has no dormant memory with that id (another tenant's id reads as absent).
72
+ */
73
+ export declare function readDormantSnapshot(db: DatabaseSyncLike, tenantId: string, id: string): DormantSnapshot | null;
74
+ /** Whether a tenant has a dormant memory with this id (snapshot readable or not). */
75
+ export declare function hasDormantRow(db: DatabaseSyncLike, tenantId: string, id: string): boolean;
76
+ /** Delete a tenant's dormant memory. Returns false when there was none. */
77
+ export declare function deleteDormantRow(db: DatabaseSyncLike, tenantId: string, id: string): boolean;
78
+ /**
79
+ * Delete every dormant memory in the tenant whose content has this rejection
80
+ * digest, so a rejected value cannot linger in dormant storage. Returns the
81
+ * ids removed. Same O(N) scan as the live-row sweep in reject-flow.ts.
82
+ */
83
+ export declare function purgeDormantByDigest(db: DatabaseSyncLike, tenantId: string, digest: string): string[];
84
+ /** How many dormant memories went dormant before `cutoffIso` (a dry-run count). */
85
+ export declare function countExpiredDormant(db: DatabaseSyncLike, cutoffIso: string): number;
86
+ /**
87
+ * Delete every dormant memory (all tenants) that went dormant before
88
+ * `cutoffIso`: the `dormant.retentionDays` window. Returns how many went.
89
+ */
90
+ export declare function purgeExpiredDormant(db: DatabaseSyncLike, cutoffIso: string): number;
91
+ //# sourceMappingURL=dormant.d.ts.map
@@ -0,0 +1,121 @@
1
+ import { rejectionDigest } from './rejection.js';
2
+ const DEFAULT_LIST_LIMIT = 20;
3
+ /**
4
+ * Insert (or refresh) the dormant snapshot for `move.entry`. Does not touch
5
+ * the `memories` row: the caller deletes it in the same transaction, so the
6
+ * memory is never in both places or in neither.
7
+ */
8
+ export function insertDormantRow(db, move) {
9
+ db.prepare(`
10
+ INSERT INTO dormant_memories (tenant_id, id, content, entry_json, reason, strength, dormant_at)
11
+ VALUES (?, ?, ?, ?, ?, ?, ?)
12
+ ON CONFLICT(tenant_id, id) DO UPDATE SET
13
+ content = excluded.content,
14
+ entry_json = excluded.entry_json,
15
+ reason = excluded.reason,
16
+ strength = excluded.strength,
17
+ dormant_at = excluded.dormant_at
18
+ `).run(move.entry.tenantId, move.entry.id, move.entry.content, JSON.stringify(move.entry), move.reason, move.strength, move.dormantAt);
19
+ }
20
+ /** Escape LIKE metacharacters so a search term matches literally (ESCAPE '\'). */
21
+ function escapeLike(term) {
22
+ return term.replace(/[\\%_]/g, (ch) => `\\${ch}`);
23
+ }
24
+ /**
25
+ * Parse a stored snapshot back into a MemoryEntry, or null when the row no
26
+ * longer matches the snapshot it carries (edited by hand, or truncated).
27
+ */
28
+ function parseSnapshot(row) {
29
+ try {
30
+ // SAFETY: entry_json is only written by insertDormantRow from a MemoryEntry;
31
+ // the id / tenant / content checks below reject a row edited out of shape.
32
+ const entry = JSON.parse(row.entry_json);
33
+ if (entry.id !== row.id || entry.tenantId !== row.tenant_id || entry.content !== row.content) {
34
+ return null;
35
+ }
36
+ return entry;
37
+ }
38
+ catch {
39
+ return null;
40
+ }
41
+ }
42
+ function rowToDormantMemory(row) {
43
+ const entry = parseSnapshot(row);
44
+ return {
45
+ id: row.id,
46
+ tenantId: row.tenant_id,
47
+ content: row.content,
48
+ tags: entry && Array.isArray(entry.tags) ? entry.tags.map(String) : [],
49
+ strength: row.strength,
50
+ reason: row.reason,
51
+ dormantAt: row.dormant_at,
52
+ };
53
+ }
54
+ /** A tenant's dormant memories, newest first, optionally filtered by search terms. */
55
+ export function listDormantRows(db, tenantId, opts = {}) {
56
+ const terms = (opts.query ?? '').trim().split(/\s+/).filter((t) => t.length > 0);
57
+ const limit = opts.limit !== undefined && Number.isFinite(opts.limit) && opts.limit >= 1
58
+ ? Math.floor(opts.limit)
59
+ : DEFAULT_LIST_LIMIT;
60
+ const termClauses = terms.map(() => ` AND content LIKE ? ESCAPE '\\'`).join('');
61
+ // SAFETY: rows' shape matches the seven columns named in the SELECT.
62
+ const rows = db.prepare(`SELECT tenant_id, id, content, entry_json, reason, strength, dormant_at
63
+ FROM dormant_memories
64
+ WHERE tenant_id = ?${termClauses}
65
+ ORDER BY dormant_at DESC, id ASC
66
+ LIMIT ?`).all(tenantId, ...terms.map((t) => `%${escapeLike(t)}%`), limit);
67
+ return rows.map(rowToDormantMemory);
68
+ }
69
+ /**
70
+ * The stored snapshot for a tenant's dormant memory, or null when the tenant
71
+ * has no dormant memory with that id (another tenant's id reads as absent).
72
+ */
73
+ export function readDormantSnapshot(db, tenantId, id) {
74
+ // SAFETY: row's shape matches the seven columns named in the SELECT.
75
+ const row = db.prepare(`SELECT tenant_id, id, content, entry_json, reason, strength, dormant_at
76
+ FROM dormant_memories WHERE tenant_id = ? AND id = ?`).get(tenantId, id);
77
+ const entry = row ? parseSnapshot(row) : null;
78
+ return row && entry ? { entry, reason: row.reason, strength: row.strength, dormantAt: row.dormant_at } : null;
79
+ }
80
+ /** Whether a tenant has a dormant memory with this id (snapshot readable or not). */
81
+ export function hasDormantRow(db, tenantId, id) {
82
+ return db.prepare(`SELECT 1 FROM dormant_memories WHERE tenant_id = ? AND id = ?`).get(tenantId, id) !== undefined;
83
+ }
84
+ /** Delete a tenant's dormant memory. Returns false when there was none. */
85
+ export function deleteDormantRow(db, tenantId, id) {
86
+ const result = db.prepare(`DELETE FROM dormant_memories WHERE tenant_id = ? AND id = ?`).run(tenantId, id);
87
+ return Number(result.changes ?? 0) > 0;
88
+ }
89
+ /**
90
+ * Delete every dormant memory in the tenant whose content has this rejection
91
+ * digest, so a rejected value cannot linger in dormant storage. Returns the
92
+ * ids removed. Same O(N) scan as the live-row sweep in reject-flow.ts.
93
+ */
94
+ export function purgeDormantByDigest(db, tenantId, digest) {
95
+ // SAFETY: rows' shape matches the two columns named in the SELECT.
96
+ const rows = db.prepare(`SELECT id, content FROM dormant_memories WHERE tenant_id = ?`)
97
+ .all(tenantId);
98
+ const removed = [];
99
+ for (const row of rows) {
100
+ if (rejectionDigest(row.content) !== digest)
101
+ continue;
102
+ deleteDormantRow(db, tenantId, row.id);
103
+ removed.push(row.id);
104
+ }
105
+ return removed;
106
+ }
107
+ /** How many dormant memories went dormant before `cutoffIso` (a dry-run count). */
108
+ export function countExpiredDormant(db, cutoffIso) {
109
+ // SAFETY: row's shape matches the single aliased COUNT column in the SELECT.
110
+ const row = db.prepare(`SELECT COUNT(*) AS n FROM dormant_memories WHERE dormant_at < ?`).get(cutoffIso);
111
+ return Number(row.n);
112
+ }
113
+ /**
114
+ * Delete every dormant memory (all tenants) that went dormant before
115
+ * `cutoffIso`: the `dormant.retentionDays` window. Returns how many went.
116
+ */
117
+ export function purgeExpiredDormant(db, cutoffIso) {
118
+ const result = db.prepare(`DELETE FROM dormant_memories WHERE dormant_at < ?`).run(cutoffIso);
119
+ return Number(result.changes ?? 0);
120
+ }
121
+ //# sourceMappingURL=dormant.js.map
@@ -0,0 +1,123 @@
1
+ /**
2
+ * Statistics and cost accounting for the token-efficiency evals (ROADMAP
3
+ * Part IX, TE3-TE5).
4
+ *
5
+ * - Cost: price provider usage over four buckets (uncached input, cache
6
+ * write, cache read, output). Raw token counts overstate savings when most
7
+ * input is already cache reads, so every dollar claim goes through here.
8
+ * - Uncertainty: paired bootstrap over tasks, cluster bootstrap (tasks that
9
+ * share a repository are not independent), and a paired ratio bootstrap for
10
+ * dollars per resolved task.
11
+ *
12
+ * Deterministic: every resampling function takes a seed, so a published
13
+ * result can be reproduced exactly.
14
+ */
15
+ /** Token usage for one model call or one whole task, split by how it is billed. */
16
+ export interface Usage {
17
+ /** Input tokens billed at the base input price (not read from or written to a cache). */
18
+ inputTokens: number;
19
+ /** Input tokens written to a prompt cache. */
20
+ cacheWriteTokens: number;
21
+ /** Input tokens read from a prompt cache. */
22
+ cacheReadTokens: number;
23
+ /** Output tokens, including any reasoning tokens the provider bills as output. */
24
+ outputTokens: number;
25
+ }
26
+ /**
27
+ * Prices in dollars per million tokens. Take them from the provider's
28
+ * current price page for the exact model; this module has no built-in
29
+ * prices because they change.
30
+ */
31
+ export interface Prices {
32
+ inputPerMTok: number;
33
+ cacheWritePerMTok: number;
34
+ cacheReadPerMTok: number;
35
+ outputPerMTok: number;
36
+ }
37
+ /** Sum several usages bucket by bucket. */
38
+ export declare function addUsage(...usages: Usage[]): Usage;
39
+ /** Dollar cost of `usage` at `prices`. */
40
+ export declare function priceUsage(usage: Usage, prices: Prices): number;
41
+ /**
42
+ * Relative prices of the cache buckets against the base input price, for
43
+ * cost in "uncached-equivalent tokens" when no dollar prices are given.
44
+ * Defaults follow Anthropic's published ratios (5-minute cache write 1.25x,
45
+ * cache read 0.1x); pass the ratios for another provider when needed.
46
+ */
47
+ export interface CacheRatios {
48
+ write: number;
49
+ read: number;
50
+ }
51
+ /** Default cache price ratios (write 1.25x, read 0.1x of base input). */
52
+ export declare const DEFAULT_CACHE_RATIOS: Readonly<CacheRatios>;
53
+ /** Input cost of `usage` in uncached-equivalent tokens (output excluded). */
54
+ export declare function uncachedEquivalentInput(usage: Usage, ratios?: CacheRatios): number;
55
+ /**
56
+ * Mulberry32: a small seeded PRNG returning floats in [0, 1). Same seed,
57
+ * same stream, on every platform.
58
+ */
59
+ export declare function seededRandom(seed: number): () => number;
60
+ /** A point estimate with a percentile bootstrap confidence interval. */
61
+ export interface Estimate {
62
+ estimate: number;
63
+ low: number;
64
+ high: number;
65
+ /** Resamples drawn. */
66
+ iterations: number;
67
+ }
68
+ /** Options shared by the bootstrap functions. */
69
+ export interface BootstrapOpts {
70
+ /** Resamples. Default 5000. */
71
+ iterations?: number;
72
+ /** Two-sided level, e.g. 0.05 for a 95% interval. Default 0.05. */
73
+ alpha?: number;
74
+ /** PRNG seed. Default 1. */
75
+ seed?: number;
76
+ }
77
+ /**
78
+ * Paired bootstrap for the mean of per-task differences (treatment minus
79
+ * control on the same task). An interval that excludes zero is the bar for
80
+ * calling a difference real.
81
+ */
82
+ export declare function pairedBootstrap(diffs: number[], opts?: BootstrapOpts): Estimate;
83
+ /**
84
+ * Cluster bootstrap for the mean of per-task differences: resamples whole
85
+ * clusters (for example all tasks from one repository), because tasks in a
86
+ * cluster share causes and are not independent draws.
87
+ */
88
+ export declare function clusteredPairedBootstrap(diffsByCluster: ReadonlyMap<string, number[]>, opts?: BootstrapOpts): Estimate;
89
+ /** One task's outcome in one arm, for {@link costPerResolvedDelta}. */
90
+ export interface ArmOutcome {
91
+ /** Dollars (or uncached-equivalent tokens) spent on the task. */
92
+ cost: number;
93
+ /** Whether the task was resolved. */
94
+ resolved: boolean;
95
+ }
96
+ /** Result of {@link costPerResolvedDelta}. */
97
+ export interface CostPerResolvedDelta {
98
+ control: number;
99
+ treatment: number;
100
+ /** treatment minus control, with its interval. */
101
+ delta: Estimate;
102
+ /** (treatment minus control) / control, with its interval. Negative is a saving. */
103
+ relative: Estimate;
104
+ }
105
+ /**
106
+ * Cost per resolved task in two arms run on the same tasks, with a paired
107
+ * bootstrap over tasks (a task is resampled with both of its arm outcomes).
108
+ * `control[i]` and `treatment[i]` must be the same task. Resamples in which
109
+ * an arm resolves nothing are dropped; `iterations` reports how many were
110
+ * kept.
111
+ */
112
+ export declare function costPerResolvedDelta(control: ArmOutcome[], treatment: ArmOutcome[], opts?: BootstrapOpts): CostPerResolvedDelta;
113
+ /**
114
+ * pass@k: share of tasks with at least one success in their first k runs.
115
+ * NaN when no task has k runs (not measured, which is not the same as 0).
116
+ */
117
+ export declare function passAtK(runsByTask: boolean[][], k: number): number;
118
+ /**
119
+ * pass^k: share of tasks whose first k runs all succeed (consistency).
120
+ * NaN when no task has k runs.
121
+ */
122
+ export declare function passHatK(runsByTask: boolean[][], k: number): number;
123
+ //# sourceMappingURL=eval-stats.d.ts.map
@@ -0,0 +1,187 @@
1
+ /**
2
+ * Statistics and cost accounting for the token-efficiency evals (ROADMAP
3
+ * Part IX, TE3-TE5).
4
+ *
5
+ * - Cost: price provider usage over four buckets (uncached input, cache
6
+ * write, cache read, output). Raw token counts overstate savings when most
7
+ * input is already cache reads, so every dollar claim goes through here.
8
+ * - Uncertainty: paired bootstrap over tasks, cluster bootstrap (tasks that
9
+ * share a repository are not independent), and a paired ratio bootstrap for
10
+ * dollars per resolved task.
11
+ *
12
+ * Deterministic: every resampling function takes a seed, so a published
13
+ * result can be reproduced exactly.
14
+ */
15
+ /** Sum several usages bucket by bucket. */
16
+ export function addUsage(...usages) {
17
+ const total = { inputTokens: 0, cacheWriteTokens: 0, cacheReadTokens: 0, outputTokens: 0 };
18
+ for (const u of usages) {
19
+ total.inputTokens += u.inputTokens;
20
+ total.cacheWriteTokens += u.cacheWriteTokens;
21
+ total.cacheReadTokens += u.cacheReadTokens;
22
+ total.outputTokens += u.outputTokens;
23
+ }
24
+ return total;
25
+ }
26
+ /** Dollar cost of `usage` at `prices`. */
27
+ export function priceUsage(usage, prices) {
28
+ return (usage.inputTokens * prices.inputPerMTok
29
+ + usage.cacheWriteTokens * prices.cacheWritePerMTok
30
+ + usage.cacheReadTokens * prices.cacheReadPerMTok
31
+ + usage.outputTokens * prices.outputPerMTok) / 1_000_000;
32
+ }
33
+ /** Default cache price ratios (write 1.25x, read 0.1x of base input). */
34
+ export const DEFAULT_CACHE_RATIOS = { write: 1.25, read: 0.1 };
35
+ /** Input cost of `usage` in uncached-equivalent tokens (output excluded). */
36
+ export function uncachedEquivalentInput(usage, ratios = DEFAULT_CACHE_RATIOS) {
37
+ return usage.inputTokens + usage.cacheWriteTokens * ratios.write + usage.cacheReadTokens * ratios.read;
38
+ }
39
+ /**
40
+ * Mulberry32: a small seeded PRNG returning floats in [0, 1). Same seed,
41
+ * same stream, on every platform.
42
+ */
43
+ export function seededRandom(seed) {
44
+ let a = seed >>> 0;
45
+ return () => {
46
+ a = (a + 0x6d2b79f5) >>> 0;
47
+ let t = a;
48
+ t = Math.imul(t ^ (t >>> 15), t | 1);
49
+ t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
50
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
51
+ };
52
+ }
53
+ function percentileInterval(samples, alpha) {
54
+ const sorted = [...samples].sort((x, y) => x - y);
55
+ const lowIdx = Math.max(0, Math.floor((alpha / 2) * sorted.length));
56
+ const highIdx = Math.min(sorted.length - 1, Math.ceil((1 - alpha / 2) * sorted.length) - 1);
57
+ return { low: sorted[lowIdx], high: sorted[highIdx] };
58
+ }
59
+ function mean(xs) {
60
+ return xs.length === 0 ? 0 : xs.reduce((s, x) => s + x, 0) / xs.length;
61
+ }
62
+ /**
63
+ * Paired bootstrap for the mean of per-task differences (treatment minus
64
+ * control on the same task). An interval that excludes zero is the bar for
65
+ * calling a difference real.
66
+ */
67
+ export function pairedBootstrap(diffs, opts = {}) {
68
+ const iterations = opts.iterations ?? 5000;
69
+ const alpha = opts.alpha ?? 0.05;
70
+ if (diffs.length === 0)
71
+ return { estimate: 0, low: 0, high: 0, iterations: 0 };
72
+ const rand = seededRandom(opts.seed ?? 1);
73
+ const n = diffs.length;
74
+ const samples = [];
75
+ for (let b = 0; b < iterations; b++) {
76
+ let s = 0;
77
+ for (let i = 0; i < n; i++)
78
+ s += diffs[Math.floor(rand() * n)];
79
+ samples.push(s / n);
80
+ }
81
+ return { estimate: mean(diffs), ...percentileInterval(samples, alpha), iterations };
82
+ }
83
+ /**
84
+ * Cluster bootstrap for the mean of per-task differences: resamples whole
85
+ * clusters (for example all tasks from one repository), because tasks in a
86
+ * cluster share causes and are not independent draws.
87
+ */
88
+ export function clusteredPairedBootstrap(diffsByCluster, opts = {}) {
89
+ const iterations = opts.iterations ?? 5000;
90
+ const alpha = opts.alpha ?? 0.05;
91
+ const clusters = [...diffsByCluster.values()].filter((c) => c.length > 0);
92
+ const all = clusters.flat();
93
+ if (all.length === 0)
94
+ return { estimate: 0, low: 0, high: 0, iterations: 0 };
95
+ const rand = seededRandom(opts.seed ?? 1);
96
+ const k = clusters.length;
97
+ const samples = [];
98
+ for (let b = 0; b < iterations; b++) {
99
+ let sum = 0;
100
+ let count = 0;
101
+ for (let i = 0; i < k; i++) {
102
+ const c = clusters[Math.floor(rand() * k)];
103
+ for (const d of c)
104
+ sum += d;
105
+ count += c.length;
106
+ }
107
+ samples.push(count === 0 ? 0 : sum / count);
108
+ }
109
+ return { estimate: mean(all), ...percentileInterval(samples, alpha), iterations };
110
+ }
111
+ function costPerResolved(outcomes) {
112
+ const resolved = outcomes.filter((o) => o.resolved).length;
113
+ const cost = outcomes.reduce((s, o) => s + o.cost, 0);
114
+ return resolved === 0 ? Number.POSITIVE_INFINITY : cost / resolved;
115
+ }
116
+ /**
117
+ * Cost per resolved task in two arms run on the same tasks, with a paired
118
+ * bootstrap over tasks (a task is resampled with both of its arm outcomes).
119
+ * `control[i]` and `treatment[i]` must be the same task. Resamples in which
120
+ * an arm resolves nothing are dropped; `iterations` reports how many were
121
+ * kept.
122
+ */
123
+ export function costPerResolvedDelta(control, treatment, opts = {}) {
124
+ if (control.length !== treatment.length) {
125
+ throw new Error('control and treatment must list the same tasks in the same order');
126
+ }
127
+ const iterations = opts.iterations ?? 5000;
128
+ const alpha = opts.alpha ?? 0.05;
129
+ const c = costPerResolved(control);
130
+ const t = costPerResolved(treatment);
131
+ const rand = seededRandom(opts.seed ?? 1);
132
+ const n = control.length;
133
+ const deltas = [];
134
+ const relatives = [];
135
+ for (let b = 0; b < iterations && n > 0; b++) {
136
+ const cs = [];
137
+ const ts = [];
138
+ for (let i = 0; i < n; i++) {
139
+ const j = Math.floor(rand() * n);
140
+ cs.push(control[j]);
141
+ ts.push(treatment[j]);
142
+ }
143
+ const cc = costPerResolved(cs);
144
+ const tt = costPerResolved(ts);
145
+ if (!Number.isFinite(cc) || !Number.isFinite(tt))
146
+ continue;
147
+ deltas.push(tt - cc);
148
+ relatives.push((tt - cc) / cc);
149
+ }
150
+ const finite = Number.isFinite(c) && Number.isFinite(t);
151
+ const empty = { low: Number.NaN, high: Number.NaN };
152
+ return {
153
+ control: c,
154
+ treatment: t,
155
+ delta: {
156
+ estimate: finite ? t - c : Number.NaN,
157
+ ...(deltas.length > 0 ? percentileInterval(deltas, alpha) : empty),
158
+ iterations: deltas.length,
159
+ },
160
+ relative: {
161
+ estimate: finite ? (t - c) / c : Number.NaN,
162
+ ...(relatives.length > 0 ? percentileInterval(relatives, alpha) : empty),
163
+ iterations: relatives.length,
164
+ },
165
+ };
166
+ }
167
+ /**
168
+ * pass@k: share of tasks with at least one success in their first k runs.
169
+ * NaN when no task has k runs (not measured, which is not the same as 0).
170
+ */
171
+ export function passAtK(runsByTask, k) {
172
+ const eligible = runsByTask.filter((r) => r.length >= k);
173
+ if (eligible.length === 0)
174
+ return Number.NaN;
175
+ return eligible.filter((r) => r.slice(0, k).some(Boolean)).length / eligible.length;
176
+ }
177
+ /**
178
+ * pass^k: share of tasks whose first k runs all succeed (consistency).
179
+ * NaN when no task has k runs.
180
+ */
181
+ export function passHatK(runsByTask, k) {
182
+ const eligible = runsByTask.filter((r) => r.length >= k);
183
+ if (eligible.length === 0)
184
+ return Number.NaN;
185
+ return eligible.filter((r) => r.slice(0, k).every(Boolean)).length / eligible.length;
186
+ }
187
+ //# sourceMappingURL=eval-stats.js.map
@@ -0,0 +1,49 @@
1
+ /** Failure log (ROADMAP CD13): every failed tool call the capture-error hook sees, stored or not. */
2
+ import type { CaptureErrorOutcome, RoutineRule } from './capture-error.js';
3
+ import type { DatabaseSyncLike } from './db.js';
4
+ /** Rows older than this are pruned on write, which also bounds how far back a repeat can be found. */
5
+ export declare const FAILURE_LOG_RETENTION_DAYS = 90;
6
+ /** A capture-error outcome, or `store-failed` when storing the lesson threw. */
7
+ export type FailureOutcome = CaptureErrorOutcome | 'store-failed';
8
+ /** One failed tool call, for {@link recordFailure}. Never the failure text: it can carry paths and secrets. */
9
+ export interface FailureEvent {
10
+ tenantId: string;
11
+ /** Host session id from the hook payload; null when it had none. */
12
+ sessionId?: string | null;
13
+ tool?: string | null;
14
+ outcome: FailureOutcome;
15
+ /** The routine check that skipped it, for `skipped-routine`. */
16
+ rule?: RoutineRule | null;
17
+ /** Hash of the lesson text's signature, the key dedupe uses; null when the payload had no readable error. */
18
+ sigHash?: string | null;
19
+ /** Hash of the untruncated error plus the command's first two words, finer than `sigHash`. */
20
+ detailHash?: string | null;
21
+ /** Override the timestamp (tests). ISO string. */
22
+ now?: string;
23
+ }
24
+ /** Append one failure row and prune rows past {@link FAILURE_LOG_RETENTION_DAYS}. */
25
+ export declare function recordFailure(db: DatabaseSyncLike, event: FailureEvent): void;
26
+ /** Rated failures and repeats in one session, for {@link failuresBySession}. */
27
+ export interface SessionFailures {
28
+ sessionId: string;
29
+ /** Failures hippo treats as lessons (stored, duplicate or store-failed) in the window. */
30
+ failures: number;
31
+ /** Of those, failures whose signature another session hit first. */
32
+ repeats: number;
33
+ }
34
+ /** Rated failures per session since `sinceIso`, the input for repeat-error rate per arm (CD11, CD12). */
35
+ export declare function failuresBySession(db: DatabaseSyncLike, tenantId: string, sinceIso: string): SessionFailures[];
36
+ /** Failure log totals over a window, for {@link summarizeFailures}. Counts only: a rate needs a holdout arm (CD11). */
37
+ export interface FailureSummary {
38
+ /** ISO start of the window (inclusive). */
39
+ since: string;
40
+ outcomes: Record<FailureOutcome, number>;
41
+ total: number;
42
+ /** Rated failures from sessions with an id: the failures a repeat is counted among. */
43
+ rated: number;
44
+ repeats: number;
45
+ sessions: number;
46
+ }
47
+ /** Sum the failure log for one tenant since `sinceIso`. */
48
+ export declare function summarizeFailures(db: DatabaseSyncLike, tenantId: string, sinceIso: string): FailureSummary;
49
+ //# sourceMappingURL=failure-log.d.ts.map
@@ -0,0 +1,58 @@
1
+ /** Rows older than this are pruned on write, which also bounds how far back a repeat can be found. */
2
+ export const FAILURE_LOG_RETENTION_DAYS = 90;
3
+ /** Longest session id or tool name kept; the hook payload is not trusted to be short. */
4
+ const MAX_FIELD = 128;
5
+ /** Append one failure row and prune rows past {@link FAILURE_LOG_RETENTION_DAYS}. */
6
+ export function recordFailure(db, event) {
7
+ // Normalised, because the window and prune compare timestamps as strings.
8
+ const now = new Date(event.now ?? Date.now()).toISOString();
9
+ db.prepare(`INSERT INTO failure_log (ts, tenant_id, session_id, tool, outcome, skip_rule, sig_hash, detail_hash)
10
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)`).run(now, event.tenantId, event.sessionId?.slice(0, MAX_FIELD) ?? null, event.tool?.slice(0, MAX_FIELD) ?? null, event.outcome, event.rule ?? null, event.sigHash ?? null, event.detailHash ?? null);
11
+ const cutoff = new Date(Date.parse(now) - FAILURE_LOG_RETENTION_DAYS * 86_400_000).toISOString();
12
+ db.prepare(`DELETE FROM failure_log WHERE ts < ?`).run(cutoff);
13
+ }
14
+ /** Rated failures per session since `sinceIso`, the input for repeat-error rate per arm (CD11, CD12). */
15
+ export function failuresBySession(db, tenantId, sinceIso) {
16
+ // SAFETY: the SELECT names exactly these three TEXT columns.
17
+ const rows = db.prepare(`SELECT ts, session_id, sig_hash FROM failure_log
18
+ WHERE tenant_id = ? AND session_id IS NOT NULL AND sig_hash IS NOT NULL
19
+ AND outcome IN ('stored', 'duplicate', 'store-failed')
20
+ ORDER BY id`).all(tenantId);
21
+ const firstSession = new Map();
22
+ const bySession = new Map();
23
+ for (const row of rows) {
24
+ if (!firstSession.has(row.sig_hash))
25
+ firstSession.set(row.sig_hash, row.session_id);
26
+ // Rows before the window are not counted but still decide which session hit a signature first.
27
+ if (row.ts < sinceIso)
28
+ continue;
29
+ const repeat = firstSession.get(row.sig_hash) !== row.session_id;
30
+ const s = bySession.get(row.session_id) ?? { sessionId: row.session_id, failures: 0, repeats: 0 };
31
+ bySession.set(row.session_id, { ...s, failures: s.failures + 1, repeats: s.repeats + (repeat ? 1 : 0) });
32
+ }
33
+ return [...bySession.values()];
34
+ }
35
+ /** Sum the failure log for one tenant since `sinceIso`. */
36
+ export function summarizeFailures(db, tenantId, sinceIso) {
37
+ // SAFETY: the SELECT names exactly these two columns, TEXT and an aggregate.
38
+ const rows = db.prepare(`SELECT outcome, COUNT(*) AS n FROM failure_log WHERE tenant_id = ? AND ts >= ? GROUP BY outcome`).all(tenantId, sinceIso);
39
+ const counts = new Map(rows.map((r) => [r.outcome, Number(r.n)]));
40
+ const outcomes = {
41
+ stored: counts.get('stored') ?? 0,
42
+ duplicate: counts.get('duplicate') ?? 0,
43
+ 'store-failed': counts.get('store-failed') ?? 0,
44
+ 'skipped-interrupt': counts.get('skipped-interrupt') ?? 0,
45
+ 'skipped-routine': counts.get('skipped-routine') ?? 0,
46
+ 'skipped-invalid': counts.get('skipped-invalid') ?? 0,
47
+ };
48
+ const sessions = failuresBySession(db, tenantId, sinceIso);
49
+ return {
50
+ since: sinceIso,
51
+ outcomes,
52
+ total: Object.values(outcomes).reduce((sum, n) => sum + n, 0),
53
+ rated: sessions.reduce((sum, s) => sum + s.failures, 0),
54
+ repeats: sessions.reduce((sum, s) => sum + s.repeats, 0),
55
+ sessions: sessions.length,
56
+ };
57
+ }
58
+ //# sourceMappingURL=failure-log.js.map