hippo-memory 1.45.0 → 1.46.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +57 -15
- package/dist/ablation.d.ts +10 -1
- package/dist/ablation.js +17 -1
- package/dist/api.d.ts +60 -1
- package/dist/api.js +189 -7
- package/dist/audit.d.ts +1 -1
- package/dist/capture-error.d.ts +20 -0
- package/dist/capture-error.js +82 -0
- package/dist/capture.d.ts +25 -8
- package/dist/capture.js +100 -5
- package/dist/cli.d.ts +6 -1
- package/dist/cli.js +381 -48
- package/dist/config.d.ts +20 -0
- package/dist/config.js +35 -0
- package/dist/consolidate.d.ts +6 -0
- package/dist/consolidate.js +98 -13
- package/dist/db.js +57 -1
- package/dist/doctor.d.ts +34 -0
- package/dist/doctor.js +174 -0
- package/dist/dormant.d.ts +91 -0
- package/dist/dormant.js +121 -0
- package/dist/eval-stats.d.ts +123 -0
- package/dist/eval-stats.js +187 -0
- package/dist/half-life-migration.d.ts +55 -0
- package/dist/half-life-migration.js +111 -0
- package/dist/hooks.d.ts +4 -0
- package/dist/hooks.js +47 -0
- package/dist/mcp/server.d.ts +6 -0
- package/dist/mcp/server.js +70 -13
- package/dist/memory.d.ts +16 -2
- package/dist/memory.js +27 -5
- package/dist/physics-config.js +5 -1
- package/dist/recall-scope.d.ts +24 -0
- package/dist/recall-scope.js +41 -0
- package/dist/reject-flow.d.ts +3 -3
- package/dist/reject-flow.js +10 -3
- package/dist/search.d.ts +4 -4
- package/dist/search.js +23 -18
- package/dist/server.js +11 -1
- package/dist/store.d.ts +12 -1
- package/dist/store.js +58 -12
- package/dist/token-ledger.d.ts +119 -0
- package/dist/token-ledger.js +181 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/extensions/openclaw-plugin/openclaw.plugin.json +1 -1
- package/extensions/openclaw-plugin/package.json +1 -1
- package/openclaw.plugin.json +1 -1
- package/package.json +2 -1
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Dormant memories: what sleep does with a faded memory instead of deleting
|
|
3
|
+
* it (config `dormant.enabled`, on by default; `retentionDays` bounds how
|
|
4
|
+
* long one is kept).
|
|
5
|
+
*
|
|
6
|
+
* A dormant memory keeps its full content in `dormant_memories` (schema v44)
|
|
7
|
+
* but is no longer a `memories` row, so recall, context, every sleep pass and
|
|
8
|
+
* every other reader of `memories` stop seeing it exactly as if it had been
|
|
9
|
+
* deleted. `hippo dormant` lists and searches them, `hippo dormant restore`
|
|
10
|
+
* brings one back, `hippo dormant forget` deletes one for good.
|
|
11
|
+
*
|
|
12
|
+
* Not to be confused with the raw archive (`raw_archive`, `archiveRawMemory`,
|
|
13
|
+
* `hippo forget --archive`): that path removes a raw receipt's content and
|
|
14
|
+
* keeps only its metadata. A dormant memory keeps its content.
|
|
15
|
+
*
|
|
16
|
+
* DB-only helpers: the caller owns the handle and any transaction. The
|
|
17
|
+
* tenant-scoped entry points are api.listDormant / restoreDormant /
|
|
18
|
+
* forgetDormant.
|
|
19
|
+
*/
|
|
20
|
+
import type { DatabaseSyncLike } from './db.js';
|
|
21
|
+
import type { MemoryEntry } from './memory.js';
|
|
22
|
+
/** Why sleep made a memory dormant. Only the decay pass does today. */
|
|
23
|
+
export type DormantReason = 'decay';
|
|
24
|
+
/** One memory that sleep is moving out of active memory into the dormant store. */
|
|
25
|
+
export interface DormantMove {
|
|
26
|
+
/** The memory as it stood when it faded; restored verbatim apart from its recall clock. */
|
|
27
|
+
entry: MemoryEntry;
|
|
28
|
+
/** Live strength at the moment it went dormant (below the decay threshold). */
|
|
29
|
+
strength: number;
|
|
30
|
+
reason: DormantReason;
|
|
31
|
+
/** ISO time of the sleep that made it dormant. */
|
|
32
|
+
dormantAt: string;
|
|
33
|
+
}
|
|
34
|
+
/** A dormant memory as listed to a user. */
|
|
35
|
+
export interface DormantMemory {
|
|
36
|
+
id: string;
|
|
37
|
+
tenantId: string;
|
|
38
|
+
content: string;
|
|
39
|
+
tags: string[];
|
|
40
|
+
/** Live strength when it went dormant. */
|
|
41
|
+
strength: number;
|
|
42
|
+
/** Why it went dormant (`decay`). */
|
|
43
|
+
reason: string;
|
|
44
|
+
/** ISO time it went dormant. */
|
|
45
|
+
dormantAt: string;
|
|
46
|
+
}
|
|
47
|
+
/** Options for {@link listDormantRows}. */
|
|
48
|
+
export interface ListDormantOpts {
|
|
49
|
+
/** Whitespace-separated terms; a row must contain every term (case-insensitive substring). */
|
|
50
|
+
query?: string;
|
|
51
|
+
/** Maximum rows returned, newest first. Default 20. */
|
|
52
|
+
limit?: number;
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Insert (or refresh) the dormant snapshot for `move.entry`. Does not touch
|
|
56
|
+
* the `memories` row: the caller deletes it in the same transaction, so the
|
|
57
|
+
* memory is never in both places or in neither.
|
|
58
|
+
*/
|
|
59
|
+
export declare function insertDormantRow(db: DatabaseSyncLike, move: DormantMove): void;
|
|
60
|
+
/** A tenant's dormant memories, newest first, optionally filtered by search terms. */
|
|
61
|
+
export declare function listDormantRows(db: DatabaseSyncLike, tenantId: string, opts?: ListDormantOpts): DormantMemory[];
|
|
62
|
+
/** A dormant memory's stored snapshot plus when and why it went dormant. */
|
|
63
|
+
export interface DormantSnapshot {
|
|
64
|
+
entry: MemoryEntry;
|
|
65
|
+
reason: string;
|
|
66
|
+
strength: number;
|
|
67
|
+
dormantAt: string;
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* The stored snapshot for a tenant's dormant memory, or null when the tenant
|
|
71
|
+
* has no dormant memory with that id (another tenant's id reads as absent).
|
|
72
|
+
*/
|
|
73
|
+
export declare function readDormantSnapshot(db: DatabaseSyncLike, tenantId: string, id: string): DormantSnapshot | null;
|
|
74
|
+
/** Whether a tenant has a dormant memory with this id (snapshot readable or not). */
|
|
75
|
+
export declare function hasDormantRow(db: DatabaseSyncLike, tenantId: string, id: string): boolean;
|
|
76
|
+
/** Delete a tenant's dormant memory. Returns false when there was none. */
|
|
77
|
+
export declare function deleteDormantRow(db: DatabaseSyncLike, tenantId: string, id: string): boolean;
|
|
78
|
+
/**
|
|
79
|
+
* Delete every dormant memory in the tenant whose content has this rejection
|
|
80
|
+
* digest, so a rejected value cannot linger in dormant storage. Returns the
|
|
81
|
+
* ids removed. Same O(N) scan as the live-row sweep in reject-flow.ts.
|
|
82
|
+
*/
|
|
83
|
+
export declare function purgeDormantByDigest(db: DatabaseSyncLike, tenantId: string, digest: string): string[];
|
|
84
|
+
/** How many dormant memories went dormant before `cutoffIso` (a dry-run count). */
|
|
85
|
+
export declare function countExpiredDormant(db: DatabaseSyncLike, cutoffIso: string): number;
|
|
86
|
+
/**
|
|
87
|
+
* Delete every dormant memory (all tenants) that went dormant before
|
|
88
|
+
* `cutoffIso`: the `dormant.retentionDays` window. Returns how many went.
|
|
89
|
+
*/
|
|
90
|
+
export declare function purgeExpiredDormant(db: DatabaseSyncLike, cutoffIso: string): number;
|
|
91
|
+
//# sourceMappingURL=dormant.d.ts.map
|
package/dist/dormant.js
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
import { rejectionDigest } from './rejection.js';
|
|
2
|
+
const DEFAULT_LIST_LIMIT = 20;
|
|
3
|
+
/**
|
|
4
|
+
* Insert (or refresh) the dormant snapshot for `move.entry`. Does not touch
|
|
5
|
+
* the `memories` row: the caller deletes it in the same transaction, so the
|
|
6
|
+
* memory is never in both places or in neither.
|
|
7
|
+
*/
|
|
8
|
+
export function insertDormantRow(db, move) {
|
|
9
|
+
db.prepare(`
|
|
10
|
+
INSERT INTO dormant_memories (tenant_id, id, content, entry_json, reason, strength, dormant_at)
|
|
11
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
12
|
+
ON CONFLICT(tenant_id, id) DO UPDATE SET
|
|
13
|
+
content = excluded.content,
|
|
14
|
+
entry_json = excluded.entry_json,
|
|
15
|
+
reason = excluded.reason,
|
|
16
|
+
strength = excluded.strength,
|
|
17
|
+
dormant_at = excluded.dormant_at
|
|
18
|
+
`).run(move.entry.tenantId, move.entry.id, move.entry.content, JSON.stringify(move.entry), move.reason, move.strength, move.dormantAt);
|
|
19
|
+
}
|
|
20
|
+
/** Escape LIKE metacharacters so a search term matches literally (ESCAPE '\'). */
|
|
21
|
+
function escapeLike(term) {
|
|
22
|
+
return term.replace(/[\\%_]/g, (ch) => `\\${ch}`);
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* Parse a stored snapshot back into a MemoryEntry, or null when the row no
|
|
26
|
+
* longer matches the snapshot it carries (edited by hand, or truncated).
|
|
27
|
+
*/
|
|
28
|
+
function parseSnapshot(row) {
|
|
29
|
+
try {
|
|
30
|
+
// SAFETY: entry_json is only written by insertDormantRow from a MemoryEntry;
|
|
31
|
+
// the id / tenant / content checks below reject a row edited out of shape.
|
|
32
|
+
const entry = JSON.parse(row.entry_json);
|
|
33
|
+
if (entry.id !== row.id || entry.tenantId !== row.tenant_id || entry.content !== row.content) {
|
|
34
|
+
return null;
|
|
35
|
+
}
|
|
36
|
+
return entry;
|
|
37
|
+
}
|
|
38
|
+
catch {
|
|
39
|
+
return null;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
function rowToDormantMemory(row) {
|
|
43
|
+
const entry = parseSnapshot(row);
|
|
44
|
+
return {
|
|
45
|
+
id: row.id,
|
|
46
|
+
tenantId: row.tenant_id,
|
|
47
|
+
content: row.content,
|
|
48
|
+
tags: entry && Array.isArray(entry.tags) ? entry.tags.map(String) : [],
|
|
49
|
+
strength: row.strength,
|
|
50
|
+
reason: row.reason,
|
|
51
|
+
dormantAt: row.dormant_at,
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
/** A tenant's dormant memories, newest first, optionally filtered by search terms. */
|
|
55
|
+
export function listDormantRows(db, tenantId, opts = {}) {
|
|
56
|
+
const terms = (opts.query ?? '').trim().split(/\s+/).filter((t) => t.length > 0);
|
|
57
|
+
const limit = opts.limit !== undefined && Number.isFinite(opts.limit) && opts.limit >= 1
|
|
58
|
+
? Math.floor(opts.limit)
|
|
59
|
+
: DEFAULT_LIST_LIMIT;
|
|
60
|
+
const termClauses = terms.map(() => ` AND content LIKE ? ESCAPE '\\'`).join('');
|
|
61
|
+
// SAFETY: rows' shape matches the seven columns named in the SELECT.
|
|
62
|
+
const rows = db.prepare(`SELECT tenant_id, id, content, entry_json, reason, strength, dormant_at
|
|
63
|
+
FROM dormant_memories
|
|
64
|
+
WHERE tenant_id = ?${termClauses}
|
|
65
|
+
ORDER BY dormant_at DESC, id ASC
|
|
66
|
+
LIMIT ?`).all(tenantId, ...terms.map((t) => `%${escapeLike(t)}%`), limit);
|
|
67
|
+
return rows.map(rowToDormantMemory);
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* The stored snapshot for a tenant's dormant memory, or null when the tenant
|
|
71
|
+
* has no dormant memory with that id (another tenant's id reads as absent).
|
|
72
|
+
*/
|
|
73
|
+
export function readDormantSnapshot(db, tenantId, id) {
|
|
74
|
+
// SAFETY: row's shape matches the seven columns named in the SELECT.
|
|
75
|
+
const row = db.prepare(`SELECT tenant_id, id, content, entry_json, reason, strength, dormant_at
|
|
76
|
+
FROM dormant_memories WHERE tenant_id = ? AND id = ?`).get(tenantId, id);
|
|
77
|
+
const entry = row ? parseSnapshot(row) : null;
|
|
78
|
+
return row && entry ? { entry, reason: row.reason, strength: row.strength, dormantAt: row.dormant_at } : null;
|
|
79
|
+
}
|
|
80
|
+
/** Whether a tenant has a dormant memory with this id (snapshot readable or not). */
|
|
81
|
+
export function hasDormantRow(db, tenantId, id) {
|
|
82
|
+
return db.prepare(`SELECT 1 FROM dormant_memories WHERE tenant_id = ? AND id = ?`).get(tenantId, id) !== undefined;
|
|
83
|
+
}
|
|
84
|
+
/** Delete a tenant's dormant memory. Returns false when there was none. */
|
|
85
|
+
export function deleteDormantRow(db, tenantId, id) {
|
|
86
|
+
const result = db.prepare(`DELETE FROM dormant_memories WHERE tenant_id = ? AND id = ?`).run(tenantId, id);
|
|
87
|
+
return Number(result.changes ?? 0) > 0;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Delete every dormant memory in the tenant whose content has this rejection
|
|
91
|
+
* digest, so a rejected value cannot linger in dormant storage. Returns the
|
|
92
|
+
* ids removed. Same O(N) scan as the live-row sweep in reject-flow.ts.
|
|
93
|
+
*/
|
|
94
|
+
export function purgeDormantByDigest(db, tenantId, digest) {
|
|
95
|
+
// SAFETY: rows' shape matches the two columns named in the SELECT.
|
|
96
|
+
const rows = db.prepare(`SELECT id, content FROM dormant_memories WHERE tenant_id = ?`)
|
|
97
|
+
.all(tenantId);
|
|
98
|
+
const removed = [];
|
|
99
|
+
for (const row of rows) {
|
|
100
|
+
if (rejectionDigest(row.content) !== digest)
|
|
101
|
+
continue;
|
|
102
|
+
deleteDormantRow(db, tenantId, row.id);
|
|
103
|
+
removed.push(row.id);
|
|
104
|
+
}
|
|
105
|
+
return removed;
|
|
106
|
+
}
|
|
107
|
+
/** How many dormant memories went dormant before `cutoffIso` (a dry-run count). */
|
|
108
|
+
export function countExpiredDormant(db, cutoffIso) {
|
|
109
|
+
// SAFETY: row's shape matches the single aliased COUNT column in the SELECT.
|
|
110
|
+
const row = db.prepare(`SELECT COUNT(*) AS n FROM dormant_memories WHERE dormant_at < ?`).get(cutoffIso);
|
|
111
|
+
return Number(row.n);
|
|
112
|
+
}
|
|
113
|
+
/**
|
|
114
|
+
* Delete every dormant memory (all tenants) that went dormant before
|
|
115
|
+
* `cutoffIso`: the `dormant.retentionDays` window. Returns how many went.
|
|
116
|
+
*/
|
|
117
|
+
export function purgeExpiredDormant(db, cutoffIso) {
|
|
118
|
+
const result = db.prepare(`DELETE FROM dormant_memories WHERE dormant_at < ?`).run(cutoffIso);
|
|
119
|
+
return Number(result.changes ?? 0);
|
|
120
|
+
}
|
|
121
|
+
//# sourceMappingURL=dormant.js.map
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Statistics and cost accounting for the token-efficiency evals (ROADMAP
|
|
3
|
+
* Part IX, TE3-TE5).
|
|
4
|
+
*
|
|
5
|
+
* - Cost: price provider usage over four buckets (uncached input, cache
|
|
6
|
+
* write, cache read, output). Raw token counts overstate savings when most
|
|
7
|
+
* input is already cache reads, so every dollar claim goes through here.
|
|
8
|
+
* - Uncertainty: paired bootstrap over tasks, cluster bootstrap (tasks that
|
|
9
|
+
* share a repository are not independent), and a paired ratio bootstrap for
|
|
10
|
+
* dollars per resolved task.
|
|
11
|
+
*
|
|
12
|
+
* Deterministic: every resampling function takes a seed, so a published
|
|
13
|
+
* result can be reproduced exactly.
|
|
14
|
+
*/
|
|
15
|
+
/** Token usage for one model call or one whole task, split by how it is billed. */
|
|
16
|
+
export interface Usage {
|
|
17
|
+
/** Input tokens billed at the base input price (not read from or written to a cache). */
|
|
18
|
+
inputTokens: number;
|
|
19
|
+
/** Input tokens written to a prompt cache. */
|
|
20
|
+
cacheWriteTokens: number;
|
|
21
|
+
/** Input tokens read from a prompt cache. */
|
|
22
|
+
cacheReadTokens: number;
|
|
23
|
+
/** Output tokens, including any reasoning tokens the provider bills as output. */
|
|
24
|
+
outputTokens: number;
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Prices in dollars per million tokens. Take them from the provider's
|
|
28
|
+
* current price page for the exact model; this module has no built-in
|
|
29
|
+
* prices because they change.
|
|
30
|
+
*/
|
|
31
|
+
export interface Prices {
|
|
32
|
+
inputPerMTok: number;
|
|
33
|
+
cacheWritePerMTok: number;
|
|
34
|
+
cacheReadPerMTok: number;
|
|
35
|
+
outputPerMTok: number;
|
|
36
|
+
}
|
|
37
|
+
/** Sum several usages bucket by bucket. */
|
|
38
|
+
export declare function addUsage(...usages: Usage[]): Usage;
|
|
39
|
+
/** Dollar cost of `usage` at `prices`. */
|
|
40
|
+
export declare function priceUsage(usage: Usage, prices: Prices): number;
|
|
41
|
+
/**
|
|
42
|
+
* Relative prices of the cache buckets against the base input price, for
|
|
43
|
+
* cost in "uncached-equivalent tokens" when no dollar prices are given.
|
|
44
|
+
* Defaults follow Anthropic's published ratios (5-minute cache write 1.25x,
|
|
45
|
+
* cache read 0.1x); pass the ratios for another provider when needed.
|
|
46
|
+
*/
|
|
47
|
+
export interface CacheRatios {
|
|
48
|
+
write: number;
|
|
49
|
+
read: number;
|
|
50
|
+
}
|
|
51
|
+
/** Default cache price ratios (write 1.25x, read 0.1x of base input). */
|
|
52
|
+
export declare const DEFAULT_CACHE_RATIOS: Readonly<CacheRatios>;
|
|
53
|
+
/** Input cost of `usage` in uncached-equivalent tokens (output excluded). */
|
|
54
|
+
export declare function uncachedEquivalentInput(usage: Usage, ratios?: CacheRatios): number;
|
|
55
|
+
/**
|
|
56
|
+
* Mulberry32: a small seeded PRNG returning floats in [0, 1). Same seed,
|
|
57
|
+
* same stream, on every platform.
|
|
58
|
+
*/
|
|
59
|
+
export declare function seededRandom(seed: number): () => number;
|
|
60
|
+
/** A point estimate with a percentile bootstrap confidence interval. */
|
|
61
|
+
export interface Estimate {
|
|
62
|
+
estimate: number;
|
|
63
|
+
low: number;
|
|
64
|
+
high: number;
|
|
65
|
+
/** Resamples drawn. */
|
|
66
|
+
iterations: number;
|
|
67
|
+
}
|
|
68
|
+
/** Options shared by the bootstrap functions. */
|
|
69
|
+
export interface BootstrapOpts {
|
|
70
|
+
/** Resamples. Default 5000. */
|
|
71
|
+
iterations?: number;
|
|
72
|
+
/** Two-sided level, e.g. 0.05 for a 95% interval. Default 0.05. */
|
|
73
|
+
alpha?: number;
|
|
74
|
+
/** PRNG seed. Default 1. */
|
|
75
|
+
seed?: number;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Paired bootstrap for the mean of per-task differences (treatment minus
|
|
79
|
+
* control on the same task). An interval that excludes zero is the bar for
|
|
80
|
+
* calling a difference real.
|
|
81
|
+
*/
|
|
82
|
+
export declare function pairedBootstrap(diffs: number[], opts?: BootstrapOpts): Estimate;
|
|
83
|
+
/**
|
|
84
|
+
* Cluster bootstrap for the mean of per-task differences: resamples whole
|
|
85
|
+
* clusters (for example all tasks from one repository), because tasks in a
|
|
86
|
+
* cluster share causes and are not independent draws.
|
|
87
|
+
*/
|
|
88
|
+
export declare function clusteredPairedBootstrap(diffsByCluster: ReadonlyMap<string, number[]>, opts?: BootstrapOpts): Estimate;
|
|
89
|
+
/** One task's outcome in one arm, for {@link costPerResolvedDelta}. */
|
|
90
|
+
export interface ArmOutcome {
|
|
91
|
+
/** Dollars (or uncached-equivalent tokens) spent on the task. */
|
|
92
|
+
cost: number;
|
|
93
|
+
/** Whether the task was resolved. */
|
|
94
|
+
resolved: boolean;
|
|
95
|
+
}
|
|
96
|
+
/** Result of {@link costPerResolvedDelta}. */
|
|
97
|
+
export interface CostPerResolvedDelta {
|
|
98
|
+
control: number;
|
|
99
|
+
treatment: number;
|
|
100
|
+
/** treatment minus control, with its interval. */
|
|
101
|
+
delta: Estimate;
|
|
102
|
+
/** (treatment minus control) / control, with its interval. Negative is a saving. */
|
|
103
|
+
relative: Estimate;
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* Cost per resolved task in two arms run on the same tasks, with a paired
|
|
107
|
+
* bootstrap over tasks (a task is resampled with both of its arm outcomes).
|
|
108
|
+
* `control[i]` and `treatment[i]` must be the same task. Resamples in which
|
|
109
|
+
* an arm resolves nothing are dropped; `iterations` reports how many were
|
|
110
|
+
* kept.
|
|
111
|
+
*/
|
|
112
|
+
export declare function costPerResolvedDelta(control: ArmOutcome[], treatment: ArmOutcome[], opts?: BootstrapOpts): CostPerResolvedDelta;
|
|
113
|
+
/**
|
|
114
|
+
* pass@k: share of tasks with at least one success in their first k runs.
|
|
115
|
+
* NaN when no task has k runs (not measured, which is not the same as 0).
|
|
116
|
+
*/
|
|
117
|
+
export declare function passAtK(runsByTask: boolean[][], k: number): number;
|
|
118
|
+
/**
|
|
119
|
+
* pass^k: share of tasks whose first k runs all succeed (consistency).
|
|
120
|
+
* NaN when no task has k runs.
|
|
121
|
+
*/
|
|
122
|
+
export declare function passHatK(runsByTask: boolean[][], k: number): number;
|
|
123
|
+
//# sourceMappingURL=eval-stats.d.ts.map
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Statistics and cost accounting for the token-efficiency evals (ROADMAP
|
|
3
|
+
* Part IX, TE3-TE5).
|
|
4
|
+
*
|
|
5
|
+
* - Cost: price provider usage over four buckets (uncached input, cache
|
|
6
|
+
* write, cache read, output). Raw token counts overstate savings when most
|
|
7
|
+
* input is already cache reads, so every dollar claim goes through here.
|
|
8
|
+
* - Uncertainty: paired bootstrap over tasks, cluster bootstrap (tasks that
|
|
9
|
+
* share a repository are not independent), and a paired ratio bootstrap for
|
|
10
|
+
* dollars per resolved task.
|
|
11
|
+
*
|
|
12
|
+
* Deterministic: every resampling function takes a seed, so a published
|
|
13
|
+
* result can be reproduced exactly.
|
|
14
|
+
*/
|
|
15
|
+
/** Sum several usages bucket by bucket. */
|
|
16
|
+
export function addUsage(...usages) {
|
|
17
|
+
const total = { inputTokens: 0, cacheWriteTokens: 0, cacheReadTokens: 0, outputTokens: 0 };
|
|
18
|
+
for (const u of usages) {
|
|
19
|
+
total.inputTokens += u.inputTokens;
|
|
20
|
+
total.cacheWriteTokens += u.cacheWriteTokens;
|
|
21
|
+
total.cacheReadTokens += u.cacheReadTokens;
|
|
22
|
+
total.outputTokens += u.outputTokens;
|
|
23
|
+
}
|
|
24
|
+
return total;
|
|
25
|
+
}
|
|
26
|
+
/** Dollar cost of `usage` at `prices`. */
|
|
27
|
+
export function priceUsage(usage, prices) {
|
|
28
|
+
return (usage.inputTokens * prices.inputPerMTok
|
|
29
|
+
+ usage.cacheWriteTokens * prices.cacheWritePerMTok
|
|
30
|
+
+ usage.cacheReadTokens * prices.cacheReadPerMTok
|
|
31
|
+
+ usage.outputTokens * prices.outputPerMTok) / 1_000_000;
|
|
32
|
+
}
|
|
33
|
+
/** Default cache price ratios (write 1.25x, read 0.1x of base input). */
|
|
34
|
+
export const DEFAULT_CACHE_RATIOS = { write: 1.25, read: 0.1 };
|
|
35
|
+
/** Input cost of `usage` in uncached-equivalent tokens (output excluded). */
|
|
36
|
+
export function uncachedEquivalentInput(usage, ratios = DEFAULT_CACHE_RATIOS) {
|
|
37
|
+
return usage.inputTokens + usage.cacheWriteTokens * ratios.write + usage.cacheReadTokens * ratios.read;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Mulberry32: a small seeded PRNG returning floats in [0, 1). Same seed,
|
|
41
|
+
* same stream, on every platform.
|
|
42
|
+
*/
|
|
43
|
+
export function seededRandom(seed) {
|
|
44
|
+
let a = seed >>> 0;
|
|
45
|
+
return () => {
|
|
46
|
+
a = (a + 0x6d2b79f5) >>> 0;
|
|
47
|
+
let t = a;
|
|
48
|
+
t = Math.imul(t ^ (t >>> 15), t | 1);
|
|
49
|
+
t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
|
|
50
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
function percentileInterval(samples, alpha) {
|
|
54
|
+
const sorted = [...samples].sort((x, y) => x - y);
|
|
55
|
+
const lowIdx = Math.max(0, Math.floor((alpha / 2) * sorted.length));
|
|
56
|
+
const highIdx = Math.min(sorted.length - 1, Math.ceil((1 - alpha / 2) * sorted.length) - 1);
|
|
57
|
+
return { low: sorted[lowIdx], high: sorted[highIdx] };
|
|
58
|
+
}
|
|
59
|
+
function mean(xs) {
|
|
60
|
+
return xs.length === 0 ? 0 : xs.reduce((s, x) => s + x, 0) / xs.length;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Paired bootstrap for the mean of per-task differences (treatment minus
|
|
64
|
+
* control on the same task). An interval that excludes zero is the bar for
|
|
65
|
+
* calling a difference real.
|
|
66
|
+
*/
|
|
67
|
+
export function pairedBootstrap(diffs, opts = {}) {
|
|
68
|
+
const iterations = opts.iterations ?? 5000;
|
|
69
|
+
const alpha = opts.alpha ?? 0.05;
|
|
70
|
+
if (diffs.length === 0)
|
|
71
|
+
return { estimate: 0, low: 0, high: 0, iterations: 0 };
|
|
72
|
+
const rand = seededRandom(opts.seed ?? 1);
|
|
73
|
+
const n = diffs.length;
|
|
74
|
+
const samples = [];
|
|
75
|
+
for (let b = 0; b < iterations; b++) {
|
|
76
|
+
let s = 0;
|
|
77
|
+
for (let i = 0; i < n; i++)
|
|
78
|
+
s += diffs[Math.floor(rand() * n)];
|
|
79
|
+
samples.push(s / n);
|
|
80
|
+
}
|
|
81
|
+
return { estimate: mean(diffs), ...percentileInterval(samples, alpha), iterations };
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* Cluster bootstrap for the mean of per-task differences: resamples whole
|
|
85
|
+
* clusters (for example all tasks from one repository), because tasks in a
|
|
86
|
+
* cluster share causes and are not independent draws.
|
|
87
|
+
*/
|
|
88
|
+
export function clusteredPairedBootstrap(diffsByCluster, opts = {}) {
|
|
89
|
+
const iterations = opts.iterations ?? 5000;
|
|
90
|
+
const alpha = opts.alpha ?? 0.05;
|
|
91
|
+
const clusters = [...diffsByCluster.values()].filter((c) => c.length > 0);
|
|
92
|
+
const all = clusters.flat();
|
|
93
|
+
if (all.length === 0)
|
|
94
|
+
return { estimate: 0, low: 0, high: 0, iterations: 0 };
|
|
95
|
+
const rand = seededRandom(opts.seed ?? 1);
|
|
96
|
+
const k = clusters.length;
|
|
97
|
+
const samples = [];
|
|
98
|
+
for (let b = 0; b < iterations; b++) {
|
|
99
|
+
let sum = 0;
|
|
100
|
+
let count = 0;
|
|
101
|
+
for (let i = 0; i < k; i++) {
|
|
102
|
+
const c = clusters[Math.floor(rand() * k)];
|
|
103
|
+
for (const d of c)
|
|
104
|
+
sum += d;
|
|
105
|
+
count += c.length;
|
|
106
|
+
}
|
|
107
|
+
samples.push(count === 0 ? 0 : sum / count);
|
|
108
|
+
}
|
|
109
|
+
return { estimate: mean(all), ...percentileInterval(samples, alpha), iterations };
|
|
110
|
+
}
|
|
111
|
+
function costPerResolved(outcomes) {
|
|
112
|
+
const resolved = outcomes.filter((o) => o.resolved).length;
|
|
113
|
+
const cost = outcomes.reduce((s, o) => s + o.cost, 0);
|
|
114
|
+
return resolved === 0 ? Number.POSITIVE_INFINITY : cost / resolved;
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Cost per resolved task in two arms run on the same tasks, with a paired
|
|
118
|
+
* bootstrap over tasks (a task is resampled with both of its arm outcomes).
|
|
119
|
+
* `control[i]` and `treatment[i]` must be the same task. Resamples in which
|
|
120
|
+
* an arm resolves nothing are dropped; `iterations` reports how many were
|
|
121
|
+
* kept.
|
|
122
|
+
*/
|
|
123
|
+
export function costPerResolvedDelta(control, treatment, opts = {}) {
|
|
124
|
+
if (control.length !== treatment.length) {
|
|
125
|
+
throw new Error('control and treatment must list the same tasks in the same order');
|
|
126
|
+
}
|
|
127
|
+
const iterations = opts.iterations ?? 5000;
|
|
128
|
+
const alpha = opts.alpha ?? 0.05;
|
|
129
|
+
const c = costPerResolved(control);
|
|
130
|
+
const t = costPerResolved(treatment);
|
|
131
|
+
const rand = seededRandom(opts.seed ?? 1);
|
|
132
|
+
const n = control.length;
|
|
133
|
+
const deltas = [];
|
|
134
|
+
const relatives = [];
|
|
135
|
+
for (let b = 0; b < iterations && n > 0; b++) {
|
|
136
|
+
const cs = [];
|
|
137
|
+
const ts = [];
|
|
138
|
+
for (let i = 0; i < n; i++) {
|
|
139
|
+
const j = Math.floor(rand() * n);
|
|
140
|
+
cs.push(control[j]);
|
|
141
|
+
ts.push(treatment[j]);
|
|
142
|
+
}
|
|
143
|
+
const cc = costPerResolved(cs);
|
|
144
|
+
const tt = costPerResolved(ts);
|
|
145
|
+
if (!Number.isFinite(cc) || !Number.isFinite(tt))
|
|
146
|
+
continue;
|
|
147
|
+
deltas.push(tt - cc);
|
|
148
|
+
relatives.push((tt - cc) / cc);
|
|
149
|
+
}
|
|
150
|
+
const finite = Number.isFinite(c) && Number.isFinite(t);
|
|
151
|
+
const empty = { low: Number.NaN, high: Number.NaN };
|
|
152
|
+
return {
|
|
153
|
+
control: c,
|
|
154
|
+
treatment: t,
|
|
155
|
+
delta: {
|
|
156
|
+
estimate: finite ? t - c : Number.NaN,
|
|
157
|
+
...(deltas.length > 0 ? percentileInterval(deltas, alpha) : empty),
|
|
158
|
+
iterations: deltas.length,
|
|
159
|
+
},
|
|
160
|
+
relative: {
|
|
161
|
+
estimate: finite ? (t - c) / c : Number.NaN,
|
|
162
|
+
...(relatives.length > 0 ? percentileInterval(relatives, alpha) : empty),
|
|
163
|
+
iterations: relatives.length,
|
|
164
|
+
},
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* pass@k: share of tasks with at least one success in their first k runs.
|
|
169
|
+
* NaN when no task has k runs (not measured, which is not the same as 0).
|
|
170
|
+
*/
|
|
171
|
+
export function passAtK(runsByTask, k) {
|
|
172
|
+
const eligible = runsByTask.filter((r) => r.length >= k);
|
|
173
|
+
if (eligible.length === 0)
|
|
174
|
+
return Number.NaN;
|
|
175
|
+
return eligible.filter((r) => r.slice(0, k).some(Boolean)).length / eligible.length;
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* pass^k: share of tasks whose first k runs all succeed (consistency).
|
|
179
|
+
* NaN when no task has k runs.
|
|
180
|
+
*/
|
|
181
|
+
export function passHatK(runsByTask, k) {
|
|
182
|
+
const eligible = runsByTask.filter((r) => r.length >= k);
|
|
183
|
+
if (eligible.length === 0)
|
|
184
|
+
return Number.NaN;
|
|
185
|
+
return eligible.filter((r) => r.slice(0, k).every(Boolean)).length / eligible.length;
|
|
186
|
+
}
|
|
187
|
+
//# sourceMappingURL=eval-stats.js.map
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Moving a store's memories to a new default half-life.
|
|
3
|
+
*
|
|
4
|
+
* Each memory stores its own `half_life_days`, set at write from the
|
|
5
|
+
* default base and a few write-time multipliers (`deriveHalfLife`). Changing
|
|
6
|
+
* the default therefore reaches only new memories; without this migration a
|
|
7
|
+
* store would mix old-base and new-base memories, a state the decay
|
|
8
|
+
* evaluation never tested (docs/evals/2026-09-24-decay-default-prereg.md,
|
|
9
|
+
* Migration). The rule, declared there before any run:
|
|
10
|
+
*
|
|
11
|
+
* - only a memory still on the old base is rescaled: its half-life is
|
|
12
|
+
* `deriveHalfLife(from, entry)` plus its recall bonus. A memory hippo shortened since
|
|
13
|
+
* (invalidated, superseded, a merge source, marked bad) or one with its own
|
|
14
|
+
* fixed half-life (decisions, incidents, customer notes) keeps its value;
|
|
15
|
+
* - every rescale is written to the audit log with the ids, so it can be
|
|
16
|
+
* undone, and the store records the base it is on (`meta`), so the
|
|
17
|
+
* migration runs once.
|
|
18
|
+
*
|
|
19
|
+
* `hippo sleep` runs it before its decay pass, from the base the store is on
|
|
20
|
+
* (7 days when never recorded) to the configured `defaultHalfLifeDays`.
|
|
21
|
+
*/
|
|
22
|
+
import { type MemoryEntry } from './memory.js';
|
|
23
|
+
import { HALF_LIFE_BASE_META_KEY } from './store.js';
|
|
24
|
+
/** The base every store used before the base was recorded. */
|
|
25
|
+
export declare const LEGACY_HALF_LIFE_BASE = 7;
|
|
26
|
+
export { HALF_LIFE_BASE_META_KEY };
|
|
27
|
+
/** What {@link migrateDefaultHalfLife} did, or would do under `dryRun`. */
|
|
28
|
+
export interface HalfLifeMigrationResult {
|
|
29
|
+
from: number;
|
|
30
|
+
to: number;
|
|
31
|
+
/** Memories moved to the new base. */
|
|
32
|
+
rescaled: number;
|
|
33
|
+
/** Memories left alone because they are not on the old base. */
|
|
34
|
+
kept: number;
|
|
35
|
+
dryRun: boolean;
|
|
36
|
+
/** New half-life per moved id, so a dry run can preview decay at the new base. */
|
|
37
|
+
halfLives: ReadonlyMap<string, number>;
|
|
38
|
+
}
|
|
39
|
+
type HalfLifeFields = Pick<MemoryEntry, 'half_life_days' | 'tags' | 'schema_fit' | 'retrieval_count' | 'superseded_by'>;
|
|
40
|
+
/** Recall bonus over what `base` gave `entry`, or null when off that base; pre-1.46 recalls each added 2 days. */
|
|
41
|
+
export declare function halfLifeRecallBonus(entry: HalfLifeFields, base: number): number | null;
|
|
42
|
+
/** The entries to rescale from `from` to `to`, as copies that keep their recall bonus. Pure. */
|
|
43
|
+
export declare function planHalfLifeMigration(entries: readonly MemoryEntry[], from: number, to: number): MemoryEntry[];
|
|
44
|
+
/** The base this store's memories are on. */
|
|
45
|
+
export declare function storeHalfLifeBase(hippoRoot: string): number;
|
|
46
|
+
/**
|
|
47
|
+
* Move the store's memories from the base they are on to `to`. A no-op when
|
|
48
|
+
* they are already on it. Under `dryRun` nothing is written, the recorded
|
|
49
|
+
* base included.
|
|
50
|
+
*/
|
|
51
|
+
export declare function migrateDefaultHalfLife(hippoRoot: string, to: number, opts?: {
|
|
52
|
+
dryRun?: boolean;
|
|
53
|
+
actor?: string;
|
|
54
|
+
}): HalfLifeMigrationResult;
|
|
55
|
+
//# sourceMappingURL=half-life-migration.d.ts.map
|