openmerit 0.1.3 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -312
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -388
  52. package/dist/diagnostics.js +0 -194
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -203
  60. package/dist/pi-trials.js +0 -366
  61. package/dist/policy.js +0 -115
  62. package/dist/providers.js +0 -1
  63. package/dist/recommend.js +0 -76
  64. package/dist/routes.js +0 -59
  65. package/dist/standalone.js +0 -220
  66. package/dist/store.js +0 -89
  67. package/dist/strategist.js +0 -68
  68. package/dist/task-input.js +0 -54
  69. package/dist/traces.js +0 -127
  70. package/dist/trials.js +0 -140
  71. package/dist/types.js +0 -2
  72. package/examples/invoice-prompt.txt +0 -19
  73. package/examples/task.example.json +0 -7
  74. package/extension/openmerit.ts +0 -820
  75. package/instructions/OPENMERIT.md +0 -54
  76. package/instructions/openmerit.policy.json +0 -33
  77. package/rules.md +0 -39
@@ -1,220 +0,0 @@
1
- /** Provider-neutral standalone task trials executed through Pi routes. */
2
- import { readFileSync } from "node:fs";
3
- import { openRouterBenchmarkCandidates, relevantBenchmarks } from "./benchmarks.js";
4
- import { fetchCatalog, saveSnapshot } from "./catalog.js";
5
- import { paretoFrontier, pickBest, pickFallback } from "./frontier.js";
6
- import { JUDGE_PREFS } from "./judge.js";
7
- import { loadKey, PiCliChatClient } from "./llm.js";
8
- import { availablePiRoutes, recordedActiveTask, runPiTrial, scorePiRun } from "./pi-trials.js";
9
- import { loadPolicy, providerAllowed } from "./policy.js";
10
- import { buildRecommendation } from "./recommend.js";
11
- import { enrichRouteEntry, routeCatalog, routeKey, routeLabel } from "./routes.js";
12
- import { appendJsonl, paths, readJson, taskKey } from "./store.js";
13
- import { pickNext, STRAT_PREFS } from "./strategist.js";
14
- import { budgetOk, recordMeritSpend, recordTrialSpend, trialBudgetOk } from "./trials.js";
15
- function optionalOpenRouterKey() {
16
- try {
17
- return loadKey();
18
- }
19
- catch {
20
- return null;
21
- }
22
- }
23
- function configuredRoutes() {
24
- const state = readJson(paths.harnessState(), {});
25
- let routes = state.routes?.length ? state.routes : availablePiRoutes();
26
- if (state.currentRoute && !routes.some((route) => routeKey(route) === routeKey(state.currentRoute)))
27
- routes = [state.currentRoute, ...routes];
28
- const unique = new Map(routes.map((route) => [routeKey(route), route]));
29
- return { routes: [...unique.values()], currentRoute: state.currentRoute ?? null };
30
- }
31
- function selectorKey(selector) {
32
- if (!selector)
33
- return null;
34
- if (typeof selector === "string")
35
- return selector;
36
- const modelId = selector.modelId ?? selector.model_id;
37
- return selector.provider && modelId ? `${selector.provider}:${modelId}` : null;
38
- }
39
- function matchingEntries(cat, selector) {
40
- const exact = cat.get(selector);
41
- if (exact)
42
- return [exact];
43
- return [...cat.values()].filter((entry) => entry.id === selector || entry.route?.meritId === selector ||
44
- (entry.route && `${entry.route.provider}/${entry.route.modelId}` === selector));
45
- }
46
- function resolveInitial(cat, cfg, currentRoute) {
47
- const exactKey = selectorKey(cfg.initial_route);
48
- if (cfg.initial_route && !exactKey)
49
- throw new Error("initial_route must include provider and modelId");
50
- if (exactKey) {
51
- const exact = cat.get(exactKey);
52
- if (!exact)
53
- throw new Error(`initial route ${exactKey} is not eligible in Pi`);
54
- if (cfg.initial_model !== exact.id && cfg.initial_model !== exactKey)
55
- throw new Error(`initial_route ${exactKey} resolves to ${exact.id}, not initial_model ${cfg.initial_model}`);
56
- return exact;
57
- }
58
- const matches = matchingEntries(cat, cfg.initial_model);
59
- if (matches.length === 0)
60
- throw new Error(`initial_model ${cfg.initial_model} is not eligible in Pi`);
61
- if (matches.length === 1)
62
- return matches[0];
63
- if (currentRoute) {
64
- const current = matches.find((entry) => entry.route && routeKey(entry.route) === routeKey(currentRoute));
65
- if (current)
66
- return current;
67
- }
68
- throw new Error(`initial_model ${cfg.initial_model} has multiple Pi routes; set initial_route to one of: ` +
69
- matches.map((entry) => routeKey(entry.route)).join(", "));
70
- }
71
- function resolveMeritRoute(cat, prefs, label) {
72
- for (const pref of prefs) {
73
- if (!pref)
74
- continue;
75
- const matches = matchingEntries(cat, pref);
76
- if (matches.length)
77
- return matches.sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
78
- }
79
- for (const pref of prefs) {
80
- if (!pref)
81
- continue;
82
- const fuzzy = [...cat.values()].find((entry) => entry.id.includes(pref));
83
- if (fuzzy)
84
- return fuzzy;
85
- }
86
- const fallback = [...cat.values()].sort((a, b) => (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price)[0];
87
- if (fallback)
88
- return fallback;
89
- throw new Error(`could not resolve ${label} route from Pi's eligible models`);
90
- }
91
- async function buildCatalog(routes, key) {
92
- let cat = routeCatalog(routes);
93
- if (!key)
94
- return cat;
95
- try {
96
- const live = await fetchCatalog(key);
97
- saveSnapshot(live);
98
- cat = new Map([...cat].map(([id, entry]) => [id,
99
- entry.route?.provider === "openrouter" ? enrichRouteEntry(entry, live.get(entry.id)) : entry]));
100
- }
101
- catch (error) {
102
- console.log(`OpenRouter enrichment unavailable: ${error.message}`);
103
- }
104
- return cat;
105
- }
106
- export async function runStandaloneTrial(taskFile, rounds) {
107
- const cfg = JSON.parse(readFileSync(taskFile, "utf8"));
108
- if (!cfg.task?.trim() || !cfg.eval?.trim() || !cfg.initial_model?.trim())
109
- throw new Error("task JSON requires task, eval, and initial_model");
110
- if (!Number.isInteger(rounds) || rounds < 1)
111
- throw new Error("--rounds must be a positive integer");
112
- const policy = loadPolicy(paths.policy());
113
- const key = optionalOpenRouterKey();
114
- const { routes, currentRoute } = configuredRoutes();
115
- let cat = await buildCatalog(routes, key);
116
- for (const [id, entry] of [...cat]) {
117
- if (!entry.route || !providerAllowed(policy, entry.id, entry.route.provider) ||
118
- (entry.priceKnown !== false && entry.price > (cfg.max_usd_per_m ?? policy.max_usd_per_m)))
119
- cat.delete(id);
120
- }
121
- if (cat.size === 0)
122
- throw new Error("Pi exposes no model routes allowed by the current policy");
123
- const initial = resolveInitial(cat, cfg, currentRoute);
124
- const judge = resolveMeritRoute(cat, [cfg.judge_model, policy.judge_model, ...JUDGE_PREFS, initial.id], "judge");
125
- const strategist = resolveMeritRoute(cat, [cfg.strategist_model, policy.strategist_model, ...STRAT_PREFS, initial.id], "strategist");
126
- const client = new PiCliChatClient(undefined, recordMeritSpend);
127
- const { category, benchmarks } = relevantBenchmarks(cfg.task);
128
- let benchmarkCandidates = [];
129
- if (key) {
130
- try {
131
- benchmarkCandidates = await openRouterBenchmarkCandidates(key, category, new Map([...cat.values()].map((entry) => [entry.id, entry])));
132
- }
133
- catch (error) {
134
- console.log(`OpenRouter benchmark shortlist unavailable: ${error.message}`);
135
- }
136
- }
137
- console.log(`routes: ${cat.size} eligible in Pi${key ? " (OpenRouter enrichment enabled)" : ""}`);
138
- console.log(`initial: ${routeLabel(initial.route)} | judge: ${routeLabel(judge.route)} | ` +
139
- `strategist: ${routeLabel(strategist.route)} | benchmarks: ${benchmarks.join(", ")}`);
140
- const tKey = taskKey(cfg.task);
141
- const tried = new Set();
142
- const results = [];
143
- const failedProviders = new Set();
144
- for (let i = 0; i < rounds; i++) {
145
- const daily = budgetOk(policy);
146
- if (!daily.ok) {
147
- console.log(`budget: ${daily.reason}; stopping`);
148
- break;
149
- }
150
- let entry;
151
- let why;
152
- if (i === 0) {
153
- entry = initial;
154
- why = "initial route";
155
- }
156
- else {
157
- const pick = await pickNext(key ?? "", strategist.route, cfg.task, cfg.eval, benchmarks, results, cat, tried, cfg.max_usd_per_m ?? policy.max_usd_per_m, failedProviders, benchmarkCandidates, client);
158
- if (!pick)
159
- break;
160
- entry = pick.route ? cat.get(routeKey(pick.route)) : cat.get(pick.model);
161
- why = pick.why;
162
- }
163
- const route = entry.route;
164
- const keyForRoute = routeKey(route);
165
- if (tried.has(keyForRoute))
166
- break;
167
- if (i > 0) {
168
- const admission = trialBudgetOk(policy, entry, cfg.task.length);
169
- if (!admission.ok) {
170
- tried.add(keyForRoute);
171
- console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} skipped: ${admission.reason}`);
172
- continue;
173
- }
174
- }
175
- console.log(`[round ${i + 1}/${rounds}] ${routeLabel(route)} (${why}) ...`);
176
- const observed = i === 0 ? recordedActiveTask(cfg.task, route) : null;
177
- if (i === 0)
178
- console.log(observed
179
- ? " using the exact provider route from Pi's active session trace"
180
- : " active trace unavailable; running the initial route in a fresh Pi session");
181
- const outcome = observed
182
- ? await scorePiRun(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, observed, "trace", [], client)
183
- : await runPiTrial(key ?? "", judge.route, cfg.task, cfg.eval, entry.id, entry, cfg.cwd ?? process.cwd(), [], [], undefined, client);
184
- tried.add(keyForRoute);
185
- results.push(outcome.point);
186
- // Reusing the active Pi answer is observation, not a new candidate call.
187
- if (!observed)
188
- recordTrialSpend(outcome.costUsd);
189
- appendJsonl(paths.trials(), { ...outcome.point, taskKey: tKey });
190
- if (outcome.sessionId)
191
- console.log(` pi trace session: ${outcome.sessionId}`);
192
- if (outcome.error) {
193
- failedProviders.add(route.provider);
194
- console.log(` FAILED: ${outcome.error.slice(0, 160)}`);
195
- }
196
- else {
197
- const price = outcome.point.priceKnown === false ? "price unknown" : `$${outcome.point.price}/M`;
198
- console.log(` score=${outcome.point.score.toFixed(2)} ${price} ` +
199
- `run=$${outcome.costUsd.toFixed(4)} ${(outcome.point.why ?? "").slice(0, 100)}`);
200
- }
201
- }
202
- const scored = results.filter((point) => point.score > 0);
203
- if (scored.length === 0)
204
- throw new Error("no trials completed");
205
- const frontier = paretoFrontier(scored);
206
- const best = pickBest(scored);
207
- const fallback = pickFallback(scored, best, policy.fallback.min_score);
208
- console.log("\n=== pareto frontier (quality up, price and latency down) ===");
209
- for (const point of frontier)
210
- console.log(` ${(point.route ? routeLabel(point.route) : point.model).padEnd(52)} ` +
211
- `score=${point.score.toFixed(2)} ${point.priceKnown === false ? "price unknown" : `$${point.price.toFixed(2)}/M`}`);
212
- console.log(`\nBEST FIT : ${best.route ? routeLabel(best.route) : best.model}`);
213
- console.log(fallback ? `FALLBACK : ${fallback.route ? routeLabel(fallback.route) : fallback.model}` : "FALLBACK : n/a");
214
- const recommendation = buildRecommendation(tKey, cfg.task.slice(0, 120), initial.id, results, policy, null, initial.route);
215
- if (recommendation) {
216
- appendJsonl(paths.recommendations(), recommendation);
217
- console.log(`\nrecommendation -> ${paths.recommendations()} (auto=${recommendation.policy.autoApply})`);
218
- }
219
- return { points: results, recommendation };
220
- }
package/dist/store.js DELETED
@@ -1,89 +0,0 @@
1
- /** State directory layout + JSONL persistence. All state lives under OPENMERIT_HOME (default ~/.openmerit). */
2
- import { createHash } from "node:crypto";
3
- import { appendFileSync, existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync, } from "node:fs";
4
- import { homedir } from "node:os";
5
- import { join } from "node:path";
6
- export function stateDir() {
7
- const dir = process.env.OPENMERIT_HOME ?? join(homedir(), ".openmerit");
8
- if (!existsSync(dir))
9
- mkdirSync(dir, { recursive: true });
10
- return dir;
11
- }
12
- export const paths = {
13
- policy: () => join(stateDir(), "policy.json"),
14
- trials: () => join(stateDir(), "trials.jsonl"),
15
- recommendations: () => join(stateDir(), "recommendations.jsonl"),
16
- catalogSnapshot: () => join(stateDir(), "catalog", "snapshot.json"),
17
- candidates: () => join(stateDir(), "catalog", "candidates.json"),
18
- benchmarksDigest: () => join(stateDir(), "benchmarks", "digest.json"),
19
- traceCursor: () => join(stateDir(), "traces", "cursor.json"),
20
- observations: () => join(stateDir(), "traces", "observations.jsonl"),
21
- events: () => join(stateDir(), "events.jsonl"),
22
- harnessState: () => join(stateDir(), "harness-state.json"),
23
- ledger: () => join(stateDir(), "ledger.json"),
24
- watchProcessed: () => join(stateDir(), "watch", "processed.json"),
25
- sessionJob: (marker) => join(stateDir(), "watch", "jobs", sha1(marker) + ".json"),
26
- trialTrace: (id) => join(stateDir(), "traces", "trials", `${sha1(id)}.jsonl`),
27
- envFile: () => join(stateDir(), ".env"),
28
- };
29
- export function sha1(text) {
30
- return createHash("sha1").update(text).digest("hex").slice(0, 12);
31
- }
32
- /** Stable task key from a task statement. */
33
- export function taskKey(taskText) {
34
- const normalized = taskText.toLowerCase().replace(/\s+/g, " ").trim();
35
- return sha1(normalized);
36
- }
37
- export function appendJsonl(file, obj) {
38
- mkdirSync(join(file, ".."), { recursive: true });
39
- appendFileSync(file, JSON.stringify(obj) + "\n");
40
- }
41
- /**
42
- * Read append-only state without letting one interrupted or malformed line
43
- * hide the remaining history. Diagnostics can surface invalid line numbers.
44
- */
45
- export function readJsonlReport(file) {
46
- if (!existsSync(file))
47
- return { records: [], invalidLines: [] };
48
- const records = [];
49
- const invalidLines = [];
50
- for (const [index, line] of readFileSync(file, "utf8").split("\n").entries()) {
51
- if (!line.trim())
52
- continue;
53
- try {
54
- records.push(JSON.parse(line));
55
- }
56
- catch {
57
- invalidLines.push(index + 1);
58
- }
59
- }
60
- return { records, invalidLines };
61
- }
62
- export function readJsonl(file) {
63
- return readJsonlReport(file).records;
64
- }
65
- export function readJsonReport(file, fallback) {
66
- if (!existsSync(file))
67
- return { value: fallback, exists: false, valid: true };
68
- try {
69
- return { value: JSON.parse(readFileSync(file, "utf8")), exists: true, valid: true };
70
- }
71
- catch (error) {
72
- return { value: fallback, exists: true, valid: false, error: error.message };
73
- }
74
- }
75
- export function readJson(file, fallback) {
76
- return readJsonReport(file, fallback).value;
77
- }
78
- export function writeJson(file, obj) {
79
- mkdirSync(join(file, ".."), { recursive: true });
80
- const temp = `${file}.tmp-${process.pid}-${Date.now()}`;
81
- try {
82
- writeFileSync(temp, JSON.stringify(obj, null, 2) + "\n");
83
- renameSync(temp, file);
84
- }
85
- finally {
86
- if (existsSync(temp))
87
- rmSync(temp, { force: true });
88
- }
89
- }
@@ -1,68 +0,0 @@
1
- /** Strategist: pick the next untried model to trial, benchmark-aware. */
2
- import { directChatClient } from "./llm.js";
3
- import { parseObj } from "./judge.js";
4
- import { routeKey, routeLabel } from "./routes.js";
5
- export const STRAT_PREFS = [
6
- "anthropic/claude-3.7-sonnet",
7
- "anthropic/claude-3.5-sonnet",
8
- "openai/gpt-4o",
9
- ];
10
- const STRAT_PROMPT = `We are finding the best configured model route for a task via iterative trials.
11
- TASK: {task}
12
- RUBRIC: {rubric}
13
- RELEVANT PUBLIC BENCHMARKS FOR THIS TASK: {benchmarks}
14
- RESULTS SO FAR (model=score): {results}
15
- Pick ONE untried model from the catalog below that is likely to do well on this
16
- task, based on the benchmarks relevant to it. Balance quality against price so
17
- a pareto frontier emerges.
18
- Return ONLY json: {{"next_model": "<route id>", "why": "<one line>"}}
19
- ROUTES (route id | logical model | blended $/1M tokens | ctx):
20
- {catalog}`;
21
- /** Pick the next model to trial, or null when the catalog is exhausted. */
22
- export async function pickNext(key, stratModel, task, rubric, benchmarks, results, cat, tried, maxPrice, failedVendors, extraCandidates, client = directChatClient(key)) {
23
- const entryKey = (c) => c.route ? routeKey(c.route) : c.id;
24
- const ok = (c) => c.priceKnown !== false && c.price <= maxPrice &&
25
- !tried.has(entryKey(c)) &&
26
- c.ctx >= 4096 &&
27
- !failedVendors.has(c.route?.provider ?? c.id.split("/")[0]);
28
- // Benchmark-shortlisted models get priority in the listing.
29
- const pool = [...cat.values()].filter(ok);
30
- if (pool.length === 0)
31
- return null;
32
- const priority = new Set(extraCandidates ?? []);
33
- pool.sort((a, b) => {
34
- const pa = priority.has(a.id) ? 0 : 1;
35
- const pb = priority.has(b.id) ? 0 : 1;
36
- return pa - pb || (a.priceKnown === false ? 1 : 0) - (b.priceKnown === false ? 1 : 0) || a.price - b.price;
37
- });
38
- const lines = pool
39
- .map((c) => `${entryKey(c)} | ${c.route ? routeLabel(c.route) : c.id} | ` +
40
- `${c.priceKnown === false ? "unknown" : c.price.toFixed(2)} | ${c.ctx}${priority.has(c.id) ? " | BENCHMARK" : ""}`)
41
- .join("\n");
42
- const res = results
43
- .map((r) => `${r.model}=${r.score.toFixed(2)}`)
44
- .join(" ") || "none yet";
45
- const { content } = await client.chat(stratModel, STRAT_PROMPT.replace("{task}", task)
46
- .replace("{rubric}", rubric)
47
- .replace("{benchmarks}", benchmarks.join(", "))
48
- .replace("{results}", res)
49
- .replace("{catalog}", lines), 1024, 0.2);
50
- let pick = null;
51
- let why = "unparseable strategist output";
52
- try {
53
- const obj = parseObj(content);
54
- pick = typeof obj.next_model === "string" ? obj.next_model : null;
55
- why = String(obj.why ?? "");
56
- }
57
- catch {
58
- /* fall through to cheap pick */
59
- }
60
- if (pick) {
61
- const entry = cat.get(pick) ?? [...cat.values()].find((candidate) => candidate.id === pick);
62
- if (entry && ok(entry))
63
- return { model: entry.id, route: entry.route, why };
64
- }
65
- const cheapest = pool[0];
66
- return { model: cheapest.id, route: cheapest.route,
67
- why: `fallback pick: cheapest untried (strategist pick ${pick ?? "none"} unusable)` };
68
- }
@@ -1,54 +0,0 @@
1
- /** A pi text task may carry images embedded in its saved session message. */
2
- import { createHash } from "node:crypto";
3
- import { existsSync, readFileSync, statSync } from "node:fs";
4
- import { basename } from "node:path";
5
- import { taskKey } from "./store.js";
6
- /** Recover files Pi expanded into a task message (for example @invoice.pdf). */
7
- export function sessionFiles(text) {
8
- const files = [];
9
- const seen = new Set();
10
- const re = /<file\s+name=["']([^"']+)["'][^>]*>/g;
11
- for (const match of text.matchAll(re)) {
12
- const file = match[1];
13
- if (!file || seen.has(file) || !existsSync(file))
14
- continue;
15
- try {
16
- const stat = statSync(file);
17
- if (!stat.isFile() || stat.size > 100_000_000)
18
- continue;
19
- files.push({ path: file, name: basename(file), size: stat.size,
20
- sha256: createHash("sha256").update(readFileSync(file)).digest("hex") });
21
- seen.add(file);
22
- }
23
- catch { /* inaccessible attachment */ }
24
- }
25
- return files;
26
- }
27
- const SUPPORTED = new Set(["image/jpeg", "image/png", "image/webp", "image/gif"]);
28
- export function sessionImages(content) {
29
- if (!Array.isArray(content))
30
- return [];
31
- const images = [];
32
- for (const part of content) {
33
- if (!part || typeof part !== "object")
34
- return null;
35
- if (part.type === "text")
36
- continue;
37
- if (part.type !== "image" || typeof part.data !== "string" ||
38
- typeof part.mimeType !== "string" || !SUPPORTED.has(part.mimeType) ||
39
- !/^[A-Za-z0-9+/]+={0,2}$/.test(part.data))
40
- return null;
41
- images.push({ type: "image", data: part.data, mimeType: part.mimeType });
42
- }
43
- return images;
44
- }
45
- /** Text-only keys remain compatible; image bytes distinguish same-prompt documents. */
46
- export function taskInputKey(text, images, files = []) {
47
- if (images.length === 0 && files.length === 0)
48
- return taskKey(text);
49
- const normalized = text.toLowerCase().replace(/\s+/g, " ").trim();
50
- const hashes = images.map((image) => `${image.mimeType}:` +
51
- createHash("sha256").update(Buffer.from(image.data, "base64")).digest("hex"));
52
- const fileHashes = files.map((file) => `${file.name}:${file.sha256}`).join(",");
53
- return taskKey(`${normalized}\nimages:${hashes.join(",")}\nfiles:${fileHashes}`);
54
- }
package/dist/traces.js DELETED
@@ -1,127 +0,0 @@
1
- /**
2
- * Trace ingestion: read pi session files (JSONL entry trees) and extract
3
- * per-task observations (model, cost, tokens, latency, errors). Incremental
4
- * via a per-file byte-offset cursor, so the background track never re-reads.
5
- */
6
- import { readdirSync, readFileSync, existsSync, statSync } from "node:fs";
7
- import { homedir } from "node:os";
8
- import { join } from "node:path";
9
- import { paths, readJson, taskKey, writeJson } from "./store.js";
10
- import { meritModelId, modelRoute } from "./routes.js";
11
- export function defaultSessionsDir() {
12
- return join(homedir(), ".pi", "agent", "sessions");
13
- }
14
- function listSessionFiles(dir) {
15
- if (!existsSync(dir))
16
- return [];
17
- const out = [];
18
- for (const entry of readdirSync(dir, { withFileTypes: true })) {
19
- const p = join(dir, entry.name);
20
- if (entry.isDirectory())
21
- out.push(...listSessionFiles(p));
22
- else if (entry.name.endsWith(".jsonl"))
23
- out.push(p);
24
- }
25
- return out;
26
- }
27
- function textOf(content) {
28
- if (typeof content === "string")
29
- return content;
30
- if (Array.isArray(content)) {
31
- return content
32
- .map((b) => (b && typeof b === "object" && b.type === "text"
33
- ? String(b.text ?? "")
34
- : ""))
35
- .filter(Boolean)
36
- .join(" ");
37
- }
38
- return "";
39
- }
40
- /** Extract observations from the entries of one session file. */
41
- export function observationsFromEntries(entries, sessionFile) {
42
- const obs = [];
43
- let current = null;
44
- for (const e of entries) {
45
- if (e.type !== "message" || !e.message)
46
- continue;
47
- const m = e.message;
48
- if (m.role === "user") {
49
- if (current)
50
- obs.push(current);
51
- const text = textOf(m.content);
52
- if (!text.trim()) {
53
- current = null;
54
- continue;
55
- }
56
- current = {
57
- schemaVersion: 1,
58
- taskKey: taskKey(text),
59
- taskLabel: text.replace(/\s+/g, " ").trim().slice(0, 120),
60
- model: null,
61
- costUsd: 0,
62
- tokens: 0,
63
- latencyMs: null,
64
- errors: 0,
65
- ts: e.timestamp ?? new Date().toISOString(),
66
- sessionFile,
67
- };
68
- continue;
69
- }
70
- if (!current)
71
- continue;
72
- if (m.role === "assistant") {
73
- if (!current.model && m.provider && m.model) {
74
- current.model = meritModelId(m.provider, m.model);
75
- current.route = modelRoute(m.provider, m.model);
76
- }
77
- current.costUsd += m.usage?.cost?.total ?? 0;
78
- current.tokens += m.usage?.totalTokens ?? 0;
79
- if (m.stopReason === "error")
80
- current.errors += 1;
81
- if (typeof m.timestamp === "number") {
82
- const start = Date.parse(current.ts);
83
- if (!Number.isNaN(start))
84
- current.latencyMs = Math.max(0, m.timestamp - start);
85
- }
86
- }
87
- else if (m.role === "toolResult") {
88
- const tr = m;
89
- if (tr.isError)
90
- current.errors += 1;
91
- }
92
- }
93
- if (current)
94
- obs.push(current);
95
- return obs;
96
- }
97
- /** Read new bytes of every session file since the last run; returns fresh observations. */
98
- export function ingestNewTraces(sessionsDir = defaultSessionsDir()) {
99
- const cursor = readJson(paths.traceCursor(), { offsets: {} });
100
- const fresh = [];
101
- for (const file of listSessionFiles(sessionsDir)) {
102
- const size = statSync(file).size;
103
- const from = cursor.offsets[file] ?? 0;
104
- if (size <= from)
105
- continue;
106
- const buf = readFileSync(file);
107
- const chunk = buf.subarray(from).toString("utf8");
108
- // Drop a trailing partial line; it will be re-read next tick.
109
- const lastNl = chunk.lastIndexOf("\n");
110
- const complete = lastNl >= 0 ? chunk.slice(0, lastNl + 1) : "";
111
- const entries = [];
112
- for (const line of complete.split("\n")) {
113
- if (!line.trim())
114
- continue;
115
- try {
116
- entries.push(JSON.parse(line));
117
- }
118
- catch {
119
- /* skip corrupt line */
120
- }
121
- }
122
- fresh.push(...observationsFromEntries(entries, file));
123
- cursor.offsets[file] = from + Buffer.byteLength(complete);
124
- }
125
- writeJson(paths.traceCursor(), cursor);
126
- return fresh;
127
- }