homegraph 1.5.0 → 1.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +305 -305
  3. package/dist/bin/command-supervision.d.ts.map +1 -1
  4. package/dist/bin/command-supervision.js +7 -4
  5. package/dist/bin/command-supervision.js.map +1 -1
  6. package/dist/bin/homegraph.js +9 -9
  7. package/dist/db/index.js +36 -36
  8. package/dist/db/migrations.js +37 -37
  9. package/dist/db/queries.js +156 -156
  10. package/dist/db/schema.sql +203 -203
  11. package/dist/directory.js +5 -5
  12. package/dist/extraction/languages/arkts-viewtree.d.ts +2 -4
  13. package/dist/extraction/languages/arkts-viewtree.d.ts.map +1 -1
  14. package/dist/extraction/languages/arkts-viewtree.js +6 -21
  15. package/dist/extraction/languages/arkts-viewtree.js.map +1 -1
  16. package/dist/extraction/languages/arkts.d.ts +16 -6
  17. package/dist/extraction/languages/arkts.d.ts.map +1 -1
  18. package/dist/extraction/languages/arkts.js +174 -21
  19. package/dist/extraction/languages/arkts.js.map +1 -1
  20. package/dist/extraction/wasm/tree-sitter-c_sharp.wasm +0 -0
  21. package/dist/extraction/wasm/tree-sitter-cfml.wasm +0 -0
  22. package/dist/extraction/wasm/tree-sitter-cfquery.wasm +0 -0
  23. package/dist/extraction/wasm/tree-sitter-cfscript.wasm +0 -0
  24. package/dist/extraction/wasm/tree-sitter-cobol.wasm +0 -0
  25. package/dist/extraction/wasm/tree-sitter-erlang.wasm +0 -0
  26. package/dist/extraction/wasm/tree-sitter-nix.wasm +0 -0
  27. package/dist/extraction/wasm/tree-sitter-pascal.wasm +0 -0
  28. package/dist/extraction/wasm/tree-sitter-vbnet.wasm +0 -0
  29. package/dist/installer/instructions-template.js +9 -9
  30. package/dist/mcp/liveness-watchdog.d.ts +18 -0
  31. package/dist/mcp/liveness-watchdog.d.ts.map +1 -1
  32. package/dist/mcp/liveness-watchdog.js +185 -73
  33. package/dist/mcp/liveness-watchdog.js.map +1 -1
  34. package/dist/mcp/server-instructions.js +47 -47
  35. package/dist/reasoning/reasoner.js +32 -32
  36. package/dist/spec/db/commit-node.js +4 -4
  37. package/dist/spec/db/fragment-node.js +10 -10
  38. package/dist/spec/db/fts.js +8 -8
  39. package/dist/spec/db/relations.js +88 -88
  40. package/dist/spec/db/schema.js +6 -6
  41. package/dist/spec/db/schema.sql +121 -121
  42. package/dist/spec/db/spec-node.js +4 -4
  43. package/dist/spec/llm/prompts.js +53 -53
  44. package/dist/spec/utils.d.ts +16 -4
  45. package/dist/spec/utils.d.ts.map +1 -1
  46. package/dist/spec/utils.js +58 -6
  47. package/dist/spec/utils.js.map +1 -1
  48. package/package.json +62 -62
  49. package/scripts/_tmp-cfwk-resolve.log +0 -0
  50. package/scripts/_tmp-cfwk-sig.log +0 -0
  51. package/scripts/_tmp-cfwk-vt.log +0 -0
  52. package/scripts/add-lang/bench.sh +60 -60
  53. package/scripts/add-lang/check-grammar.mjs +75 -75
  54. package/scripts/add-lang/dump-ast.mjs +103 -103
  55. package/scripts/add-lang/verify-extraction.mjs +70 -70
  56. package/scripts/agent-eval/ab-adoption.sh +91 -91
  57. package/scripts/agent-eval/ab-hook.sh +86 -86
  58. package/scripts/agent-eval/ab-impl.sh +78 -78
  59. package/scripts/agent-eval/ab-new-vs-baseline.sh +102 -102
  60. package/scripts/agent-eval/ab-sufficiency.sh +78 -78
  61. package/scripts/agent-eval/arms-F.sh +21 -21
  62. package/scripts/agent-eval/arms-matrix.sh +37 -37
  63. package/scripts/agent-eval/audit.sh +68 -68
  64. package/scripts/agent-eval/bench-readme.sh +28 -28
  65. package/scripts/agent-eval/bench-why-repo.sh +22 -22
  66. package/scripts/agent-eval/block-read-hook.sh +19 -19
  67. package/scripts/agent-eval/hook-settings.json +15 -15
  68. package/scripts/agent-eval/itrun.sh +120 -120
  69. package/scripts/agent-eval/offload-eval-3arm.sh +72 -72
  70. package/scripts/agent-eval/offload-eval-cost.mjs +133 -133
  71. package/scripts/agent-eval/offload-eval-effort.mjs +108 -108
  72. package/scripts/agent-eval/offload-eval-frontload-matrix.sh +25 -25
  73. package/scripts/agent-eval/offload-eval-frontload.sh +47 -47
  74. package/scripts/agent-eval/offload-eval-ground-truth.json +18 -18
  75. package/scripts/agent-eval/offload-eval-hook.mjs +84 -84
  76. package/scripts/agent-eval/offload-eval-judge.mjs +103 -103
  77. package/scripts/agent-eval/offload-eval-matrix.sh +20 -20
  78. package/scripts/agent-eval/offload-eval-metrics.mjs +94 -94
  79. package/scripts/agent-eval/offload-eval-refs1.sh +50 -50
  80. package/scripts/agent-eval/offload-eval-setup.sh +24 -24
  81. package/scripts/agent-eval/offload-eval-styles.sh +71 -71
  82. package/scripts/agent-eval/offload-eval-summarize.mjs +68 -68
  83. package/scripts/agent-eval/offload-eval.md +76 -76
  84. package/scripts/agent-eval/parse-arms.mjs +116 -116
  85. package/scripts/agent-eval/parse-bench-readme.mjs +84 -84
  86. package/scripts/agent-eval/parse-run.mjs +45 -45
  87. package/scripts/agent-eval/parse-session.mjs +93 -93
  88. package/scripts/agent-eval/probe-context.mjs +21 -21
  89. package/scripts/agent-eval/probe-explore.mjs +40 -40
  90. package/scripts/agent-eval/probe-node.mjs +20 -20
  91. package/scripts/agent-eval/probe-sweep.mjs +119 -119
  92. package/scripts/agent-eval/probe-trace.mjs +20 -20
  93. package/scripts/agent-eval/redirect-read-hook.sh +38 -38
  94. package/scripts/agent-eval/repro-concurrent-explore.mjs +119 -119
  95. package/scripts/agent-eval/repro-daemon-clients.mjs +125 -125
  96. package/scripts/agent-eval/run-agent.sh +34 -34
  97. package/scripts/agent-eval/run-all.sh +75 -75
  98. package/scripts/agent-eval/run-arms.sh +56 -56
  99. package/scripts/agent-eval/seq-matrix.mjs +137 -137
  100. package/scripts/bench-arkts-init-rss.log +0 -0
  101. package/scripts/build-bundle.sh +123 -123
  102. package/scripts/exp_boundary_eval/README.md +247 -247
  103. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-310.pyc +0 -0
  104. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-38.pyc +0 -0
  105. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-310.pyc +0 -0
  106. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-38.pyc +0 -0
  107. package/scripts/exp_boundary_eval/__pycache__/deveco_arm.cpython-38.pyc +0 -0
  108. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-310.pyc +0 -0
  109. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-38.pyc +0 -0
  110. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-310.pyc +0 -0
  111. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-38.pyc +0 -0
  112. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-310.pyc +0 -0
  113. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-38.pyc +0 -0
  114. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-310.pyc +0 -0
  115. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-38.pyc +0 -0
  116. package/scripts/exp_boundary_eval/__pycache__/win_mcp_launcher.cpython-38.pyc +0 -0
  117. package/scripts/exp_boundary_eval/_test_mcp_chain.py +78 -78
  118. package/scripts/exp_boundary_eval/_test_stdin.py +8 -8
  119. package/scripts/exp_boundary_eval/_utils.py +1116 -1116
  120. package/scripts/exp_boundary_eval/analyze.py +1313 -1313
  121. package/scripts/exp_boundary_eval/deveco_arm.py +519 -519
  122. package/scripts/exp_boundary_eval/run_all.py +378 -378
  123. package/scripts/exp_boundary_eval/run_one.py +165 -165
  124. package/scripts/exp_boundary_eval/run_session.py +158 -158
  125. package/scripts/exp_boundary_eval/setup.py +120 -120
  126. package/scripts/exp_boundary_eval/win_mcp_launcher.py +73 -73
  127. package/scripts/exp_boundary_eval/win_mcp_stdio_wrap.js +36 -36
  128. package/scripts/exp_boundary_eval/win_node_launcher.py +24 -24
  129. package/scripts/extract-release-notes.mjs +130 -130
  130. package/scripts/local-install.sh +41 -41
  131. package/scripts/npm-sdk.js +75 -75
  132. package/scripts/npm-shim.js +275 -275
  133. package/scripts/ohos-sdk-publish.mjs +133 -133
  134. package/scripts/pack-npm.sh +119 -119
  135. package/scripts/prepare-release.mjs +270 -270
  136. package/scripts/probe-arkts-mem-why-run.log +0 -0
  137. package/scripts/probe-banner-livecard-ir.log +0 -0
  138. package/scripts/probe-cfwk-ast.stderr.log +0 -0
  139. package/scripts/probe-cfwk-ast.stdout.log +0 -0
  140. package/scripts/probe-cfwk-attr-shape.log +0 -0
  141. package/scripts/probe-cfwk-cfgdump.stderr.log +0 -0
  142. package/scripts/probe-cfwk-cfgdump.stdout.log +0 -0
  143. package/scripts/probe-cfwk-cvc.stderr.log +0 -0
  144. package/scripts/probe-cfwk-cvc.stdout.log +0 -0
  145. package/scripts/probe-cfwk-diag2.log +0 -0
  146. package/scripts/probe-cfwk-diag3.log +0 -0
  147. package/scripts/probe-cfwk-fedbg.stderr.log +0 -0
  148. package/scripts/probe-cfwk-fedbg.stdout.log +0 -0
  149. package/scripts/probe-cfwk-fileresult.log +0 -0
  150. package/scripts/probe-cfwk-fix.stderr.log +0 -0
  151. package/scripts/probe-cfwk-fix.stdout.log +0 -0
  152. package/scripts/probe-cfwk-fix2.stderr.log +0 -0
  153. package/scripts/probe-cfwk-fix2.stdout.log +0 -0
  154. package/scripts/probe-cfwk-fix3.stderr.log +0 -0
  155. package/scripts/probe-cfwk-fix3.stdout.log +0 -0
  156. package/scripts/probe-cfwk-foreach.log +0 -0
  157. package/scripts/probe-cfwk-getmethod-throw.log +0 -0
  158. package/scripts/probe-cfwk-hg-extract.log +0 -0
  159. package/scripts/probe-cfwk-pr1003-noprior.stderr.log +0 -0
  160. package/scripts/probe-cfwk-pr1003-noprior.stdout.log +0 -0
  161. package/scripts/probe-cfwk-pr1003.stderr.log +0 -0
  162. package/scripts/probe-cfwk-pr1003.stdout.log +0 -0
  163. package/scripts/probe-cfwk-preroot-noprior.stderr.log +0 -0
  164. package/scripts/probe-cfwk-preroot-noprior.stdout.log +0 -0
  165. package/scripts/probe-cfwk-preroot-prior.stderr.log +0 -0
  166. package/scripts/probe-cfwk-preroot-prior.stdout.log +0 -0
  167. package/scripts/probe-cfwk-preroot-skipstate.stderr.log +0 -0
  168. package/scripts/probe-cfwk-preroot-skipstate.stdout.log +0 -0
  169. package/scripts/probe-cfwk-resolve-sim.log +0 -0
  170. package/scripts/probe-cfwk-tree-shape.log +0 -0
  171. package/scripts/probe-cfwk-walk-abort.log +0 -0
  172. package/scripts/probe-cfwk.log +0 -0
  173. package/scripts/probe-force-index.log +0 -0
  174. package/scripts/probe-no-force-index.log +0 -0
  175. package/scripts/probe-samefile-ir.stderr.log +0 -0
  176. package/scripts/probe-samefile-ir.stdout.log +224 -0
  177. package/scripts/probe-sdk-vs-project.log +0 -0
  178. package/scripts/probe-viewtree-downgrade.log +0 -0
  179. package/dist/arkts/ohos-api-index.d.ts +0 -15
  180. package/dist/arkts/ohos-api-index.d.ts.map +0 -1
  181. package/dist/arkts/ohos-api-index.js +0 -190
  182. package/dist/arkts/ohos-api-index.js.map +0 -1
  183. package/dist/arkts/ohos-sdk-input.d.ts +0 -36
  184. package/dist/arkts/ohos-sdk-input.d.ts.map +0 -1
  185. package/dist/arkts/ohos-sdk-input.js +0 -214
  186. package/dist/arkts/ohos-sdk-input.js.map +0 -1
  187. package/dist/extraction/languages/arkts-state-decorators.d.ts +0 -13
  188. package/dist/extraction/languages/arkts-state-decorators.d.ts.map +0 -1
  189. package/dist/extraction/languages/arkts-state-decorators.js +0 -26
  190. package/dist/extraction/languages/arkts-state-decorators.js.map +0 -1
  191. package/dist/extraction/languages/ohos-api-consumer.d.ts +0 -34
  192. package/dist/extraction/languages/ohos-api-consumer.d.ts.map +0 -1
  193. package/dist/extraction/languages/ohos-api-consumer.js +0 -283
  194. package/dist/extraction/languages/ohos-api-consumer.js.map +0 -1
  195. package/dist/spec/build/git-scanner.d.ts +0 -93
  196. package/dist/spec/build/git-scanner.d.ts.map +0 -1
  197. package/dist/spec/build/git-scanner.js +0 -254
  198. package/dist/spec/build/git-scanner.js.map +0 -1
  199. package/dist/spec/git-utils.d.ts +0 -8
  200. package/dist/spec/git-utils.d.ts.map +0 -1
  201. package/dist/spec/git-utils.js +0 -14
  202. package/dist/spec/git-utils.js.map +0 -1
  203. package/dist/spec/mine/clusterer.d.ts +0 -63
  204. package/dist/spec/mine/clusterer.d.ts.map +0 -1
  205. package/dist/spec/mine/clusterer.js +0 -904
  206. package/dist/spec/mine/clusterer.js.map +0 -1
  207. package/dist/spec/mine/progress-handler.d.ts +0 -22
  208. package/dist/spec/mine/progress-handler.d.ts.map +0 -1
  209. package/dist/spec/mine/progress-handler.js +0 -108
  210. package/dist/spec/mine/progress-handler.js.map +0 -1
  211. package/dist/spec/mine/progress.d.ts +0 -23
  212. package/dist/spec/mine/progress.d.ts.map +0 -1
  213. package/dist/spec/mine/progress.js +0 -12
  214. package/dist/spec/mine/progress.js.map +0 -1
@@ -1,133 +1,133 @@
1
- #!/usr/bin/env node
2
- // Cost/token analysis for the 3-arm offload eval, with a MAIN-vs-SUBAGENT split.
3
- //
4
- // The explore-subagent question. With delegation ALLOWED, the nocg arm spawns a
5
- // Claude Code Explore subagent; the homegraph arms do all work in the main agent.
6
- // Two facts make naive accounting wrong:
7
- // 1. The Explore subagent runs on HAIKU 4.5; the main agent on SONNET 4.6.
8
- // So per-token cost differs ~3x between them — you cannot price both the same.
9
- // 2. The subagent's consumption is ~95% cache-reads. At Haiku's $0.10/MTok
10
- // cache-read rate, a huge TOKEN volume is a small DOLLAR cost.
11
- //
12
- // Rather than re-derive cost from raw token counts (and guess the cache TTL —
13
- // Claude Code uses 1-hour ephemeral cache here, 2x write, not 5-min), we read
14
- // Claude Code's OWN authoritative accounting from the `result` event:
15
- // result.modelUsage[model].costUSD — per-model cost CC itself billed
16
- // result.total_cost_usd — their sum (INCLUDES the Haiku subagent;
17
- // the handoff's "excludes subagent" was wrong)
18
- // The model split IS the agent split here: sonnet => main, haiku => Explore subagent
19
- // (only nocg spawns one, and only nocg shows haiku usage). Token volume is still
20
- // summed per-model from modelUsage for the separate "tokens" story.
21
- //
22
- // Usage: offload-eval-cost.mjs <runs-dir> <repo> [reps]
23
- // e.g. offload-eval-cost.mjs /tmp/cg-offload-eval/runs trezor 3
24
- import { readFileSync, existsSync } from 'fs';
25
-
26
- const MAIN_TIER = /sonnet/; // main agent
27
- const SUB_TIER = /haiku/; // Claude Code Explore subagent
28
-
29
- const [,, runsDir, repo, repsArg] = process.argv;
30
- if (!runsDir || !repo) { console.error('usage: offload-eval-cost.mjs <runs-dir> <repo> [reps] (env ARMS=nocg,raw,offload)'); process.exit(1); }
31
- const REPS = Number(repsArg || 3);
32
- // Arms to analyze (file stems `<repo>-<arm>-<rep>.jsonl`). Override for the style A/B:
33
- // ARMS=raw,refs,map,src. nocg's Haiku subagent is the only sub-tier; the rest are main-only.
34
- const ARMS = (process.env.ARMS || 'nocg,raw,offload').split(',').map((s) => s.trim()).filter(Boolean);
35
-
36
- const toks = (u) => (u.inputTokens||0)+(u.outputTokens||0)+(u.cacheReadInputTokens||0)+(u.cacheCreationInputTokens||0);
37
-
38
- function analyzeRun(file) {
39
- let result = null, agentCalls = 0;
40
- const tools = {}, subPids = new Set();
41
- for (const line of readFileSync(file, 'utf8').split('\n')) {
42
- if (!line) continue;
43
- let e; try { e = JSON.parse(line); } catch { continue; }
44
- if (e.parent_tool_use_id && e.message?.usage) subPids.add(e.parent_tool_use_id);
45
- if (e.type === 'assistant' && Array.isArray(e.message?.content))
46
- for (const b of e.message.content)
47
- if (b.type === 'tool_use') { tools[b.name] = (tools[b.name]||0)+1; if (b.name === 'Agent') agentCalls++; }
48
- if (e.type === 'result') result = e;
49
- }
50
- // Authoritative cost + tokens from Claude Code's per-model accounting.
51
- const mu = result?.modelUsage || {};
52
- const main = { cost: 0, tok: 0 }, sub = { cost: 0, tok: 0 };
53
- for (const [model, u] of Object.entries(mu)) {
54
- const bucket = SUB_TIER.test(model) ? sub : main; // sonnet/anything-else => main
55
- bucket.cost += u.costUSD || 0;
56
- bucket.tok += toks(u);
57
- }
58
- return {
59
- main, sub, subagents: subPids.size, agentCalls,
60
- ccTotal: result?.total_cost_usd ?? null,
61
- ok: result?.subtype === 'success',
62
- durationSec: result?.duration_ms ? +(result.duration_ms/1000).toFixed(1) : null,
63
- models: Object.keys(mu), tools,
64
- };
65
- }
66
-
67
- const k = (n) => (n/1000).toFixed(0).padStart(5) + 'K';
68
- const d = (n) => '$' + n.toFixed(3);
69
- const cost = (b) => b.cost;
70
- const tot = (b) => b.tok;
71
-
72
- const byArm = {};
73
- for (const arm of ARMS) {
74
- const runs = [];
75
- for (let r = 1; r <= REPS; r++) {
76
- const f = `${runsDir}/${repo}-${arm}-${r}.jsonl`;
77
- if (existsSync(f)) runs.push({ rep: r, ...analyzeRun(f) });
78
- }
79
- byArm[arm] = runs;
80
- }
81
-
82
- // Per-run detail. Cost is Claude Code's own modelUsage.costUSD (authoritative,
83
- // per-model pricing + correct cache TTL). MAIN=Sonnet, SUB=Haiku Explore subagent.
84
- // cc-check: main$+sub$ must equal result.total_cost_usd (delta should be ~0).
85
- console.log(`\n=== ${repo}: per-run main(Sonnet)/sub(Haiku) split — Claude Code's own cost accounting ===`);
86
- console.log('arm rep | subAg | MAIN(sonnet) tok / $ | SUB(haiku) tok / $ | TOTAL tok / $ | cc_total Δ | dur reads');
87
- for (const arm of ARMS) for (const r of byArm[arm]) {
88
- const mC = cost(r.main), sC = cost(r.sub), mT = tot(r.main), sT = tot(r.sub);
89
- const reads = r.tools['Read'] || 0, grep = (r.tools['Grep']||0)+(r.tools['Bash']||0)+(r.tools['Glob']||0);
90
- const explore = r.tools['mcp__homegraph__homegraph_explore'] || 0;
91
- const delta = (mC + sC) - (r.ccTotal || 0); // should be ~0
92
- console.log(
93
- `${arm.padEnd(8)} #${r.rep} | ${String(r.subagents).padStart(2)} | ${k(mT)} ${d(mC).padStart(7)} | ${k(sT)} ${d(sC).padStart(7)} | ${k(mT+sT)} ${d(mC+sC).padStart(7)} | ${d(r.ccTotal||0).padStart(7)} ${(delta>=0?'+':'')+delta.toFixed(4)} | ${String(r.durationSec).padStart(5)} r=${reads} g=${grep} x=${explore}`
94
- );
95
- }
96
-
97
- // Per-arm means
98
- const mean = (arr, f) => arr.length ? arr.reduce((s,x)=>s+f(x),0)/arr.length : 0;
99
- console.log(`\n=== ${repo}: per-arm MEANS (n per arm) ===`);
100
- console.log('arm n | main $ sub $ TOTAL $ | main tok sub tok TOTAL tok | %$ in sub | %tok in sub');
101
- for (const arm of ARMS) {
102
- const runs = byArm[arm]; if (!runs.length) continue;
103
- const mC = mean(runs, r=>cost(r.main)), sC = mean(runs, r=>cost(r.sub));
104
- const mT = mean(runs, r=>tot(r.main)), sT = mean(runs, r=>tot(r.sub));
105
- const pctSubC = (mC+sC) ? (100*sC/(mC+sC)) : 0;
106
- const pctSubT = (mT+sT) ? (100*sT/(mT+sT)) : 0;
107
- console.log(
108
- `${arm.padEnd(8)} ${runs.length} | ${d(mC).padStart(7)} ${d(sC).padStart(7)} ${d(mC+sC).padStart(7)} | ${k(mT)} ${k(sT)} ${k(mT+sT)} | ${pctSubC.toFixed(0).padStart(3)}% | ${pctSubT.toFixed(0).padStart(3)}%`
109
- );
110
- }
111
-
112
- // Headline ladders — cost, tokens, duration, all vs a baseline (nocg if present, else first arm).
113
- console.log(`\n=== Ladders (mean, incl. subagent) ===`);
114
- const totals = ARMS.map(a => ({ a, c: mean(byArm[a], r=>cost(r.main)+cost(r.sub)), t: mean(byArm[a], r=>tot(r.main)+tot(r.sub)) })).filter(x=>byArm[x.a].length);
115
- const base = totals.find(x=>x.a==='nocg') ?? totals[0];
116
- const bn = base?.a ?? '?';
117
- console.log(` COST (vs ${bn}):`);
118
- for (const x of totals) {
119
- const vs = base && base.c ? ` (${((x.c/base.c-1)*100>=0?'+':'')}${((x.c/base.c-1)*100).toFixed(0)}%)` : '';
120
- console.log(` ${x.a.padEnd(8)} ${d(x.c)}${vs}`);
121
- }
122
- console.log(` TOKENS (vs ${bn}):`);
123
- for (const x of totals) {
124
- const vs = base && base.t ? ` (${((x.t/base.t-1)*100>=0?'+':'')}${((x.t/base.t-1)*100).toFixed(0)}%)` : '';
125
- console.log(` ${x.a.padEnd(8)} ${k(x.t)}${vs}`);
126
- }
127
- console.log(` DURATION (wall-clock, vs ${bn}):`);
128
- const durs = ARMS.map(a => ({ a, s: mean(byArm[a].filter(r=>r.durationSec!=null), r=>r.durationSec) })).filter(x=>byArm[x.a].length);
129
- const dbase = durs.find(x=>x.a==='nocg') ?? durs[0];
130
- for (const x of durs) {
131
- const vs = dbase && dbase.s ? ` (${((x.s/dbase.s-1)*100>=0?'+':'')}${((x.s/dbase.s-1)*100).toFixed(0)}%)` : '';
132
- console.log(` ${x.a.padEnd(8)} ${x.s.toFixed(0)}s${vs}`);
133
- }
1
+ #!/usr/bin/env node
2
+ // Cost/token analysis for the 3-arm offload eval, with a MAIN-vs-SUBAGENT split.
3
+ //
4
+ // The explore-subagent question. With delegation ALLOWED, the nocg arm spawns a
5
+ // Claude Code Explore subagent; the homegraph arms do all work in the main agent.
6
+ // Two facts make naive accounting wrong:
7
+ // 1. The Explore subagent runs on HAIKU 4.5; the main agent on SONNET 4.6.
8
+ // So per-token cost differs ~3x between them — you cannot price both the same.
9
+ // 2. The subagent's consumption is ~95% cache-reads. At Haiku's $0.10/MTok
10
+ // cache-read rate, a huge TOKEN volume is a small DOLLAR cost.
11
+ //
12
+ // Rather than re-derive cost from raw token counts (and guess the cache TTL —
13
+ // Claude Code uses 1-hour ephemeral cache here, 2x write, not 5-min), we read
14
+ // Claude Code's OWN authoritative accounting from the `result` event:
15
+ // result.modelUsage[model].costUSD — per-model cost CC itself billed
16
+ // result.total_cost_usd — their sum (INCLUDES the Haiku subagent;
17
+ // the handoff's "excludes subagent" was wrong)
18
+ // The model split IS the agent split here: sonnet => main, haiku => Explore subagent
19
+ // (only nocg spawns one, and only nocg shows haiku usage). Token volume is still
20
+ // summed per-model from modelUsage for the separate "tokens" story.
21
+ //
22
+ // Usage: offload-eval-cost.mjs <runs-dir> <repo> [reps]
23
+ // e.g. offload-eval-cost.mjs /tmp/cg-offload-eval/runs trezor 3
24
+ import { readFileSync, existsSync } from 'fs';
25
+
26
+ const MAIN_TIER = /sonnet/; // main agent
27
+ const SUB_TIER = /haiku/; // Claude Code Explore subagent
28
+
29
+ const [,, runsDir, repo, repsArg] = process.argv;
30
+ if (!runsDir || !repo) { console.error('usage: offload-eval-cost.mjs <runs-dir> <repo> [reps] (env ARMS=nocg,raw,offload)'); process.exit(1); }
31
+ const REPS = Number(repsArg || 3);
32
+ // Arms to analyze (file stems `<repo>-<arm>-<rep>.jsonl`). Override for the style A/B:
33
+ // ARMS=raw,refs,map,src. nocg's Haiku subagent is the only sub-tier; the rest are main-only.
34
+ const ARMS = (process.env.ARMS || 'nocg,raw,offload').split(',').map((s) => s.trim()).filter(Boolean);
35
+
36
+ const toks = (u) => (u.inputTokens||0)+(u.outputTokens||0)+(u.cacheReadInputTokens||0)+(u.cacheCreationInputTokens||0);
37
+
38
+ function analyzeRun(file) {
39
+ let result = null, agentCalls = 0;
40
+ const tools = {}, subPids = new Set();
41
+ for (const line of readFileSync(file, 'utf8').split('\n')) {
42
+ if (!line) continue;
43
+ let e; try { e = JSON.parse(line); } catch { continue; }
44
+ if (e.parent_tool_use_id && e.message?.usage) subPids.add(e.parent_tool_use_id);
45
+ if (e.type === 'assistant' && Array.isArray(e.message?.content))
46
+ for (const b of e.message.content)
47
+ if (b.type === 'tool_use') { tools[b.name] = (tools[b.name]||0)+1; if (b.name === 'Agent') agentCalls++; }
48
+ if (e.type === 'result') result = e;
49
+ }
50
+ // Authoritative cost + tokens from Claude Code's per-model accounting.
51
+ const mu = result?.modelUsage || {};
52
+ const main = { cost: 0, tok: 0 }, sub = { cost: 0, tok: 0 };
53
+ for (const [model, u] of Object.entries(mu)) {
54
+ const bucket = SUB_TIER.test(model) ? sub : main; // sonnet/anything-else => main
55
+ bucket.cost += u.costUSD || 0;
56
+ bucket.tok += toks(u);
57
+ }
58
+ return {
59
+ main, sub, subagents: subPids.size, agentCalls,
60
+ ccTotal: result?.total_cost_usd ?? null,
61
+ ok: result?.subtype === 'success',
62
+ durationSec: result?.duration_ms ? +(result.duration_ms/1000).toFixed(1) : null,
63
+ models: Object.keys(mu), tools,
64
+ };
65
+ }
66
+
67
+ const k = (n) => (n/1000).toFixed(0).padStart(5) + 'K';
68
+ const d = (n) => '$' + n.toFixed(3);
69
+ const cost = (b) => b.cost;
70
+ const tot = (b) => b.tok;
71
+
72
+ const byArm = {};
73
+ for (const arm of ARMS) {
74
+ const runs = [];
75
+ for (let r = 1; r <= REPS; r++) {
76
+ const f = `${runsDir}/${repo}-${arm}-${r}.jsonl`;
77
+ if (existsSync(f)) runs.push({ rep: r, ...analyzeRun(f) });
78
+ }
79
+ byArm[arm] = runs;
80
+ }
81
+
82
+ // Per-run detail. Cost is Claude Code's own modelUsage.costUSD (authoritative,
83
+ // per-model pricing + correct cache TTL). MAIN=Sonnet, SUB=Haiku Explore subagent.
84
+ // cc-check: main$+sub$ must equal result.total_cost_usd (delta should be ~0).
85
+ console.log(`\n=== ${repo}: per-run main(Sonnet)/sub(Haiku) split — Claude Code's own cost accounting ===`);
86
+ console.log('arm rep | subAg | MAIN(sonnet) tok / $ | SUB(haiku) tok / $ | TOTAL tok / $ | cc_total Δ | dur reads');
87
+ for (const arm of ARMS) for (const r of byArm[arm]) {
88
+ const mC = cost(r.main), sC = cost(r.sub), mT = tot(r.main), sT = tot(r.sub);
89
+ const reads = r.tools['Read'] || 0, grep = (r.tools['Grep']||0)+(r.tools['Bash']||0)+(r.tools['Glob']||0);
90
+ const explore = r.tools['mcp__homegraph__homegraph_explore'] || 0;
91
+ const delta = (mC + sC) - (r.ccTotal || 0); // should be ~0
92
+ console.log(
93
+ `${arm.padEnd(8)} #${r.rep} | ${String(r.subagents).padStart(2)} | ${k(mT)} ${d(mC).padStart(7)} | ${k(sT)} ${d(sC).padStart(7)} | ${k(mT+sT)} ${d(mC+sC).padStart(7)} | ${d(r.ccTotal||0).padStart(7)} ${(delta>=0?'+':'')+delta.toFixed(4)} | ${String(r.durationSec).padStart(5)} r=${reads} g=${grep} x=${explore}`
94
+ );
95
+ }
96
+
97
+ // Per-arm means
98
+ const mean = (arr, f) => arr.length ? arr.reduce((s,x)=>s+f(x),0)/arr.length : 0;
99
+ console.log(`\n=== ${repo}: per-arm MEANS (n per arm) ===`);
100
+ console.log('arm n | main $ sub $ TOTAL $ | main tok sub tok TOTAL tok | %$ in sub | %tok in sub');
101
+ for (const arm of ARMS) {
102
+ const runs = byArm[arm]; if (!runs.length) continue;
103
+ const mC = mean(runs, r=>cost(r.main)), sC = mean(runs, r=>cost(r.sub));
104
+ const mT = mean(runs, r=>tot(r.main)), sT = mean(runs, r=>tot(r.sub));
105
+ const pctSubC = (mC+sC) ? (100*sC/(mC+sC)) : 0;
106
+ const pctSubT = (mT+sT) ? (100*sT/(mT+sT)) : 0;
107
+ console.log(
108
+ `${arm.padEnd(8)} ${runs.length} | ${d(mC).padStart(7)} ${d(sC).padStart(7)} ${d(mC+sC).padStart(7)} | ${k(mT)} ${k(sT)} ${k(mT+sT)} | ${pctSubC.toFixed(0).padStart(3)}% | ${pctSubT.toFixed(0).padStart(3)}%`
109
+ );
110
+ }
111
+
112
+ // Headline ladders — cost, tokens, duration, all vs a baseline (nocg if present, else first arm).
113
+ console.log(`\n=== Ladders (mean, incl. subagent) ===`);
114
+ const totals = ARMS.map(a => ({ a, c: mean(byArm[a], r=>cost(r.main)+cost(r.sub)), t: mean(byArm[a], r=>tot(r.main)+tot(r.sub)) })).filter(x=>byArm[x.a].length);
115
+ const base = totals.find(x=>x.a==='nocg') ?? totals[0];
116
+ const bn = base?.a ?? '?';
117
+ console.log(` COST (vs ${bn}):`);
118
+ for (const x of totals) {
119
+ const vs = base && base.c ? ` (${((x.c/base.c-1)*100>=0?'+':'')}${((x.c/base.c-1)*100).toFixed(0)}%)` : '';
120
+ console.log(` ${x.a.padEnd(8)} ${d(x.c)}${vs}`);
121
+ }
122
+ console.log(` TOKENS (vs ${bn}):`);
123
+ for (const x of totals) {
124
+ const vs = base && base.t ? ` (${((x.t/base.t-1)*100>=0?'+':'')}${((x.t/base.t-1)*100).toFixed(0)}%)` : '';
125
+ console.log(` ${x.a.padEnd(8)} ${k(x.t)}${vs}`);
126
+ }
127
+ console.log(` DURATION (wall-clock, vs ${bn}):`);
128
+ const durs = ARMS.map(a => ({ a, s: mean(byArm[a].filter(r=>r.durationSec!=null), r=>r.durationSec) })).filter(x=>byArm[x.a].length);
129
+ const dbase = durs.find(x=>x.a==='nocg') ?? durs[0];
130
+ for (const x of durs) {
131
+ const vs = dbase && dbase.s ? ` (${((x.s/dbase.s-1)*100>=0?'+':'')}${((x.s/dbase.s-1)*100).toFixed(0)}%)` : '';
132
+ console.log(` ${x.a.padEnd(8)} ${x.s.toFixed(0)}s${vs}`);
133
+ }
@@ -1,108 +1,108 @@
1
- #!/usr/bin/env node
2
- // Effort A/B — does HOMEGRAPH_OFFLOAD_EFFORT=high improve offload SYNTHESIS FIDELITY vs low?
3
- // Probe-based (no agent): for each repo × effort × rep, run homegraph_explore with the offload
4
- // ON on the canonical question, capture the synthesized answer + AI tokens/cost/latency, then
5
- // Sonnet-judge that answer's fidelity vs source-verified ground truth. Isolates the synthesis
6
- // from agent/adoption noise. Requires `homegraph login` (managed offload) + indexed repos.
7
- //
8
- // Env: REPS (default 3) · CG_ENGINE (engine repo) · AGENT_EVAL_OUT (repos under /repos) · CONC (judge concurrency)
9
- import { pathToFileURL, fileURLToPath } from 'node:url';
10
- import { resolve, dirname, join } from 'node:path';
11
- import { readFileSync, writeFileSync, existsSync, rmSync } from 'node:fs';
12
- import { execFile } from 'node:child_process';
13
- import { tmpdir } from 'node:os';
14
-
15
- const HERE = dirname(fileURLToPath(import.meta.url));
16
- const ENGINE = process.env.CG_ENGINE || resolve(HERE, '..', '..');
17
- const OUT = process.env.AGENT_EVAL_OUT || '/tmp/cg-offload-eval';
18
- const REPOS = join(OUT, 'repos');
19
- const GT = JSON.parse(readFileSync(resolve(HERE, 'offload-eval-ground-truth.json'), 'utf8'));
20
- const REPS = Number(process.env.REPS || 3);
21
- const CONC = Number(process.env.CONC || 4);
22
- const EFFORTS = (process.env.EFFORTS_FILTER || 'low,high').split(',');
23
- const ONLY = process.env.REPOS_FILTER ? new Set(process.env.REPOS_FILTER.split(',')) : null;
24
- const TIER = { mtkruto: 'small', postybirb: 'medium', shapeshift: 'complex', trezor: 'large' };
25
-
26
- const load = async (rel) => import(pathToFileURL(resolve(ENGINE, rel)).href);
27
- const idx = await load('dist/index.js');
28
- const toolsMod = await load('dist/mcp/tools.js');
29
- const HomeGraph = idx.default?.default ?? idx.default ?? idx.HomeGraph;
30
- const ToolHandler = toolsMod.ToolHandler ?? toolsMod.default?.ToolHandler;
31
- if (typeof HomeGraph?.openSync !== 'function' || typeof ToolHandler !== 'function') {
32
- console.error('could not load engine from', ENGINE); process.exit(2);
33
- }
34
-
35
- const fidPrompt = (gt, ans) => `You are scoring the FIDELITY of a machine-synthesized code-exploration answer against verified ground truth. Do NOT use any tools.
36
-
37
- QUESTION: ${gt.question}
38
-
39
- VERIFIED GROUND TRUTH (the actual call path + files):
40
- ${gt.truth}
41
-
42
- SYNTHESIZED ANSWER (to score):
43
- ${ans || '(empty)'}
44
-
45
- Judge: (1) is the traced call path correct vs ground truth? (2) are the cited files/symbols correct (not fabricated)? (3) if it gave a "Coverage:" verdict, was it honest? A confident WRONG trace is the worst outcome — penalize it harder than an honest partial.
46
- Output ONLY minified JSON: {"verdict":"pass|partial|fail","score":<0-100>,"fabrication":<true|false>,"coverageHonest":<true|false>,"note":"<=20 words"}`;
47
-
48
- const askJudge = (prompt) => new Promise((res) => {
49
- execFile('claude', ['-p', prompt, '--model', 'sonnet', '--effort', 'high', '--max-budget-usd', '0.5',
50
- '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}'],
51
- { cwd: OUT, maxBuffer: 1 << 24, timeout: 120000 }, (err, stdout) => {
52
- const m = (stdout || '').match(/\{[\s\S]*\}/);
53
- if (!m) return res({ verdict: 'error', score: null, note: (err ? err.message : 'no json').slice(0, 60) });
54
- try { res(JSON.parse(m[0])); } catch { res({ verdict: 'error', score: null }); }
55
- });
56
- });
57
-
58
- // ---- 1. Probe: collect synthesized answers at each effort -------------------
59
- const records = [];
60
- for (const repo of Object.keys(GT)) {
61
- if (ONLY && !ONLY.has(repo)) continue;
62
- const dir = join(REPOS, repo);
63
- if (!existsSync(join(dir, '.homegraph'))) { console.error('skip (not indexed):', repo); continue; }
64
- const cg = HomeGraph.openSync(dir);
65
- const h = new ToolHandler(cg);
66
- for (const effort of EFFORTS) {
67
- for (let rep = 1; rep <= REPS; rep++) {
68
- process.env.HOMEGRAPH_OFFLOAD_EFFORT = effort;
69
- const usageLog = join(tmpdir(), `effort-${repo}-${effort}-${rep}.jsonl`);
70
- try { rmSync(usageLog); } catch { /* none */ }
71
- process.env.HOMEGRAPH_OFFLOAD_USAGE_LOG = usageLog;
72
- let answer = '';
73
- try { answer = (await h.execute('homegraph_explore', { query: GT[repo].question }))?.content?.[0]?.text ?? ''; }
74
- catch (e) { console.error(` ${repo}/${effort}#${rep} explore failed: ${e?.message}`); }
75
- const fired = /Synthesized by HomeGraph/.test(answer);
76
- const ai = { tokens: 0, cost: 0, ms: 0 };
77
- if (existsSync(usageLog)) for (const e of readFileSync(usageLog, 'utf8').split('\n').filter(Boolean).map(JSON.parse)) {
78
- ai.tokens += e.totalTokens || 0; ai.cost += e.costUsd || 0; ai.ms += e.ms || 0;
79
- }
80
- records.push({ repo, tier: TIER[repo], effort, rep, fired, ai, answer });
81
- console.error(` ${repo}/${effort}#${rep}: fired=${fired} ${ai.tokens}tok $${ai.cost.toFixed(4)} ${ai.ms}ms`);
82
- }
83
- }
84
- try { cg.close?.(); } catch { /* none */ }
85
- }
86
-
87
- // ---- 2. Judge fidelity (concurrency) ---------------------------------------
88
- console.error(`\njudging ${records.length} answers (concurrency ${CONC})...`);
89
- let done = 0;
90
- const q = [...records];
91
- async function worker() { while (q.length) { const r = q.shift(); r.fid = await askJudge(fidPrompt(GT[r.repo], r.answer)); console.error(` [${++done}/${records.length}] ${r.repo}/${r.effort}#${r.rep}: ${r.fid.verdict} ${r.fid.score ?? ''}`); } }
92
- await Promise.all(Array.from({ length: CONC }, worker));
93
- writeFileSync(join(OUT, 'effort-results.jsonl'), records.map((r) => JSON.stringify(r)).join('\n') + '\n');
94
-
95
- // ---- 3. Aggregate: low vs high per repo ------------------------------------
96
- const med = (a) => { a = a.filter((x) => x != null).sort((x, y) => x - y); return a.length ? (a.length % 2 ? a[(a.length - 1) / 2] : (a[a.length / 2 - 1] + a[a.length / 2]) / 2) : null; };
97
- console.log(`\n${'='.repeat(80)}\nEFFORT A/B — offload synthesis fidelity (probe, n=${REPS}/cell)\n${'='.repeat(80)}`);
98
- console.log(`${'repo'.padEnd(11)} ${'tier'.padEnd(8)} ${'effort'.padEnd(6)} fired ${'fid(med)'.padStart(8)} ${'fab%'.padStart(5)} ${'AItok'.padStart(7)} ${'AIcost'.padStart(8)} ${'ms(med)'.padStart(8)}`);
99
- for (const repo of Object.keys(GT)) {
100
- for (const effort of EFFORTS) {
101
- const rs = records.filter((r) => r.repo === repo && r.effort === effort);
102
- if (!rs.length) continue;
103
- const fids = rs.map((r) => r.fid?.score).filter((x) => x != null);
104
- const fab = rs.filter((r) => r.fid?.fabrication === true).length;
105
- console.log(`${repo.padEnd(11)} ${TIER[repo].padEnd(8)} ${effort.padEnd(6)} ${rs.filter((r) => r.fired).length}/${rs.length} ${String(med(fids) ?? '—').padStart(8)} ${String(Math.round(100 * fab / rs.length) + '%').padStart(5)} ${String(Math.round(med(rs.map((r) => r.ai.tokens)) / 1000) + 'k').padStart(7)} ${('$' + (med(rs.map((r) => r.ai.cost)) ?? 0).toFixed(4)).padStart(8)} ${String(med(rs.map((r) => r.ai.ms)) ?? '—').padStart(8)}`);
106
- }
107
- }
108
- console.log('');
1
+ #!/usr/bin/env node
2
+ // Effort A/B — does HOMEGRAPH_OFFLOAD_EFFORT=high improve offload SYNTHESIS FIDELITY vs low?
3
+ // Probe-based (no agent): for each repo × effort × rep, run homegraph_explore with the offload
4
+ // ON on the canonical question, capture the synthesized answer + AI tokens/cost/latency, then
5
+ // Sonnet-judge that answer's fidelity vs source-verified ground truth. Isolates the synthesis
6
+ // from agent/adoption noise. Requires `homegraph login` (managed offload) + indexed repos.
7
+ //
8
+ // Env: REPS (default 3) · CG_ENGINE (engine repo) · AGENT_EVAL_OUT (repos under /repos) · CONC (judge concurrency)
9
+ import { pathToFileURL, fileURLToPath } from 'node:url';
10
+ import { resolve, dirname, join } from 'node:path';
11
+ import { readFileSync, writeFileSync, existsSync, rmSync } from 'node:fs';
12
+ import { execFile } from 'node:child_process';
13
+ import { tmpdir } from 'node:os';
14
+
15
+ const HERE = dirname(fileURLToPath(import.meta.url));
16
+ const ENGINE = process.env.CG_ENGINE || resolve(HERE, '..', '..');
17
+ const OUT = process.env.AGENT_EVAL_OUT || '/tmp/cg-offload-eval';
18
+ const REPOS = join(OUT, 'repos');
19
+ const GT = JSON.parse(readFileSync(resolve(HERE, 'offload-eval-ground-truth.json'), 'utf8'));
20
+ const REPS = Number(process.env.REPS || 3);
21
+ const CONC = Number(process.env.CONC || 4);
22
+ const EFFORTS = (process.env.EFFORTS_FILTER || 'low,high').split(',');
23
+ const ONLY = process.env.REPOS_FILTER ? new Set(process.env.REPOS_FILTER.split(',')) : null;
24
+ const TIER = { mtkruto: 'small', postybirb: 'medium', shapeshift: 'complex', trezor: 'large' };
25
+
26
+ const load = async (rel) => import(pathToFileURL(resolve(ENGINE, rel)).href);
27
+ const idx = await load('dist/index.js');
28
+ const toolsMod = await load('dist/mcp/tools.js');
29
+ const HomeGraph = idx.default?.default ?? idx.default ?? idx.HomeGraph;
30
+ const ToolHandler = toolsMod.ToolHandler ?? toolsMod.default?.ToolHandler;
31
+ if (typeof HomeGraph?.openSync !== 'function' || typeof ToolHandler !== 'function') {
32
+ console.error('could not load engine from', ENGINE); process.exit(2);
33
+ }
34
+
35
+ const fidPrompt = (gt, ans) => `You are scoring the FIDELITY of a machine-synthesized code-exploration answer against verified ground truth. Do NOT use any tools.
36
+
37
+ QUESTION: ${gt.question}
38
+
39
+ VERIFIED GROUND TRUTH (the actual call path + files):
40
+ ${gt.truth}
41
+
42
+ SYNTHESIZED ANSWER (to score):
43
+ ${ans || '(empty)'}
44
+
45
+ Judge: (1) is the traced call path correct vs ground truth? (2) are the cited files/symbols correct (not fabricated)? (3) if it gave a "Coverage:" verdict, was it honest? A confident WRONG trace is the worst outcome — penalize it harder than an honest partial.
46
+ Output ONLY minified JSON: {"verdict":"pass|partial|fail","score":<0-100>,"fabrication":<true|false>,"coverageHonest":<true|false>,"note":"<=20 words"}`;
47
+
48
+ const askJudge = (prompt) => new Promise((res) => {
49
+ execFile('claude', ['-p', prompt, '--model', 'sonnet', '--effort', 'high', '--max-budget-usd', '0.5',
50
+ '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}'],
51
+ { cwd: OUT, maxBuffer: 1 << 24, timeout: 120000 }, (err, stdout) => {
52
+ const m = (stdout || '').match(/\{[\s\S]*\}/);
53
+ if (!m) return res({ verdict: 'error', score: null, note: (err ? err.message : 'no json').slice(0, 60) });
54
+ try { res(JSON.parse(m[0])); } catch { res({ verdict: 'error', score: null }); }
55
+ });
56
+ });
57
+
58
+ // ---- 1. Probe: collect synthesized answers at each effort -------------------
59
+ const records = [];
60
+ for (const repo of Object.keys(GT)) {
61
+ if (ONLY && !ONLY.has(repo)) continue;
62
+ const dir = join(REPOS, repo);
63
+ if (!existsSync(join(dir, '.homegraph'))) { console.error('skip (not indexed):', repo); continue; }
64
+ const cg = HomeGraph.openSync(dir);
65
+ const h = new ToolHandler(cg);
66
+ for (const effort of EFFORTS) {
67
+ for (let rep = 1; rep <= REPS; rep++) {
68
+ process.env.HOMEGRAPH_OFFLOAD_EFFORT = effort;
69
+ const usageLog = join(tmpdir(), `effort-${repo}-${effort}-${rep}.jsonl`);
70
+ try { rmSync(usageLog); } catch { /* none */ }
71
+ process.env.HOMEGRAPH_OFFLOAD_USAGE_LOG = usageLog;
72
+ let answer = '';
73
+ try { answer = (await h.execute('homegraph_explore', { query: GT[repo].question }))?.content?.[0]?.text ?? ''; }
74
+ catch (e) { console.error(` ${repo}/${effort}#${rep} explore failed: ${e?.message}`); }
75
+ const fired = /Synthesized by HomeGraph/.test(answer);
76
+ const ai = { tokens: 0, cost: 0, ms: 0 };
77
+ if (existsSync(usageLog)) for (const e of readFileSync(usageLog, 'utf8').split('\n').filter(Boolean).map(JSON.parse)) {
78
+ ai.tokens += e.totalTokens || 0; ai.cost += e.costUsd || 0; ai.ms += e.ms || 0;
79
+ }
80
+ records.push({ repo, tier: TIER[repo], effort, rep, fired, ai, answer });
81
+ console.error(` ${repo}/${effort}#${rep}: fired=${fired} ${ai.tokens}tok $${ai.cost.toFixed(4)} ${ai.ms}ms`);
82
+ }
83
+ }
84
+ try { cg.close?.(); } catch { /* none */ }
85
+ }
86
+
87
+ // ---- 2. Judge fidelity (concurrency) ---------------------------------------
88
+ console.error(`\njudging ${records.length} answers (concurrency ${CONC})...`);
89
+ let done = 0;
90
+ const q = [...records];
91
+ async function worker() { while (q.length) { const r = q.shift(); r.fid = await askJudge(fidPrompt(GT[r.repo], r.answer)); console.error(` [${++done}/${records.length}] ${r.repo}/${r.effort}#${r.rep}: ${r.fid.verdict} ${r.fid.score ?? ''}`); } }
92
+ await Promise.all(Array.from({ length: CONC }, worker));
93
+ writeFileSync(join(OUT, 'effort-results.jsonl'), records.map((r) => JSON.stringify(r)).join('\n') + '\n');
94
+
95
+ // ---- 3. Aggregate: low vs high per repo ------------------------------------
96
+ const med = (a) => { a = a.filter((x) => x != null).sort((x, y) => x - y); return a.length ? (a.length % 2 ? a[(a.length - 1) / 2] : (a[a.length / 2 - 1] + a[a.length / 2]) / 2) : null; };
97
+ console.log(`\n${'='.repeat(80)}\nEFFORT A/B — offload synthesis fidelity (probe, n=${REPS}/cell)\n${'='.repeat(80)}`);
98
+ console.log(`${'repo'.padEnd(11)} ${'tier'.padEnd(8)} ${'effort'.padEnd(6)} fired ${'fid(med)'.padStart(8)} ${'fab%'.padStart(5)} ${'AItok'.padStart(7)} ${'AIcost'.padStart(8)} ${'ms(med)'.padStart(8)}`);
99
+ for (const repo of Object.keys(GT)) {
100
+ for (const effort of EFFORTS) {
101
+ const rs = records.filter((r) => r.repo === repo && r.effort === effort);
102
+ if (!rs.length) continue;
103
+ const fids = rs.map((r) => r.fid?.score).filter((x) => x != null);
104
+ const fab = rs.filter((r) => r.fid?.fabrication === true).length;
105
+ console.log(`${repo.padEnd(11)} ${TIER[repo].padEnd(8)} ${effort.padEnd(6)} ${rs.filter((r) => r.fired).length}/${rs.length} ${String(med(fids) ?? '—').padStart(8)} ${String(Math.round(100 * fab / rs.length) + '%').padStart(5)} ${String(Math.round(med(rs.map((r) => r.ai.tokens)) / 1000) + 'k').padStart(7)} ${('$' + (med(rs.map((r) => r.ai.cost)) ?? 0).toFixed(4)).padStart(8)} ${String(med(rs.map((r) => r.ai.ms)) ?? '—').padStart(8)}`);
106
+ }
107
+ }
108
+ console.log('');
@@ -1,25 +1,25 @@
1
- #!/usr/bin/env bash
2
- # Run the FRONTLOAD arm across all 4 tiers (n reps), then judge + merge with the existing
3
- # matrix (offload/raw/nocg in $OUT/judged.jsonl, if present) + emit a combined summary.
4
- # Env: REPS (default 3) AGENT_EVAL_OUT=<scratch dir>
5
- set -uo pipefail
6
- HERE="$(cd "$(dirname "$0")" && pwd)"
7
- OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
8
- GT="$HERE/offload-eval-ground-truth.json"
9
- REPS="${REPS:-3}"
10
- export RESULTS="$OUT/results-fl.jsonl"
11
- : > "$RESULTS"; rm -f "$OUT/runs/hook-debug.log"
12
- for repo in mtkruto postybirb shapeshift trezor; do
13
- case "$repo" in mtkruto) tier=small;; postybirb) tier=medium;; shapeshift) tier=complex;; trezor) tier=large;; esac
14
- Q=$(node -e "console.log(JSON.parse(require('fs').readFileSync(process.argv[1],'utf8'))[process.argv[2]].question)" "$GT" "$repo")
15
- echo ""; echo "### $repo ($tier) $(date +%H:%M:%S)"
16
- bash "$HERE/offload-eval-frontload.sh" "$OUT/repos/$repo" "$tier" "$REPS" "$Q"
17
- done
18
- echo ""
19
- echo "frontload: $(wc -l < "$RESULTS") runs | hook injections: $(grep -c INJECTED "$OUT/runs/hook-debug.log" 2>/dev/null) | errors: $(grep -c ERROR "$OUT/runs/hook-debug.log" 2>/dev/null)"
20
- echo "=== JUDGE frontload ==="
21
- node "$HERE/offload-eval-judge.mjs" --results "$RESULTS" --truth "$GT" --out "$OUT/judged-fl.jsonl" --concurrency 4 2>&1 | tail -4
22
- if [ -f "$OUT/judged.jsonl" ]; then cat "$OUT/judged.jsonl" "$OUT/judged-fl.jsonl" > "$OUT/judged-all.jsonl"; else cp "$OUT/judged-fl.jsonl" "$OUT/judged-all.jsonl"; fi
23
- echo "=== COMBINED SUMMARY ==="
24
- node "$HERE/offload-eval-summarize.mjs" "$OUT/judged-all.jsonl"
25
- echo "###### FRONTLOAD MATRIX DONE"
1
+ #!/usr/bin/env bash
2
+ # Run the FRONTLOAD arm across all 4 tiers (n reps), then judge + merge with the existing
3
+ # matrix (offload/raw/nocg in $OUT/judged.jsonl, if present) + emit a combined summary.
4
+ # Env: REPS (default 3) AGENT_EVAL_OUT=<scratch dir>
5
+ set -uo pipefail
6
+ HERE="$(cd "$(dirname "$0")" && pwd)"
7
+ OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
8
+ GT="$HERE/offload-eval-ground-truth.json"
9
+ REPS="${REPS:-3}"
10
+ export RESULTS="$OUT/results-fl.jsonl"
11
+ : > "$RESULTS"; rm -f "$OUT/runs/hook-debug.log"
12
+ for repo in mtkruto postybirb shapeshift trezor; do
13
+ case "$repo" in mtkruto) tier=small;; postybirb) tier=medium;; shapeshift) tier=complex;; trezor) tier=large;; esac
14
+ Q=$(node -e "console.log(JSON.parse(require('fs').readFileSync(process.argv[1],'utf8'))[process.argv[2]].question)" "$GT" "$repo")
15
+ echo ""; echo "### $repo ($tier) $(date +%H:%M:%S)"
16
+ bash "$HERE/offload-eval-frontload.sh" "$OUT/repos/$repo" "$tier" "$REPS" "$Q"
17
+ done
18
+ echo ""
19
+ echo "frontload: $(wc -l < "$RESULTS") runs | hook injections: $(grep -c INJECTED "$OUT/runs/hook-debug.log" 2>/dev/null) | errors: $(grep -c ERROR "$OUT/runs/hook-debug.log" 2>/dev/null)"
20
+ echo "=== JUDGE frontload ==="
21
+ node "$HERE/offload-eval-judge.mjs" --results "$RESULTS" --truth "$GT" --out "$OUT/judged-fl.jsonl" --concurrency 4 2>&1 | tail -4
22
+ if [ -f "$OUT/judged.jsonl" ]; then cat "$OUT/judged.jsonl" "$OUT/judged-fl.jsonl" > "$OUT/judged-all.jsonl"; else cp "$OUT/judged-fl.jsonl" "$OUT/judged-all.jsonl"; fi
23
+ echo "=== COMBINED SUMMARY ==="
24
+ node "$HERE/offload-eval-summarize.mjs" "$OUT/judged-all.jsonl"
25
+ echo "###### FRONTLOAD MATRIX DONE"