homegraph 1.5.0 → 1.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +305 -305
  3. package/dist/bin/command-supervision.d.ts.map +1 -1
  4. package/dist/bin/command-supervision.js +7 -4
  5. package/dist/bin/command-supervision.js.map +1 -1
  6. package/dist/bin/homegraph.js +9 -9
  7. package/dist/db/index.js +36 -36
  8. package/dist/db/migrations.js +37 -37
  9. package/dist/db/queries.js +156 -156
  10. package/dist/db/schema.sql +203 -203
  11. package/dist/directory.js +5 -5
  12. package/dist/extraction/languages/arkts-viewtree.d.ts +2 -4
  13. package/dist/extraction/languages/arkts-viewtree.d.ts.map +1 -1
  14. package/dist/extraction/languages/arkts-viewtree.js +6 -21
  15. package/dist/extraction/languages/arkts-viewtree.js.map +1 -1
  16. package/dist/extraction/languages/arkts.d.ts +16 -6
  17. package/dist/extraction/languages/arkts.d.ts.map +1 -1
  18. package/dist/extraction/languages/arkts.js +174 -21
  19. package/dist/extraction/languages/arkts.js.map +1 -1
  20. package/dist/extraction/wasm/tree-sitter-c_sharp.wasm +0 -0
  21. package/dist/extraction/wasm/tree-sitter-cfml.wasm +0 -0
  22. package/dist/extraction/wasm/tree-sitter-cfquery.wasm +0 -0
  23. package/dist/extraction/wasm/tree-sitter-cfscript.wasm +0 -0
  24. package/dist/extraction/wasm/tree-sitter-cobol.wasm +0 -0
  25. package/dist/extraction/wasm/tree-sitter-erlang.wasm +0 -0
  26. package/dist/extraction/wasm/tree-sitter-nix.wasm +0 -0
  27. package/dist/extraction/wasm/tree-sitter-pascal.wasm +0 -0
  28. package/dist/extraction/wasm/tree-sitter-vbnet.wasm +0 -0
  29. package/dist/installer/instructions-template.js +9 -9
  30. package/dist/mcp/liveness-watchdog.d.ts +18 -0
  31. package/dist/mcp/liveness-watchdog.d.ts.map +1 -1
  32. package/dist/mcp/liveness-watchdog.js +185 -73
  33. package/dist/mcp/liveness-watchdog.js.map +1 -1
  34. package/dist/mcp/server-instructions.js +47 -47
  35. package/dist/reasoning/reasoner.js +32 -32
  36. package/dist/spec/db/commit-node.js +4 -4
  37. package/dist/spec/db/fragment-node.js +10 -10
  38. package/dist/spec/db/fts.js +8 -8
  39. package/dist/spec/db/relations.js +88 -88
  40. package/dist/spec/db/schema.js +6 -6
  41. package/dist/spec/db/schema.sql +121 -121
  42. package/dist/spec/db/spec-node.js +4 -4
  43. package/dist/spec/llm/prompts.js +53 -53
  44. package/dist/spec/utils.d.ts +16 -4
  45. package/dist/spec/utils.d.ts.map +1 -1
  46. package/dist/spec/utils.js +58 -6
  47. package/dist/spec/utils.js.map +1 -1
  48. package/package.json +62 -62
  49. package/scripts/_tmp-cfwk-resolve.log +0 -0
  50. package/scripts/_tmp-cfwk-sig.log +0 -0
  51. package/scripts/_tmp-cfwk-vt.log +0 -0
  52. package/scripts/add-lang/bench.sh +60 -60
  53. package/scripts/add-lang/check-grammar.mjs +75 -75
  54. package/scripts/add-lang/dump-ast.mjs +103 -103
  55. package/scripts/add-lang/verify-extraction.mjs +70 -70
  56. package/scripts/agent-eval/ab-adoption.sh +91 -91
  57. package/scripts/agent-eval/ab-hook.sh +86 -86
  58. package/scripts/agent-eval/ab-impl.sh +78 -78
  59. package/scripts/agent-eval/ab-new-vs-baseline.sh +102 -102
  60. package/scripts/agent-eval/ab-sufficiency.sh +78 -78
  61. package/scripts/agent-eval/arms-F.sh +21 -21
  62. package/scripts/agent-eval/arms-matrix.sh +37 -37
  63. package/scripts/agent-eval/audit.sh +68 -68
  64. package/scripts/agent-eval/bench-readme.sh +28 -28
  65. package/scripts/agent-eval/bench-why-repo.sh +22 -22
  66. package/scripts/agent-eval/block-read-hook.sh +19 -19
  67. package/scripts/agent-eval/hook-settings.json +15 -15
  68. package/scripts/agent-eval/itrun.sh +120 -120
  69. package/scripts/agent-eval/offload-eval-3arm.sh +72 -72
  70. package/scripts/agent-eval/offload-eval-cost.mjs +133 -133
  71. package/scripts/agent-eval/offload-eval-effort.mjs +108 -108
  72. package/scripts/agent-eval/offload-eval-frontload-matrix.sh +25 -25
  73. package/scripts/agent-eval/offload-eval-frontload.sh +47 -47
  74. package/scripts/agent-eval/offload-eval-ground-truth.json +18 -18
  75. package/scripts/agent-eval/offload-eval-hook.mjs +84 -84
  76. package/scripts/agent-eval/offload-eval-judge.mjs +103 -103
  77. package/scripts/agent-eval/offload-eval-matrix.sh +20 -20
  78. package/scripts/agent-eval/offload-eval-metrics.mjs +94 -94
  79. package/scripts/agent-eval/offload-eval-refs1.sh +50 -50
  80. package/scripts/agent-eval/offload-eval-setup.sh +24 -24
  81. package/scripts/agent-eval/offload-eval-styles.sh +71 -71
  82. package/scripts/agent-eval/offload-eval-summarize.mjs +68 -68
  83. package/scripts/agent-eval/offload-eval.md +76 -76
  84. package/scripts/agent-eval/parse-arms.mjs +116 -116
  85. package/scripts/agent-eval/parse-bench-readme.mjs +84 -84
  86. package/scripts/agent-eval/parse-run.mjs +45 -45
  87. package/scripts/agent-eval/parse-session.mjs +93 -93
  88. package/scripts/agent-eval/probe-context.mjs +21 -21
  89. package/scripts/agent-eval/probe-explore.mjs +40 -40
  90. package/scripts/agent-eval/probe-node.mjs +20 -20
  91. package/scripts/agent-eval/probe-sweep.mjs +119 -119
  92. package/scripts/agent-eval/probe-trace.mjs +20 -20
  93. package/scripts/agent-eval/redirect-read-hook.sh +38 -38
  94. package/scripts/agent-eval/repro-concurrent-explore.mjs +119 -119
  95. package/scripts/agent-eval/repro-daemon-clients.mjs +125 -125
  96. package/scripts/agent-eval/run-agent.sh +34 -34
  97. package/scripts/agent-eval/run-all.sh +75 -75
  98. package/scripts/agent-eval/run-arms.sh +56 -56
  99. package/scripts/agent-eval/seq-matrix.mjs +137 -137
  100. package/scripts/bench-arkts-init-rss.log +0 -0
  101. package/scripts/build-bundle.sh +123 -123
  102. package/scripts/exp_boundary_eval/README.md +247 -247
  103. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-310.pyc +0 -0
  104. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-38.pyc +0 -0
  105. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-310.pyc +0 -0
  106. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-38.pyc +0 -0
  107. package/scripts/exp_boundary_eval/__pycache__/deveco_arm.cpython-38.pyc +0 -0
  108. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-310.pyc +0 -0
  109. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-38.pyc +0 -0
  110. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-310.pyc +0 -0
  111. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-38.pyc +0 -0
  112. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-310.pyc +0 -0
  113. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-38.pyc +0 -0
  114. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-310.pyc +0 -0
  115. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-38.pyc +0 -0
  116. package/scripts/exp_boundary_eval/__pycache__/win_mcp_launcher.cpython-38.pyc +0 -0
  117. package/scripts/exp_boundary_eval/_test_mcp_chain.py +78 -78
  118. package/scripts/exp_boundary_eval/_test_stdin.py +8 -8
  119. package/scripts/exp_boundary_eval/_utils.py +1116 -1116
  120. package/scripts/exp_boundary_eval/analyze.py +1313 -1313
  121. package/scripts/exp_boundary_eval/deveco_arm.py +519 -519
  122. package/scripts/exp_boundary_eval/run_all.py +378 -378
  123. package/scripts/exp_boundary_eval/run_one.py +165 -165
  124. package/scripts/exp_boundary_eval/run_session.py +158 -158
  125. package/scripts/exp_boundary_eval/setup.py +120 -120
  126. package/scripts/exp_boundary_eval/win_mcp_launcher.py +73 -73
  127. package/scripts/exp_boundary_eval/win_mcp_stdio_wrap.js +36 -36
  128. package/scripts/exp_boundary_eval/win_node_launcher.py +24 -24
  129. package/scripts/extract-release-notes.mjs +130 -130
  130. package/scripts/local-install.sh +41 -41
  131. package/scripts/npm-sdk.js +75 -75
  132. package/scripts/npm-shim.js +275 -275
  133. package/scripts/ohos-sdk-publish.mjs +133 -133
  134. package/scripts/pack-npm.sh +119 -119
  135. package/scripts/prepare-release.mjs +270 -270
  136. package/scripts/probe-arkts-mem-why-run.log +0 -0
  137. package/scripts/probe-banner-livecard-ir.log +0 -0
  138. package/scripts/probe-cfwk-ast.stderr.log +0 -0
  139. package/scripts/probe-cfwk-ast.stdout.log +0 -0
  140. package/scripts/probe-cfwk-attr-shape.log +0 -0
  141. package/scripts/probe-cfwk-cfgdump.stderr.log +0 -0
  142. package/scripts/probe-cfwk-cfgdump.stdout.log +0 -0
  143. package/scripts/probe-cfwk-cvc.stderr.log +0 -0
  144. package/scripts/probe-cfwk-cvc.stdout.log +0 -0
  145. package/scripts/probe-cfwk-diag2.log +0 -0
  146. package/scripts/probe-cfwk-diag3.log +0 -0
  147. package/scripts/probe-cfwk-fedbg.stderr.log +0 -0
  148. package/scripts/probe-cfwk-fedbg.stdout.log +0 -0
  149. package/scripts/probe-cfwk-fileresult.log +0 -0
  150. package/scripts/probe-cfwk-fix.stderr.log +0 -0
  151. package/scripts/probe-cfwk-fix.stdout.log +0 -0
  152. package/scripts/probe-cfwk-fix2.stderr.log +0 -0
  153. package/scripts/probe-cfwk-fix2.stdout.log +0 -0
  154. package/scripts/probe-cfwk-fix3.stderr.log +0 -0
  155. package/scripts/probe-cfwk-fix3.stdout.log +0 -0
  156. package/scripts/probe-cfwk-foreach.log +0 -0
  157. package/scripts/probe-cfwk-getmethod-throw.log +0 -0
  158. package/scripts/probe-cfwk-hg-extract.log +0 -0
  159. package/scripts/probe-cfwk-pr1003-noprior.stderr.log +0 -0
  160. package/scripts/probe-cfwk-pr1003-noprior.stdout.log +0 -0
  161. package/scripts/probe-cfwk-pr1003.stderr.log +0 -0
  162. package/scripts/probe-cfwk-pr1003.stdout.log +0 -0
  163. package/scripts/probe-cfwk-preroot-noprior.stderr.log +0 -0
  164. package/scripts/probe-cfwk-preroot-noprior.stdout.log +0 -0
  165. package/scripts/probe-cfwk-preroot-prior.stderr.log +0 -0
  166. package/scripts/probe-cfwk-preroot-prior.stdout.log +0 -0
  167. package/scripts/probe-cfwk-preroot-skipstate.stderr.log +0 -0
  168. package/scripts/probe-cfwk-preroot-skipstate.stdout.log +0 -0
  169. package/scripts/probe-cfwk-resolve-sim.log +0 -0
  170. package/scripts/probe-cfwk-tree-shape.log +0 -0
  171. package/scripts/probe-cfwk-walk-abort.log +0 -0
  172. package/scripts/probe-cfwk.log +0 -0
  173. package/scripts/probe-force-index.log +0 -0
  174. package/scripts/probe-no-force-index.log +0 -0
  175. package/scripts/probe-samefile-ir.stderr.log +0 -0
  176. package/scripts/probe-samefile-ir.stdout.log +224 -0
  177. package/scripts/probe-sdk-vs-project.log +0 -0
  178. package/scripts/probe-viewtree-downgrade.log +0 -0
  179. package/dist/arkts/ohos-api-index.d.ts +0 -15
  180. package/dist/arkts/ohos-api-index.d.ts.map +0 -1
  181. package/dist/arkts/ohos-api-index.js +0 -190
  182. package/dist/arkts/ohos-api-index.js.map +0 -1
  183. package/dist/arkts/ohos-sdk-input.d.ts +0 -36
  184. package/dist/arkts/ohos-sdk-input.d.ts.map +0 -1
  185. package/dist/arkts/ohos-sdk-input.js +0 -214
  186. package/dist/arkts/ohos-sdk-input.js.map +0 -1
  187. package/dist/extraction/languages/arkts-state-decorators.d.ts +0 -13
  188. package/dist/extraction/languages/arkts-state-decorators.d.ts.map +0 -1
  189. package/dist/extraction/languages/arkts-state-decorators.js +0 -26
  190. package/dist/extraction/languages/arkts-state-decorators.js.map +0 -1
  191. package/dist/extraction/languages/ohos-api-consumer.d.ts +0 -34
  192. package/dist/extraction/languages/ohos-api-consumer.d.ts.map +0 -1
  193. package/dist/extraction/languages/ohos-api-consumer.js +0 -283
  194. package/dist/extraction/languages/ohos-api-consumer.js.map +0 -1
  195. package/dist/spec/build/git-scanner.d.ts +0 -93
  196. package/dist/spec/build/git-scanner.d.ts.map +0 -1
  197. package/dist/spec/build/git-scanner.js +0 -254
  198. package/dist/spec/build/git-scanner.js.map +0 -1
  199. package/dist/spec/git-utils.d.ts +0 -8
  200. package/dist/spec/git-utils.d.ts.map +0 -1
  201. package/dist/spec/git-utils.js +0 -14
  202. package/dist/spec/git-utils.js.map +0 -1
  203. package/dist/spec/mine/clusterer.d.ts +0 -63
  204. package/dist/spec/mine/clusterer.d.ts.map +0 -1
  205. package/dist/spec/mine/clusterer.js +0 -904
  206. package/dist/spec/mine/clusterer.js.map +0 -1
  207. package/dist/spec/mine/progress-handler.d.ts +0 -22
  208. package/dist/spec/mine/progress-handler.d.ts.map +0 -1
  209. package/dist/spec/mine/progress-handler.js +0 -108
  210. package/dist/spec/mine/progress-handler.js.map +0 -1
  211. package/dist/spec/mine/progress.d.ts +0 -23
  212. package/dist/spec/mine/progress.d.ts.map +0 -1
  213. package/dist/spec/mine/progress.js +0 -12
  214. package/dist/spec/mine/progress.js.map +0 -1
@@ -1,94 +1,94 @@
1
- #!/usr/bin/env node
2
- // Extract one eval run's metrics from its Claude stream-json transcript + the
3
- // offload usage sidecar log, emit ONE merged JSON line.
4
- //
5
- // Usage: extract-metrics.mjs --run <run.jsonl> --usage <usage.jsonl|-> \
6
- // --arm <a> --rep <n> --repo <r> --tier <t> --q <question>
7
- import { readFileSync, existsSync } from 'fs';
8
-
9
- const args = {};
10
- for (let i = 2; i < process.argv.length; i += 2) args[process.argv[i].replace(/^--/, '')] = process.argv[i + 1];
11
-
12
- const runFile = args.run;
13
- const lines = existsSync(runFile) ? readFileSync(runFile, 'utf8').split('\n').filter(Boolean) : [];
14
-
15
- const toolCounts = {};
16
- let result = null;
17
- const tok = { gen: 0, fresh: 0, cached: 0 };
18
- const offloadAnswers = [];
19
- let exploreResults = 0; // tool_results from explore (offload or raw)
20
- let lastAssistantText = '';
21
-
22
- for (const line of lines) {
23
- let ev; try { ev = JSON.parse(line); } catch { continue; }
24
-
25
- // per-turn token usage (authoritative token measure; result.usage is last-turn only)
26
- const u = ev.message?.usage;
27
- if (u) {
28
- tok.gen += u.output_tokens || 0;
29
- tok.fresh += (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0);
30
- tok.cached += u.cache_read_input_tokens || 0;
31
- }
32
-
33
- if (ev.type === 'assistant' && Array.isArray(ev.message?.content)) {
34
- for (const b of ev.message.content) {
35
- if (b.type === 'tool_use') toolCounts[b.name] = (toolCounts[b.name] || 0) + 1;
36
- if (b.type === 'text' && b.text?.trim()) lastAssistantText = b.text.trim();
37
- }
38
- }
39
- // tool_results arrive in user messages
40
- if (ev.type === 'user' && Array.isArray(ev.message?.content)) {
41
- for (const b of ev.message.content) {
42
- if (b.type !== 'tool_result') continue;
43
- const text = Array.isArray(b.content)
44
- ? b.content.map(c => (typeof c === 'string' ? c : c.text || '')).join('')
45
- : (typeof b.content === 'string' ? b.content : '');
46
- // An offload answer is either the 'plain'/'report' synthesis (carries the
47
- // "Synthesized by HomeGraph" footer) or a 'refs' answer (carries the re-expanded
48
- // "### Referenced source — verbatim" appendix). A refs call that cited nothing
49
- // valid falls back to RAW source, which is correctly counted as a raw explore below.
50
- if (/Synthesized by HomeGraph|### Referenced source — verbatim/.test(text)) { offloadAnswers.push(text); exploreResults++; }
51
- else if (/Found \d+ symbols? across|\*\*Exploration:/.test(text)) exploreResults++;
52
- }
53
- }
54
- if (ev.type === 'result') result = ev;
55
- }
56
-
57
- // offload usage sidecar (HomeGraph AI tokens + cost) — one JSON line per offload call
58
- const ai = { calls: 0, promptTokens: 0, completionTokens: 0, totalTokens: 0, credits: 0, costUsd: 0, ms: 0 };
59
- if (args.usage && args.usage !== '-' && existsSync(args.usage)) {
60
- for (const line of readFileSync(args.usage, 'utf8').split('\n').filter(Boolean)) {
61
- let e; try { e = JSON.parse(line); } catch { continue; }
62
- ai.calls++;
63
- ai.promptTokens += e.promptTokens || 0;
64
- ai.completionTokens += e.completionTokens || 0;
65
- ai.totalTokens += e.totalTokens || 0;
66
- ai.credits += e.creditsCharged || 0;
67
- ai.costUsd += e.costUsd || 0;
68
- ai.ms += e.ms || 0;
69
- }
70
- }
71
-
72
- // front-load hook fired iff its injected header appears in the transcript
73
- const frontload = lines.some(l => l.includes('auto-retrieved for this question'));
74
- const get = (n) => toolCounts[n] || 0;
75
- const read = get('Read');
76
- const grep = get('Grep') + get('Bash') + get('Glob');
77
- const explore = get('mcp__homegraph__homegraph_explore');
78
- const cgAny = Object.keys(toolCounts).filter(k => /mcp__homegraph__/.test(k)).reduce((s, k) => s + toolCounts[k], 0);
79
-
80
- const out = {
81
- repo: args.repo, tier: args.tier, arm: args.arm, rep: Number(args.rep), question: args.q,
82
- ok: result?.subtype === 'success',
83
- durationSec: result ? +(result.duration_ms / 1000).toFixed(1) : null,
84
- numTurns: result?.num_turns ?? null,
85
- costUsdMain: result ? +(result.total_cost_usd || 0).toFixed(4) : null,
86
- tokGen: tok.gen, tokFresh: tok.fresh, tokCached: tok.cached, tokBillable: tok.gen + tok.fresh,
87
- read, grep, explore, cgAny, frontload,
88
- offloadFired: offloadAnswers.length,
89
- ai,
90
- // text payloads for the accuracy judge (kept separate; large)
91
- finalAnswer: (result?.result || lastAssistantText || '').slice(0, 8000),
92
- offloadAnswers: offloadAnswers.map(a => a.slice(0, 6000)),
93
- };
94
- process.stdout.write(JSON.stringify(out) + '\n');
1
+ #!/usr/bin/env node
2
+ // Extract one eval run's metrics from its Claude stream-json transcript + the
3
+ // offload usage sidecar log, emit ONE merged JSON line.
4
+ //
5
+ // Usage: extract-metrics.mjs --run <run.jsonl> --usage <usage.jsonl|-> \
6
+ // --arm <a> --rep <n> --repo <r> --tier <t> --q <question>
7
+ import { readFileSync, existsSync } from 'fs';
8
+
9
+ const args = {};
10
+ for (let i = 2; i < process.argv.length; i += 2) args[process.argv[i].replace(/^--/, '')] = process.argv[i + 1];
11
+
12
+ const runFile = args.run;
13
+ const lines = existsSync(runFile) ? readFileSync(runFile, 'utf8').split('\n').filter(Boolean) : [];
14
+
15
+ const toolCounts = {};
16
+ let result = null;
17
+ const tok = { gen: 0, fresh: 0, cached: 0 };
18
+ const offloadAnswers = [];
19
+ let exploreResults = 0; // tool_results from explore (offload or raw)
20
+ let lastAssistantText = '';
21
+
22
+ for (const line of lines) {
23
+ let ev; try { ev = JSON.parse(line); } catch { continue; }
24
+
25
+ // per-turn token usage (authoritative token measure; result.usage is last-turn only)
26
+ const u = ev.message?.usage;
27
+ if (u) {
28
+ tok.gen += u.output_tokens || 0;
29
+ tok.fresh += (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0);
30
+ tok.cached += u.cache_read_input_tokens || 0;
31
+ }
32
+
33
+ if (ev.type === 'assistant' && Array.isArray(ev.message?.content)) {
34
+ for (const b of ev.message.content) {
35
+ if (b.type === 'tool_use') toolCounts[b.name] = (toolCounts[b.name] || 0) + 1;
36
+ if (b.type === 'text' && b.text?.trim()) lastAssistantText = b.text.trim();
37
+ }
38
+ }
39
+ // tool_results arrive in user messages
40
+ if (ev.type === 'user' && Array.isArray(ev.message?.content)) {
41
+ for (const b of ev.message.content) {
42
+ if (b.type !== 'tool_result') continue;
43
+ const text = Array.isArray(b.content)
44
+ ? b.content.map(c => (typeof c === 'string' ? c : c.text || '')).join('')
45
+ : (typeof b.content === 'string' ? b.content : '');
46
+ // An offload answer is either the 'plain'/'report' synthesis (carries the
47
+ // "Synthesized by HomeGraph" footer) or a 'refs' answer (carries the re-expanded
48
+ // "### Referenced source — verbatim" appendix). A refs call that cited nothing
49
+ // valid falls back to RAW source, which is correctly counted as a raw explore below.
50
+ if (/Synthesized by HomeGraph|### Referenced source — verbatim/.test(text)) { offloadAnswers.push(text); exploreResults++; }
51
+ else if (/Found \d+ symbols? across|\*\*Exploration:/.test(text)) exploreResults++;
52
+ }
53
+ }
54
+ if (ev.type === 'result') result = ev;
55
+ }
56
+
57
+ // offload usage sidecar (HomeGraph AI tokens + cost) — one JSON line per offload call
58
+ const ai = { calls: 0, promptTokens: 0, completionTokens: 0, totalTokens: 0, credits: 0, costUsd: 0, ms: 0 };
59
+ if (args.usage && args.usage !== '-' && existsSync(args.usage)) {
60
+ for (const line of readFileSync(args.usage, 'utf8').split('\n').filter(Boolean)) {
61
+ let e; try { e = JSON.parse(line); } catch { continue; }
62
+ ai.calls++;
63
+ ai.promptTokens += e.promptTokens || 0;
64
+ ai.completionTokens += e.completionTokens || 0;
65
+ ai.totalTokens += e.totalTokens || 0;
66
+ ai.credits += e.creditsCharged || 0;
67
+ ai.costUsd += e.costUsd || 0;
68
+ ai.ms += e.ms || 0;
69
+ }
70
+ }
71
+
72
+ // front-load hook fired iff its injected header appears in the transcript
73
+ const frontload = lines.some(l => l.includes('auto-retrieved for this question'));
74
+ const get = (n) => toolCounts[n] || 0;
75
+ const read = get('Read');
76
+ const grep = get('Grep') + get('Bash') + get('Glob');
77
+ const explore = get('mcp__homegraph__homegraph_explore');
78
+ const cgAny = Object.keys(toolCounts).filter(k => /mcp__homegraph__/.test(k)).reduce((s, k) => s + toolCounts[k], 0);
79
+
80
+ const out = {
81
+ repo: args.repo, tier: args.tier, arm: args.arm, rep: Number(args.rep), question: args.q,
82
+ ok: result?.subtype === 'success',
83
+ durationSec: result ? +(result.duration_ms / 1000).toFixed(1) : null,
84
+ numTurns: result?.num_turns ?? null,
85
+ costUsdMain: result ? +(result.total_cost_usd || 0).toFixed(4) : null,
86
+ tokGen: tok.gen, tokFresh: tok.fresh, tokCached: tok.cached, tokBillable: tok.gen + tok.fresh,
87
+ read, grep, explore, cgAny, frontload,
88
+ offloadFired: offloadAnswers.length,
89
+ ai,
90
+ // text payloads for the accuracy judge (kept separate; large)
91
+ finalAnswer: (result?.result || lastAssistantText || '').slice(0, 8000),
92
+ offloadAnswers: offloadAnswers.map(a => a.slice(0, 6000)),
93
+ };
94
+ process.stdout.write(JSON.stringify(out) + '\n');
@@ -1,50 +1,50 @@
1
- #!/usr/bin/env bash
2
- # ONE offload run on ONE indexed repo at a given offload STYLE (plain|refs), so we can
3
- # watch a single agent transcript at a time (the user's one-run-at-a-time methodology).
4
- # The OFFLOAD reasoning runs in the prewarmed DAEMON process, so the style env must be
5
- # set on BOTH the daemon and the client MCP config. Writes one metrics line to RESULTS
6
- # and leaves the raw stream-json at $RUNS/<repo>-<style>-<n>.jsonl for inspection.
7
- #
8
- # Usage: offload-eval-refs1.sh <indexed-repo> <style> <n> "<question>"
9
- set -uo pipefail
10
- HERE="$(cd "$(dirname "$0")" && pwd)"; ENGINE="$(cd "$HERE/../.." && pwd)"; BIN="$ENGINE/dist/bin/homegraph.js"
11
- OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"; RUNS="$OUT/runs"; EXTRACT="$HERE/offload-eval-metrics.mjs"
12
- TARGET="${1:?repo}"; STYLE="${2:?style}"; N="${3:?run-tag}"; Q="${4:?question}"
13
- RESULTS="${RESULTS:-$OUT/results-refs.jsonl}"; REPO=$(basename "$TARGET"); TARGET=$(cd "$TARGET" && pwd -P)
14
- mkdir -p "$RUNS"; command -v claude >/dev/null || { echo "no claude"; exit 1; }
15
- USAGE="$RUNS/$REPO-$STYLE-usage.jsonl"; : > "$USAGE"
16
- CFG="$RUNS/mcp-$REPO-$STYLE.json"
17
- # `raw` is a pseudo-style: homegraph attached but the offload DISABLED (the ceiling —
18
- # verbatim source, no reasoning model). Any other value is an offload style (plain|refs).
19
- if [ "$STYLE" = "raw" ]; then
20
- DAEMON_ENV="HOMEGRAPH_OFFLOAD_DISABLE=1"
21
- printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_DISABLE=1","node","%s","serve","--mcp","--path","%s"]}}}' \
22
- "$BIN" "$TARGET" > "$CFG"
23
- USAGE="-"
24
- else
25
- DAEMON_ENV="HOMEGRAPH_OFFLOAD_STYLE=$STYLE HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
26
- printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_STYLE=%s","HOMEGRAPH_OFFLOAD_USAGE_LOG=%s","node","%s","serve","--mcp","--path","%s"]}}}' \
27
- "$STYLE" "$USAGE" "$BIN" "$TARGET" > "$CFG"
28
- fi
29
-
30
- # Prewarm a persistent daemon carrying the SAME offload config (it does the reasoning).
31
- pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
32
- env $DAEMON_ENV HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 \
33
- node "$BIN" serve --mcp --path "$TARGET" </dev/null >/dev/null 2>&1 &
34
- node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$TARGET" \
35
- && echo "daemon warm ($STYLE)" || echo "WARN daemon never bound"
36
-
37
- tag="$REPO-$STYLE-$N"
38
- echo "== run $tag =="
39
- # DISALLOW (optional): block tools that confound the offload-sufficiency signal —
40
- # chiefly "Agent" (sub-agent delegation: the spawned Explore subagent has low MCP
41
- # salience, ignores homegraph, and thrashes via Bash+Read, making the A/B noise).
42
- ( cd "$TARGET" && claude -p "$Q" --output-format stream-json --verbose --permission-mode bypassPermissions \
43
- --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
44
- ${DISALLOW:+--disallowedTools "$DISALLOW"} \
45
- --strict-mcp-config --mcp-config "$CFG" </dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
46
- node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$USAGE" --arm "offload-$STYLE" --rep "$N" \
47
- --repo "$REPO" --tier "complex" --q "$Q" >> "$RESULTS"
48
- node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | main $${o.costUsdMain} ${o.tokBillable} tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}call/${o.ai.totalTokens}tok/$${o.ai.costUsd.toFixed(4)} | ok=${o.ok}`)' "$RESULTS"
49
- pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null
50
- echo "raw transcript: $RUNS/$tag.jsonl"
1
+ #!/usr/bin/env bash
2
+ # ONE offload run on ONE indexed repo at a given offload STYLE (plain|refs), so we can
3
+ # watch a single agent transcript at a time (the user's one-run-at-a-time methodology).
4
+ # The OFFLOAD reasoning runs in the prewarmed DAEMON process, so the style env must be
5
+ # set on BOTH the daemon and the client MCP config. Writes one metrics line to RESULTS
6
+ # and leaves the raw stream-json at $RUNS/<repo>-<style>-<n>.jsonl for inspection.
7
+ #
8
+ # Usage: offload-eval-refs1.sh <indexed-repo> <style> <n> "<question>"
9
+ set -uo pipefail
10
+ HERE="$(cd "$(dirname "$0")" && pwd)"; ENGINE="$(cd "$HERE/../.." && pwd)"; BIN="$ENGINE/dist/bin/homegraph.js"
11
+ OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"; RUNS="$OUT/runs"; EXTRACT="$HERE/offload-eval-metrics.mjs"
12
+ TARGET="${1:?repo}"; STYLE="${2:?style}"; N="${3:?run-tag}"; Q="${4:?question}"
13
+ RESULTS="${RESULTS:-$OUT/results-refs.jsonl}"; REPO=$(basename "$TARGET"); TARGET=$(cd "$TARGET" && pwd -P)
14
+ mkdir -p "$RUNS"; command -v claude >/dev/null || { echo "no claude"; exit 1; }
15
+ USAGE="$RUNS/$REPO-$STYLE-usage.jsonl"; : > "$USAGE"
16
+ CFG="$RUNS/mcp-$REPO-$STYLE.json"
17
+ # `raw` is a pseudo-style: homegraph attached but the offload DISABLED (the ceiling —
18
+ # verbatim source, no reasoning model). Any other value is an offload style (plain|refs).
19
+ if [ "$STYLE" = "raw" ]; then
20
+ DAEMON_ENV="HOMEGRAPH_OFFLOAD_DISABLE=1"
21
+ printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_DISABLE=1","node","%s","serve","--mcp","--path","%s"]}}}' \
22
+ "$BIN" "$TARGET" > "$CFG"
23
+ USAGE="-"
24
+ else
25
+ DAEMON_ENV="HOMEGRAPH_OFFLOAD_STYLE=$STYLE HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
26
+ printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_STYLE=%s","HOMEGRAPH_OFFLOAD_USAGE_LOG=%s","node","%s","serve","--mcp","--path","%s"]}}}' \
27
+ "$STYLE" "$USAGE" "$BIN" "$TARGET" > "$CFG"
28
+ fi
29
+
30
+ # Prewarm a persistent daemon carrying the SAME offload config (it does the reasoning).
31
+ pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
32
+ env $DAEMON_ENV HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 \
33
+ node "$BIN" serve --mcp --path "$TARGET" </dev/null >/dev/null 2>&1 &
34
+ node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$TARGET" \
35
+ && echo "daemon warm ($STYLE)" || echo "WARN daemon never bound"
36
+
37
+ tag="$REPO-$STYLE-$N"
38
+ echo "== run $tag =="
39
+ # DISALLOW (optional): block tools that confound the offload-sufficiency signal —
40
+ # chiefly "Agent" (sub-agent delegation: the spawned Explore subagent has low MCP
41
+ # salience, ignores homegraph, and thrashes via Bash+Read, making the A/B noise).
42
+ ( cd "$TARGET" && claude -p "$Q" --output-format stream-json --verbose --permission-mode bypassPermissions \
43
+ --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
44
+ ${DISALLOW:+--disallowedTools "$DISALLOW"} \
45
+ --strict-mcp-config --mcp-config "$CFG" </dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
46
+ node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$USAGE" --arm "offload-$STYLE" --rep "$N" \
47
+ --repo "$REPO" --tier "complex" --q "$Q" >> "$RESULTS"
48
+ node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | main $${o.costUsdMain} ${o.tokBillable} tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}call/${o.ai.totalTokens}tok/$${o.ai.costUsd.toFixed(4)} | ok=${o.ok}`)' "$RESULTS"
49
+ pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null
50
+ echo "raw transcript: $RUNS/$tag.jsonl"
@@ -1,24 +1,24 @@
1
- #!/usr/bin/env bash
2
- # Clone + index the 4 "not-trained-on" eval repos into $AGENT_EVAL_OUT/repos. These were
3
- # selected via a no-tools memory-probe gate (Sonnet cannot answer their flow questions from
4
- # memory — so the no-homegraph baseline is honest). Env: AGENT_EVAL_OUT=<scratch dir>
5
- set -uo pipefail
6
- HERE="$(cd "$(dirname "$0")" && pwd)"
7
- ENGINE="$(cd "$HERE/../.." && pwd)"
8
- BIN="$ENGINE/dist/bin/homegraph.js"
9
- OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
10
- ROOT="$OUT/repos"; mkdir -p "$ROOT"
11
- export HOMEGRAPH_TELEMETRY=0 DO_NOT_TRACK=1
12
- [ -f "$BIN" ] || { echo "engine not built: run 'npm run build' in $ENGINE first"; exit 1; }
13
-
14
- clone_index() { # url name
15
- echo "=== $2: clone ==="; rm -rf "$ROOT/$2"
16
- git clone --quiet --depth 1 "$1" "$ROOT/$2" || { echo " clone FAILED"; return 1; }
17
- echo "=== $2: index ==="
18
- node "$BIN" init "$ROOT/$2" 2>&1 | grep -iE 'indexed|nodes|edges|error' | tail -2
19
- }
20
- clone_index https://github.com/MTKruto/MTKruto.git mtkruto # small (~322 TS)
21
- clone_index https://github.com/mvdicarlo/postybirb-plus.git postybirb # medium (~608 TS)
22
- clone_index https://github.com/shapeshift/web.git shapeshift # complex (~3.2k TS, 35-pkg monorepo)
23
- clone_index https://github.com/trezor/trezor-suite.git trezor # large (~8k TS monorepo)
24
- echo "###### SETUP DONE -> $ROOT"
1
+ #!/usr/bin/env bash
2
+ # Clone + index the 4 "not-trained-on" eval repos into $AGENT_EVAL_OUT/repos. These were
3
+ # selected via a no-tools memory-probe gate (Sonnet cannot answer their flow questions from
4
+ # memory — so the no-homegraph baseline is honest). Env: AGENT_EVAL_OUT=<scratch dir>
5
+ set -uo pipefail
6
+ HERE="$(cd "$(dirname "$0")" && pwd)"
7
+ ENGINE="$(cd "$HERE/../.." && pwd)"
8
+ BIN="$ENGINE/dist/bin/homegraph.js"
9
+ OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
10
+ ROOT="$OUT/repos"; mkdir -p "$ROOT"
11
+ export HOMEGRAPH_TELEMETRY=0 DO_NOT_TRACK=1
12
+ [ -f "$BIN" ] || { echo "engine not built: run 'npm run build' in $ENGINE first"; exit 1; }
13
+
14
+ clone_index() { # url name
15
+ echo "=== $2: clone ==="; rm -rf "$ROOT/$2"
16
+ git clone --quiet --depth 1 "$1" "$ROOT/$2" || { echo " clone FAILED"; return 1; }
17
+ echo "=== $2: index ==="
18
+ node "$BIN" init "$ROOT/$2" 2>&1 | grep -iE 'indexed|nodes|edges|error' | tail -2
19
+ }
20
+ clone_index https://github.com/MTKruto/MTKruto.git mtkruto # small (~322 TS)
21
+ clone_index https://github.com/mvdicarlo/postybirb-plus.git postybirb # medium (~608 TS)
22
+ clone_index https://github.com/shapeshift/web.git shapeshift # complex (~3.2k TS, 35-pkg monorepo)
23
+ clone_index https://github.com/trezor/trezor-suite.git trezor # large (~8k TS monorepo)
24
+ echo "###### SETUP DONE -> $ROOT"
@@ -1,72 +1,72 @@
1
- #!/usr/bin/env bash
2
- # Offload reasoning-OUTPUT-STYLE A/B — all homegraph-on, isolating the Worker's
3
- # output shape's effect on main-session tokens / latency / accuracy:
4
- # raw : HOMEGRAPH_OFFLOAD_DISABLE=1 (verbatim explore source, the floor)
5
- # refs : managed offload, default (Cerebras map re-expanded to verbatim, ~24K)
6
- # map : managed offload, STYLE=map (compact reasoned map + file:line anchors, ~1-3K)
7
- # src : managed offload, STYLE=src (map + cited line ranges only, ~1-5K)
8
- # Delegation BLOCKED by default (DISALLOW=Agent) so we measure the offload payload's
9
- # effect on the main Sonnet agent, not whether it spawns a Haiku Explore subagent.
10
- #
11
- # Usage: offload-eval-styles.sh <indexed-repo> <reps> "<question>"
12
- # Env: RESULTS=<file> AGENT_EVAL_OUT=<dir> REP_START=1 DISALLOW=Agent MODEL/EFFORT
13
- set -uo pipefail
14
- HERE="$(cd "$(dirname "$0")" && pwd)"
15
- ENGINE="$(cd "$HERE/../.." && pwd)"
16
- BIN="$ENGINE/dist/bin/homegraph.js"
17
- OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
18
- TARGET="${1:?usage: offload-eval-styles.sh <indexed-repo> <reps> \"<question>\"}"
19
- REPS="${2:?reps}"; Q="${3:?question}"
20
- RUNS="$OUT/runs"; EXTRACT="$HERE/offload-eval-metrics.mjs"
21
- RESULTS="${RESULTS:-$OUT/results-styles.jsonl}"
22
- REPO=$(basename "$TARGET")
23
- DISALLOW="${DISALLOW-Agent}" # default: block delegation. `DISALLOW= ` to allow.
24
- START="${REP_START:-1}"; END=$((START + REPS - 1))
25
- mkdir -p "$RUNS"
26
- command -v claude >/dev/null || { echo "no claude on PATH"; exit 1; }
27
- [ -d "$TARGET/.homegraph" ] || { echo "not indexed: $TARGET"; exit 1; }
28
- TARGET=$(cd "$TARGET" && pwd -P)
29
-
30
- prewarm() { # path extra-env
31
- pkill -9 -f "serve --mcp --path $1" 2>/dev/null; rm -f "$1/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
32
- env ${2:-} HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 node "$BIN" serve --mcp --path "$1" </dev/null >/dev/null 2>&1 &
33
- node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$1" \
34
- && echo " daemon warm" || echo " WARN daemon never bound"
35
- }
36
- kill_daemon() { pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1; }
37
-
38
- run() { # arm rep mcp-config usage-log-or-dash
39
- local arm="$1" rep="$2" cfg="$3" usage="$4" tag="$REPO-$1-$2"
40
- [ "$usage" != "-" ] && : > "$usage"
41
- ( cd "$TARGET" && claude -p "$Q" \
42
- --output-format stream-json --verbose --permission-mode bypassPermissions \
43
- --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
44
- ${DISALLOW:+--disallowedTools "$DISALLOW"} \
45
- --strict-mcp-config --mcp-config "$cfg" \
46
- </dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
47
- node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$usage" --arm "$arm" --rep "$rep" \
48
- --repo "$REPO" --tier styles --q "$Q" >> "$RESULTS"
49
- node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | ${o.tokBillable} billable tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}c/${o.ai.totalTokens}t | ok=${o.ok}`)' "$RESULTS"
50
- }
51
-
52
- # MCP configs: env baked into the daemon-spawn command claude uses.
53
- USAGE="$RUNS/$REPO-usage.jsonl"
54
- mkcfg() { # file extra-env-pairs(JSON array entries, comma-led or empty)
55
- printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1"%s,"node","%s","serve","--mcp","--path","%s"]}}}' "$1" "$BIN" "$TARGET"
56
- }
57
- CFG_RAW="$RUNS/mcp-sty-raw-$REPO.json"; mkcfg ',"HOMEGRAPH_OFFLOAD_DISABLE=1"' > "$CFG_RAW"
58
- CFG_REFS="$RUNS/mcp-sty-refs-$REPO.json"; mkcfg ",\"HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE\"" > "$CFG_REFS"
59
- CFG_MAP="$RUNS/mcp-sty-map-$REPO.json"; mkcfg ",\"HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE\",\"HOMEGRAPH_OFFLOAD_STYLE=map\"" > "$CFG_MAP"
60
- CFG_SRC="$RUNS/mcp-sty-src-$REPO.json"; mkcfg ",\"HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE\",\"HOMEGRAPH_OFFLOAD_STYLE=src\"" > "$CFG_SRC"
61
-
62
- echo "###### repo=$REPO reps=$START..$END model=${MODEL:-sonnet}/${EFFORT:-high} disallow=${DISALLOW:-<none>}"
63
- echo "###### Q=$Q"
64
- echo "== ARM raw =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_DISABLE=1"
65
- for r in $(seq "$START" "$END"); do run raw "$r" "$CFG_RAW" "-"; done; kill_daemon
66
- echo "== ARM refs =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
67
- for r in $(seq "$START" "$END"); do run refs "$r" "$CFG_REFS" "$USAGE"; done; kill_daemon
68
- echo "== ARM map =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE HOMEGRAPH_OFFLOAD_STYLE=map"
69
- for r in $(seq "$START" "$END"); do run map "$r" "$CFG_MAP" "$USAGE"; done; kill_daemon
70
- echo "== ARM src =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE HOMEGRAPH_OFFLOAD_STYLE=src"
71
- for r in $(seq "$START" "$END"); do run src "$r" "$CFG_SRC" "$USAGE"; done; kill_daemon
1
+ #!/usr/bin/env bash
2
+ # Offload reasoning-OUTPUT-STYLE A/B — all homegraph-on, isolating the Worker's
3
+ # output shape's effect on main-session tokens / latency / accuracy:
4
+ # raw : HOMEGRAPH_OFFLOAD_DISABLE=1 (verbatim explore source, the floor)
5
+ # refs : managed offload, default (Cerebras map re-expanded to verbatim, ~24K)
6
+ # map : managed offload, STYLE=map (compact reasoned map + file:line anchors, ~1-3K)
7
+ # src : managed offload, STYLE=src (map + cited line ranges only, ~1-5K)
8
+ # Delegation BLOCKED by default (DISALLOW=Agent) so we measure the offload payload's
9
+ # effect on the main Sonnet agent, not whether it spawns a Haiku Explore subagent.
10
+ #
11
+ # Usage: offload-eval-styles.sh <indexed-repo> <reps> "<question>"
12
+ # Env: RESULTS=<file> AGENT_EVAL_OUT=<dir> REP_START=1 DISALLOW=Agent MODEL/EFFORT
13
+ set -uo pipefail
14
+ HERE="$(cd "$(dirname "$0")" && pwd)"
15
+ ENGINE="$(cd "$HERE/../.." && pwd)"
16
+ BIN="$ENGINE/dist/bin/homegraph.js"
17
+ OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
18
+ TARGET="${1:?usage: offload-eval-styles.sh <indexed-repo> <reps> \"<question>\"}"
19
+ REPS="${2:?reps}"; Q="${3:?question}"
20
+ RUNS="$OUT/runs"; EXTRACT="$HERE/offload-eval-metrics.mjs"
21
+ RESULTS="${RESULTS:-$OUT/results-styles.jsonl}"
22
+ REPO=$(basename "$TARGET")
23
+ DISALLOW="${DISALLOW-Agent}" # default: block delegation. `DISALLOW= ` to allow.
24
+ START="${REP_START:-1}"; END=$((START + REPS - 1))
25
+ mkdir -p "$RUNS"
26
+ command -v claude >/dev/null || { echo "no claude on PATH"; exit 1; }
27
+ [ -d "$TARGET/.homegraph" ] || { echo "not indexed: $TARGET"; exit 1; }
28
+ TARGET=$(cd "$TARGET" && pwd -P)
29
+
30
+ prewarm() { # path extra-env
31
+ pkill -9 -f "serve --mcp --path $1" 2>/dev/null; rm -f "$1/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
32
+ env ${2:-} HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 node "$BIN" serve --mcp --path "$1" </dev/null >/dev/null 2>&1 &
33
+ node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$1" \
34
+ && echo " daemon warm" || echo " WARN daemon never bound"
35
+ }
36
+ kill_daemon() { pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1; }
37
+
38
+ run() { # arm rep mcp-config usage-log-or-dash
39
+ local arm="$1" rep="$2" cfg="$3" usage="$4" tag="$REPO-$1-$2"
40
+ [ "$usage" != "-" ] && : > "$usage"
41
+ ( cd "$TARGET" && claude -p "$Q" \
42
+ --output-format stream-json --verbose --permission-mode bypassPermissions \
43
+ --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
44
+ ${DISALLOW:+--disallowedTools "$DISALLOW"} \
45
+ --strict-mcp-config --mcp-config "$cfg" \
46
+ </dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
47
+ node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$usage" --arm "$arm" --rep "$rep" \
48
+ --repo "$REPO" --tier styles --q "$Q" >> "$RESULTS"
49
+ node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | ${o.tokBillable} billable tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}c/${o.ai.totalTokens}t | ok=${o.ok}`)' "$RESULTS"
50
+ }
51
+
52
+ # MCP configs: env baked into the daemon-spawn command claude uses.
53
+ USAGE="$RUNS/$REPO-usage.jsonl"
54
+ mkcfg() { # file extra-env-pairs(JSON array entries, comma-led or empty)
55
+ printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1"%s,"node","%s","serve","--mcp","--path","%s"]}}}' "$1" "$BIN" "$TARGET"
56
+ }
57
+ CFG_RAW="$RUNS/mcp-sty-raw-$REPO.json"; mkcfg ',"HOMEGRAPH_OFFLOAD_DISABLE=1"' > "$CFG_RAW"
58
+ CFG_REFS="$RUNS/mcp-sty-refs-$REPO.json"; mkcfg ",\"HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE\"" > "$CFG_REFS"
59
+ CFG_MAP="$RUNS/mcp-sty-map-$REPO.json"; mkcfg ",\"HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE\",\"HOMEGRAPH_OFFLOAD_STYLE=map\"" > "$CFG_MAP"
60
+ CFG_SRC="$RUNS/mcp-sty-src-$REPO.json"; mkcfg ",\"HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE\",\"HOMEGRAPH_OFFLOAD_STYLE=src\"" > "$CFG_SRC"
61
+
62
+ echo "###### repo=$REPO reps=$START..$END model=${MODEL:-sonnet}/${EFFORT:-high} disallow=${DISALLOW:-<none>}"
63
+ echo "###### Q=$Q"
64
+ echo "== ARM raw =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_DISABLE=1"
65
+ for r in $(seq "$START" "$END"); do run raw "$r" "$CFG_RAW" "-"; done; kill_daemon
66
+ echo "== ARM refs =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
67
+ for r in $(seq "$START" "$END"); do run refs "$r" "$CFG_REFS" "$USAGE"; done; kill_daemon
68
+ echo "== ARM map =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE HOMEGRAPH_OFFLOAD_STYLE=map"
69
+ for r in $(seq "$START" "$END"); do run map "$r" "$CFG_MAP" "$USAGE"; done; kill_daemon
70
+ echo "== ARM src =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE HOMEGRAPH_OFFLOAD_STYLE=src"
71
+ for r in $(seq "$START" "$END"); do run src "$r" "$CFG_SRC" "$USAGE"; done; kill_daemon
72
72
  echo "###### DONE $REPO — judge: node $HERE/offload-eval-judge.mjs --results $RESULTS --truth $HERE/offload-eval-ground-truth.json --out $OUT/judged-styles.jsonl"