homegraph 1.5.1 → 1.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (258) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +305 -305
  3. package/dist/bin/homegraph.js +9 -9
  4. package/dist/db/index.js +36 -36
  5. package/dist/db/migrations.js +37 -37
  6. package/dist/db/queries.js +156 -156
  7. package/dist/db/schema.sql +203 -203
  8. package/dist/directory.js +5 -5
  9. package/dist/extraction/index.d.ts +1 -1
  10. package/dist/extraction/index.d.ts.map +1 -1
  11. package/dist/extraction/index.js +54 -43
  12. package/dist/extraction/index.js.map +1 -1
  13. package/dist/extraction/languages/arkts.d.ts +53 -6
  14. package/dist/extraction/languages/arkts.d.ts.map +1 -1
  15. package/dist/extraction/languages/arkts.js +565 -65
  16. package/dist/extraction/languages/arkts.js.map +1 -1
  17. package/dist/extraction/wasm/tree-sitter-c_sharp.wasm +0 -0
  18. package/dist/extraction/wasm/tree-sitter-cfml.wasm +0 -0
  19. package/dist/extraction/wasm/tree-sitter-cfquery.wasm +0 -0
  20. package/dist/extraction/wasm/tree-sitter-cfscript.wasm +0 -0
  21. package/dist/extraction/wasm/tree-sitter-cobol.wasm +0 -0
  22. package/dist/extraction/wasm/tree-sitter-erlang.wasm +0 -0
  23. package/dist/extraction/wasm/tree-sitter-nix.wasm +0 -0
  24. package/dist/extraction/wasm/tree-sitter-pascal.wasm +0 -0
  25. package/dist/extraction/wasm/tree-sitter-vbnet.wasm +0 -0
  26. package/dist/index.d.ts.map +1 -1
  27. package/dist/index.js +3 -2
  28. package/dist/index.js.map +1 -1
  29. package/dist/installer/instructions-template.js +9 -9
  30. package/dist/mcp/diff-impact.d.ts +131 -0
  31. package/dist/mcp/diff-impact.d.ts.map +1 -0
  32. package/dist/mcp/diff-impact.js +385 -0
  33. package/dist/mcp/diff-impact.js.map +1 -0
  34. package/dist/mcp/liveness-watchdog.js +51 -51
  35. package/dist/mcp/server-instructions.d.ts +1 -1
  36. package/dist/mcp/server-instructions.d.ts.map +1 -1
  37. package/dist/mcp/server-instructions.js +49 -47
  38. package/dist/mcp/server-instructions.js.map +1 -1
  39. package/dist/mcp/tools.d.ts +21 -5
  40. package/dist/mcp/tools.d.ts.map +1 -1
  41. package/dist/mcp/tools.js +305 -55
  42. package/dist/mcp/tools.js.map +1 -1
  43. package/dist/resolution/callback-synthesizer.d.ts +2 -1
  44. package/dist/resolution/callback-synthesizer.d.ts.map +1 -1
  45. package/dist/resolution/callback-synthesizer.js +153 -53
  46. package/dist/resolution/callback-synthesizer.js.map +1 -1
  47. package/dist/resolution/frameworks/arkts-entry.d.ts.map +1 -1
  48. package/dist/resolution/frameworks/arkts-entry.js +20 -9
  49. package/dist/resolution/frameworks/arkts-entry.js.map +1 -1
  50. package/dist/resolution/index.d.ts.map +1 -1
  51. package/dist/resolution/index.js +1 -1
  52. package/dist/resolution/index.js.map +1 -1
  53. package/dist/resolution/memory-budget.d.ts +33 -0
  54. package/dist/resolution/memory-budget.d.ts.map +1 -1
  55. package/dist/resolution/memory-budget.js +58 -0
  56. package/dist/resolution/memory-budget.js.map +1 -1
  57. package/dist/resolution/resolver-pool.d.ts +2 -0
  58. package/dist/resolution/resolver-pool.d.ts.map +1 -1
  59. package/dist/resolution/resolver-pool.js +4 -0
  60. package/dist/resolution/resolver-pool.js.map +1 -1
  61. package/dist/spec/db/commit-node.js +4 -4
  62. package/dist/spec/db/fragment-node.js +10 -10
  63. package/dist/spec/db/fts.js +8 -8
  64. package/dist/spec/db/relations.js +88 -88
  65. package/dist/spec/db/schema.js +6 -6
  66. package/dist/spec/db/schema.sql +121 -121
  67. package/dist/spec/db/spec-node.js +4 -4
  68. package/dist/spec/llm/prompts.js +53 -53
  69. package/dist/ui/shimmer-progress.d.ts.map +1 -1
  70. package/dist/ui/shimmer-progress.js +4 -1
  71. package/dist/ui/shimmer-progress.js.map +1 -1
  72. package/package.json +61 -62
  73. package/dist/extraction/languages/arkts-viewtree.d.ts +0 -22
  74. package/dist/extraction/languages/arkts-viewtree.d.ts.map +0 -1
  75. package/dist/extraction/languages/arkts-viewtree.js +0 -133
  76. package/dist/extraction/languages/arkts-viewtree.js.map +0 -1
  77. package/dist/reasoning/config.d.ts +0 -45
  78. package/dist/reasoning/config.d.ts.map +0 -1
  79. package/dist/reasoning/config.js +0 -171
  80. package/dist/reasoning/config.js.map +0 -1
  81. package/dist/reasoning/credentials.d.ts +0 -5
  82. package/dist/reasoning/credentials.d.ts.map +0 -1
  83. package/dist/reasoning/credentials.js +0 -83
  84. package/dist/reasoning/credentials.js.map +0 -1
  85. package/dist/reasoning/login.d.ts +0 -21
  86. package/dist/reasoning/login.d.ts.map +0 -1
  87. package/dist/reasoning/login.js +0 -85
  88. package/dist/reasoning/login.js.map +0 -1
  89. package/dist/reasoning/reasoner.d.ts +0 -43
  90. package/dist/reasoning/reasoner.d.ts.map +0 -1
  91. package/dist/reasoning/reasoner.js +0 -308
  92. package/dist/reasoning/reasoner.js.map +0 -1
  93. package/dist/spec/evolve/llm-client.d.ts +0 -50
  94. package/dist/spec/evolve/llm-client.d.ts.map +0 -1
  95. package/dist/spec/evolve/llm-client.js +0 -176
  96. package/dist/spec/evolve/llm-client.js.map +0 -1
  97. package/dist/spec/evolve/logic-checker.d.ts +0 -12
  98. package/dist/spec/evolve/logic-checker.d.ts.map +0 -1
  99. package/dist/spec/evolve/logic-checker.js +0 -24
  100. package/dist/spec/evolve/logic-checker.js.map +0 -1
  101. package/dist/spec/llm/index.d.ts +0 -3
  102. package/dist/spec/llm/index.d.ts.map +0 -1
  103. package/dist/spec/llm/index.js +0 -11
  104. package/dist/spec/llm/index.js.map +0 -1
  105. package/dist/spec/mining/diff-parser.d.ts +0 -33
  106. package/dist/spec/mining/diff-parser.d.ts.map +0 -1
  107. package/dist/spec/mining/diff-parser.js +0 -166
  108. package/dist/spec/mining/diff-parser.js.map +0 -1
  109. package/dist/spec/mining/git-scanner.d.ts +0 -103
  110. package/dist/spec/mining/git-scanner.d.ts.map +0 -1
  111. package/dist/spec/mining/git-scanner.js +0 -307
  112. package/dist/spec/mining/git-scanner.js.map +0 -1
  113. package/dist/spec/mining/pipeline.d.ts +0 -53
  114. package/dist/spec/mining/pipeline.d.ts.map +0 -1
  115. package/dist/spec/mining/pipeline.js +0 -178
  116. package/dist/spec/mining/pipeline.js.map +0 -1
  117. package/dist/spec/mining/scope-resolver.d.ts +0 -45
  118. package/dist/spec/mining/scope-resolver.d.ts.map +0 -1
  119. package/dist/spec/mining/scope-resolver.js +0 -103
  120. package/dist/spec/mining/scope-resolver.js.map +0 -1
  121. package/dist/spec/mining/spec-extractor.d.ts +0 -69
  122. package/dist/spec/mining/spec-extractor.d.ts.map +0 -1
  123. package/dist/spec/mining/spec-extractor.js +0 -369
  124. package/dist/spec/mining/spec-extractor.js.map +0 -1
  125. package/dist/spec/utils.d.ts +0 -167
  126. package/dist/spec/utils.d.ts.map +0 -1
  127. package/dist/spec/utils.js +0 -463
  128. package/dist/spec/utils.js.map +0 -1
  129. package/scripts/_tmp-cfwk-resolve.log +0 -0
  130. package/scripts/_tmp-cfwk-sig.log +0 -0
  131. package/scripts/_tmp-cfwk-vt.log +0 -0
  132. package/scripts/add-lang/bench.sh +0 -60
  133. package/scripts/add-lang/check-grammar.mjs +0 -75
  134. package/scripts/add-lang/dump-ast.mjs +0 -103
  135. package/scripts/add-lang/verify-extraction.mjs +0 -70
  136. package/scripts/agent-eval/ab-adoption.sh +0 -91
  137. package/scripts/agent-eval/ab-hook.sh +0 -86
  138. package/scripts/agent-eval/ab-impl.sh +0 -78
  139. package/scripts/agent-eval/ab-new-vs-baseline.sh +0 -102
  140. package/scripts/agent-eval/ab-sufficiency.sh +0 -78
  141. package/scripts/agent-eval/arms-F.sh +0 -21
  142. package/scripts/agent-eval/arms-matrix.sh +0 -37
  143. package/scripts/agent-eval/audit.sh +0 -68
  144. package/scripts/agent-eval/bench-readme.sh +0 -28
  145. package/scripts/agent-eval/bench-why-repo.sh +0 -22
  146. package/scripts/agent-eval/block-read-hook.sh +0 -19
  147. package/scripts/agent-eval/hook-settings.json +0 -15
  148. package/scripts/agent-eval/itrun.sh +0 -120
  149. package/scripts/agent-eval/offload-eval-3arm.sh +0 -72
  150. package/scripts/agent-eval/offload-eval-cost.mjs +0 -133
  151. package/scripts/agent-eval/offload-eval-effort.mjs +0 -108
  152. package/scripts/agent-eval/offload-eval-frontload-matrix.sh +0 -25
  153. package/scripts/agent-eval/offload-eval-frontload.sh +0 -47
  154. package/scripts/agent-eval/offload-eval-ground-truth.json +0 -18
  155. package/scripts/agent-eval/offload-eval-hook.mjs +0 -84
  156. package/scripts/agent-eval/offload-eval-judge.mjs +0 -103
  157. package/scripts/agent-eval/offload-eval-matrix.sh +0 -20
  158. package/scripts/agent-eval/offload-eval-metrics.mjs +0 -94
  159. package/scripts/agent-eval/offload-eval-refs1.sh +0 -50
  160. package/scripts/agent-eval/offload-eval-setup.sh +0 -24
  161. package/scripts/agent-eval/offload-eval-styles.sh +0 -72
  162. package/scripts/agent-eval/offload-eval-summarize.mjs +0 -68
  163. package/scripts/agent-eval/offload-eval.md +0 -76
  164. package/scripts/agent-eval/parse-arms.mjs +0 -116
  165. package/scripts/agent-eval/parse-bench-readme.mjs +0 -84
  166. package/scripts/agent-eval/parse-run.mjs +0 -45
  167. package/scripts/agent-eval/parse-session.mjs +0 -93
  168. package/scripts/agent-eval/probe-context.mjs +0 -21
  169. package/scripts/agent-eval/probe-explore.mjs +0 -40
  170. package/scripts/agent-eval/probe-node.mjs +0 -20
  171. package/scripts/agent-eval/probe-sweep.mjs +0 -119
  172. package/scripts/agent-eval/probe-trace.mjs +0 -20
  173. package/scripts/agent-eval/redirect-read-hook.sh +0 -38
  174. package/scripts/agent-eval/repro-concurrent-explore.mjs +0 -119
  175. package/scripts/agent-eval/repro-daemon-clients.mjs +0 -125
  176. package/scripts/agent-eval/run-agent.sh +0 -34
  177. package/scripts/agent-eval/run-all.sh +0 -75
  178. package/scripts/agent-eval/run-arms.sh +0 -56
  179. package/scripts/agent-eval/seq-matrix.mjs +0 -137
  180. package/scripts/bench-arkts-init-rss.log +0 -0
  181. package/scripts/build-bundle.sh +0 -123
  182. package/scripts/exp_boundary_eval/README.md +0 -247
  183. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-310.pyc +0 -0
  184. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-38.pyc +0 -0
  185. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-310.pyc +0 -0
  186. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-38.pyc +0 -0
  187. package/scripts/exp_boundary_eval/__pycache__/deveco_arm.cpython-38.pyc +0 -0
  188. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-310.pyc +0 -0
  189. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-38.pyc +0 -0
  190. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-310.pyc +0 -0
  191. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-38.pyc +0 -0
  192. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-310.pyc +0 -0
  193. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-38.pyc +0 -0
  194. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-310.pyc +0 -0
  195. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-38.pyc +0 -0
  196. package/scripts/exp_boundary_eval/__pycache__/win_mcp_launcher.cpython-38.pyc +0 -0
  197. package/scripts/exp_boundary_eval/_test_mcp_chain.py +0 -78
  198. package/scripts/exp_boundary_eval/_test_stdin.py +0 -8
  199. package/scripts/exp_boundary_eval/_utils.py +0 -1116
  200. package/scripts/exp_boundary_eval/analyze.py +0 -1313
  201. package/scripts/exp_boundary_eval/deveco_arm.py +0 -519
  202. package/scripts/exp_boundary_eval/run_all.py +0 -378
  203. package/scripts/exp_boundary_eval/run_one.py +0 -165
  204. package/scripts/exp_boundary_eval/run_session.py +0 -158
  205. package/scripts/exp_boundary_eval/setup.py +0 -120
  206. package/scripts/exp_boundary_eval/win_mcp_launcher.py +0 -73
  207. package/scripts/exp_boundary_eval/win_mcp_stdio_wrap.js +0 -36
  208. package/scripts/exp_boundary_eval/win_node_launcher.py +0 -24
  209. package/scripts/extract-release-notes.mjs +0 -130
  210. package/scripts/local-install.sh +0 -41
  211. package/scripts/npm-sdk.js +0 -75
  212. package/scripts/npm-shim.js +0 -275
  213. package/scripts/ohos-sdk-publish.mjs +0 -133
  214. package/scripts/pack-npm.sh +0 -119
  215. package/scripts/prepare-release.mjs +0 -270
  216. package/scripts/probe-arkts-mem-why-run.log +0 -0
  217. package/scripts/probe-banner-livecard-ir.log +0 -0
  218. package/scripts/probe-cfwk-ast.stderr.log +0 -0
  219. package/scripts/probe-cfwk-ast.stdout.log +0 -0
  220. package/scripts/probe-cfwk-attr-shape.log +0 -0
  221. package/scripts/probe-cfwk-cfgdump.stderr.log +0 -0
  222. package/scripts/probe-cfwk-cfgdump.stdout.log +0 -0
  223. package/scripts/probe-cfwk-cvc.stderr.log +0 -0
  224. package/scripts/probe-cfwk-cvc.stdout.log +0 -0
  225. package/scripts/probe-cfwk-diag2.log +0 -0
  226. package/scripts/probe-cfwk-diag3.log +0 -0
  227. package/scripts/probe-cfwk-fedbg.stderr.log +0 -0
  228. package/scripts/probe-cfwk-fedbg.stdout.log +0 -0
  229. package/scripts/probe-cfwk-fileresult.log +0 -0
  230. package/scripts/probe-cfwk-fix.stderr.log +0 -0
  231. package/scripts/probe-cfwk-fix.stdout.log +0 -0
  232. package/scripts/probe-cfwk-fix2.stderr.log +0 -0
  233. package/scripts/probe-cfwk-fix2.stdout.log +0 -0
  234. package/scripts/probe-cfwk-fix3.stderr.log +0 -0
  235. package/scripts/probe-cfwk-fix3.stdout.log +0 -0
  236. package/scripts/probe-cfwk-foreach.log +0 -0
  237. package/scripts/probe-cfwk-getmethod-throw.log +0 -0
  238. package/scripts/probe-cfwk-hg-extract.log +0 -0
  239. package/scripts/probe-cfwk-pr1003-noprior.stderr.log +0 -0
  240. package/scripts/probe-cfwk-pr1003-noprior.stdout.log +0 -0
  241. package/scripts/probe-cfwk-pr1003.stderr.log +0 -0
  242. package/scripts/probe-cfwk-pr1003.stdout.log +0 -0
  243. package/scripts/probe-cfwk-preroot-noprior.stderr.log +0 -0
  244. package/scripts/probe-cfwk-preroot-noprior.stdout.log +0 -0
  245. package/scripts/probe-cfwk-preroot-prior.stderr.log +0 -0
  246. package/scripts/probe-cfwk-preroot-prior.stdout.log +0 -0
  247. package/scripts/probe-cfwk-preroot-skipstate.stderr.log +0 -0
  248. package/scripts/probe-cfwk-preroot-skipstate.stdout.log +0 -0
  249. package/scripts/probe-cfwk-resolve-sim.log +0 -0
  250. package/scripts/probe-cfwk-tree-shape.log +0 -0
  251. package/scripts/probe-cfwk-walk-abort.log +0 -0
  252. package/scripts/probe-cfwk.log +0 -0
  253. package/scripts/probe-force-index.log +0 -0
  254. package/scripts/probe-no-force-index.log +0 -0
  255. package/scripts/probe-samefile-ir.stderr.log +0 -0
  256. package/scripts/probe-samefile-ir.stdout.log +0 -224
  257. package/scripts/probe-sdk-vs-project.log +0 -0
  258. package/scripts/probe-viewtree-downgrade.log +0 -0
@@ -1,37 +0,0 @@
1
- #!/usr/bin/env bash
2
- # Drive the tool-surface ablation across the chosen repos × arms (A–E).
3
- # Arms A–D ask the canonical FLOW question; arm E asks a NON-flow survey
4
- # question (the control probe — should degrade without explore+context).
5
- # Output: /tmp/arms/<repo>/<arm>-r<n>.jsonl (parse with parse-arms.mjs).
6
- set -uo pipefail
7
- HARNESS="$(cd "$(dirname "$0")" && pwd)"
8
- RUNS="${RUNS:-2}"
9
- C="${CORPUS:-/tmp/homegraph-corpus}"
10
- NFQ='What are the main modules/components of this codebase and what does each one do? Give an overview of how it is organized.'
11
-
12
- # repo-path|flow-question (2 small, 2 medium, 2 large — spans the size range)
13
- ROWS=(
14
- "$C/flutter-samples/add_to_app/books/flutter_module_books|How does the books UI build and what child widgets does it show?"
15
- "$C/aspnet-realworld|How is creating an article handled? Trace the controller to the service."
16
- "$C/spring-mall|How is a product-list request handled? Trace the controller to the service."
17
- "$C/vapor-spi|How is a package-show request handled? Name the route and controller."
18
- "$C/excalidraw|How does updating an element re-render the canvas on screen? Trace the flow."
19
- "$C/spring-halo|How is publishing a post handled? Trace the controller to the service."
20
- )
21
-
22
- echo "### ARMS MATRIX START $(date) RUNS=$RUNS"
23
- for row in "${ROWS[@]}"; do
24
- repo="${row%%|*}"; q="${row#*|}"
25
- for arm in A B C D; do
26
- for r in $(seq 1 "$RUNS"); do
27
- bash "$HARNESS/run-arms.sh" "$repo" "$q" "$arm" "$r"
28
- done
29
- done
30
- done
31
- # E: non-flow control probe on two repos (must degrade without explore+context)
32
- for repo in "$C/excalidraw" "$C/spring-mall"; do
33
- for r in $(seq 1 "$RUNS"); do
34
- bash "$HARNESS/run-arms.sh" "$repo" "$NFQ" E "$r"
35
- done
36
- done
37
- echo "### ARMS MATRIX COMPLETE $(date)"
@@ -1,68 +0,0 @@
1
- #!/usr/bin/env bash
2
- # One-shot HomeGraph quality audit:
3
- # set version -> ensure corpus repo -> wipe+reindex with that version ->
4
- # run with/without A/B -> restore the local dev link.
5
- #
6
- # Usage: audit.sh <version> <repo-name> <repo-url> "<question>" [headless|all]
7
- # <version> "local" (build + npm link this repo) | "latest" | a version (e.g. 0.7.10)
8
- # <repo-name> dir name under the corpus dir
9
- # <repo-url> git URL (cloned --depth 1 when the repo dir is missing)
10
- # [mode] headless (default) | all (also the interactive tmux arms)
11
- # Env: CORPUS corpus dir (default: /tmp/homegraph-corpus)
12
- set -uo pipefail
13
-
14
- VERSION="${1:?usage: audit.sh <version> <repo-name> <repo-url> \"<question>\" [mode]}"
15
- NAME="${2:?repo-name required}"
16
- URL="${3:?repo-url required}"
17
- Q="${4:?question required}"
18
- MODE="${5:-headless}"
19
-
20
- HARNESS="$(cd "$(dirname "$0")" && pwd)"
21
- REPO_ROOT="$(cd "$HARNESS/../.." && pwd)" # homegraph repo root
22
- CORPUS="${CORPUS:-/tmp/homegraph-corpus}"
23
- REPO="$CORPUS/$NAME"
24
- PKG="homegraph"
25
-
26
- echo "==================== HomeGraph audit ===================="
27
- echo "version=$VERSION repo=$NAME mode=$MODE corpus=$CORPUS"
28
- echo
29
-
30
- # 1. Set the homegraph version under test (mutates the global install).
31
- if [ "$VERSION" = local ]; then
32
- echo "→ [1/4] building + linking local dev build (local-install.sh)"
33
- ( cd "$REPO_ROOT" && ./scripts/local-install.sh ) || { echo "local-install.sh failed"; exit 1; }
34
- else
35
- echo "→ [1/4] installing $PKG@$VERSION globally"
36
- npm install -g "$PKG@$VERSION" || { echo "npm install -g $PKG@$VERSION failed"; exit 1; }
37
- fi
38
- ACTUAL="$(homegraph --version 2>/dev/null || echo '?')"
39
- echo " homegraph on PATH: $(command -v homegraph) -> $ACTUAL"
40
-
41
- # 2. Ensure the corpus repo exists (clone shallow if missing, reuse if present).
42
- mkdir -p "$CORPUS"
43
- if [ -d "$REPO/.git" ]; then
44
- echo "→ [2/4] reusing existing checkout: $REPO"
45
- else
46
- echo "→ [2/4] cloning $URL"
47
- git clone --depth 1 "$URL" "$REPO" || { echo "git clone failed"; exit 1; }
48
- fi
49
-
50
- # 3. Wipe + re-index with THIS version (the index must be built by the same
51
- # binary that serves it — different versions extract differently).
52
- echo "→ [3/4] wiping .homegraph and re-indexing with $ACTUAL"
53
- rm -rf "$REPO/.homegraph"
54
- ( cd "$REPO" && homegraph init -i ) || { echo "indexing failed"; exit 1; }
55
-
56
- # 4. Run the with/without A/B.
57
- echo "→ [4/4] running A/B harness (mode=$MODE)"
58
- bash "$HARNESS/run-all.sh" "$REPO" "$Q" "$MODE"
59
-
60
- # Restore the dev link (the normal working state in this repo).
61
- echo
62
- echo "→ restoring local dev link (local-install.sh)"
63
- if ( cd "$REPO_ROOT" && ./scripts/local-install.sh >/dev/null 2>&1 ); then
64
- echo " global homegraph restored to dev build"
65
- else
66
- echo " WARN: restore failed — run ./scripts/local-install.sh manually"
67
- fi
68
- echo "==================== audit complete ===================="
@@ -1,28 +0,0 @@
1
- #!/usr/bin/env bash
2
- # Re-run the README "Benchmark Results" A/B (with vs without homegraph) on the
3
- # current build: the 7 README repos, same queries, RUNS per arm (default 4).
4
- # Output → /tmp/ab-readme/<repo>/run<n>/run-headless-{with,without}.jsonl
5
- # Aggregate with parse-bench-readme.mjs. Repos must be cloned + indexed under
6
- # $CORPUS (default /tmp/homegraph-corpus) by the build under test.
7
- set -uo pipefail
8
- H="$(cd "$(dirname "$0")" && pwd)"
9
- C="${CORPUS:-/tmp/homegraph-corpus}"
10
- RUNS="${RUNS:-4}"
11
- ROWS=(
12
- "vscode|How does the extension host communicate with the main process?"
13
- "excalidraw|How does Excalidraw render and update canvas elements?"
14
- "django|How does Django's ORM build and execute a query from a QuerySet?"
15
- "tokio|How does tokio schedule and run async tasks on its runtime?"
16
- "okhttp|How does OkHttp process a request through its interceptor chain?"
17
- "gin|How does gin route requests through its middleware chain?"
18
- "alamofire|How does Alamofire build, send, and validate a request?"
19
- )
20
- echo "### README A/B START $(date) RUNS=$RUNS"
21
- for row in "${ROWS[@]}"; do
22
- repo="${row%%|*}"; q="${row#*|}"
23
- echo "===== $repo ====="
24
- for run in $(seq 1 "$RUNS"); do
25
- AGENT_EVAL_OUT="/tmp/ab-readme/$repo/run$run" bash "$H/run-all.sh" "$C/$repo" "$q" headless 2>&1 | grep -E "exit [0-9]" || echo " run$run: (no exit line)"
26
- done
27
- done
28
- echo "### README A/B DONE $(date)"
@@ -1,22 +0,0 @@
1
- #!/usr/bin/env bash
2
- # One README repo, WITH-homegraph only, N runs. Each run appends a why-Read
3
- # diagnostic so the agent explains any Read/Grep. (The WITHOUT baseline is
4
- # homegraph-independent and already in the README — no point re-running it.)
5
- # Output -> /tmp/ab-why/<repo>/with<n>.jsonl
6
- # Usage: bench-why-repo.sh <repo-path> "<query>" [N]
7
- set -uo pipefail
8
- REPO="$1"; Q="$2"; N="${3:-4}"
9
- NAME="$(basename "$REPO")"
10
- CG="/Users/colby/Development/Personal/homegraph/dist/bin/homegraph.js"
11
- OUT="/tmp/ab-why/$NAME"; mkdir -p "$OUT"
12
- WHY=$'\n\nIMPORTANT — diagnostic: if you use the Read or Grep tool at ANY point, for EACH such call explain why homegraph_explore / homegraph_node did not already give you what you needed. End your entire answer with a section titled exactly "## Why I read" listing every Read and Grep you made and the precise reason homegraph fell short for it. If you used neither, write "## Why I read" then "none — homegraph was sufficient."'
13
- printf '{"mcpServers":{"homegraph":{"command":"%s","args":["serve","--mcp","--path","%s"]}}}' "$CG" "$REPO" > "$OUT/cg.json"
14
-
15
- for i in $(seq 1 "$N"); do
16
- pkill -f "serve --mcp" 2>/dev/null; sleep 1; rm -f "$REPO/.homegraph/daemon.sock"
17
- ( cd "$REPO" && claude -p "$Q$WHY" --output-format stream-json --verbose \
18
- --permission-mode bypassPermissions --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
19
- --strict-mcp-config --mcp-config "$OUT/cg.json" > "$OUT/with$i.jsonl" 2>"$OUT/with$i.err" )
20
- echo "WITH run $i: exit $? ($(wc -l < "$OUT/with$i.jsonl" | tr -d ' ') lines)"
21
- done
22
- echo "DONE $NAME"
@@ -1,19 +0,0 @@
1
- #!/usr/bin/env bash
2
- # PreToolUse hook (experiment): deny Read of homegraph-indexed source files and
3
- # steer the agent to homegraph_explore/homegraph_node instead. Tests whether
4
- # homegraph can FULLY replace Read for code-understanding once the escape hatch
5
- # is removed. Non-source reads (config, .env, markdown, new files) pass through.
6
- #
7
- # Wire via: claude ... --settings scripts/agent-eval/hook-settings.json
8
- set -uo pipefail
9
- input="$(cat)"
10
- fp="$(printf '%s' "$input" | jq -r '.tool_input.file_path // empty' 2>/dev/null)"
11
-
12
- case "$fp" in
13
- *.ts|*.tsx|*.js|*.jsx|*.mjs|*.cjs|*.py|*.go|*.rs|*.java|*.rb|*.php|*.swift|*.kt|*.kts|*.c|*.cc|*.cpp|*.h|*.hpp|*.cs|*.lua|*.vue|*.svelte)
14
- msg="Read is disabled for source files in this session — homegraph already has this file indexed (with line numbers, kept in sync on every change). Use homegraph_explore (several related symbols at once) or homegraph_node (one symbol's full source). If a symbol you need wasn't in a prior explore, run ANOTHER homegraph_explore with its exact name instead of reading the file."
15
- jq -n --arg m "$msg" '{reason:$m, hookSpecificOutput:{hookEventName:"PreToolUse",permissionDecision:"deny",permissionDecisionReason:$m}}'
16
- exit 0
17
- ;;
18
- esac
19
- exit 0
@@ -1,15 +0,0 @@
1
- {
2
- "hooks": {
3
- "PreToolUse": [
4
- {
5
- "matcher": "Read",
6
- "hooks": [
7
- {
8
- "type": "command",
9
- "command": "bash /Users/colby/Development/Personal/homegraph/scripts/agent-eval/block-read-hook.sh"
10
- }
11
- ]
12
- }
13
- ]
14
- }
15
- }
@@ -1,120 +0,0 @@
1
- #!/usr/bin/env bash
2
- # Drive an INTERACTIVE Claude Code session in tmux, send a prompt, wait for the
3
- # agent to finish, then print the tool-call breakdown from the session logs.
4
- #
5
- # Why interactive (not `claude -p`): headless print-mode picks the
6
- # general-purpose subagent, while real interactive sessions delegate to the
7
- # Explore subagent (or drive homegraph from the main thread). Only the
8
- # interactive TUI reproduces the behavior users actually see. (Idle-detection
9
- # technique borrowed from devpit's WaitForIdle.)
10
- #
11
- # Usage: itrun.sh <repo-path> <label> "<prompt>"
12
- # Output dir: $AGENT_EVAL_OUT (default /tmp/agent-eval)
13
- # Requires: tmux 3.0+, a logged-in `claude` CLI, homegraph MCP configured.
14
- set -uo pipefail
15
- REPO="$1"; LABEL="$2"; PROMPT="$3"
16
- SESSION="cgt_${LABEL}"
17
- OUT_DIR="${AGENT_EVAL_OUT:-/tmp/agent-eval}"; mkdir -p "$OUT_DIR"
18
- OUT="$OUT_DIR/itrun-${LABEL}.txt"
19
- HERE="$(cd "$(dirname "$0")" && pwd)"
20
-
21
- cap() { tmux capture-pane -p -t "$SESSION" -S -40; }
22
-
23
- tmux kill-session -t "$SESSION" 2>/dev/null
24
-
25
- # Wide pane so the TUI doesn't hard-wrap tool lines.
26
- tmux new-session -d -s "$SESSION" -x 230 -y 60
27
- tmux send-keys -t "$SESSION" "cd $REPO && claude --dangerously-skip-permissions ${CLAUDE_EXTRA_ARGS:-}" Enter
28
-
29
- # Wait for the ❯ prompt (claude drew its UI), up to 60s. NOTE: ❯ appears on the
30
- # welcome screen seconds before the input actually accepts keystrokes, so this is
31
- # necessary but NOT sufficient — the type-and-verify loop below is what proves
32
- # the input is live.
33
- ready=0
34
- for _ in $(seq 1 120); do
35
- cap | grep -q "❯" && { ready=1; break; }
36
- sleep 0.5
37
- done
38
- [ "$ready" = 1 ] || { echo "claude never drew its UI"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
39
-
40
- # Accept the per-folder "Is this a project you trust?" dialog if it shows (first
41
- # time claude opens a given repo). Option 1 ("Yes, I trust this folder") is
42
- # pre-selected, so Enter accepts. This dialog also contains ❯, so it must be
43
- # cleared before the type-and-verify loop or keystrokes land on the menu.
44
- for _ in $(seq 1 20); do
45
- cap | grep -q "trust this folder" || break
46
- tmux send-keys -t "$SESSION" Enter
47
- sleep 1
48
- done
49
-
50
- # Type-and-verify: send the prompt, confirm a distinctive chunk of it actually
51
- # landed in the input box, retry if it didn't (handles the early-❯ race where
52
- # the welcome screen shows the prompt glyph but MCP init is still eating keys).
53
- needle="${PROMPT:0:24}"
54
- typed=0
55
- for _ in $(seq 1 30); do
56
- tmux send-keys -l -t "$SESSION" "$PROMPT"
57
- sleep 1
58
- if cap | grep -Fq "$needle"; then typed=1; break; fi
59
- # Clear whatever partial text may have landed, then retry.
60
- tmux send-keys -t "$SESSION" C-u
61
- sleep 1
62
- done
63
- [ "$typed" = 1 ] || { echo "prompt never landed in the input box"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
64
- sleep 0.5
65
- tmux send-keys -t "$SESSION" Enter
66
-
67
- # Busy signals. The robust one is the spinner's elapsed-time-in-parens, which
68
- # EVERY working state shows — both the pre-stream thinking phase
69
- # "(8s · thinking with max effort)" and the streaming phase
70
- # "(24s · ↑ 2.5k tokens · …)", and it survives the 32s→"1m 3s" rollover. We OR
71
- # in the token arrows, "esc to interrupt", and "Initializing" as belt-and-braces
72
- # (some TUI versions/states show one but not the others).
73
- BUSY_RE='esc to interrupt|↓ [0-9]|↑ [0-9]|Initializing|\(([0-9]+m )?[0-9]+s ·'
74
-
75
- # Wait for work to START (busy indicator appears), up to 60s. If it never starts,
76
- # fail loudly rather than silently reporting an empty run.
77
- started=0
78
- for _ in $(seq 1 120); do
79
- cap | grep -qE "$BUSY_RE" && { started=1; break; }
80
- sleep 0.5
81
- done
82
- [ "$started" = 1 ] || { echo "agent never started working"; cap; tmux kill-session -t "$SESSION" 2>/dev/null; exit 1; }
83
-
84
- # Poll for idle. CRITICAL: Opus 4.8 (extended thinking) renders NO spinner /
85
- # "esc to interrupt" / timer while it STREAMS its final answer — those appear
86
- # only during the thinking + tool-use phases ("✻ Marinating… (32s · ↓ 1.3k
87
- # tokens · thinking with max effort)"). So BUSY_RE reads "not busy" for the whole
88
- # 10-30s answer stream, and any short not-busy threshold kills the run mid-answer
89
- # (the truncation bug). We therefore detect "done" by CONTENT STABILITY, not by a
90
- # spinner string: while the agent streams, the captured pane changes every poll,
91
- # so stability never accrues; it accrues only once the agent has finished and the
92
- # static "✻ Brewed for 1m 9s" summary is all that is left. BUSY_RE still hard-
93
- # resets stability (covers thinking/tool-use/live-timer, where text can briefly
94
- # sit still). Need STABLE_NEEDED polls (~8s) of zero pane change + ❯ present.
95
- # Content-stability is model-agnostic — it survives future spinner re-wordings.
96
- STABLE_NEEDED=16
97
- prev=""; stable=0
98
- for _ in $(seq 1 2400); do # up to ~20 min
99
- pane="$(cap)"
100
- sig="$(printf '%s' "$pane" | tr -s '[:space:]' ' ')"
101
- if printf '%s' "$pane" | grep -qE "$BUSY_RE"; then
102
- stable=0 # thinking / tool use / live timer → busy
103
- elif [ -n "$sig" ] && [ "$sig" = "$prev" ] && printf '%s' "$pane" | grep -q "❯"; then
104
- stable=$((stable+1)); [ "$stable" -ge "$STABLE_NEEDED" ] && break
105
- else
106
- stable=0 # answer still streaming → pane changing
107
- fi
108
- prev="$sig"
109
- sleep 0.5
110
- done
111
- sleep 1
112
-
113
- tmux capture-pane -p -t "$SESSION" -S - > "$OUT"
114
- echo "captured $(wc -l < "$OUT") lines -> $OUT"
115
- grep -oE "Done \([^)]*\)|[A-Z][a-z]+ for ([0-9]+m )?[0-9]+s" "$OUT" | tail -1
116
- grep -oE "[0-9.]+k?/[0-9.]+M" "$OUT" | tail -1 | sed 's/^/Context /'
117
- tmux kill-session -t "$SESSION" 2>/dev/null
118
-
119
- # Clean tool breakdown from the session logs (main + subagents).
120
- node "$HERE/parse-session.mjs" "$REPO" 2>/dev/null || true
@@ -1,72 +0,0 @@
1
- #!/usr/bin/env bash
2
- # 3-arm offload eval for ONE indexed repo + ONE question, n reps each.
3
- # ARM offload : homegraph attached, managed offload ON (per-run AI usage log)
4
- # ARM raw : homegraph attached, HOMEGRAPH_OFFLOAD_DISABLE=1 (raw source)
5
- # ARM nocg : no homegraph (empty MCP config) -> Read/Grep baseline
6
- # All arms: claude -p sonnet --effort high. One JSON metrics line/run -> $RESULTS.
7
- #
8
- # Usage: offload-eval-3arm.sh <indexed-repo> <tier> <reps> "<question>"
9
- # Env: MODEL=sonnet EFFORT=high RESULTS=<file> AGENT_EVAL_OUT=<scratch dir>
10
- set -uo pipefail
11
- HERE="$(cd "$(dirname "$0")" && pwd)"
12
- ENGINE="$(cd "$HERE/../.." && pwd)"
13
- BIN="$ENGINE/dist/bin/homegraph.js"
14
- OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
15
- TARGET="${1:?usage: offload-eval-3arm.sh <indexed-repo> <tier> <reps> \"<question>\"}"
16
- TIER="${2:?tier}"; REPS="${3:?reps}"; Q="${4:?question}"
17
- RUNS="$OUT/runs"
18
- EXTRACT="$HERE/offload-eval-metrics.mjs"
19
- RESULTS="${RESULTS:-$OUT/results.jsonl}"
20
- REPO=$(basename "$TARGET")
21
- mkdir -p "$RUNS"
22
- command -v claude >/dev/null || { echo "no claude on PATH"; exit 1; }
23
- [ -d "$TARGET/.homegraph" ] || { echo "not indexed: $TARGET (run offload-eval-setup.sh first)"; exit 1; }
24
- # Physical path so pkill matches the daemon's real cmdline (macOS /tmp->/private/tmp symlink
25
- # otherwise makes the kill miss the daemon, and the next arm connects to the SURVIVING daemon
26
- # — contaminating the raw arm with offload).
27
- TARGET=$(cd "$TARGET" && pwd -P)
28
-
29
- prewarm() { # path extra-env (e.g. "FOO=bar")
30
- pkill -9 -f "serve --mcp --path $1" 2>/dev/null; rm -f "$1/.homegraph/daemon.sock" 2>/dev/null; sleep 0.6
31
- env ${2:-} HOMEGRAPH_DAEMON_IDLE_TIMEOUT_MS=1800000 node "$BIN" serve --mcp --path "$1" </dev/null >/dev/null 2>&1 &
32
- node -e 'const fs=require("fs");let n=0;const t=setInterval(()=>{if(fs.existsSync(process.argv[1]+"/.homegraph/daemon.sock")){clearInterval(t);process.exit(0)}if(n++>150){clearInterval(t);process.exit(1)}},100)' "$1" \
33
- && echo " daemon warm" || echo " WARN daemon never bound"
34
- }
35
-
36
- run() { # arm rep mcp-config usage-log-or-dash
37
- local arm="$1" rep="$2" cfg="$3" usage="$4" tag="$REPO-$1-$2"
38
- [ "$usage" != "-" ] && : > "$usage"
39
- # DISALLOW (optional): block sub-agent delegation across all arms so the A/B
40
- # measures the retrieval mode, not whether Sonnet decides to spawn a homegraph-blind
41
- # Explore subagent (which thrashes regardless and adds huge variance).
42
- ( cd "$TARGET" && claude -p "$Q" \
43
- --output-format stream-json --verbose --permission-mode bypassPermissions \
44
- --model "${MODEL:-sonnet}" --effort "${EFFORT:-high}" --max-budget-usd 4 \
45
- ${DISALLOW:+--disallowedTools "$DISALLOW"} \
46
- --strict-mcp-config --mcp-config "$cfg" \
47
- </dev/null > "$RUNS/$tag.jsonl" 2>"$RUNS/$tag.err" )
48
- node "$EXTRACT" --run "$RUNS/$tag.jsonl" --usage "$usage" --arm "$arm" --rep "$rep" \
49
- --repo "$REPO" --tier "$TIER" --q "$Q" >> "$RESULTS"
50
- node -e 'const o=JSON.parse(require("fs").readFileSync(process.argv[1],"utf8").trim().split("\n").pop());console.log(` [${o.arm} #${o.rep}] ${o.durationSec}s | main $${o.costUsdMain} ${o.tokBillable} tok | read=${o.read} grep=${o.grep} explore=${o.explore} offload=${o.offloadFired} | AI ${o.ai.calls}call/${o.ai.totalTokens}tok/$${o.ai.costUsd.toFixed(4)} | ok=${o.ok}`)' "$RESULTS"
51
- }
52
-
53
- CFG_OFF="$RUNS/mcp-offload-$REPO.json"; CFG_RAW="$RUNS/mcp-raw-$REPO.json"; CFG_NOCG="$RUNS/mcp-nocg.json"
54
- USAGE="$RUNS/$REPO-usage.jsonl"
55
- printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_USAGE_LOG=%s","node","%s","serve","--mcp","--path","%s"]}}}' "$USAGE" "$BIN" "$TARGET" > "$CFG_OFF"
56
- printf '{"mcpServers":{"homegraph":{"command":"env","args":["HOMEGRAPH_WASM_RELAUNCHED=1","HOMEGRAPH_OFFLOAD_DISABLE=1","node","%s","serve","--mcp","--path","%s"]}}}' "$BIN" "$TARGET" > "$CFG_RAW"
57
- printf '{"mcpServers":{}}' > "$CFG_NOCG"
58
-
59
- # REP_START lets a later batch ADD reps without clobbering earlier jsonls
60
- # (e.g. REP_START=4 REPS=3 -> reps 4,5,6; default starts at 1).
61
- START="${REP_START:-1}"; END=$((START + REPS - 1))
62
- echo "###### repo=$REPO tier=$TIER reps=$START..$END model=${MODEL:-sonnet}/${EFFORT:-high}"
63
- echo "###### Q=$Q"
64
- echo "== ARM offload =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_USAGE_LOG=$USAGE"
65
- for r in $(seq "$START" "$END"); do run offload "$r" "$CFG_OFF" "$USAGE"; done
66
- pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1
67
- echo "== ARM raw =="; prewarm "$TARGET" "HOMEGRAPH_OFFLOAD_DISABLE=1"
68
- for r in $(seq "$START" "$END"); do run raw "$r" "$CFG_RAW" "-"; done
69
- pkill -9 -f "serve --mcp --path $TARGET" 2>/dev/null; rm -f "$TARGET/.homegraph/daemon.sock" 2>/dev/null; sleep 1
70
- echo "== ARM nocg =="
71
- for r in $(seq "$START" "$END"); do run nocg "$r" "$CFG_NOCG" "-"; done
72
- echo "###### DONE $REPO"
@@ -1,133 +0,0 @@
1
- #!/usr/bin/env node
2
- // Cost/token analysis for the 3-arm offload eval, with a MAIN-vs-SUBAGENT split.
3
- //
4
- // The explore-subagent question. With delegation ALLOWED, the nocg arm spawns a
5
- // Claude Code Explore subagent; the homegraph arms do all work in the main agent.
6
- // Two facts make naive accounting wrong:
7
- // 1. The Explore subagent runs on HAIKU 4.5; the main agent on SONNET 4.6.
8
- // So per-token cost differs ~3x between them — you cannot price both the same.
9
- // 2. The subagent's consumption is ~95% cache-reads. At Haiku's $0.10/MTok
10
- // cache-read rate, a huge TOKEN volume is a small DOLLAR cost.
11
- //
12
- // Rather than re-derive cost from raw token counts (and guess the cache TTL —
13
- // Claude Code uses 1-hour ephemeral cache here, 2x write, not 5-min), we read
14
- // Claude Code's OWN authoritative accounting from the `result` event:
15
- // result.modelUsage[model].costUSD — per-model cost CC itself billed
16
- // result.total_cost_usd — their sum (INCLUDES the Haiku subagent;
17
- // the handoff's "excludes subagent" was wrong)
18
- // The model split IS the agent split here: sonnet => main, haiku => Explore subagent
19
- // (only nocg spawns one, and only nocg shows haiku usage). Token volume is still
20
- // summed per-model from modelUsage for the separate "tokens" story.
21
- //
22
- // Usage: offload-eval-cost.mjs <runs-dir> <repo> [reps]
23
- // e.g. offload-eval-cost.mjs /tmp/cg-offload-eval/runs trezor 3
24
- import { readFileSync, existsSync } from 'fs';
25
-
26
- const MAIN_TIER = /sonnet/; // main agent
27
- const SUB_TIER = /haiku/; // Claude Code Explore subagent
28
-
29
- const [,, runsDir, repo, repsArg] = process.argv;
30
- if (!runsDir || !repo) { console.error('usage: offload-eval-cost.mjs <runs-dir> <repo> [reps] (env ARMS=nocg,raw,offload)'); process.exit(1); }
31
- const REPS = Number(repsArg || 3);
32
- // Arms to analyze (file stems `<repo>-<arm>-<rep>.jsonl`). Override for the style A/B:
33
- // ARMS=raw,refs,map,src. nocg's Haiku subagent is the only sub-tier; the rest are main-only.
34
- const ARMS = (process.env.ARMS || 'nocg,raw,offload').split(',').map((s) => s.trim()).filter(Boolean);
35
-
36
- const toks = (u) => (u.inputTokens||0)+(u.outputTokens||0)+(u.cacheReadInputTokens||0)+(u.cacheCreationInputTokens||0);
37
-
38
- function analyzeRun(file) {
39
- let result = null, agentCalls = 0;
40
- const tools = {}, subPids = new Set();
41
- for (const line of readFileSync(file, 'utf8').split('\n')) {
42
- if (!line) continue;
43
- let e; try { e = JSON.parse(line); } catch { continue; }
44
- if (e.parent_tool_use_id && e.message?.usage) subPids.add(e.parent_tool_use_id);
45
- if (e.type === 'assistant' && Array.isArray(e.message?.content))
46
- for (const b of e.message.content)
47
- if (b.type === 'tool_use') { tools[b.name] = (tools[b.name]||0)+1; if (b.name === 'Agent') agentCalls++; }
48
- if (e.type === 'result') result = e;
49
- }
50
- // Authoritative cost + tokens from Claude Code's per-model accounting.
51
- const mu = result?.modelUsage || {};
52
- const main = { cost: 0, tok: 0 }, sub = { cost: 0, tok: 0 };
53
- for (const [model, u] of Object.entries(mu)) {
54
- const bucket = SUB_TIER.test(model) ? sub : main; // sonnet/anything-else => main
55
- bucket.cost += u.costUSD || 0;
56
- bucket.tok += toks(u);
57
- }
58
- return {
59
- main, sub, subagents: subPids.size, agentCalls,
60
- ccTotal: result?.total_cost_usd ?? null,
61
- ok: result?.subtype === 'success',
62
- durationSec: result?.duration_ms ? +(result.duration_ms/1000).toFixed(1) : null,
63
- models: Object.keys(mu), tools,
64
- };
65
- }
66
-
67
- const k = (n) => (n/1000).toFixed(0).padStart(5) + 'K';
68
- const d = (n) => '$' + n.toFixed(3);
69
- const cost = (b) => b.cost;
70
- const tot = (b) => b.tok;
71
-
72
- const byArm = {};
73
- for (const arm of ARMS) {
74
- const runs = [];
75
- for (let r = 1; r <= REPS; r++) {
76
- const f = `${runsDir}/${repo}-${arm}-${r}.jsonl`;
77
- if (existsSync(f)) runs.push({ rep: r, ...analyzeRun(f) });
78
- }
79
- byArm[arm] = runs;
80
- }
81
-
82
- // Per-run detail. Cost is Claude Code's own modelUsage.costUSD (authoritative,
83
- // per-model pricing + correct cache TTL). MAIN=Sonnet, SUB=Haiku Explore subagent.
84
- // cc-check: main$+sub$ must equal result.total_cost_usd (delta should be ~0).
85
- console.log(`\n=== ${repo}: per-run main(Sonnet)/sub(Haiku) split — Claude Code's own cost accounting ===`);
86
- console.log('arm rep | subAg | MAIN(sonnet) tok / $ | SUB(haiku) tok / $ | TOTAL tok / $ | cc_total Δ | dur reads');
87
- for (const arm of ARMS) for (const r of byArm[arm]) {
88
- const mC = cost(r.main), sC = cost(r.sub), mT = tot(r.main), sT = tot(r.sub);
89
- const reads = r.tools['Read'] || 0, grep = (r.tools['Grep']||0)+(r.tools['Bash']||0)+(r.tools['Glob']||0);
90
- const explore = r.tools['mcp__homegraph__homegraph_explore'] || 0;
91
- const delta = (mC + sC) - (r.ccTotal || 0); // should be ~0
92
- console.log(
93
- `${arm.padEnd(8)} #${r.rep} | ${String(r.subagents).padStart(2)} | ${k(mT)} ${d(mC).padStart(7)} | ${k(sT)} ${d(sC).padStart(7)} | ${k(mT+sT)} ${d(mC+sC).padStart(7)} | ${d(r.ccTotal||0).padStart(7)} ${(delta>=0?'+':'')+delta.toFixed(4)} | ${String(r.durationSec).padStart(5)} r=${reads} g=${grep} x=${explore}`
94
- );
95
- }
96
-
97
- // Per-arm means
98
- const mean = (arr, f) => arr.length ? arr.reduce((s,x)=>s+f(x),0)/arr.length : 0;
99
- console.log(`\n=== ${repo}: per-arm MEANS (n per arm) ===`);
100
- console.log('arm n | main $ sub $ TOTAL $ | main tok sub tok TOTAL tok | %$ in sub | %tok in sub');
101
- for (const arm of ARMS) {
102
- const runs = byArm[arm]; if (!runs.length) continue;
103
- const mC = mean(runs, r=>cost(r.main)), sC = mean(runs, r=>cost(r.sub));
104
- const mT = mean(runs, r=>tot(r.main)), sT = mean(runs, r=>tot(r.sub));
105
- const pctSubC = (mC+sC) ? (100*sC/(mC+sC)) : 0;
106
- const pctSubT = (mT+sT) ? (100*sT/(mT+sT)) : 0;
107
- console.log(
108
- `${arm.padEnd(8)} ${runs.length} | ${d(mC).padStart(7)} ${d(sC).padStart(7)} ${d(mC+sC).padStart(7)} | ${k(mT)} ${k(sT)} ${k(mT+sT)} | ${pctSubC.toFixed(0).padStart(3)}% | ${pctSubT.toFixed(0).padStart(3)}%`
109
- );
110
- }
111
-
112
- // Headline ladders — cost, tokens, duration, all vs a baseline (nocg if present, else first arm).
113
- console.log(`\n=== Ladders (mean, incl. subagent) ===`);
114
- const totals = ARMS.map(a => ({ a, c: mean(byArm[a], r=>cost(r.main)+cost(r.sub)), t: mean(byArm[a], r=>tot(r.main)+tot(r.sub)) })).filter(x=>byArm[x.a].length);
115
- const base = totals.find(x=>x.a==='nocg') ?? totals[0];
116
- const bn = base?.a ?? '?';
117
- console.log(` COST (vs ${bn}):`);
118
- for (const x of totals) {
119
- const vs = base && base.c ? ` (${((x.c/base.c-1)*100>=0?'+':'')}${((x.c/base.c-1)*100).toFixed(0)}%)` : '';
120
- console.log(` ${x.a.padEnd(8)} ${d(x.c)}${vs}`);
121
- }
122
- console.log(` TOKENS (vs ${bn}):`);
123
- for (const x of totals) {
124
- const vs = base && base.t ? ` (${((x.t/base.t-1)*100>=0?'+':'')}${((x.t/base.t-1)*100).toFixed(0)}%)` : '';
125
- console.log(` ${x.a.padEnd(8)} ${k(x.t)}${vs}`);
126
- }
127
- console.log(` DURATION (wall-clock, vs ${bn}):`);
128
- const durs = ARMS.map(a => ({ a, s: mean(byArm[a].filter(r=>r.durationSec!=null), r=>r.durationSec) })).filter(x=>byArm[x.a].length);
129
- const dbase = durs.find(x=>x.a==='nocg') ?? durs[0];
130
- for (const x of durs) {
131
- const vs = dbase && dbase.s ? ` (${((x.s/dbase.s-1)*100>=0?'+':'')}${((x.s/dbase.s-1)*100).toFixed(0)}%)` : '';
132
- console.log(` ${x.a.padEnd(8)} ${x.s.toFixed(0)}s${vs}`);
133
- }
@@ -1,108 +0,0 @@
1
- #!/usr/bin/env node
2
- // Effort A/B — does HOMEGRAPH_OFFLOAD_EFFORT=high improve offload SYNTHESIS FIDELITY vs low?
3
- // Probe-based (no agent): for each repo × effort × rep, run homegraph_explore with the offload
4
- // ON on the canonical question, capture the synthesized answer + AI tokens/cost/latency, then
5
- // Sonnet-judge that answer's fidelity vs source-verified ground truth. Isolates the synthesis
6
- // from agent/adoption noise. Requires `homegraph login` (managed offload) + indexed repos.
7
- //
8
- // Env: REPS (default 3) · CG_ENGINE (engine repo) · AGENT_EVAL_OUT (repos under /repos) · CONC (judge concurrency)
9
- import { pathToFileURL, fileURLToPath } from 'node:url';
10
- import { resolve, dirname, join } from 'node:path';
11
- import { readFileSync, writeFileSync, existsSync, rmSync } from 'node:fs';
12
- import { execFile } from 'node:child_process';
13
- import { tmpdir } from 'node:os';
14
-
15
- const HERE = dirname(fileURLToPath(import.meta.url));
16
- const ENGINE = process.env.CG_ENGINE || resolve(HERE, '..', '..');
17
- const OUT = process.env.AGENT_EVAL_OUT || '/tmp/cg-offload-eval';
18
- const REPOS = join(OUT, 'repos');
19
- const GT = JSON.parse(readFileSync(resolve(HERE, 'offload-eval-ground-truth.json'), 'utf8'));
20
- const REPS = Number(process.env.REPS || 3);
21
- const CONC = Number(process.env.CONC || 4);
22
- const EFFORTS = (process.env.EFFORTS_FILTER || 'low,high').split(',');
23
- const ONLY = process.env.REPOS_FILTER ? new Set(process.env.REPOS_FILTER.split(',')) : null;
24
- const TIER = { mtkruto: 'small', postybirb: 'medium', shapeshift: 'complex', trezor: 'large' };
25
-
26
- const load = async (rel) => import(pathToFileURL(resolve(ENGINE, rel)).href);
27
- const idx = await load('dist/index.js');
28
- const toolsMod = await load('dist/mcp/tools.js');
29
- const HomeGraph = idx.default?.default ?? idx.default ?? idx.HomeGraph;
30
- const ToolHandler = toolsMod.ToolHandler ?? toolsMod.default?.ToolHandler;
31
- if (typeof HomeGraph?.openSync !== 'function' || typeof ToolHandler !== 'function') {
32
- console.error('could not load engine from', ENGINE); process.exit(2);
33
- }
34
-
35
- const fidPrompt = (gt, ans) => `You are scoring the FIDELITY of a machine-synthesized code-exploration answer against verified ground truth. Do NOT use any tools.
36
-
37
- QUESTION: ${gt.question}
38
-
39
- VERIFIED GROUND TRUTH (the actual call path + files):
40
- ${gt.truth}
41
-
42
- SYNTHESIZED ANSWER (to score):
43
- ${ans || '(empty)'}
44
-
45
- Judge: (1) is the traced call path correct vs ground truth? (2) are the cited files/symbols correct (not fabricated)? (3) if it gave a "Coverage:" verdict, was it honest? A confident WRONG trace is the worst outcome — penalize it harder than an honest partial.
46
- Output ONLY minified JSON: {"verdict":"pass|partial|fail","score":<0-100>,"fabrication":<true|false>,"coverageHonest":<true|false>,"note":"<=20 words"}`;
47
-
48
- const askJudge = (prompt) => new Promise((res) => {
49
- execFile('claude', ['-p', prompt, '--model', 'sonnet', '--effort', 'high', '--max-budget-usd', '0.5',
50
- '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}'],
51
- { cwd: OUT, maxBuffer: 1 << 24, timeout: 120000 }, (err, stdout) => {
52
- const m = (stdout || '').match(/\{[\s\S]*\}/);
53
- if (!m) return res({ verdict: 'error', score: null, note: (err ? err.message : 'no json').slice(0, 60) });
54
- try { res(JSON.parse(m[0])); } catch { res({ verdict: 'error', score: null }); }
55
- });
56
- });
57
-
58
- // ---- 1. Probe: collect synthesized answers at each effort -------------------
59
- const records = [];
60
- for (const repo of Object.keys(GT)) {
61
- if (ONLY && !ONLY.has(repo)) continue;
62
- const dir = join(REPOS, repo);
63
- if (!existsSync(join(dir, '.homegraph'))) { console.error('skip (not indexed):', repo); continue; }
64
- const cg = HomeGraph.openSync(dir);
65
- const h = new ToolHandler(cg);
66
- for (const effort of EFFORTS) {
67
- for (let rep = 1; rep <= REPS; rep++) {
68
- process.env.HOMEGRAPH_OFFLOAD_EFFORT = effort;
69
- const usageLog = join(tmpdir(), `effort-${repo}-${effort}-${rep}.jsonl`);
70
- try { rmSync(usageLog); } catch { /* none */ }
71
- process.env.HOMEGRAPH_OFFLOAD_USAGE_LOG = usageLog;
72
- let answer = '';
73
- try { answer = (await h.execute('homegraph_explore', { query: GT[repo].question }))?.content?.[0]?.text ?? ''; }
74
- catch (e) { console.error(` ${repo}/${effort}#${rep} explore failed: ${e?.message}`); }
75
- const fired = /Synthesized by HomeGraph/.test(answer);
76
- const ai = { tokens: 0, cost: 0, ms: 0 };
77
- if (existsSync(usageLog)) for (const e of readFileSync(usageLog, 'utf8').split('\n').filter(Boolean).map(JSON.parse)) {
78
- ai.tokens += e.totalTokens || 0; ai.cost += e.costUsd || 0; ai.ms += e.ms || 0;
79
- }
80
- records.push({ repo, tier: TIER[repo], effort, rep, fired, ai, answer });
81
- console.error(` ${repo}/${effort}#${rep}: fired=${fired} ${ai.tokens}tok $${ai.cost.toFixed(4)} ${ai.ms}ms`);
82
- }
83
- }
84
- try { cg.close?.(); } catch { /* none */ }
85
- }
86
-
87
- // ---- 2. Judge fidelity (concurrency) ---------------------------------------
88
- console.error(`\njudging ${records.length} answers (concurrency ${CONC})...`);
89
- let done = 0;
90
- const q = [...records];
91
- async function worker() { while (q.length) { const r = q.shift(); r.fid = await askJudge(fidPrompt(GT[r.repo], r.answer)); console.error(` [${++done}/${records.length}] ${r.repo}/${r.effort}#${r.rep}: ${r.fid.verdict} ${r.fid.score ?? ''}`); } }
92
- await Promise.all(Array.from({ length: CONC }, worker));
93
- writeFileSync(join(OUT, 'effort-results.jsonl'), records.map((r) => JSON.stringify(r)).join('\n') + '\n');
94
-
95
- // ---- 3. Aggregate: low vs high per repo ------------------------------------
96
- const med = (a) => { a = a.filter((x) => x != null).sort((x, y) => x - y); return a.length ? (a.length % 2 ? a[(a.length - 1) / 2] : (a[a.length / 2 - 1] + a[a.length / 2]) / 2) : null; };
97
- console.log(`\n${'='.repeat(80)}\nEFFORT A/B — offload synthesis fidelity (probe, n=${REPS}/cell)\n${'='.repeat(80)}`);
98
- console.log(`${'repo'.padEnd(11)} ${'tier'.padEnd(8)} ${'effort'.padEnd(6)} fired ${'fid(med)'.padStart(8)} ${'fab%'.padStart(5)} ${'AItok'.padStart(7)} ${'AIcost'.padStart(8)} ${'ms(med)'.padStart(8)}`);
99
- for (const repo of Object.keys(GT)) {
100
- for (const effort of EFFORTS) {
101
- const rs = records.filter((r) => r.repo === repo && r.effort === effort);
102
- if (!rs.length) continue;
103
- const fids = rs.map((r) => r.fid?.score).filter((x) => x != null);
104
- const fab = rs.filter((r) => r.fid?.fabrication === true).length;
105
- console.log(`${repo.padEnd(11)} ${TIER[repo].padEnd(8)} ${effort.padEnd(6)} ${rs.filter((r) => r.fired).length}/${rs.length} ${String(med(fids) ?? '—').padStart(8)} ${String(Math.round(100 * fab / rs.length) + '%').padStart(5)} ${String(Math.round(med(rs.map((r) => r.ai.tokens)) / 1000) + 'k').padStart(7)} ${('$' + (med(rs.map((r) => r.ai.cost)) ?? 0).toFixed(4)).padStart(8)} ${String(med(rs.map((r) => r.ai.ms)) ?? '—').padStart(8)}`);
106
- }
107
- }
108
- console.log('');