homegraph 1.4.1 → 1.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (483) hide show
  1. package/LICENSE +21 -21
  2. package/README.md +305 -268
  3. package/dist/bin/command-supervision.d.ts.map +1 -1
  4. package/dist/bin/command-supervision.js +7 -4
  5. package/dist/bin/command-supervision.js.map +1 -1
  6. package/dist/bin/homegraph.js +304 -81
  7. package/dist/bin/homegraph.js.map +1 -1
  8. package/dist/context/index.d.ts.map +1 -1
  9. package/dist/context/index.js +32 -11
  10. package/dist/context/index.js.map +1 -1
  11. package/dist/db/index.d.ts +143 -1
  12. package/dist/db/index.d.ts.map +1 -1
  13. package/dist/db/index.js +305 -28
  14. package/dist/db/index.js.map +1 -1
  15. package/dist/db/migrations.js +37 -37
  16. package/dist/db/queries.d.ts +43 -4
  17. package/dist/db/queries.d.ts.map +1 -1
  18. package/dist/db/queries.js +294 -170
  19. package/dist/db/queries.js.map +1 -1
  20. package/dist/db/schema.sql +203 -203
  21. package/dist/db/sqlite-adapter.d.ts +24 -17
  22. package/dist/db/sqlite-adapter.d.ts.map +1 -1
  23. package/dist/db/sqlite-adapter.js +292 -36
  24. package/dist/db/sqlite-adapter.js.map +1 -1
  25. package/dist/db/wal-valve.d.ts +15 -2
  26. package/dist/db/wal-valve.d.ts.map +1 -1
  27. package/dist/db/wal-valve.js +88 -6
  28. package/dist/db/wal-valve.js.map +1 -1
  29. package/dist/directory.js +5 -5
  30. package/dist/extraction/arkts-batch-worker.js +3 -2
  31. package/dist/extraction/arkts-batch-worker.js.map +1 -1
  32. package/dist/extraction/default-ignore.d.ts +39 -0
  33. package/dist/extraction/default-ignore.d.ts.map +1 -0
  34. package/dist/extraction/default-ignore.js +221 -0
  35. package/dist/extraction/default-ignore.js.map +1 -0
  36. package/dist/extraction/index.d.ts +9 -10
  37. package/dist/extraction/index.d.ts.map +1 -1
  38. package/dist/extraction/index.js +227 -323
  39. package/dist/extraction/index.js.map +1 -1
  40. package/dist/extraction/languages/arkts-viewtree.d.ts +2 -4
  41. package/dist/extraction/languages/arkts-viewtree.d.ts.map +1 -1
  42. package/dist/extraction/languages/arkts-viewtree.js +6 -21
  43. package/dist/extraction/languages/arkts-viewtree.js.map +1 -1
  44. package/dist/extraction/languages/arkts.d.ts +100 -6
  45. package/dist/extraction/languages/arkts.d.ts.map +1 -1
  46. package/dist/extraction/languages/arkts.js +1777 -173
  47. package/dist/extraction/languages/arkts.js.map +1 -1
  48. package/dist/extraction/store-worker.d.ts +20 -0
  49. package/dist/extraction/store-worker.d.ts.map +1 -0
  50. package/dist/extraction/store-worker.js +102 -0
  51. package/dist/extraction/store-worker.js.map +1 -0
  52. package/dist/extraction/store-writer.d.ts +48 -0
  53. package/dist/extraction/store-writer.d.ts.map +1 -0
  54. package/dist/extraction/store-writer.js +167 -0
  55. package/dist/extraction/store-writer.js.map +1 -0
  56. package/dist/extraction/tree-sitter.d.ts.map +1 -1
  57. package/dist/extraction/tree-sitter.js +47 -0
  58. package/dist/extraction/tree-sitter.js.map +1 -1
  59. package/dist/extraction/wasm/tree-sitter-c_sharp.wasm +0 -0
  60. package/dist/extraction/wasm/tree-sitter-cfml.wasm +0 -0
  61. package/dist/extraction/wasm/tree-sitter-cfquery.wasm +0 -0
  62. package/dist/extraction/wasm/tree-sitter-cfscript.wasm +0 -0
  63. package/dist/extraction/wasm/tree-sitter-cobol.wasm +0 -0
  64. package/dist/extraction/wasm/tree-sitter-erlang.wasm +0 -0
  65. package/dist/extraction/wasm/tree-sitter-nix.wasm +0 -0
  66. package/dist/extraction/wasm/tree-sitter-pascal.wasm +0 -0
  67. package/dist/extraction/wasm/tree-sitter-vbnet.wasm +0 -0
  68. package/dist/extraction/wasm-runtime-flags.d.ts +14 -4
  69. package/dist/extraction/wasm-runtime-flags.d.ts.map +1 -1
  70. package/dist/extraction/wasm-runtime-flags.js +55 -7
  71. package/dist/extraction/wasm-runtime-flags.js.map +1 -1
  72. package/dist/index.d.ts +16 -5
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +135 -97
  75. package/dist/index.js.map +1 -1
  76. package/dist/installer/instructions-template.js +9 -9
  77. package/dist/mcp/liveness-watchdog.d.ts +18 -0
  78. package/dist/mcp/liveness-watchdog.d.ts.map +1 -1
  79. package/dist/mcp/liveness-watchdog.js +185 -73
  80. package/dist/mcp/liveness-watchdog.js.map +1 -1
  81. package/dist/mcp/query-worker.d.ts +2 -2
  82. package/dist/mcp/query-worker.js +2 -2
  83. package/dist/mcp/server-instructions.js +47 -47
  84. package/dist/mcp/tools.d.ts.map +1 -1
  85. package/dist/mcp/tools.js +42 -26
  86. package/dist/mcp/tools.js.map +1 -1
  87. package/dist/reasoning/reasoner.js +32 -32
  88. package/dist/resolution/callback-synthesizer.d.ts +52 -16
  89. package/dist/resolution/callback-synthesizer.d.ts.map +1 -1
  90. package/dist/resolution/callback-synthesizer.js +234 -180
  91. package/dist/resolution/callback-synthesizer.js.map +1 -1
  92. package/dist/resolution/cooperative-yield.d.ts +10 -2
  93. package/dist/resolution/cooperative-yield.d.ts.map +1 -1
  94. package/dist/resolution/cooperative-yield.js +6 -4
  95. package/dist/resolution/cooperative-yield.js.map +1 -1
  96. package/dist/resolution/frameworks/go.d.ts.map +1 -1
  97. package/dist/resolution/frameworks/go.js +29 -6
  98. package/dist/resolution/frameworks/go.js.map +1 -1
  99. package/dist/resolution/import-resolver.d.ts +2 -3
  100. package/dist/resolution/import-resolver.d.ts.map +1 -1
  101. package/dist/resolution/import-resolver.js +166 -7
  102. package/dist/resolution/import-resolver.js.map +1 -1
  103. package/dist/resolution/index.d.ts +87 -2
  104. package/dist/resolution/index.d.ts.map +1 -1
  105. package/dist/resolution/index.js +739 -131
  106. package/dist/resolution/index.js.map +1 -1
  107. package/dist/resolution/memory-budget.d.ts +48 -0
  108. package/dist/resolution/memory-budget.d.ts.map +1 -0
  109. package/dist/resolution/memory-budget.js +162 -0
  110. package/dist/resolution/memory-budget.js.map +1 -0
  111. package/dist/resolution/name-matcher.d.ts +12 -1
  112. package/dist/resolution/name-matcher.d.ts.map +1 -1
  113. package/dist/resolution/name-matcher.js +439 -97
  114. package/dist/resolution/name-matcher.js.map +1 -1
  115. package/dist/resolution/resolver-pool.d.ts +106 -0
  116. package/dist/resolution/resolver-pool.d.ts.map +1 -0
  117. package/dist/resolution/resolver-pool.js +344 -0
  118. package/dist/resolution/resolver-pool.js.map +1 -0
  119. package/dist/resolution/resolver-worker.d.ts +17 -0
  120. package/dist/resolution/resolver-worker.d.ts.map +1 -0
  121. package/dist/resolution/resolver-worker.js +141 -0
  122. package/dist/resolution/resolver-worker.js.map +1 -0
  123. package/dist/resolution/types.d.ts +3 -0
  124. package/dist/resolution/types.d.ts.map +1 -1
  125. package/dist/spec/build/diff-parser.d.ts +55 -0
  126. package/dist/spec/build/diff-parser.d.ts.map +1 -0
  127. package/dist/spec/build/diff-parser.js +205 -0
  128. package/dist/spec/build/diff-parser.js.map +1 -0
  129. package/dist/spec/build/pipeline.d.ts +54 -0
  130. package/dist/spec/build/pipeline.d.ts.map +1 -0
  131. package/dist/spec/build/pipeline.js +170 -0
  132. package/dist/spec/build/pipeline.js.map +1 -0
  133. package/dist/spec/build/scan.d.ts +48 -0
  134. package/dist/spec/build/scan.d.ts.map +1 -0
  135. package/dist/spec/build/scan.js +89 -0
  136. package/dist/spec/build/scan.js.map +1 -0
  137. package/dist/spec/build/scope-resolver.d.ts +50 -0
  138. package/dist/spec/build/scope-resolver.d.ts.map +1 -0
  139. package/dist/spec/build/scope-resolver.js +105 -0
  140. package/dist/spec/build/scope-resolver.js.map +1 -0
  141. package/dist/spec/build/spec-extractor.d.ts +74 -0
  142. package/dist/spec/build/spec-extractor.d.ts.map +1 -0
  143. package/dist/spec/build/spec-extractor.js +344 -0
  144. package/dist/spec/build/spec-extractor.js.map +1 -0
  145. package/dist/spec/config.d.ts +35 -1
  146. package/dist/spec/config.d.ts.map +1 -1
  147. package/dist/spec/config.js +30 -5
  148. package/dist/spec/config.js.map +1 -1
  149. package/dist/spec/db/commit-node.js +4 -4
  150. package/dist/spec/db/fragment-node.js +10 -10
  151. package/dist/spec/db/fts.d.ts.map +1 -1
  152. package/dist/spec/db/fts.js +27 -58
  153. package/dist/spec/db/fts.js.map +1 -1
  154. package/dist/spec/db/index.d.ts +5 -3
  155. package/dist/spec/db/index.d.ts.map +1 -1
  156. package/dist/spec/db/index.js +16 -1
  157. package/dist/spec/db/index.js.map +1 -1
  158. package/dist/spec/db/persist.d.ts +45 -0
  159. package/dist/spec/db/persist.d.ts.map +1 -0
  160. package/dist/spec/db/persist.js +66 -0
  161. package/dist/spec/db/persist.js.map +1 -0
  162. package/dist/spec/db/relations.d.ts +80 -1
  163. package/dist/spec/db/relations.d.ts.map +1 -1
  164. package/dist/spec/db/relations.js +218 -47
  165. package/dist/spec/db/relations.js.map +1 -1
  166. package/dist/spec/db/schema.d.ts +4 -0
  167. package/dist/spec/db/schema.d.ts.map +1 -1
  168. package/dist/spec/db/schema.js +10 -6
  169. package/dist/spec/db/schema.js.map +1 -1
  170. package/dist/spec/db/schema.sql +121 -117
  171. package/dist/spec/db/spec-node.d.ts +5 -0
  172. package/dist/spec/db/spec-node.d.ts.map +1 -1
  173. package/dist/spec/db/spec-node.js +15 -9
  174. package/dist/spec/db/spec-node.js.map +1 -1
  175. package/dist/spec/db/sql-utils.d.ts +11 -0
  176. package/dist/spec/db/sql-utils.d.ts.map +1 -0
  177. package/dist/spec/db/sql-utils.js +19 -0
  178. package/dist/spec/db/sql-utils.js.map +1 -0
  179. package/dist/spec/evolve/cluster-context.d.ts +30 -0
  180. package/dist/spec/evolve/cluster-context.d.ts.map +1 -0
  181. package/dist/spec/evolve/cluster-context.js +78 -0
  182. package/dist/spec/evolve/cluster-context.js.map +1 -0
  183. package/dist/spec/evolve/commit-spec-analyzer.d.ts +51 -0
  184. package/dist/spec/evolve/commit-spec-analyzer.d.ts.map +1 -0
  185. package/dist/spec/evolve/commit-spec-analyzer.js +88 -0
  186. package/dist/spec/evolve/commit-spec-analyzer.js.map +1 -0
  187. package/dist/spec/evolve/commit-spec-persister.d.ts +53 -0
  188. package/dist/spec/evolve/commit-spec-persister.d.ts.map +1 -0
  189. package/dist/spec/evolve/commit-spec-persister.js +93 -0
  190. package/dist/spec/evolve/commit-spec-persister.js.map +1 -0
  191. package/dist/spec/evolve/impact-locator.d.ts +23 -7
  192. package/dist/spec/evolve/impact-locator.d.ts.map +1 -1
  193. package/dist/spec/evolve/impact-locator.js +58 -14
  194. package/dist/spec/evolve/impact-locator.js.map +1 -1
  195. package/dist/spec/evolve/pipeline.d.ts +62 -25
  196. package/dist/spec/evolve/pipeline.d.ts.map +1 -1
  197. package/dist/spec/evolve/pipeline.js +436 -457
  198. package/dist/spec/evolve/pipeline.js.map +1 -1
  199. package/dist/spec/evolve/spec-rewriter.d.ts +13 -13
  200. package/dist/spec/evolve/spec-rewriter.d.ts.map +1 -1
  201. package/dist/spec/evolve/spec-rewriter.js +45 -45
  202. package/dist/spec/evolve/spec-rewriter.js.map +1 -1
  203. package/dist/spec/git/commits.d.ts +85 -0
  204. package/dist/spec/git/commits.d.ts.map +1 -0
  205. package/dist/spec/git/commits.js +218 -0
  206. package/dist/spec/git/commits.js.map +1 -0
  207. package/dist/spec/git/exec.d.ts +13 -0
  208. package/dist/spec/git/exec.d.ts.map +1 -0
  209. package/dist/spec/git/exec.js +19 -0
  210. package/dist/spec/git/exec.js.map +1 -0
  211. package/dist/spec/git/index.d.ts +9 -0
  212. package/dist/spec/git/index.d.ts.map +1 -0
  213. package/dist/spec/git/index.js +22 -0
  214. package/dist/spec/git/index.js.map +1 -0
  215. package/dist/spec/graph/index.d.ts +8 -0
  216. package/dist/spec/graph/index.d.ts.map +1 -0
  217. package/dist/spec/graph/index.js +17 -0
  218. package/dist/spec/graph/index.js.map +1 -0
  219. package/dist/spec/graph/queries.d.ts +6 -35
  220. package/dist/spec/graph/queries.d.ts.map +1 -1
  221. package/dist/spec/graph/queries.js +71 -248
  222. package/dist/spec/graph/queries.js.map +1 -1
  223. package/dist/spec/llm/agent-client.d.ts +59 -0
  224. package/dist/spec/llm/agent-client.d.ts.map +1 -0
  225. package/dist/spec/llm/agent-client.js +213 -0
  226. package/dist/spec/llm/agent-client.js.map +1 -0
  227. package/dist/spec/llm/agents/claude-code.d.ts +32 -0
  228. package/dist/spec/llm/agents/claude-code.d.ts.map +1 -0
  229. package/dist/spec/llm/agents/claude-code.js +125 -0
  230. package/dist/spec/llm/agents/claude-code.js.map +1 -0
  231. package/dist/spec/llm/agents/codex.d.ts +32 -0
  232. package/dist/spec/llm/agents/codex.d.ts.map +1 -0
  233. package/dist/spec/llm/agents/codex.js +145 -0
  234. package/dist/spec/llm/agents/codex.js.map +1 -0
  235. package/dist/spec/llm/agents/detect-utils.d.ts +28 -0
  236. package/dist/spec/llm/agents/detect-utils.d.ts.map +1 -0
  237. package/dist/spec/llm/agents/detect-utils.js +91 -0
  238. package/dist/spec/llm/agents/detect-utils.js.map +1 -0
  239. package/dist/spec/llm/agents/deveco-code.d.ts +33 -0
  240. package/dist/spec/llm/agents/deveco-code.d.ts.map +1 -0
  241. package/dist/spec/llm/agents/deveco-code.js +165 -0
  242. package/dist/spec/llm/agents/deveco-code.js.map +1 -0
  243. package/dist/spec/llm/agents/index.d.ts +31 -0
  244. package/dist/spec/llm/agents/index.d.ts.map +1 -0
  245. package/dist/spec/llm/agents/index.js +65 -0
  246. package/dist/spec/llm/agents/index.js.map +1 -0
  247. package/dist/spec/llm/agents/types.d.ts +76 -0
  248. package/dist/spec/llm/agents/types.d.ts.map +1 -0
  249. package/dist/spec/llm/agents/types.js +15 -0
  250. package/dist/spec/llm/agents/types.js.map +1 -0
  251. package/dist/spec/llm/client.d.ts +31 -11
  252. package/dist/spec/llm/client.d.ts.map +1 -1
  253. package/dist/spec/llm/client.js +143 -81
  254. package/dist/spec/llm/client.js.map +1 -1
  255. package/dist/spec/llm/factory.d.ts +41 -0
  256. package/dist/spec/llm/factory.d.ts.map +1 -0
  257. package/dist/spec/llm/factory.js +80 -0
  258. package/dist/spec/llm/factory.js.map +1 -0
  259. package/dist/spec/llm/prompts.d.ts +24 -8
  260. package/dist/spec/llm/prompts.d.ts.map +1 -1
  261. package/dist/spec/llm/prompts.js +112 -62
  262. package/dist/spec/llm/prompts.js.map +1 -1
  263. package/dist/spec/llm/retry.d.ts +53 -0
  264. package/dist/spec/llm/retry.d.ts.map +1 -0
  265. package/dist/spec/llm/retry.js +153 -0
  266. package/dist/spec/llm/retry.js.map +1 -0
  267. package/dist/spec/mine/clustering/features.d.ts +38 -0
  268. package/dist/spec/mine/clustering/features.d.ts.map +1 -0
  269. package/dist/spec/mine/clustering/features.js +108 -0
  270. package/dist/spec/mine/clustering/features.js.map +1 -0
  271. package/dist/spec/mine/clustering/index.d.ts +62 -0
  272. package/dist/spec/mine/clustering/index.d.ts.map +1 -0
  273. package/dist/spec/mine/clustering/index.js +259 -0
  274. package/dist/spec/mine/clustering/index.js.map +1 -0
  275. package/dist/spec/mine/clustering/leiden.d.ts +61 -0
  276. package/dist/spec/mine/clustering/leiden.d.ts.map +1 -0
  277. package/dist/spec/mine/clustering/leiden.js +489 -0
  278. package/dist/spec/mine/clustering/leiden.js.map +1 -0
  279. package/dist/spec/mine/clustering/text-similarity.d.ts +22 -0
  280. package/dist/spec/mine/clustering/text-similarity.d.ts.map +1 -0
  281. package/dist/spec/mine/clustering/text-similarity.js +98 -0
  282. package/dist/spec/mine/clustering/text-similarity.js.map +1 -0
  283. package/dist/spec/mine/generator.d.ts +54 -0
  284. package/dist/spec/mine/generator.d.ts.map +1 -0
  285. package/dist/spec/mine/generator.js +267 -0
  286. package/dist/spec/mine/generator.js.map +1 -0
  287. package/dist/spec/mine/persist.d.ts +37 -0
  288. package/dist/spec/mine/persist.d.ts.map +1 -0
  289. package/dist/spec/mine/persist.js +170 -0
  290. package/dist/spec/mine/persist.js.map +1 -0
  291. package/dist/spec/mine/pipeline.d.ts +41 -0
  292. package/dist/spec/mine/pipeline.d.ts.map +1 -0
  293. package/dist/spec/mine/pipeline.js +236 -0
  294. package/dist/spec/mine/pipeline.js.map +1 -0
  295. package/dist/spec/mine/scanner.d.ts +59 -0
  296. package/dist/spec/mine/scanner.d.ts.map +1 -0
  297. package/dist/spec/mine/scanner.js +383 -0
  298. package/dist/spec/mine/scanner.js.map +1 -0
  299. package/dist/spec/types.d.ts +32 -0
  300. package/dist/spec/types.d.ts.map +1 -1
  301. package/dist/spec/ui/index.d.ts +9 -0
  302. package/dist/spec/ui/index.d.ts.map +1 -0
  303. package/dist/spec/ui/index.js +16 -0
  304. package/dist/spec/ui/index.js.map +1 -0
  305. package/dist/spec/ui/progress-handler.d.ts +33 -0
  306. package/dist/spec/ui/progress-handler.d.ts.map +1 -0
  307. package/dist/spec/ui/progress-handler.js +132 -0
  308. package/dist/spec/ui/progress-handler.js.map +1 -0
  309. package/dist/spec/ui/progress.d.ts +24 -0
  310. package/dist/spec/ui/progress.d.ts.map +1 -0
  311. package/dist/spec/ui/progress.js +13 -0
  312. package/dist/spec/ui/progress.js.map +1 -0
  313. package/dist/spec/utils/fs.d.ts +45 -0
  314. package/dist/spec/utils/fs.d.ts.map +1 -0
  315. package/dist/spec/utils/fs.js +163 -0
  316. package/dist/spec/utils/fs.js.map +1 -0
  317. package/dist/spec/utils/index.d.ts +12 -0
  318. package/dist/spec/utils/index.d.ts.map +1 -0
  319. package/dist/spec/utils/index.js +26 -0
  320. package/dist/spec/utils/index.js.map +1 -0
  321. package/dist/spec/utils/meta.d.ts +32 -0
  322. package/dist/spec/utils/meta.d.ts.map +1 -0
  323. package/dist/spec/utils/meta.js +122 -0
  324. package/dist/spec/utils/meta.js.map +1 -0
  325. package/dist/spec/utils/truncate.d.ts +86 -0
  326. package/dist/spec/utils/truncate.d.ts.map +1 -0
  327. package/dist/spec/utils/truncate.js +172 -0
  328. package/dist/spec/utils/truncate.js.map +1 -0
  329. package/dist/sync/watcher.d.ts +25 -1
  330. package/dist/sync/watcher.d.ts.map +1 -1
  331. package/dist/sync/watcher.js +76 -3
  332. package/dist/sync/watcher.js.map +1 -1
  333. package/dist/telemetry/index.js +3 -3
  334. package/dist/telemetry/index.js.map +1 -1
  335. package/dist/types.d.ts +3 -0
  336. package/dist/types.d.ts.map +1 -1
  337. package/package.json +62 -58
  338. package/scripts/_tmp-cfwk-resolve.log +0 -0
  339. package/scripts/_tmp-cfwk-sig.log +0 -0
  340. package/scripts/_tmp-cfwk-vt.log +0 -0
  341. package/scripts/add-lang/bench.sh +60 -60
  342. package/scripts/add-lang/check-grammar.mjs +75 -75
  343. package/scripts/add-lang/dump-ast.mjs +103 -103
  344. package/scripts/add-lang/verify-extraction.mjs +70 -70
  345. package/scripts/agent-eval/ab-adoption.sh +91 -91
  346. package/scripts/agent-eval/ab-hook.sh +86 -86
  347. package/scripts/agent-eval/ab-impl.sh +78 -78
  348. package/scripts/agent-eval/ab-new-vs-baseline.sh +102 -102
  349. package/scripts/agent-eval/ab-sufficiency.sh +78 -78
  350. package/scripts/agent-eval/arms-F.sh +21 -21
  351. package/scripts/agent-eval/arms-matrix.sh +37 -37
  352. package/scripts/agent-eval/audit.sh +68 -68
  353. package/scripts/agent-eval/bench-readme.sh +28 -28
  354. package/scripts/agent-eval/bench-why-repo.sh +22 -22
  355. package/scripts/agent-eval/block-read-hook.sh +19 -19
  356. package/scripts/agent-eval/hook-settings.json +15 -15
  357. package/scripts/agent-eval/itrun.sh +120 -120
  358. package/scripts/agent-eval/offload-eval-3arm.sh +72 -72
  359. package/scripts/agent-eval/offload-eval-cost.mjs +133 -133
  360. package/scripts/agent-eval/offload-eval-effort.mjs +108 -108
  361. package/scripts/agent-eval/offload-eval-frontload-matrix.sh +25 -25
  362. package/scripts/agent-eval/offload-eval-frontload.sh +47 -47
  363. package/scripts/agent-eval/offload-eval-ground-truth.json +18 -18
  364. package/scripts/agent-eval/offload-eval-hook.mjs +84 -84
  365. package/scripts/agent-eval/offload-eval-judge.mjs +103 -103
  366. package/scripts/agent-eval/offload-eval-matrix.sh +20 -20
  367. package/scripts/agent-eval/offload-eval-metrics.mjs +94 -94
  368. package/scripts/agent-eval/offload-eval-refs1.sh +50 -50
  369. package/scripts/agent-eval/offload-eval-setup.sh +24 -24
  370. package/scripts/agent-eval/offload-eval-styles.sh +71 -71
  371. package/scripts/agent-eval/offload-eval-summarize.mjs +68 -68
  372. package/scripts/agent-eval/offload-eval.md +76 -76
  373. package/scripts/agent-eval/parse-arms.mjs +116 -116
  374. package/scripts/agent-eval/parse-bench-readme.mjs +84 -84
  375. package/scripts/agent-eval/parse-run.mjs +45 -45
  376. package/scripts/agent-eval/parse-session.mjs +93 -93
  377. package/scripts/agent-eval/probe-context.mjs +21 -21
  378. package/scripts/agent-eval/probe-explore.mjs +40 -40
  379. package/scripts/agent-eval/probe-node.mjs +20 -20
  380. package/scripts/agent-eval/probe-sweep.mjs +119 -119
  381. package/scripts/agent-eval/probe-trace.mjs +20 -20
  382. package/scripts/agent-eval/redirect-read-hook.sh +38 -38
  383. package/scripts/agent-eval/repro-concurrent-explore.mjs +119 -119
  384. package/scripts/agent-eval/repro-daemon-clients.mjs +125 -125
  385. package/scripts/agent-eval/run-agent.sh +34 -34
  386. package/scripts/agent-eval/run-all.sh +75 -75
  387. package/scripts/agent-eval/run-arms.sh +56 -56
  388. package/scripts/agent-eval/seq-matrix.mjs +137 -137
  389. package/scripts/bench-arkts-init-rss.log +0 -0
  390. package/scripts/build-bundle.sh +123 -123
  391. package/scripts/exp_boundary_eval/README.md +247 -247
  392. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-310.pyc +0 -0
  393. package/scripts/exp_boundary_eval/__pycache__/_utils.cpython-38.pyc +0 -0
  394. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-310.pyc +0 -0
  395. package/scripts/exp_boundary_eval/__pycache__/analyze.cpython-38.pyc +0 -0
  396. package/scripts/exp_boundary_eval/__pycache__/deveco_arm.cpython-38.pyc +0 -0
  397. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-310.pyc +0 -0
  398. package/scripts/exp_boundary_eval/__pycache__/run_all.cpython-38.pyc +0 -0
  399. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-310.pyc +0 -0
  400. package/scripts/exp_boundary_eval/__pycache__/run_one.cpython-38.pyc +0 -0
  401. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-310.pyc +0 -0
  402. package/scripts/exp_boundary_eval/__pycache__/run_session.cpython-38.pyc +0 -0
  403. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-310.pyc +0 -0
  404. package/scripts/exp_boundary_eval/__pycache__/setup.cpython-38.pyc +0 -0
  405. package/scripts/exp_boundary_eval/__pycache__/win_mcp_launcher.cpython-38.pyc +0 -0
  406. package/scripts/exp_boundary_eval/_test_mcp_chain.py +78 -78
  407. package/scripts/exp_boundary_eval/_test_stdin.py +8 -8
  408. package/scripts/exp_boundary_eval/_utils.py +1116 -1116
  409. package/scripts/exp_boundary_eval/analyze.py +1313 -1313
  410. package/scripts/exp_boundary_eval/deveco_arm.py +519 -519
  411. package/scripts/exp_boundary_eval/run_all.py +378 -378
  412. package/scripts/exp_boundary_eval/run_one.py +165 -165
  413. package/scripts/exp_boundary_eval/run_session.py +158 -158
  414. package/scripts/exp_boundary_eval/setup.py +120 -120
  415. package/scripts/exp_boundary_eval/win_mcp_launcher.py +73 -73
  416. package/scripts/exp_boundary_eval/win_mcp_stdio_wrap.js +36 -36
  417. package/scripts/exp_boundary_eval/win_node_launcher.py +24 -24
  418. package/scripts/extract-release-notes.mjs +130 -130
  419. package/scripts/local-install.sh +41 -41
  420. package/scripts/npm-sdk.js +75 -75
  421. package/scripts/npm-shim.js +275 -275
  422. package/scripts/ohos-sdk-publish.mjs +133 -133
  423. package/scripts/pack-npm.sh +119 -119
  424. package/scripts/prepare-release.mjs +270 -270
  425. package/scripts/probe-arkts-mem-why-run.log +0 -0
  426. package/scripts/probe-banner-livecard-ir.log +0 -0
  427. package/scripts/probe-cfwk-ast.stderr.log +0 -0
  428. package/scripts/probe-cfwk-ast.stdout.log +0 -0
  429. package/scripts/probe-cfwk-attr-shape.log +0 -0
  430. package/scripts/probe-cfwk-cfgdump.stderr.log +0 -0
  431. package/scripts/probe-cfwk-cfgdump.stdout.log +0 -0
  432. package/scripts/probe-cfwk-cvc.stderr.log +0 -0
  433. package/scripts/probe-cfwk-cvc.stdout.log +0 -0
  434. package/scripts/probe-cfwk-diag2.log +0 -0
  435. package/scripts/probe-cfwk-diag3.log +0 -0
  436. package/scripts/probe-cfwk-fedbg.stderr.log +0 -0
  437. package/scripts/probe-cfwk-fedbg.stdout.log +0 -0
  438. package/scripts/probe-cfwk-fileresult.log +0 -0
  439. package/scripts/probe-cfwk-fix.stderr.log +0 -0
  440. package/scripts/probe-cfwk-fix.stdout.log +0 -0
  441. package/scripts/probe-cfwk-fix2.stderr.log +0 -0
  442. package/scripts/probe-cfwk-fix2.stdout.log +0 -0
  443. package/scripts/probe-cfwk-fix3.stderr.log +0 -0
  444. package/scripts/probe-cfwk-fix3.stdout.log +0 -0
  445. package/scripts/probe-cfwk-foreach.log +0 -0
  446. package/scripts/probe-cfwk-getmethod-throw.log +0 -0
  447. package/scripts/probe-cfwk-hg-extract.log +0 -0
  448. package/scripts/probe-cfwk-pr1003-noprior.stderr.log +0 -0
  449. package/scripts/probe-cfwk-pr1003-noprior.stdout.log +0 -0
  450. package/scripts/probe-cfwk-pr1003.stderr.log +0 -0
  451. package/scripts/probe-cfwk-pr1003.stdout.log +0 -0
  452. package/scripts/probe-cfwk-preroot-noprior.stderr.log +0 -0
  453. package/scripts/probe-cfwk-preroot-noprior.stdout.log +0 -0
  454. package/scripts/probe-cfwk-preroot-prior.stderr.log +0 -0
  455. package/scripts/probe-cfwk-preroot-prior.stdout.log +0 -0
  456. package/scripts/probe-cfwk-preroot-skipstate.stderr.log +0 -0
  457. package/scripts/probe-cfwk-preroot-skipstate.stdout.log +0 -0
  458. package/scripts/probe-cfwk-resolve-sim.log +0 -0
  459. package/scripts/probe-cfwk-tree-shape.log +0 -0
  460. package/scripts/probe-cfwk-walk-abort.log +0 -0
  461. package/scripts/probe-cfwk.log +0 -0
  462. package/scripts/probe-force-index.log +0 -0
  463. package/scripts/probe-no-force-index.log +0 -0
  464. package/scripts/probe-samefile-ir.stderr.log +0 -0
  465. package/scripts/probe-samefile-ir.stdout.log +224 -0
  466. package/scripts/probe-sdk-vs-project.log +0 -0
  467. package/scripts/probe-viewtree-downgrade.log +0 -0
  468. package/dist/arkts/ohos-api-index.d.ts +0 -15
  469. package/dist/arkts/ohos-api-index.d.ts.map +0 -1
  470. package/dist/arkts/ohos-api-index.js +0 -190
  471. package/dist/arkts/ohos-api-index.js.map +0 -1
  472. package/dist/arkts/ohos-sdk-input.d.ts +0 -36
  473. package/dist/arkts/ohos-sdk-input.d.ts.map +0 -1
  474. package/dist/arkts/ohos-sdk-input.js +0 -214
  475. package/dist/arkts/ohos-sdk-input.js.map +0 -1
  476. package/dist/extraction/languages/arkts-state-decorators.d.ts +0 -13
  477. package/dist/extraction/languages/arkts-state-decorators.d.ts.map +0 -1
  478. package/dist/extraction/languages/arkts-state-decorators.js +0 -26
  479. package/dist/extraction/languages/arkts-state-decorators.js.map +0 -1
  480. package/dist/extraction/languages/ohos-api-consumer.d.ts +0 -34
  481. package/dist/extraction/languages/ohos-api-consumer.d.ts.map +0 -1
  482. package/dist/extraction/languages/ohos-api-consumer.js +0 -283
  483. package/dist/extraction/languages/ohos-api-consumer.js.map +0 -1
@@ -1,84 +1,84 @@
1
- #!/usr/bin/env node
2
- // UserPromptSubmit hook — APPROACH 1: additive context-injection.
3
- // Front-loads homegraph's structural answer for flow/impact/"how/where" prompts so the
4
- // agent's reflex grep/read has nothing left to find. Strictly additive (never blocks),
5
- // gated to structural prompts (no cost otherwise), and uses RAW explore (offload disabled)
6
- // so the injected context is accurate — never the (currently low-fidelity) synthesis.
7
- //
8
- // Reads {prompt, cwd} as JSON on stdin; prints the explore result to stdout (which Claude
9
- // Code injects into the agent's context). Any failure -> silent exit 0 (degradable).
10
- import { pathToFileURL, fileURLToPath } from 'node:url';
11
- import { resolve, join, dirname } from 'node:path';
12
- import { existsSync, readFileSync, appendFileSync } from 'node:fs';
13
-
14
- // Resolve the engine repo from this script's own location (scripts/agent-eval/ -> ../..),
15
- // overridable with CG_ENGINE. The hook ships inside the repo, so it finds its own dist.
16
- const HERE = dirname(fileURLToPath(import.meta.url));
17
- const ENGINE = process.env.CG_ENGINE || resolve(HERE, '..', '..');
18
- const BUDGET = Number(process.env.CG_FRONTLOAD_BUDGET || 16000);
19
-
20
- // Debug log only when CG_FRONTLOAD_DEBUG is set to a file path (the harness points it at a
21
- // log to count injections); off by default so the shipped hook writes nothing extra.
22
- const DBG = process.env.CG_FRONTLOAD_DEBUG;
23
- const dbg = (m) => { if (!DBG) return; try { appendFileSync(DBG, `[${new Date().toISOString()}] ${m}\n`); } catch { /* ignore */ } };
24
-
25
- let input = {};
26
- try { input = JSON.parse(readFileSync(0, 'utf8')); } catch (e) { dbg('stdin parse fail: ' + e.message); }
27
- const prompt = String(input.prompt || '');
28
- const cwd = String(input.cwd || process.cwd());
29
- dbg(`invoked: promptLen=${prompt.length} cwd=${cwd}`);
30
-
31
- // Gate: only structural / flow / impact / where-how questions. Cheap regex; silent no-op
32
- // otherwise so non-structural prompts ("fix this typo") cost nothing.
33
- const STRUCTURAL = /\b(how|where|trace|flow|path|reach(es|ed)?|call(s|ed|er|ers|ee)?|depend|impact|affect|wire[ds]?|connect|implement|architect|structure|breaks?|what calls|why does)\b/i;
34
- if (!prompt || !STRUCTURAL.test(prompt)) { dbg('gate: non-structural, no-op'); process.exit(0); }
35
- dbg('gate: structural PASS');
36
-
37
- // Find the index: cwd, then walk up a few levels.
38
- let root = cwd, found = null;
39
- for (let i = 0; i < 6 && root; i++) {
40
- if (existsSync(join(root, '.homegraph'))) { found = root; break; }
41
- const parent = resolve(root, '..'); if (parent === root) break; root = parent;
42
- }
43
- if (!found) { dbg(`no .homegraph found from cwd=${cwd}`); process.exit(0); }
44
- dbg(`found index at ${found}`);
45
-
46
- try {
47
- process.env.HOMEGRAPH_OFFLOAD_DISABLE = '1'; // raw, accurate — never the unfixed offload
48
- process.env.HOMEGRAPH_TELEMETRY = '0'; process.env.DO_NOT_TRACK = '1';
49
- const load = async (rel) => import(pathToFileURL(resolve(ENGINE, rel)).href);
50
- const idx = await load('dist/index.js');
51
- const tools = await load('dist/mcp/tools.js');
52
- const HomeGraph = idx.default?.default ?? idx.default ?? idx.HomeGraph;
53
- const ToolHandler = tools.ToolHandler ?? tools.default?.ToolHandler;
54
- if (typeof HomeGraph?.openSync !== 'function' || typeof ToolHandler !== 'function') process.exit(0);
55
-
56
- // Retry once on a transient busy/locked index (the hook's openSync can race a
57
- // freshly-warming daemon on the first prompt of a session).
58
- let text = '';
59
- for (let attempt = 1; attempt <= 2; attempt++) {
60
- try {
61
- const cg = HomeGraph.openSync(found);
62
- const h = new ToolHandler(cg);
63
- const res = await h.execute('homegraph_explore', { query: prompt });
64
- text = res?.content?.[0]?.text ?? '';
65
- try { cg.close?.(); } catch { /* ignore */ }
66
- dbg(`explore attempt ${attempt} returned ${text.length} chars`);
67
- break;
68
- } catch (e) {
69
- dbg(`explore attempt ${attempt} failed: ${e?.message || e}`);
70
- if (attempt === 2) throw e;
71
- await new Promise((r) => setTimeout(r, 800));
72
- }
73
- }
74
- if (!text.trim()) { dbg('empty explore result, no-op'); process.exit(0); }
75
- if (text.length > BUDGET) text = text.slice(0, BUDGET) + '\n…[front-load truncated to budget]';
76
-
77
- process.stdout.write(
78
- `## HomeGraph structural context (auto-retrieved for this question)\n` +
79
- `The code graph was queried for your question; the relevant symbols, source, and call flow are below. ` +
80
- `Treat the quoted source as already read. If you need more, call homegraph_explore with specific symbol names rather than grepping or reading files.\n\n` +
81
- text + '\n'
82
- );
83
- dbg(`INJECTED ${text.length} chars`);
84
- } catch (e) { dbg('ERROR: ' + (e?.stack || e?.message || e)); process.exit(0); } // degradable
1
+ #!/usr/bin/env node
2
+ // UserPromptSubmit hook — APPROACH 1: additive context-injection.
3
+ // Front-loads homegraph's structural answer for flow/impact/"how/where" prompts so the
4
+ // agent's reflex grep/read has nothing left to find. Strictly additive (never blocks),
5
+ // gated to structural prompts (no cost otherwise), and uses RAW explore (offload disabled)
6
+ // so the injected context is accurate — never the (currently low-fidelity) synthesis.
7
+ //
8
+ // Reads {prompt, cwd} as JSON on stdin; prints the explore result to stdout (which Claude
9
+ // Code injects into the agent's context). Any failure -> silent exit 0 (degradable).
10
+ import { pathToFileURL, fileURLToPath } from 'node:url';
11
+ import { resolve, join, dirname } from 'node:path';
12
+ import { existsSync, readFileSync, appendFileSync } from 'node:fs';
13
+
14
+ // Resolve the engine repo from this script's own location (scripts/agent-eval/ -> ../..),
15
+ // overridable with CG_ENGINE. The hook ships inside the repo, so it finds its own dist.
16
+ const HERE = dirname(fileURLToPath(import.meta.url));
17
+ const ENGINE = process.env.CG_ENGINE || resolve(HERE, '..', '..');
18
+ const BUDGET = Number(process.env.CG_FRONTLOAD_BUDGET || 16000);
19
+
20
+ // Debug log only when CG_FRONTLOAD_DEBUG is set to a file path (the harness points it at a
21
+ // log to count injections); off by default so the shipped hook writes nothing extra.
22
+ const DBG = process.env.CG_FRONTLOAD_DEBUG;
23
+ const dbg = (m) => { if (!DBG) return; try { appendFileSync(DBG, `[${new Date().toISOString()}] ${m}\n`); } catch { /* ignore */ } };
24
+
25
+ let input = {};
26
+ try { input = JSON.parse(readFileSync(0, 'utf8')); } catch (e) { dbg('stdin parse fail: ' + e.message); }
27
+ const prompt = String(input.prompt || '');
28
+ const cwd = String(input.cwd || process.cwd());
29
+ dbg(`invoked: promptLen=${prompt.length} cwd=${cwd}`);
30
+
31
+ // Gate: only structural / flow / impact / where-how questions. Cheap regex; silent no-op
32
+ // otherwise so non-structural prompts ("fix this typo") cost nothing.
33
+ const STRUCTURAL = /\b(how|where|trace|flow|path|reach(es|ed)?|call(s|ed|er|ers|ee)?|depend|impact|affect|wire[ds]?|connect|implement|architect|structure|breaks?|what calls|why does)\b/i;
34
+ if (!prompt || !STRUCTURAL.test(prompt)) { dbg('gate: non-structural, no-op'); process.exit(0); }
35
+ dbg('gate: structural PASS');
36
+
37
+ // Find the index: cwd, then walk up a few levels.
38
+ let root = cwd, found = null;
39
+ for (let i = 0; i < 6 && root; i++) {
40
+ if (existsSync(join(root, '.homegraph'))) { found = root; break; }
41
+ const parent = resolve(root, '..'); if (parent === root) break; root = parent;
42
+ }
43
+ if (!found) { dbg(`no .homegraph found from cwd=${cwd}`); process.exit(0); }
44
+ dbg(`found index at ${found}`);
45
+
46
+ try {
47
+ process.env.HOMEGRAPH_OFFLOAD_DISABLE = '1'; // raw, accurate — never the unfixed offload
48
+ process.env.HOMEGRAPH_TELEMETRY = '0'; process.env.DO_NOT_TRACK = '1';
49
+ const load = async (rel) => import(pathToFileURL(resolve(ENGINE, rel)).href);
50
+ const idx = await load('dist/index.js');
51
+ const tools = await load('dist/mcp/tools.js');
52
+ const HomeGraph = idx.default?.default ?? idx.default ?? idx.HomeGraph;
53
+ const ToolHandler = tools.ToolHandler ?? tools.default?.ToolHandler;
54
+ if (typeof HomeGraph?.openSync !== 'function' || typeof ToolHandler !== 'function') process.exit(0);
55
+
56
+ // Retry once on a transient busy/locked index (the hook's openSync can race a
57
+ // freshly-warming daemon on the first prompt of a session).
58
+ let text = '';
59
+ for (let attempt = 1; attempt <= 2; attempt++) {
60
+ try {
61
+ const cg = HomeGraph.openSync(found);
62
+ const h = new ToolHandler(cg);
63
+ const res = await h.execute('homegraph_explore', { query: prompt });
64
+ text = res?.content?.[0]?.text ?? '';
65
+ try { cg.close?.(); } catch { /* ignore */ }
66
+ dbg(`explore attempt ${attempt} returned ${text.length} chars`);
67
+ break;
68
+ } catch (e) {
69
+ dbg(`explore attempt ${attempt} failed: ${e?.message || e}`);
70
+ if (attempt === 2) throw e;
71
+ await new Promise((r) => setTimeout(r, 800));
72
+ }
73
+ }
74
+ if (!text.trim()) { dbg('empty explore result, no-op'); process.exit(0); }
75
+ if (text.length > BUDGET) text = text.slice(0, BUDGET) + '\n…[front-load truncated to budget]';
76
+
77
+ process.stdout.write(
78
+ `## HomeGraph structural context (auto-retrieved for this question)\n` +
79
+ `The code graph was queried for your question; the relevant symbols, source, and call flow are below. ` +
80
+ `Treat the quoted source as already read. If you need more, call homegraph_explore with specific symbol names rather than grepping or reading files.\n\n` +
81
+ text + '\n'
82
+ );
83
+ dbg(`INJECTED ${text.length} chars`);
84
+ } catch (e) { dbg('ERROR: ' + (e?.stack || e?.message || e)); process.exit(0); } // degradable
@@ -1,103 +1,103 @@
1
- #!/usr/bin/env node
2
- // Accuracy judge. For each run in results.jsonl:
3
- // - end-to-end: agent finalAnswer vs verified ground truth (all arms)
4
- // - fidelity: offload synthesized answer vs ground truth (offload arm only)
5
- // Judge = claude -p sonnet --effort high, no tools, run from a neutral cwd,
6
- // JSON-only verdicts. Writes judged.jsonl (one line per run, verdicts merged).
7
- //
8
- // Usage: judge.mjs --results <f> --truth <f> --out <f> [--concurrency 4]
9
- import { readFileSync, writeFileSync, existsSync } from 'fs';
10
- import { execFile } from 'child_process';
11
-
12
- const A = {};
13
- for (let i = 2; i < process.argv.length; i += 2) A[process.argv[i].replace(/^--/, '')] = process.argv[i + 1];
14
- const results = readFileSync(A.results, 'utf8').split('\n').filter(Boolean).map(l => JSON.parse(l));
15
- const truth = JSON.parse(readFileSync(A.truth, 'utf8'));
16
- const OUT = A.out || '/tmp/cg-offload-eval/judged.jsonl';
17
- const CONC = Number(A.concurrency || 4);
18
-
19
- function askJudge(prompt) {
20
- return new Promise((resolve) => {
21
- execFile('claude', ['-p', prompt, '--model', 'sonnet', '--effort', 'high',
22
- '--max-budget-usd', '0.5', '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}'],
23
- // Run from a neutral dir with no repo files so the judge can't "cheat" by reading source.
24
- { cwd: process.env.AGENT_EVAL_OUT || '/tmp', maxBuffer: 1 << 24, timeout: 120000 },
25
- (err, stdout) => {
26
- const raw = (stdout || '').trim();
27
- const m = raw.match(/\{[\s\S]*\}/);
28
- if (!m) return resolve({ verdict: 'error', score: null, note: (err ? 'exec ' + err.message : 'no json').slice(0, 80) });
29
- try { resolve(JSON.parse(m[0])); } catch { resolve({ verdict: 'error', score: null, note: 'parse fail' }); }
30
- });
31
- });
32
- }
33
-
34
- const e2ePrompt = (gt, ans) => `You are scoring whether an AI coding agent correctly answered a code-flow question about a repository. Judge ONLY against the verified ground truth. Do NOT use any tools.
35
-
36
- QUESTION: ${gt.question}
37
-
38
- VERIFIED GROUND TRUTH (the actual call path + files):
39
- ${gt.truth}
40
-
41
- AGENT'S ANSWER:
42
- ${ans || '(empty)'}
43
-
44
- Score how correct the agent's answer is vs the ground truth. A "pass" means it identifies the core mechanism and the major hops with the right files/symbols and makes no materially wrong claim. "partial" = right area but misses major hops or has notable errors. "fail" = wrong layer, fabricated, or misses the mechanism.
45
- Output ONLY minified JSON, no prose, no code fences:
46
- {"verdict":"pass|partial|fail","score":<0-100>,"missedHops":["..."],"wrongClaims":["..."],"note":"<=20 words"}`;
47
-
48
- const fidPrompt = (gt, ans) => `You are scoring the FIDELITY of a machine-synthesized code-exploration answer against verified ground truth. The synthesized answer claims to trace a flow and cite file:line locations. Do NOT use any tools.
49
-
50
- QUESTION: ${gt.question}
51
-
52
- VERIFIED GROUND TRUTH (the actual call path + files):
53
- ${gt.truth}
54
-
55
- SYNTHESIZED ANSWER (to score):
56
- ${ans || '(empty)'}
57
-
58
- Judge: (1) is the traced call path correct vs ground truth? (2) are the cited files/symbols correct (not fabricated)? (3) if it gave a "Coverage:" verdict, was that verdict honest about what it actually covered? A confident WRONG trace is the worst outcome — penalize it harder than an honest "partial/not found".
59
- Output ONLY minified JSON, no prose, no code fences:
60
- {"verdict":"pass|partial|fail","score":<0-100>,"fabrication":<true|false>,"coverageHonest":<true|false>,"missedHops":["..."],"note":"<=20 words"}`;
61
-
62
- // Build the job list
63
- const jobs = [];
64
- for (const r of results) {
65
- const gt = truth[r.repo];
66
- if (!gt) { r._nojudge = true; continue; }
67
- jobs.push({ r, kind: 'e2e', prompt: e2ePrompt(gt, r.finalAnswer) });
68
- if (r.arm === 'offload' && Array.isArray(r.offloadAnswers))
69
- r.offloadAnswers.forEach((ans, i) => { if (ans && ans.trim()) jobs.push({ r, kind: 'fid', idx: i, prompt: fidPrompt(gt, ans) }); });
70
- }
71
- console.error(`judging ${jobs.length} verdicts across ${results.length} runs (concurrency ${CONC})...`);
72
-
73
- let done = 0;
74
- async function worker(queue) {
75
- while (queue.length) {
76
- const job = queue.shift();
77
- const v = await askJudge(job.prompt);
78
- if (job.kind === 'e2e') job.r.e2e = v; else (job.r._fid ??= []).push(v);
79
- console.error(` [${++done}/${jobs.length}] ${job.r.repo}/${job.r.arm}#${job.r.rep} ${job.kind}: ${v.verdict}${v.score != null ? ' ' + v.score : ''}`);
80
- }
81
- }
82
- const q = [...jobs];
83
- await Promise.all(Array.from({ length: CONC }, () => worker(q)));
84
-
85
- // Aggregate per-answer fidelity verdicts into one fidelity object per offload run.
86
- const medOf = (a) => { a = [...a].sort((x, y) => x - y); return a.length ? (a.length % 2 ? a[(a.length - 1) / 2] : (a[a.length / 2 - 1] + a[a.length / 2]) / 2) : null; };
87
- for (const r of results) {
88
- if (r._fid?.length) {
89
- const scores = r._fid.map(v => v.score).filter(x => x != null);
90
- r.fidelity = {
91
- n: r._fid.length, scores,
92
- max: scores.length ? Math.max(...scores) : null,
93
- min: scores.length ? Math.min(...scores) : null,
94
- median: medOf(scores),
95
- anyFabrication: r._fid.some(v => v.fabrication === true),
96
- allCoverageHonest: r._fid.every(v => v.coverageHonest !== false),
97
- verdicts: r._fid.map(v => v.verdict),
98
- };
99
- }
100
- delete r._fid;
101
- }
102
- writeFileSync(OUT, results.map(r => JSON.stringify(r)).join('\n') + '\n');
103
- console.error(`wrote ${OUT}`);
1
+ #!/usr/bin/env node
2
+ // Accuracy judge. For each run in results.jsonl:
3
+ // - end-to-end: agent finalAnswer vs verified ground truth (all arms)
4
+ // - fidelity: offload synthesized answer vs ground truth (offload arm only)
5
+ // Judge = claude -p sonnet --effort high, no tools, run from a neutral cwd,
6
+ // JSON-only verdicts. Writes judged.jsonl (one line per run, verdicts merged).
7
+ //
8
+ // Usage: judge.mjs --results <f> --truth <f> --out <f> [--concurrency 4]
9
+ import { readFileSync, writeFileSync, existsSync } from 'fs';
10
+ import { execFile } from 'child_process';
11
+
12
+ const A = {};
13
+ for (let i = 2; i < process.argv.length; i += 2) A[process.argv[i].replace(/^--/, '')] = process.argv[i + 1];
14
+ const results = readFileSync(A.results, 'utf8').split('\n').filter(Boolean).map(l => JSON.parse(l));
15
+ const truth = JSON.parse(readFileSync(A.truth, 'utf8'));
16
+ const OUT = A.out || '/tmp/cg-offload-eval/judged.jsonl';
17
+ const CONC = Number(A.concurrency || 4);
18
+
19
+ function askJudge(prompt) {
20
+ return new Promise((resolve) => {
21
+ execFile('claude', ['-p', prompt, '--model', 'sonnet', '--effort', 'high',
22
+ '--max-budget-usd', '0.5', '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}'],
23
+ // Run from a neutral dir with no repo files so the judge can't "cheat" by reading source.
24
+ { cwd: process.env.AGENT_EVAL_OUT || '/tmp', maxBuffer: 1 << 24, timeout: 120000 },
25
+ (err, stdout) => {
26
+ const raw = (stdout || '').trim();
27
+ const m = raw.match(/\{[\s\S]*\}/);
28
+ if (!m) return resolve({ verdict: 'error', score: null, note: (err ? 'exec ' + err.message : 'no json').slice(0, 80) });
29
+ try { resolve(JSON.parse(m[0])); } catch { resolve({ verdict: 'error', score: null, note: 'parse fail' }); }
30
+ });
31
+ });
32
+ }
33
+
34
+ const e2ePrompt = (gt, ans) => `You are scoring whether an AI coding agent correctly answered a code-flow question about a repository. Judge ONLY against the verified ground truth. Do NOT use any tools.
35
+
36
+ QUESTION: ${gt.question}
37
+
38
+ VERIFIED GROUND TRUTH (the actual call path + files):
39
+ ${gt.truth}
40
+
41
+ AGENT'S ANSWER:
42
+ ${ans || '(empty)'}
43
+
44
+ Score how correct the agent's answer is vs the ground truth. A "pass" means it identifies the core mechanism and the major hops with the right files/symbols and makes no materially wrong claim. "partial" = right area but misses major hops or has notable errors. "fail" = wrong layer, fabricated, or misses the mechanism.
45
+ Output ONLY minified JSON, no prose, no code fences:
46
+ {"verdict":"pass|partial|fail","score":<0-100>,"missedHops":["..."],"wrongClaims":["..."],"note":"<=20 words"}`;
47
+
48
+ const fidPrompt = (gt, ans) => `You are scoring the FIDELITY of a machine-synthesized code-exploration answer against verified ground truth. The synthesized answer claims to trace a flow and cite file:line locations. Do NOT use any tools.
49
+
50
+ QUESTION: ${gt.question}
51
+
52
+ VERIFIED GROUND TRUTH (the actual call path + files):
53
+ ${gt.truth}
54
+
55
+ SYNTHESIZED ANSWER (to score):
56
+ ${ans || '(empty)'}
57
+
58
+ Judge: (1) is the traced call path correct vs ground truth? (2) are the cited files/symbols correct (not fabricated)? (3) if it gave a "Coverage:" verdict, was that verdict honest about what it actually covered? A confident WRONG trace is the worst outcome — penalize it harder than an honest "partial/not found".
59
+ Output ONLY minified JSON, no prose, no code fences:
60
+ {"verdict":"pass|partial|fail","score":<0-100>,"fabrication":<true|false>,"coverageHonest":<true|false>,"missedHops":["..."],"note":"<=20 words"}`;
61
+
62
+ // Build the job list
63
+ const jobs = [];
64
+ for (const r of results) {
65
+ const gt = truth[r.repo];
66
+ if (!gt) { r._nojudge = true; continue; }
67
+ jobs.push({ r, kind: 'e2e', prompt: e2ePrompt(gt, r.finalAnswer) });
68
+ if (r.arm === 'offload' && Array.isArray(r.offloadAnswers))
69
+ r.offloadAnswers.forEach((ans, i) => { if (ans && ans.trim()) jobs.push({ r, kind: 'fid', idx: i, prompt: fidPrompt(gt, ans) }); });
70
+ }
71
+ console.error(`judging ${jobs.length} verdicts across ${results.length} runs (concurrency ${CONC})...`);
72
+
73
+ let done = 0;
74
+ async function worker(queue) {
75
+ while (queue.length) {
76
+ const job = queue.shift();
77
+ const v = await askJudge(job.prompt);
78
+ if (job.kind === 'e2e') job.r.e2e = v; else (job.r._fid ??= []).push(v);
79
+ console.error(` [${++done}/${jobs.length}] ${job.r.repo}/${job.r.arm}#${job.r.rep} ${job.kind}: ${v.verdict}${v.score != null ? ' ' + v.score : ''}`);
80
+ }
81
+ }
82
+ const q = [...jobs];
83
+ await Promise.all(Array.from({ length: CONC }, () => worker(q)));
84
+
85
+ // Aggregate per-answer fidelity verdicts into one fidelity object per offload run.
86
+ const medOf = (a) => { a = [...a].sort((x, y) => x - y); return a.length ? (a.length % 2 ? a[(a.length - 1) / 2] : (a[a.length / 2 - 1] + a[a.length / 2]) / 2) : null; };
87
+ for (const r of results) {
88
+ if (r._fid?.length) {
89
+ const scores = r._fid.map(v => v.score).filter(x => x != null);
90
+ r.fidelity = {
91
+ n: r._fid.length, scores,
92
+ max: scores.length ? Math.max(...scores) : null,
93
+ min: scores.length ? Math.min(...scores) : null,
94
+ median: medOf(scores),
95
+ anyFabrication: r._fid.some(v => v.fabrication === true),
96
+ allCoverageHonest: r._fid.every(v => v.coverageHonest !== false),
97
+ verdicts: r._fid.map(v => v.verdict),
98
+ };
99
+ }
100
+ delete r._fid;
101
+ }
102
+ writeFileSync(OUT, results.map(r => JSON.stringify(r)).join('\n') + '\n');
103
+ console.error(`wrote ${OUT}`);
@@ -1,20 +1,20 @@
1
- #!/usr/bin/env bash
2
- # Drive the 3-arm campaign (offload/raw/nocg) across all 4 tiers, n reps each, into one
3
- # results.jsonl. Reads the canonical question per repo from offload-eval-ground-truth.json.
4
- # Env: REPS (default 3) AGENT_EVAL_OUT=<scratch dir>
5
- set -uo pipefail
6
- HERE="$(cd "$(dirname "$0")" && pwd)"
7
- OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
8
- GT="$HERE/offload-eval-ground-truth.json"
9
- REPS="${REPS:-3}"
10
- export RESULTS="$OUT/results.jsonl"
11
- : > "$RESULTS"
12
- for repo in mtkruto postybirb shapeshift trezor; do
13
- case "$repo" in mtkruto) tier=small;; postybirb) tier=medium;; shapeshift) tier=complex;; trezor) tier=large;; esac
14
- Q=$(node -e "console.log(JSON.parse(require('fs').readFileSync(process.argv[1],'utf8'))[process.argv[2]].question)" "$GT" "$repo")
15
- echo ""; echo "### $repo ($tier) $(date +%H:%M:%S)"
16
- bash "$HERE/offload-eval-3arm.sh" "$OUT/repos/$repo" "$tier" "$REPS" "$Q"
17
- done
18
- echo ""; echo "###### MATRIX DONE -> $RESULTS ($(wc -l < "$RESULTS") runs). Judge + summarize with:"
19
- echo " node $HERE/offload-eval-judge.mjs --results $RESULTS --truth $GT --out $OUT/judged.jsonl"
20
- echo " node $HERE/offload-eval-summarize.mjs $OUT/judged.jsonl"
1
+ #!/usr/bin/env bash
2
+ # Drive the 3-arm campaign (offload/raw/nocg) across all 4 tiers, n reps each, into one
3
+ # results.jsonl. Reads the canonical question per repo from offload-eval-ground-truth.json.
4
+ # Env: REPS (default 3) AGENT_EVAL_OUT=<scratch dir>
5
+ set -uo pipefail
6
+ HERE="$(cd "$(dirname "$0")" && pwd)"
7
+ OUT="${AGENT_EVAL_OUT:-/tmp/cg-offload-eval}"
8
+ GT="$HERE/offload-eval-ground-truth.json"
9
+ REPS="${REPS:-3}"
10
+ export RESULTS="$OUT/results.jsonl"
11
+ : > "$RESULTS"
12
+ for repo in mtkruto postybirb shapeshift trezor; do
13
+ case "$repo" in mtkruto) tier=small;; postybirb) tier=medium;; shapeshift) tier=complex;; trezor) tier=large;; esac
14
+ Q=$(node -e "console.log(JSON.parse(require('fs').readFileSync(process.argv[1],'utf8'))[process.argv[2]].question)" "$GT" "$repo")
15
+ echo ""; echo "### $repo ($tier) $(date +%H:%M:%S)"
16
+ bash "$HERE/offload-eval-3arm.sh" "$OUT/repos/$repo" "$tier" "$REPS" "$Q"
17
+ done
18
+ echo ""; echo "###### MATRIX DONE -> $RESULTS ($(wc -l < "$RESULTS") runs). Judge + summarize with:"
19
+ echo " node $HERE/offload-eval-judge.mjs --results $RESULTS --truth $GT --out $OUT/judged.jsonl"
20
+ echo " node $HERE/offload-eval-summarize.mjs $OUT/judged.jsonl"
@@ -1,94 +1,94 @@
1
- #!/usr/bin/env node
2
- // Extract one eval run's metrics from its Claude stream-json transcript + the
3
- // offload usage sidecar log, emit ONE merged JSON line.
4
- //
5
- // Usage: extract-metrics.mjs --run <run.jsonl> --usage <usage.jsonl|-> \
6
- // --arm <a> --rep <n> --repo <r> --tier <t> --q <question>
7
- import { readFileSync, existsSync } from 'fs';
8
-
9
- const args = {};
10
- for (let i = 2; i < process.argv.length; i += 2) args[process.argv[i].replace(/^--/, '')] = process.argv[i + 1];
11
-
12
- const runFile = args.run;
13
- const lines = existsSync(runFile) ? readFileSync(runFile, 'utf8').split('\n').filter(Boolean) : [];
14
-
15
- const toolCounts = {};
16
- let result = null;
17
- const tok = { gen: 0, fresh: 0, cached: 0 };
18
- const offloadAnswers = [];
19
- let exploreResults = 0; // tool_results from explore (offload or raw)
20
- let lastAssistantText = '';
21
-
22
- for (const line of lines) {
23
- let ev; try { ev = JSON.parse(line); } catch { continue; }
24
-
25
- // per-turn token usage (authoritative token measure; result.usage is last-turn only)
26
- const u = ev.message?.usage;
27
- if (u) {
28
- tok.gen += u.output_tokens || 0;
29
- tok.fresh += (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0);
30
- tok.cached += u.cache_read_input_tokens || 0;
31
- }
32
-
33
- if (ev.type === 'assistant' && Array.isArray(ev.message?.content)) {
34
- for (const b of ev.message.content) {
35
- if (b.type === 'tool_use') toolCounts[b.name] = (toolCounts[b.name] || 0) + 1;
36
- if (b.type === 'text' && b.text?.trim()) lastAssistantText = b.text.trim();
37
- }
38
- }
39
- // tool_results arrive in user messages
40
- if (ev.type === 'user' && Array.isArray(ev.message?.content)) {
41
- for (const b of ev.message.content) {
42
- if (b.type !== 'tool_result') continue;
43
- const text = Array.isArray(b.content)
44
- ? b.content.map(c => (typeof c === 'string' ? c : c.text || '')).join('')
45
- : (typeof b.content === 'string' ? b.content : '');
46
- // An offload answer is either the 'plain'/'report' synthesis (carries the
47
- // "Synthesized by HomeGraph" footer) or a 'refs' answer (carries the re-expanded
48
- // "### Referenced source — verbatim" appendix). A refs call that cited nothing
49
- // valid falls back to RAW source, which is correctly counted as a raw explore below.
50
- if (/Synthesized by HomeGraph|### Referenced source — verbatim/.test(text)) { offloadAnswers.push(text); exploreResults++; }
51
- else if (/Found \d+ symbols? across|\*\*Exploration:/.test(text)) exploreResults++;
52
- }
53
- }
54
- if (ev.type === 'result') result = ev;
55
- }
56
-
57
- // offload usage sidecar (HomeGraph AI tokens + cost) — one JSON line per offload call
58
- const ai = { calls: 0, promptTokens: 0, completionTokens: 0, totalTokens: 0, credits: 0, costUsd: 0, ms: 0 };
59
- if (args.usage && args.usage !== '-' && existsSync(args.usage)) {
60
- for (const line of readFileSync(args.usage, 'utf8').split('\n').filter(Boolean)) {
61
- let e; try { e = JSON.parse(line); } catch { continue; }
62
- ai.calls++;
63
- ai.promptTokens += e.promptTokens || 0;
64
- ai.completionTokens += e.completionTokens || 0;
65
- ai.totalTokens += e.totalTokens || 0;
66
- ai.credits += e.creditsCharged || 0;
67
- ai.costUsd += e.costUsd || 0;
68
- ai.ms += e.ms || 0;
69
- }
70
- }
71
-
72
- // front-load hook fired iff its injected header appears in the transcript
73
- const frontload = lines.some(l => l.includes('auto-retrieved for this question'));
74
- const get = (n) => toolCounts[n] || 0;
75
- const read = get('Read');
76
- const grep = get('Grep') + get('Bash') + get('Glob');
77
- const explore = get('mcp__homegraph__homegraph_explore');
78
- const cgAny = Object.keys(toolCounts).filter(k => /mcp__homegraph__/.test(k)).reduce((s, k) => s + toolCounts[k], 0);
79
-
80
- const out = {
81
- repo: args.repo, tier: args.tier, arm: args.arm, rep: Number(args.rep), question: args.q,
82
- ok: result?.subtype === 'success',
83
- durationSec: result ? +(result.duration_ms / 1000).toFixed(1) : null,
84
- numTurns: result?.num_turns ?? null,
85
- costUsdMain: result ? +(result.total_cost_usd || 0).toFixed(4) : null,
86
- tokGen: tok.gen, tokFresh: tok.fresh, tokCached: tok.cached, tokBillable: tok.gen + tok.fresh,
87
- read, grep, explore, cgAny, frontload,
88
- offloadFired: offloadAnswers.length,
89
- ai,
90
- // text payloads for the accuracy judge (kept separate; large)
91
- finalAnswer: (result?.result || lastAssistantText || '').slice(0, 8000),
92
- offloadAnswers: offloadAnswers.map(a => a.slice(0, 6000)),
93
- };
94
- process.stdout.write(JSON.stringify(out) + '\n');
1
+ #!/usr/bin/env node
2
+ // Extract one eval run's metrics from its Claude stream-json transcript + the
3
+ // offload usage sidecar log, emit ONE merged JSON line.
4
+ //
5
+ // Usage: extract-metrics.mjs --run <run.jsonl> --usage <usage.jsonl|-> \
6
+ // --arm <a> --rep <n> --repo <r> --tier <t> --q <question>
7
+ import { readFileSync, existsSync } from 'fs';
8
+
9
+ const args = {};
10
+ for (let i = 2; i < process.argv.length; i += 2) args[process.argv[i].replace(/^--/, '')] = process.argv[i + 1];
11
+
12
+ const runFile = args.run;
13
+ const lines = existsSync(runFile) ? readFileSync(runFile, 'utf8').split('\n').filter(Boolean) : [];
14
+
15
+ const toolCounts = {};
16
+ let result = null;
17
+ const tok = { gen: 0, fresh: 0, cached: 0 };
18
+ const offloadAnswers = [];
19
+ let exploreResults = 0; // tool_results from explore (offload or raw)
20
+ let lastAssistantText = '';
21
+
22
+ for (const line of lines) {
23
+ let ev; try { ev = JSON.parse(line); } catch { continue; }
24
+
25
+ // per-turn token usage (authoritative token measure; result.usage is last-turn only)
26
+ const u = ev.message?.usage;
27
+ if (u) {
28
+ tok.gen += u.output_tokens || 0;
29
+ tok.fresh += (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0);
30
+ tok.cached += u.cache_read_input_tokens || 0;
31
+ }
32
+
33
+ if (ev.type === 'assistant' && Array.isArray(ev.message?.content)) {
34
+ for (const b of ev.message.content) {
35
+ if (b.type === 'tool_use') toolCounts[b.name] = (toolCounts[b.name] || 0) + 1;
36
+ if (b.type === 'text' && b.text?.trim()) lastAssistantText = b.text.trim();
37
+ }
38
+ }
39
+ // tool_results arrive in user messages
40
+ if (ev.type === 'user' && Array.isArray(ev.message?.content)) {
41
+ for (const b of ev.message.content) {
42
+ if (b.type !== 'tool_result') continue;
43
+ const text = Array.isArray(b.content)
44
+ ? b.content.map(c => (typeof c === 'string' ? c : c.text || '')).join('')
45
+ : (typeof b.content === 'string' ? b.content : '');
46
+ // An offload answer is either the 'plain'/'report' synthesis (carries the
47
+ // "Synthesized by HomeGraph" footer) or a 'refs' answer (carries the re-expanded
48
+ // "### Referenced source — verbatim" appendix). A refs call that cited nothing
49
+ // valid falls back to RAW source, which is correctly counted as a raw explore below.
50
+ if (/Synthesized by HomeGraph|### Referenced source — verbatim/.test(text)) { offloadAnswers.push(text); exploreResults++; }
51
+ else if (/Found \d+ symbols? across|\*\*Exploration:/.test(text)) exploreResults++;
52
+ }
53
+ }
54
+ if (ev.type === 'result') result = ev;
55
+ }
56
+
57
+ // offload usage sidecar (HomeGraph AI tokens + cost) — one JSON line per offload call
58
+ const ai = { calls: 0, promptTokens: 0, completionTokens: 0, totalTokens: 0, credits: 0, costUsd: 0, ms: 0 };
59
+ if (args.usage && args.usage !== '-' && existsSync(args.usage)) {
60
+ for (const line of readFileSync(args.usage, 'utf8').split('\n').filter(Boolean)) {
61
+ let e; try { e = JSON.parse(line); } catch { continue; }
62
+ ai.calls++;
63
+ ai.promptTokens += e.promptTokens || 0;
64
+ ai.completionTokens += e.completionTokens || 0;
65
+ ai.totalTokens += e.totalTokens || 0;
66
+ ai.credits += e.creditsCharged || 0;
67
+ ai.costUsd += e.costUsd || 0;
68
+ ai.ms += e.ms || 0;
69
+ }
70
+ }
71
+
72
+ // front-load hook fired iff its injected header appears in the transcript
73
+ const frontload = lines.some(l => l.includes('auto-retrieved for this question'));
74
+ const get = (n) => toolCounts[n] || 0;
75
+ const read = get('Read');
76
+ const grep = get('Grep') + get('Bash') + get('Glob');
77
+ const explore = get('mcp__homegraph__homegraph_explore');
78
+ const cgAny = Object.keys(toolCounts).filter(k => /mcp__homegraph__/.test(k)).reduce((s, k) => s + toolCounts[k], 0);
79
+
80
+ const out = {
81
+ repo: args.repo, tier: args.tier, arm: args.arm, rep: Number(args.rep), question: args.q,
82
+ ok: result?.subtype === 'success',
83
+ durationSec: result ? +(result.duration_ms / 1000).toFixed(1) : null,
84
+ numTurns: result?.num_turns ?? null,
85
+ costUsdMain: result ? +(result.total_cost_usd || 0).toFixed(4) : null,
86
+ tokGen: tok.gen, tokFresh: tok.fresh, tokCached: tok.cached, tokBillable: tok.gen + tok.fresh,
87
+ read, grep, explore, cgAny, frontload,
88
+ offloadFired: offloadAnswers.length,
89
+ ai,
90
+ // text payloads for the accuracy judge (kept separate; large)
91
+ finalAnswer: (result?.result || lastAssistantText || '').slice(0, 8000),
92
+ offloadAnswers: offloadAnswers.map(a => a.slice(0, 6000)),
93
+ };
94
+ process.stdout.write(JSON.stringify(out) + '\n');