akm-cli 0.9.16 → 0.9.17-alpha.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (403) hide show
  1. package/CHANGELOG.md +2101 -0
  2. package/STABILITY.md +11 -10
  3. package/dist/akm +124 -193
  4. package/dist/akm-migrate +38 -19
  5. package/dist/assets/hints/cli-hints-full.md +6 -7
  6. package/dist/assets/improve-strategies/catchup.json +0 -3
  7. package/dist/assets/improve-strategies/consolidate.json +0 -1
  8. package/dist/assets/improve-strategies/default.json +1 -2
  9. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  10. package/dist/assets/improve-strategies/quick.json +1 -2
  11. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  12. package/dist/assets/improve-strategies/thorough.json +0 -3
  13. package/dist/assets/prompts/consolidate-pair.md +20 -0
  14. package/dist/assets/prompts/consolidate-system.md +4 -11
  15. package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
  16. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +20 -20
  17. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  18. package/dist/assets/templates/html/health.html +3 -5
  19. package/dist/cli/retired-commands.js +1 -1
  20. package/dist/cli/shared.js +6 -2
  21. package/dist/cli/unknown-flags.js +24 -1
  22. package/dist/cli.js +68 -10
  23. package/dist/commands/agent/agent-dispatch.js +1 -1
  24. package/dist/commands/command/command-execution.js +24 -62
  25. package/dist/commands/feedback-cli.js +0 -1
  26. package/dist/commands/health/accept-rate.js +6 -0
  27. package/dist/commands/health/archive-usage.js +92 -0
  28. package/dist/commands/health/checks.js +83 -74
  29. package/dist/commands/health/config-skew.js +38 -0
  30. package/dist/commands/health/data-dir-usage.js +25 -13
  31. package/dist/commands/health/egress.js +54 -0
  32. package/dist/commands/health/html-report.js +1 -42
  33. package/dist/commands/health/improve-metrics.js +136 -591
  34. package/dist/commands/health/md-report.js +1 -6
  35. package/dist/commands/health/plugin-staleness.js +53 -3
  36. package/dist/commands/health/renderers.js +12 -4
  37. package/dist/commands/health/report-view-model.js +14 -120
  38. package/dist/commands/health/types-improve.js +4 -19
  39. package/dist/commands/health/windows.js +64 -74
  40. package/dist/commands/health.js +145 -143
  41. package/dist/commands/improve/consolidate/chunking.js +26 -117
  42. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  43. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  44. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  45. package/dist/commands/improve/consolidate.js +589 -1127
  46. package/dist/commands/improve/content-hash.js +16 -24
  47. package/dist/commands/improve/distill/content-repair.js +18 -100
  48. package/dist/commands/improve/distill-guards.js +20 -81
  49. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  50. package/dist/commands/improve/distill.js +608 -1041
  51. package/dist/commands/improve/eligibility.js +126 -390
  52. package/dist/commands/improve/execution.js +8 -10
  53. package/dist/commands/improve/extract-prompt.js +1 -2
  54. package/dist/commands/improve/extract.js +487 -1046
  55. package/dist/commands/improve/feedback-valence.js +0 -25
  56. package/dist/commands/improve/improve-cli.js +75 -169
  57. package/dist/commands/improve/improve-result-file.js +10 -66
  58. package/dist/commands/improve/improve-strategies.js +52 -4
  59. package/dist/commands/improve/improve-usage-report.js +18 -64
  60. package/dist/commands/improve/improve.js +480 -1074
  61. package/dist/commands/improve/ledger.js +119 -0
  62. package/dist/commands/improve/locks.js +2 -8
  63. package/dist/commands/improve/loop-stages.js +415 -1073
  64. package/dist/commands/improve/memory/derived-ref.js +12 -77
  65. package/dist/commands/improve/memory/memory-belief.js +16 -118
  66. package/dist/commands/improve/memory/memory-improve.js +266 -14
  67. package/dist/commands/improve/outcome-loop.js +28 -156
  68. package/dist/commands/improve/planner.js +5 -15
  69. package/dist/commands/improve/preparation.js +779 -2319
  70. package/dist/commands/improve/proactive-maintenance.js +34 -101
  71. package/dist/commands/improve/reflect-noise.js +104 -280
  72. package/dist/commands/improve/reflect.js +642 -1353
  73. package/dist/commands/improve/retrieval-gate.js +127 -0
  74. package/dist/commands/improve/retrieval-scope.js +92 -0
  75. package/dist/commands/improve/salience.js +41 -240
  76. package/dist/commands/improve/session-asset.js +19 -100
  77. package/dist/commands/improve/stage.js +322 -0
  78. package/dist/commands/lint/base-linter.js +37 -15
  79. package/dist/commands/proposal/drain.js +261 -578
  80. package/dist/commands/proposal/proposal-cli.js +19 -20
  81. package/dist/commands/proposal/proposal-types.js +31 -24
  82. package/dist/commands/proposal/proposal.js +38 -8
  83. package/dist/commands/proposal/propose.js +134 -160
  84. package/dist/commands/proposal/repository.js +1097 -1394
  85. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  86. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  87. package/dist/commands/proposal/validators/proposals.js +22 -89
  88. package/dist/commands/read/curate.js +105 -462
  89. package/dist/commands/read/knowledge.js +3 -2
  90. package/dist/commands/read/search-cli.js +16 -33
  91. package/dist/commands/read/search.js +17 -23
  92. package/dist/commands/read/show.js +57 -108
  93. package/dist/commands/sources/bundle-cli.js +25 -2
  94. package/dist/commands/sources/bundle-config-ops.js +4 -0
  95. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  96. package/dist/commands/sources/info.js +127 -29
  97. package/dist/commands/sources/installed-stashes.js +197 -746
  98. package/dist/commands/sources/schema-repair.js +98 -129
  99. package/dist/commands/sources/source-add.js +62 -12
  100. package/dist/commands/sources/source-manage.js +9 -2
  101. package/dist/commands/sources/stash-cli.js +24 -4
  102. package/dist/commands/tasks/explain.js +10 -13
  103. package/dist/commands/tasks/tasks-cli.js +12 -13
  104. package/dist/commands/tasks/tasks.js +350 -936
  105. package/dist/commands/tasks/validate.js +26 -24
  106. package/dist/commands/workflow/plan.js +22 -29
  107. package/dist/commands/workflow-cli.js +4 -4
  108. package/dist/core/adapter/adapters/akm-adapter.js +2 -1
  109. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  110. package/dist/core/adapter/adapters/akm-metadata.js +42 -12
  111. package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
  112. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  113. package/dist/core/adapter/execution-source.js +17 -29
  114. package/dist/core/asset/asset-placement.js +4 -13
  115. package/dist/core/asset/frontmatter.js +106 -1
  116. package/dist/core/asset/resolve-ref.js +1 -1
  117. package/dist/core/bundle-id.js +42 -5
  118. package/dist/core/bundle-rename.js +285 -0
  119. package/dist/core/config/config-io.js +1 -2
  120. package/dist/core/config/config-schema.js +9 -34
  121. package/dist/core/config/config-walker.js +1 -1
  122. package/dist/core/config/config.js +184 -111
  123. package/dist/core/config/engine-semantics.js +0 -2
  124. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  125. package/dist/core/config/schema/embedding.js +20 -5
  126. package/dist/core/config/schema/engines.js +5 -0
  127. package/dist/core/config/schema/execution.js +1 -1
  128. package/dist/core/config/schema/experimental.js +1 -1
  129. package/dist/core/config/schema/improve-processes.js +54 -125
  130. package/dist/core/config/schema/improve.js +4 -42
  131. package/dist/core/config/schema/index-config.js +9 -48
  132. package/dist/core/config/schema/scheduler.js +12 -12
  133. package/dist/core/config/schema/search.js +6 -22
  134. package/dist/core/env-secret-ref.js +0 -1
  135. package/dist/core/errors.js +8 -9
  136. package/dist/core/file-change.js +13 -5
  137. package/dist/core/file-lock.js +76 -173
  138. package/dist/core/improve-result.js +35 -7
  139. package/dist/core/improve-types.js +0 -1
  140. package/dist/core/logs-db.js +2 -2
  141. package/dist/core/loopback.js +7 -12
  142. package/dist/core/non-task-input.js +20 -0
  143. package/dist/core/parse.js +13 -16
  144. package/dist/core/paths.js +0 -24
  145. package/dist/core/redaction.js +109 -2
  146. package/dist/core/run-lock.js +2 -5
  147. package/dist/core/spawn-env.js +1 -1
  148. package/dist/core/state/migrations.js +123 -61
  149. package/dist/core/state-db-scope.js +2 -4
  150. package/dist/core/state-db.js +126 -692
  151. package/dist/core/time.js +0 -20
  152. package/dist/core/type-presentation.js +1 -9
  153. package/dist/core/write-source.js +294 -1005
  154. package/dist/execution/input-contract.js +1 -1
  155. package/dist/execution/resolved-request.js +135 -689
  156. package/dist/execution/source.js +63 -257
  157. package/dist/execution/target-ref.js +1 -1
  158. package/dist/indexer/bundle-identity-guard.js +2 -2
  159. package/dist/indexer/db/llm-cache.js +2 -2
  160. package/dist/indexer/ensure-index.js +77 -73
  161. package/dist/indexer/index-rebuild-lock.js +3 -11
  162. package/dist/indexer/index-writer-lock.js +8 -17
  163. package/dist/indexer/index-written-assets.js +141 -154
  164. package/dist/indexer/indexer.js +400 -1124
  165. package/dist/indexer/links/declared-links.js +90 -0
  166. package/dist/indexer/materialize-embeddings.js +60 -397
  167. package/dist/indexer/passes/memory-inference.js +96 -90
  168. package/dist/indexer/passes/metadata.js +132 -219
  169. package/dist/indexer/read-preflight.js +0 -7
  170. package/dist/indexer/scan/doc-to-entry.js +2 -3
  171. package/dist/indexer/scan/drain-dir.js +1 -1
  172. package/dist/indexer/search/db-search.js +190 -590
  173. package/dist/indexer/search/fts-query.js +30 -41
  174. package/dist/indexer/search/ranking.js +28 -154
  175. package/dist/indexer/search/search-attribution.js +12 -32
  176. package/dist/indexer/search/search-fields.js +11 -15
  177. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  178. package/dist/indexer/search/search-source.js +1 -4
  179. package/dist/indexer/usage/usage-events.js +36 -7
  180. package/dist/indexer/walk/walker.js +3 -4
  181. package/dist/integrations/agent/engine-fallback.js +23 -40
  182. package/dist/integrations/agent/engine-resolution.js +93 -183
  183. package/dist/integrations/agent/execution.js +507 -0
  184. package/dist/integrations/agent/model-map.js +28 -156
  185. package/dist/integrations/agent/request-lowering.js +66 -141
  186. package/dist/integrations/agent/runner-dispatch.js +143 -321
  187. package/dist/integrations/agent/runner.js +54 -14
  188. package/dist/integrations/lockfile.js +53 -101
  189. package/dist/llm/client.js +18 -6
  190. package/dist/llm/embedders/deterministic.js +2 -3
  191. package/dist/llm/embedders/profile.js +71 -0
  192. package/dist/llm/embedders/remote.js +11 -17
  193. package/dist/llm/feature-gate.js +0 -8
  194. package/dist/llm/index-passes.js +3 -5
  195. package/dist/llm/memory-infer.js +1 -2
  196. package/dist/llm/structured-call.js +5 -24
  197. package/dist/output/generic-render.js +23 -11
  198. package/dist/output/html-render.js +13 -10
  199. package/dist/output/render-registry.js +3 -32
  200. package/dist/output/shapes/helpers.js +25 -38
  201. package/dist/output/shapes/passthrough.js +1 -9
  202. package/dist/{indexer/graph/graph-types.js → output/text/bundle-rename.js} +4 -1
  203. package/dist/output/text/command-format.js +69 -31
  204. package/dist/output/text/helpers.js +1 -1
  205. package/dist/output/text/migrate.js +5 -14
  206. package/dist/output/text/proposal-format.js +48 -3
  207. package/dist/output/text/show-format.js +13 -17
  208. package/dist/output/text/workflow-format.js +0 -32
  209. package/dist/output/text.js +2 -0
  210. package/dist/registry/factory.js +4 -19
  211. package/dist/registry/network.js +66 -220
  212. package/dist/registry/providers/index.js +0 -2
  213. package/dist/registry/providers/skills-sh.js +3 -14
  214. package/dist/registry/providers/static-index.js +24 -26
  215. package/dist/registry/resolve.js +55 -131
  216. package/dist/scripts/akm-migrate-node.js +42948 -92369
  217. package/dist/scripts/akm-migrate.js +42935 -92354
  218. package/dist/setup/registry-stash-loader.js +4 -13
  219. package/dist/setup/semantic-assets.js +3 -44
  220. package/dist/setup/setup.js +1 -1
  221. package/dist/setup/steps/connection.js +5 -6
  222. package/dist/setup/steps/platforms.js +2 -2
  223. package/dist/setup/steps/tasks.js +25 -15
  224. package/dist/sources/provider-factory.js +17 -18
  225. package/dist/sources/providers/filesystem.js +2 -3
  226. package/dist/sources/providers/git-install.js +7 -1
  227. package/dist/sources/providers/git-provider.js +0 -3
  228. package/dist/sources/providers/git-stash.js +83 -21
  229. package/dist/sources/providers/npm.js +2 -4
  230. package/dist/sources/providers/provider-utils.js +5 -10
  231. package/dist/sources/providers/website.js +0 -2
  232. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  233. package/dist/sources/website-url.js +2 -2
  234. package/dist/storage/database.js +9 -35
  235. package/dist/storage/repositories/improve-ledger-repository.js +209 -0
  236. package/dist/storage/repositories/index-connection.js +39 -72
  237. package/dist/storage/repositories/index-entries-repository.js +131 -129
  238. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  239. package/dist/storage/repositories/index-entry-schema.js +101 -268
  240. package/dist/storage/repositories/index-fts-repository.js +86 -256
  241. package/dist/storage/repositories/index-links-repository.js +143 -0
  242. package/dist/storage/repositories/index-llm-cache-repository.js +7 -9
  243. package/dist/storage/repositories/index-meta-repository.js +6 -4
  244. package/dist/storage/repositories/index-schema.js +257 -325
  245. package/dist/storage/repositories/index-utility-repository.js +8 -29
  246. package/dist/storage/repositories/index-vec-repository.js +133 -414
  247. package/dist/storage/repositories/outcome-repository.js +2 -1
  248. package/dist/storage/repositories/proposals-repository.js +104 -1
  249. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  250. package/dist/storage/repositories/salience-repository.js +1 -19
  251. package/dist/storage/repositories/task-history-repository.js +26 -4
  252. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  253. package/dist/storage/sqlite-migrations.js +136 -0
  254. package/dist/storage/sqlite-pragmas.js +11 -9
  255. package/dist/storage/sqlite-transaction.js +170 -0
  256. package/dist/storage/state-db-integrity.js +130 -0
  257. package/dist/tasks/activation-config.js +134 -62
  258. package/dist/tasks/backends/cron.js +191 -302
  259. package/dist/tasks/backends/exec-utils.js +2 -5
  260. package/dist/tasks/backends/launchd.js +141 -748
  261. package/dist/tasks/backends/schtasks.js +119 -623
  262. package/dist/tasks/prepare/prepare-support.js +5 -15
  263. package/dist/tasks/prepare/prepare.js +0 -2
  264. package/dist/tasks/resolve-akm-bin.js +20 -79
  265. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  266. package/dist/tasks/run/load-task.js +1 -1
  267. package/dist/tasks/scheduler-binding.js +20 -238
  268. package/dist/tasks/scheduler-invocation.js +136 -244
  269. package/dist/tasks/scheduler-lock.js +53 -0
  270. package/dist/tasks/scheduler-sync.js +368 -679
  271. package/dist/tasks/source/parse-task-source.js +55 -9
  272. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  273. package/dist/tasks/source/task-to-v4.js +464 -88
  274. package/dist/workflows/authoring/authoring.js +3 -12
  275. package/dist/workflows/compile.js +211 -0
  276. package/dist/workflows/concurrency-policy.js +13 -74
  277. package/dist/workflows/exec/child-invocation.js +3 -17
  278. package/dist/workflows/exec/child-workflow.js +32 -141
  279. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  280. package/dist/workflows/exec/environment.js +98 -0
  281. package/dist/workflows/exec/exec-unit.js +33 -140
  282. package/dist/workflows/exec/frozen-judge.js +7 -59
  283. package/dist/workflows/exec/native-executor.js +82 -341
  284. package/dist/workflows/exec/param-secrets.js +29 -47
  285. package/dist/workflows/exec/run-workflow.js +154 -387
  286. package/dist/workflows/exec/scheduler.js +9 -36
  287. package/dist/workflows/exec/step-work.js +127 -430
  288. package/dist/workflows/exec/unit-dispatch.js +11 -63
  289. package/dist/workflows/exec/unit-writer.js +8 -52
  290. package/dist/workflows/exec/worktree.js +39 -273
  291. package/dist/workflows/freeze/child-output-references.js +4 -15
  292. package/dist/workflows/freeze/environment.js +99 -92
  293. package/dist/workflows/freeze/freeze.js +172 -0
  294. package/dist/workflows/freeze/step-values.js +19 -21
  295. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  296. package/dist/workflows/freeze/targets/command.js +10 -33
  297. package/dist/workflows/freeze/targets/script.js +5 -12
  298. package/dist/workflows/freeze/targets/shell.js +3 -6
  299. package/dist/workflows/freeze/targets/task.js +25 -80
  300. package/dist/workflows/freeze/task-bindings.js +20 -67
  301. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  302. package/dist/workflows/ir/params.js +6 -51
  303. package/dist/workflows/ir/plan-hash.js +2 -34
  304. package/dist/workflows/parser.js +140 -43
  305. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  306. package/dist/workflows/renderer.js +36 -69
  307. package/dist/workflows/resource-limits.js +12 -120
  308. package/dist/workflows/runtime/agent-identity.js +8 -40
  309. package/dist/workflows/runtime/run-outputs.js +3 -6
  310. package/dist/workflows/runtime/run-plan.js +316 -0
  311. package/dist/workflows/runtime/runs.js +48 -200
  312. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  313. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  314. package/dist/workflows/validate-summary.js +2 -7
  315. package/docs/integration/bundling-akm.md +49 -42
  316. package/docs/migration/README.md +1 -0
  317. package/docs/migration/release-notes/0.9.17.md +43 -0
  318. package/docs/migration/v0.9.1-to-v0.9.2.md +23 -7
  319. package/docs/reference/cli.md +232 -135
  320. package/docs/reference/configuration.md +71 -57
  321. package/docs/reference/data-and-telemetry.md +20 -21
  322. package/docs/reference/tasks.md +105 -39
  323. package/docs/reference/workflow-schema.md +14 -18
  324. package/docs/reference/workflows.md +6 -9
  325. package/package.json +1 -1
  326. package/schemas/akm-config.json +115 -738
  327. package/schemas/akm-workflow.json +1 -0
  328. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  329. package/dist/assets/prompts/contradiction-judge.md +0 -33
  330. package/dist/assets/prompts/graph-extract-system.md +0 -1
  331. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  332. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  333. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  334. package/dist/commands/health/advisories.js +0 -150
  335. package/dist/commands/health/metrics.js +0 -329
  336. package/dist/commands/health/surfaces.js +0 -102
  337. package/dist/commands/improve/anti-collapse.js +0 -83
  338. package/dist/commands/improve/collapse-detector.js +0 -432
  339. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  340. package/dist/commands/improve/consolidate/merge.js +0 -149
  341. package/dist/commands/improve/distill/promote-memory.js +0 -291
  342. package/dist/commands/improve/distill/quality-gate.js +0 -337
  343. package/dist/commands/improve/eval-cases.js +0 -52
  344. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  345. package/dist/commands/improve/proposal-envelope.js +0 -31
  346. package/dist/commands/improve/run-context.js +0 -123
  347. package/dist/commands/improve/shared.js +0 -31
  348. package/dist/commands/improve/source-identity.js +0 -28
  349. package/dist/commands/improve/triage.js +0 -96
  350. package/dist/commands/proposal/drain-policies.js +0 -151
  351. package/dist/commands/sources/update-transaction.js +0 -220
  352. package/dist/core/action-contributors.js +0 -28
  353. package/dist/core/config/config-version-shim.js +0 -101
  354. package/dist/core/fs-txn.js +0 -405
  355. package/dist/core/lexical-score.js +0 -25
  356. package/dist/core/maintenance-barrier.js +0 -167
  357. package/dist/execution/executable-identity.js +0 -105
  358. package/dist/execution/guarded-source.js +0 -427
  359. package/dist/indexer/db/graph-db.js +0 -444
  360. package/dist/indexer/graph/graph-boost.js +0 -427
  361. package/dist/indexer/graph/graph-dedup.js +0 -95
  362. package/dist/indexer/graph/graph-extraction.js +0 -1108
  363. package/dist/indexer/search/name-match.js +0 -35
  364. package/dist/indexer/search/ranking-contributors.js +0 -515
  365. package/dist/indexer/search/ranking-types.js +0 -4
  366. package/dist/indexer/walk/project-context.js +0 -192
  367. package/dist/integrations/agent/execution-cascade.js +0 -566
  368. package/dist/integrations/agent/execution-definitions.js +0 -202
  369. package/dist/integrations/agent/execution-lowering.js +0 -841
  370. package/dist/integrations/agent/execution-preparation.js +0 -98
  371. package/dist/integrations/agent/inline-execution.js +0 -74
  372. package/dist/llm/graph-extract.js +0 -728
  373. package/dist/llm/metadata-enhance.js +0 -96
  374. package/dist/registry/create-provider-registry.js +0 -29
  375. package/dist/registry/pinned-request-helper.js +0 -247
  376. package/dist/registry/pinned-transport.js +0 -717
  377. package/dist/sources/providers/index.js +0 -14
  378. package/dist/storage/engines/sqlite-migrations.js +0 -271
  379. package/dist/storage/repositories/canaries-repository.js +0 -107
  380. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  381. package/dist/storage/repositories/registry-cache.js +0 -113
  382. package/dist/tasks/scheduler-sync-preview.js +0 -52
  383. package/dist/tasks/source/task-to-v3.js +0 -507
  384. package/dist/workflows/freeze/resolve-steps.js +0 -86
  385. package/dist/workflows/freeze/source-freeze.js +0 -64
  386. package/dist/workflows/ir/compile.js +0 -321
  387. package/dist/workflows/ir/environment-v4.js +0 -330
  388. package/dist/workflows/ir/freeze-v4.js +0 -153
  389. package/dist/workflows/ir/schema-v4.js +0 -745
  390. package/dist/workflows/ir/schema.js +0 -354
  391. package/dist/workflows/program/schema.js +0 -77
  392. package/dist/workflows/runtime/checkin.js +0 -57
  393. package/dist/workflows/runtime/plan-classifier.js +0 -196
  394. package/dist/workflows/runtime/unit-checkin.js +0 -45
  395. package/dist/workflows/runtime/unit-phases.js +0 -20
  396. package/dist/workflows/schema.js +0 -4
  397. package/dist/workflows/source-ir/compile.js +0 -200
  398. package/dist/workflows/source-ir/program.js +0 -50
  399. package/dist/workflows/source-ir/result.js +0 -26
  400. package/dist/workflows/source-ir/schema.js +0 -786
  401. package/dist/workflows/source-ir/triggers.js +0 -79
  402. package/dist/workflows/source-ir/uses.js +0 -40
  403. package/dist/workflows/validator.js +0 -60
@@ -1,728 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * LLM helper for the `akm index` graph-extraction pass (#207).
6
- *
7
- * Given a single asset body (typically a `memory:` or `knowledge:` file),
8
- * asks the configured LLM to surface the entities mentioned in it and the
9
- * relations between them. The pass itself
10
- * (`src/indexer/graph/graph-extraction.ts`) is responsible for deciding which
11
- * files to extract, persisting the resulting nodes/edges to the index DB,
12
- * and feeding the graph data into the FTS5+boosts
13
- * search pipeline as a single boost component.
14
- *
15
- * This module is intentionally tiny and stateless so tests can stub it via
16
- * `mock.module("../src/llm/graph-extract", ...)` without hitting a network.
17
- *
18
- * The symbolic LLM runner comes from the current index-pass execution
19
- * resolution and is passed straight through.
20
- */
21
- import systemPromptTemplate from "../assets/prompts/graph-extract-system.md" with { type: "text" };
22
- import userPromptTemplate from "../assets/prompts/graph-extract-user-prompt.md" with { type: "text" };
23
- import { splitMarkdownFragmentStats } from "../core/asset/markdown-fragments.js";
24
- import { toErrorMessage } from "../core/common.js";
25
- import { ConfigError } from "../core/errors.js";
26
- import { parseEmbeddedJsonResponse } from "../core/parse.js";
27
- import { warn, warnVerbose } from "../core/warn.js";
28
- import { isContextSizeError } from "./client.js";
29
- import { tryLlmFeature } from "./feature-gate.js";
30
- import { callStructured } from "./structured-call.js";
31
- /**
32
- * Separator token used between assets in a batch prompt.
33
- * Chosen to be visually clear and unlikely to appear verbatim in asset bodies.
34
- */
35
- const BATCH_ASSET_SEPARATOR = "=== ASSET";
36
- export const GRAPH_EXTRACT_PROMPT_VERSION = "v2";
37
- /** Asset bodies longer than this are chunked instead of truncated. */
38
- const MAX_CHUNK_BODY_CHARS = 1600;
39
- /** Bodies longer than this are excluded from multi-asset batch prompts. */
40
- const MAX_BATCH_BODY_CHARS = 1600;
41
- const MIN_RELATION_CONFIDENCE = 0.5;
42
- const NON_ARRAY_BATCH_DISABLE_THRESHOLD = 2;
43
- /** Hard cap on entities returned per asset — guards against runaway LLM output. */
44
- const MAX_ENTITIES_PER_ASSET = 32;
45
- /** Hard cap on relations returned per asset. */
46
- const MAX_RELATIONS_PER_ASSET = 32;
47
- const SYSTEM_PROMPT = systemPromptTemplate;
48
- const USER_PROMPT_PREFIX = userPromptTemplate
49
- .replace("{{MAX_ENTITIES}}", String(MAX_ENTITIES_PER_ASSET))
50
- .replace("{{MAX_RELATIONS}}", String(MAX_RELATIONS_PER_ASSET));
51
- const GENERIC_ENTITIES = new Set([
52
- "agent",
53
- "application",
54
- "assistant",
55
- "code",
56
- "content",
57
- "data",
58
- "developer",
59
- "document",
60
- "file",
61
- "knowledge",
62
- "memory",
63
- "note",
64
- "notes",
65
- "project",
66
- "service",
67
- "system",
68
- "task",
69
- "team",
70
- "text",
71
- "thing",
72
- "user",
73
- ]);
74
- const GENERIC_RELATION_TYPES = new Set(["has", "is", "mentions", "references", "related to"]);
75
- function parseConfidence(raw) {
76
- if (typeof raw !== "number" || !Number.isFinite(raw))
77
- return undefined;
78
- return Math.max(0, Math.min(1, raw));
79
- }
80
- function normalizeEntityName(raw) {
81
- return raw
82
- .trim()
83
- .replace(/^[`"']+|[`"']+$/g, "")
84
- .replace(/\s+/g, " ")
85
- .replace(/[;,!?]+$/g, "")
86
- .trim();
87
- }
88
- function normalizeRelationType(raw) {
89
- const normalized = raw
90
- .trim()
91
- .toLowerCase()
92
- .replace(/^[`"']+|[`"']+$/g, "")
93
- .replace(/\s+/g, " ")
94
- .replace(/[.;,!?]+$/g, "")
95
- .trim();
96
- if (!normalized)
97
- return undefined;
98
- if (normalized === "use" || normalized === "utilizes")
99
- return "uses";
100
- if (normalized === "depend on" || normalized === "depends")
101
- return "depends on";
102
- if (normalized === "integrates" || normalized === "integration with")
103
- return "integrates with";
104
- return normalized;
105
- }
106
- function normalizeEntityKey(raw) {
107
- return normalizeEntityName(raw).toLowerCase();
108
- }
109
- function bumpTelemetry(telemetry, key, amount = 1) {
110
- if (!telemetry)
111
- return;
112
- telemetry[key] = (telemetry[key] ?? 0) + amount;
113
- }
114
- function normalizeBatchState(state) {
115
- if (!state)
116
- return undefined;
117
- state.batchingDisabled = state.batchingDisabled === true;
118
- state.nonArrayBatchFailures = Math.max(0, state.nonArrayBatchFailures ?? 0);
119
- return state;
120
- }
121
- function splitBodyIntoChunks(body, maxChars = MAX_CHUNK_BODY_CHARS) {
122
- const split = splitMarkdownFragmentStats(body, maxChars);
123
- // Graph extraction keeps its historical cost shape: adjacent safe fragments
124
- // share one LLM call whenever they fit. The fragment splitter is still the
125
- // sole boundary authority; this is only prompt packing, never a second
126
- // parser/chunker. Hard splits stay isolated and telemetry remains the core
127
- // split count rather than counting ordinary heading boundaries.
128
- const chunks = [];
129
- let current = "";
130
- for (const fragment of split.fragments) {
131
- const candidate = current ? `${current}\n\n${fragment.text}` : fragment.text;
132
- if (candidate.length <= maxChars) {
133
- current = candidate;
134
- }
135
- else {
136
- if (current)
137
- chunks.push(current);
138
- current = fragment.text;
139
- }
140
- }
141
- if (current)
142
- chunks.push(current);
143
- return { chunks, truncationCount: split.hardSplitCount };
144
- }
145
- /** Consistency weight for blending chunk-agreement with LLM confidence. */
146
- const CONSISTENCY_WEIGHT = 0.4;
147
- function mergeGraphExtractions(extractions) {
148
- const totalChunks = extractions.length;
149
- const entityCanonical = new Map();
150
- const entityChunkCounts = new Map();
151
- const relationByKey = new Map();
152
- const relationChunkCounts = new Map();
153
- let confidence;
154
- let truncationCount = 0;
155
- let filteredGenericEntities = 0;
156
- let filteredInvalidRelations = 0;
157
- let filteredLowConfidenceRelations = 0;
158
- let firstFailureReason;
159
- for (const extraction of extractions) {
160
- truncationCount += extraction.truncationCount ?? 0;
161
- filteredGenericEntities += extraction.filteredGenericEntities ?? 0;
162
- filteredInvalidRelations += extraction.filteredInvalidRelations ?? 0;
163
- filteredLowConfidenceRelations += extraction.filteredLowConfidenceRelations ?? 0;
164
- if (extraction.status === "failed" && !firstFailureReason)
165
- firstFailureReason = extraction.reason;
166
- const nextConfidence = parseConfidence(extraction.confidence);
167
- if (nextConfidence !== undefined)
168
- confidence = confidence === undefined ? nextConfidence : Math.max(confidence, nextConfidence);
169
- for (const entity of extraction.entities) {
170
- const key = normalizeEntityKey(entity);
171
- if (!key)
172
- continue;
173
- if (!entityCanonical.has(key))
174
- entityCanonical.set(key, entity);
175
- entityChunkCounts.set(key, (entityChunkCounts.get(key) ?? 0) + 1);
176
- }
177
- }
178
- for (const extraction of extractions) {
179
- for (const relation of extraction.relations) {
180
- const fromKey = normalizeEntityKey(relation.from);
181
- const toKey = normalizeEntityKey(relation.to);
182
- const type = normalizeRelationType(relation.type ?? "");
183
- if (!fromKey || !toKey || !type)
184
- continue;
185
- const from = entityCanonical.get(fromKey);
186
- const to = entityCanonical.get(toKey);
187
- if (!from || !to)
188
- continue;
189
- const key = `${fromKey}\u0000${toKey}\u0000${type}`;
190
- if (!relationByKey.has(key)) {
191
- relationByKey.set(key, {
192
- from,
193
- to,
194
- type,
195
- });
196
- relationChunkCounts.set(key, 0);
197
- }
198
- relationChunkCounts.set(key, (relationChunkCounts.get(key) ?? 0) + 1);
199
- const nextConfidence = parseConfidence(relation.confidence);
200
- const existing = relationByKey.get(key);
201
- if (existing && nextConfidence !== undefined) {
202
- const current = parseConfidence(existing.confidence) ?? 0;
203
- if (nextConfidence > current)
204
- existing.confidence = nextConfidence;
205
- }
206
- }
207
- }
208
- function blendConsistency(llmConfidence, chunkCount) {
209
- const consistency = totalChunks > 1 ? chunkCount / totalChunks : 1;
210
- if (llmConfidence === undefined)
211
- return consistency;
212
- return (1 - CONSISTENCY_WEIGHT) * llmConfidence + CONSISTENCY_WEIGHT * consistency;
213
- }
214
- const entities = [...entityCanonical.values()].slice(0, MAX_ENTITIES_PER_ASSET);
215
- const relations = [...relationByKey.values()].slice(0, MAX_RELATIONS_PER_ASSET);
216
- for (const relation of relations) {
217
- const fromKey = normalizeEntityKey(relation.from);
218
- const toKey = normalizeEntityKey(relation.to);
219
- const type = normalizeRelationType(relation.type ?? "");
220
- if (!fromKey || !toKey || !type)
221
- continue;
222
- const key = `${fromKey}\u0000${toKey}\u0000${type}`;
223
- const chunkCount = relationChunkCounts.get(key) ?? 1;
224
- relation.confidence = blendConsistency(relation.confidence, chunkCount);
225
- }
226
- const status = entities.length > 0 ? "extracted" : firstFailureReason ? "failed" : "empty";
227
- const reason = status === "extracted" ? "none" : (firstFailureReason ?? "no_graph_content");
228
- const mergedConfidence = confidence !== undefined ? blendConsistency(confidence, totalChunks) : totalChunks > 1 ? 1 : undefined;
229
- return {
230
- entities,
231
- relations,
232
- ...(mergedConfidence !== undefined ? { confidence: mergedConfidence } : {}),
233
- status,
234
- reason,
235
- chunkCount: extractions.length,
236
- truncationCount,
237
- filteredGenericEntities,
238
- filteredInvalidRelations,
239
- filteredLowConfidenceRelations,
240
- };
241
- }
242
- function parseGraphExtraction(raw) {
243
- const empty = (reason = "no_graph_content") => ({
244
- entities: [],
245
- relations: [],
246
- status: reason === "llm_error" || reason === "invalid_json" || reason === "context_limit" ? "failed" : "empty",
247
- reason,
248
- });
249
- if (typeof raw !== "object" || raw === null || Array.isArray(raw))
250
- return empty();
251
- const item = raw;
252
- const extractionConfidence = parseConfidence(item.confidence);
253
- const entityCanonical = new Map();
254
- let filteredGenericEntities = 0;
255
- if (Array.isArray(item.entities)) {
256
- for (const value of item.entities) {
257
- if (typeof value !== "string")
258
- continue;
259
- const normalized = normalizeEntityName(value);
260
- if (!normalized)
261
- continue;
262
- const normalizedKey = normalized.toLowerCase();
263
- // Drop generic/empty entities AND raw file/dir paths (anything with a
264
- // path separator) — the prompt no longer asks for them and isJunkEntity
265
- // discards them downstream, so emitting them is pure waste/junk (#632).
266
- if (!/[a-z0-9]/i.test(normalized) ||
267
- GENERIC_ENTITIES.has(normalizedKey) ||
268
- normalized.includes("/") ||
269
- normalized.includes("\\")) {
270
- filteredGenericEntities += 1;
271
- continue;
272
- }
273
- const key = normalized.toLowerCase();
274
- if (!entityCanonical.has(key))
275
- entityCanonical.set(key, normalized);
276
- if (entityCanonical.size >= MAX_ENTITIES_PER_ASSET)
277
- break;
278
- }
279
- }
280
- const entities = Array.from(entityCanonical.values());
281
- const relations = [];
282
- let filteredInvalidRelations = 0;
283
- let filteredLowConfidenceRelations = 0;
284
- if (Array.isArray(item.relations)) {
285
- for (const relation of item.relations) {
286
- if (typeof relation !== "object" || relation === null || Array.isArray(relation)) {
287
- filteredInvalidRelations += 1;
288
- continue;
289
- }
290
- const rel = relation;
291
- const fromRaw = typeof rel.from === "string" ? normalizeEntityName(rel.from) : "";
292
- const toRaw = typeof rel.to === "string" ? normalizeEntityName(rel.to) : "";
293
- if (!fromRaw || !toRaw) {
294
- filteredInvalidRelations += 1;
295
- continue;
296
- }
297
- const from = entityCanonical.get(fromRaw.toLowerCase());
298
- const to = entityCanonical.get(toRaw.toLowerCase());
299
- if (!from || !to || from.toLowerCase() === to.toLowerCase()) {
300
- filteredInvalidRelations += 1;
301
- continue;
302
- }
303
- const type = typeof rel.type === "string" ? normalizeRelationType(rel.type) : undefined;
304
- if (type !== undefined && GENERIC_RELATION_TYPES.has(type)) {
305
- filteredInvalidRelations += 1;
306
- continue;
307
- }
308
- const confidence = parseConfidence(rel.confidence);
309
- if (confidence !== undefined && confidence < MIN_RELATION_CONFIDENCE) {
310
- filteredLowConfidenceRelations += 1;
311
- continue;
312
- }
313
- relations.push({
314
- from,
315
- to,
316
- ...(type ? { type } : {}),
317
- ...(confidence !== undefined ? { confidence } : {}),
318
- });
319
- if (relations.length >= MAX_RELATIONS_PER_ASSET)
320
- break;
321
- }
322
- }
323
- const confidence = extractionConfidence;
324
- const status = entities.length > 0 ? "extracted" : "empty";
325
- const reason = entities.length > 0 ? "none" : filteredGenericEntities > 0 ? "generic_entities_only" : "no_graph_content";
326
- return {
327
- entities,
328
- relations,
329
- status,
330
- reason,
331
- filteredGenericEntities,
332
- filteredInvalidRelations,
333
- filteredLowConfidenceRelations,
334
- ...(confidence !== undefined ? { confidence } : {}),
335
- };
336
- }
337
- /**
338
- * Build the system prompt for a batched graph-extraction call.
339
- *
340
- * The prompt instructs the model to return a JSON array of exactly `count`
341
- * objects, one per asset, in input order. Index alignment is the critical
342
- * invariant — if the model drops an asset it still must emit an empty
343
- * placeholder `{"entities":[],"relations":[]}` at that position.
344
- *
345
- * Worked example (3 assets, abbreviated):
346
- *
347
- * Input user message:
348
- * Extract entities and relations from the N=3 assets below.
349
- * ...rules...
350
- * === ASSET 1 ===
351
- * ServiceA integrates with ServiceB.
352
- * === ASSET 2 ===
353
- * Terraform provisions the Prod cluster.
354
- * === ASSET 3 ===
355
- * No extractable graph content here.
356
- *
357
- * Expected model output (valid JSON array, no prose):
358
- * [
359
- * {"entities":["ServiceA","ServiceB"],"relations":[{"from":"ServiceA","to":"ServiceB","type":"integrates with"}]},
360
- * {"entities":["Terraform","Prod cluster"],"relations":[{"from":"Terraform","to":"Prod cluster","type":"provisions"}]},
361
- * {"entities":[],"relations":[]}
362
- * ]
363
- *
364
- * If the model returns fewer than 3 items (partial failure), the caller
365
- * (`extractGraphFromBodies`) falls back to individual calls for missing indices.
366
- */
367
- function buildBatchSystemPrompt() {
368
- return ("You extract knowledge graphs from developer notes. " +
369
- "Return ONLY a valid JSON array — no prose, no markdown fences, no preamble. " +
370
- "Each element of the array corresponds to one input asset, in order. " +
371
- "The array length MUST equal the number of assets provided. " +
372
- 'Use {"entities":[],"relations":[]} for assets with no extractable graph content.');
373
- }
374
- /**
375
- * Hardened system prompt for the single batch retry (#635). Used only after a
376
- * first response failed array salvage — leans harder on "raw array only" so a
377
- * model that wrapped the array in prose/fences corrects itself before we pay
378
- * the per-asset fallback.
379
- */
380
- function buildBatchRetrySystemPrompt() {
381
- return (`${buildBatchSystemPrompt()} ` +
382
- "Your previous response could NOT be parsed as a JSON array. " +
383
- "Respond with ONLY the raw JSON array — start with '[' and end with ']'. " +
384
- "No prose, no explanation, no markdown code fences, no preamble.");
385
- }
386
- function buildBatchUserPrompt(bodies) {
387
- const count = bodies.length;
388
- const assetBlocks = bodies.map((body, i) => `${BATCH_ASSET_SEPARATOR} ${i + 1} ===\n${body.trim()}`).join("\n\n");
389
- return (`Extract entities and relations from the N=${count} assets below.\n\n` +
390
- `Rules:\n` +
391
- `- Output ONLY a JSON array of exactly ${count} objects, one per asset, preserving input order.\n` +
392
- `- Each object: {"entities": ["Entity One", ...], "relations": [{"from": "A", "to": "B", "type": "uses"}, ...]}\n` +
393
- `- Entities are short, canonical noun phrases (project names, services, tools, people, file/dir names, technical concepts).\n` +
394
- `- Relations connect two entities that both appear in that asset's entities array.\n` +
395
- `- "type" is a short verb phrase (e.g. "uses", "depends on", "owns"). Optional; omit when unsure.\n` +
396
- `- Drop pleasantries, meta-commentary, and timestamps.\n` +
397
- `- Limit to at most ${MAX_ENTITIES_PER_ASSET} entities and ${MAX_RELATIONS_PER_ASSET} relations per asset.\n` +
398
- `- Use {"entities":[],"relations":[]} for assets with no extractable graph content.\n` +
399
- `- The array MUST have exactly ${count} elements — one placeholder per asset even if empty.\n\n` +
400
- assetBlocks);
401
- }
402
- function formatContextHint(llmRunner) {
403
- return llmRunner.connection.contextLength ? `, configured contextLength=${llmRunner.connection.contextLength}` : "";
404
- }
405
- /** Dispatch one raw graph prompt through the common resolved-request adapter. */
406
- async function callGraphLlm(runner, messages, request, lease, onNotices) {
407
- return callStructured({
408
- feature: "graph_extraction",
409
- runner,
410
- ...(lease ? { lease } : {}),
411
- messages,
412
- request,
413
- onNotices,
414
- parse: (raw) => raw ?? "",
415
- onError: (_cls, error) => {
416
- throw error;
417
- },
418
- fallback: "",
419
- });
420
- }
421
- /**
422
- * Parse and validate a single item from the batch response array.
423
- * Mirrors the validation logic in `extractGraphFromBody`.
424
- */
425
- function parseBatchItem(raw) {
426
- return parseGraphExtraction(raw);
427
- }
428
- function applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState) {
429
- if (batchState)
430
- batchState.nonArrayBatchFailures = 0;
431
- if (batchResult.length > nonEmptyBodies.length) {
432
- warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
433
- `ignoring ${batchResult.length - nonEmptyBodies.length} extra item(s).`);
434
- }
435
- for (let j = 0; j < nonEmptyBodies.length; j++) {
436
- const originalIndex = nonEmptyIndices[j];
437
- if (originalIndex === undefined)
438
- continue;
439
- if (j < batchResult.length)
440
- results[originalIndex] = parseBatchItem(batchResult[j]);
441
- }
442
- }
443
- /**
444
- * Extract entities and relations from multiple asset bodies in a single LLM
445
- * call (batched graph extraction).
446
- *
447
- * Sends all `bodies` as a single prompt with `=== ASSET N ===` separators
448
- * and expects a JSON array where element `i` corresponds to `bodies[i]`.
449
- *
450
- * **Partial-failure handling**: if the model returns fewer elements than
451
- * `bodies.length`, missing indices are filled by falling back to individual
452
- * `extractGraphFromBody` calls — ensuring every input always has a result.
453
- *
454
- * Returns an array of the same length as `bodies` (never shorter).
455
- * Individual elements default to `{entities:[], relations:[]}` on failure.
456
- *
457
- * Routes through `tryLlmFeature("graph_extraction", ...)` so the feature gate
458
- * and onFallback hook are honoured uniformly.
459
- *
460
- * @param llmRunner - Symbolic LLM runner selected through shared execution lowering.
461
- * @param bodies - Asset body strings to process in one batch.
462
- * @param signal - Optional AbortSignal for cancellation.
463
- * @param akmConfig - Full AKM config (for feature-gate checks).
464
- * @param onFallback - Optional fallback event sink.
465
- */
466
- export async function extractGraphFromBodies(llmRunner, bodies, signal, akmConfig, onFallback, options = {}) {
467
- const empty = () => ({ entities: [], relations: [] });
468
- const batchState = normalizeBatchState(options.batchState);
469
- // Degenerate case: no bodies → empty array (not an error).
470
- if (bodies.length === 0)
471
- return [];
472
- // Single body: delegate to the single-asset path for identical behaviour.
473
- if (bodies.length === 1) {
474
- const result = await extractGraphFromBody(llmRunner, bodies[0] ?? "", signal, akmConfig, onFallback, options);
475
- return [result];
476
- }
477
- // Filter out bodies that are empty so we don't waste tokens, but keep
478
- // index correspondence by tracking which indices were non-empty.
479
- const results = bodies.map(empty);
480
- const nonEmptyIndices = [];
481
- const nonEmptyBodies = [];
482
- const oversizedIndices = [];
483
- for (let i = 0; i < bodies.length; i++) {
484
- const trimmed = (bodies[i] ?? "").trim();
485
- if (trimmed) {
486
- if (trimmed.length > MAX_BATCH_BODY_CHARS) {
487
- oversizedIndices.push(i);
488
- }
489
- else {
490
- nonEmptyIndices.push(i);
491
- nonEmptyBodies.push(trimmed);
492
- }
493
- }
494
- }
495
- if (oversizedIndices.length > 0) {
496
- await Promise.all(oversizedIndices.map(async (index) => {
497
- results[index] = await extractGraphFromBody(llmRunner, bodies[index] ?? "", signal, akmConfig, onFallback, options);
498
- }));
499
- }
500
- if (nonEmptyBodies.length === 0)
501
- return results;
502
- if (batchState?.batchingDisabled) {
503
- return Promise.all(bodies.map((body) => extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options)));
504
- }
505
- const systemPrompt = buildBatchSystemPrompt();
506
- const userPrompt = buildBatchUserPrompt(nonEmptyBodies);
507
- const truncatedBodies = nonEmptyBodies.filter((body) => body.length > MAX_BATCH_BODY_CHARS).length;
508
- if (truncatedBodies > 0) {
509
- warnVerbose(`graph extraction (batch): ${truncatedBodies}/${nonEmptyBodies.length} asset body/bodies exceed the batch body threshold of ${MAX_BATCH_BODY_CHARS} chars.`);
510
- }
511
- let batchContextError = false;
512
- let nonArrayResponse = false;
513
- const batchOutcome = await tryLlmFeature("graph_extraction", akmConfig, async () => {
514
- try {
515
- const raw = await callGraphLlm(llmRunner, [
516
- { role: "system", content: systemPrompt },
517
- { role: "user", content: userPrompt },
518
- ], {
519
- temperature: 0.1,
520
- timeoutMs: llmRunner.timeoutMs,
521
- signal,
522
- onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
523
- }, options.lease, options.onNotices);
524
- if (!raw)
525
- return { kind: "value", value: null };
526
- // Array-preferring salvage (#635): the batch contract is a top-level
527
- // JSON array. A leading/example `{…}` object in the response must not
528
- // mask a valid `[…]` array as a false "non-array" failure.
529
- let parsed = parseEmbeddedJsonResponse(raw, { expect: "array" });
530
- if (!Array.isArray(parsed)) {
531
- // One stricter-reprompt retry before paying the per-asset fallback
532
- // (#635). Many genuine non-array responses recover when the model is
533
- // told explicitly to emit only the raw array.
534
- bumpTelemetry(options.telemetry, "retryAttempts");
535
- const retryRaw = await callGraphLlm(llmRunner, [
536
- { role: "system", content: buildBatchRetrySystemPrompt() },
537
- { role: "user", content: userPrompt },
538
- ], { temperature: 0, timeoutMs: llmRunner.timeoutMs, signal }, options.lease, options.onNotices);
539
- parsed = retryRaw ? parseEmbeddedJsonResponse(retryRaw, { expect: "array" }) : undefined;
540
- }
541
- if (!Array.isArray(parsed)) {
542
- nonArrayResponse = true;
543
- bumpTelemetry(options.telemetry, "nonArrayBatchFailures");
544
- if (batchState) {
545
- batchState.nonArrayBatchFailures += 1;
546
- if (batchState.nonArrayBatchFailures >= NON_ARRAY_BATCH_DISABLE_THRESHOLD) {
547
- batchState.batchingDisabled = true;
548
- }
549
- }
550
- warn(`graph extraction (batch): LLM response was not a JSON array for ${nonEmptyBodies.length} asset(s) ` +
551
- `even after a stricter retry; will fall back per-asset. ` +
552
- `promptChars=${userPrompt.length}${formatContextHint(llmRunner)}`);
553
- return { kind: "value", value: null };
554
- }
555
- return { kind: "value", value: parsed };
556
- }
557
- catch (err) {
558
- if (err instanceof ConfigError)
559
- return { kind: "config-error", error: err };
560
- const errMsg = toErrorMessage(err);
561
- if (isContextSizeError(errMsg)) {
562
- batchContextError = true;
563
- bumpTelemetry(options.telemetry, "contextBatchRetries");
564
- warn(`graph extraction (batch): context size exceeded for ${nonEmptyBodies.length} asset(s); ` +
565
- `skipping batch. promptChars=${userPrompt.length}${formatContextHint(llmRunner)}`);
566
- }
567
- else {
568
- warn(`graph extraction (batch) failed for ${nonEmptyBodies.length} asset(s); ` +
569
- `promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
570
- }
571
- return { kind: "value", value: null };
572
- }
573
- }, { kind: "value", value: null }, {
574
- timeoutMs: llmRunner.timeoutMs,
575
- onFallback,
576
- });
577
- if (batchOutcome.kind === "config-error")
578
- throw batchOutcome.error;
579
- const batchResult = batchOutcome.value;
580
- // Map successful batch results back to their original indices.
581
- if (batchResult !== null) {
582
- applySuccessfulBatchResults(results, batchResult, nonEmptyBodies, nonEmptyIndices, batchState);
583
- }
584
- if (batchContextError && nonEmptyBodies.length > 1) {
585
- const splitAt = Math.ceil(nonEmptyBodies.length / 2);
586
- const left = await extractGraphFromBodies(llmRunner, nonEmptyBodies.slice(0, splitAt), signal, akmConfig, onFallback, options);
587
- const right = await extractGraphFromBodies(llmRunner, nonEmptyBodies.slice(splitAt), signal, akmConfig, onFallback, options);
588
- const combined = [...left, ...right];
589
- for (let j = 0; j < nonEmptyIndices.length; j++) {
590
- const origIdx = nonEmptyIndices[j];
591
- if (origIdx === undefined)
592
- continue;
593
- results[origIdx] = combined[j] ?? empty();
594
- }
595
- return results;
596
- }
597
- // Partial-failure fallback: any non-empty body whose result is still the
598
- // empty placeholder (either because batchResult was null or the array was
599
- // shorter than expected) gets an individual retry — unless the batch failed
600
- // due to context size, in which case individual calls would also fail.
601
- const fallbackIndices = nonEmptyIndices.filter((_origIdx, j) => {
602
- if (batchContextError)
603
- return false; // skip individual retries on context error
604
- // Result is still empty → needs a fallback call.
605
- if (batchResult === null)
606
- return true;
607
- // batchResult was shorter than the number of non-empty bodies.
608
- return j >= batchResult.length;
609
- });
610
- if (fallbackIndices.length > 0) {
611
- if (batchResult !== null) {
612
- // Only warn on partial failure (not when the whole batch failed, which
613
- // already emitted a warn above).
614
- warn(`graph extraction (batch): response had ${batchResult.length} items for ${nonEmptyBodies.length} assets; ` +
615
- `falling back to individual calls for ${fallbackIndices.length} missing asset(s).`);
616
- }
617
- await Promise.all(fallbackIndices.map(async (origIdx) => {
618
- const body = bodies[origIdx] ?? "";
619
- results[origIdx] = await extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options);
620
- }));
621
- }
622
- else if (batchContextError) {
623
- warn(`graph extraction (batch): skipped ${nonEmptyBodies.length} asset(s) due to context size error; ` +
624
- `consider increasing llm.contextLength or reducing index.graph.graphExtractionBatchSize to 1.`);
625
- }
626
- else if (nonArrayResponse && batchState?.batchingDisabled) {
627
- warn("graph extraction (batch): disabling batching for the rest of this run after repeated non-array responses.");
628
- }
629
- return results;
630
- }
631
- /**
632
- * Extract entities and relations from a single asset body via the configured LLM.
633
- *
634
- * Returns `{entities: [], relations: []}` on any failure (timeout, invalid
635
- * JSON, empty response). Errors are logged via `warn()` but never thrown — a
636
- * failed extraction for one asset must not abort the rest of the index pass.
637
- *
638
- * Routes through `tryLlmFeature("graph_extraction", ...)` so the feature gate
639
- * and onFallback hook are honoured uniformly (Fix C5).
640
- */
641
- export async function extractGraphFromBody(llmRunner, body, signal, akmConfig, onFallback, options = {}) {
642
- const empty = (reason, status) => ({
643
- entities: [],
644
- relations: [],
645
- ...(status ? { status } : {}),
646
- ...(reason ? { reason } : {}),
647
- });
648
- const trimmedBody = body.trim();
649
- if (!trimmedBody)
650
- return empty();
651
- const chunked = splitBodyIntoChunks(trimmedBody, MAX_CHUNK_BODY_CHARS);
652
- if (chunked.truncationCount > 0) {
653
- bumpTelemetry(options.telemetry, "truncationCount", chunked.truncationCount);
654
- warnVerbose(`graph extraction: split a long asset into ${chunked.chunks.length} chunk(s) with ${chunked.truncationCount} hard split(s).`);
655
- }
656
- if (chunked.chunks.length > 1) {
657
- const chunkResults = [];
658
- for (const chunk of chunked.chunks) {
659
- chunkResults.push(await extractGraphFromBody(llmRunner, chunk, signal, akmConfig, onFallback, options));
660
- }
661
- const merged = mergeGraphExtractions(chunkResults);
662
- merged.truncationCount = (merged.truncationCount ?? 0) + chunked.truncationCount;
663
- return merged;
664
- }
665
- const userPrompt = `${USER_PROMPT_PREFIX}${trimmedBody}`;
666
- return callStructured({
667
- feature: "graph_extraction",
668
- akmConfig,
669
- runner: llmRunner,
670
- ...(options.lease ? { lease: options.lease } : {}),
671
- messages: [
672
- { role: "system", content: SYSTEM_PROMPT },
673
- { role: "user", content: userPrompt },
674
- ],
675
- request: {
676
- temperature: 0.1,
677
- timeoutMs: llmRunner.timeoutMs,
678
- signal,
679
- onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
680
- },
681
- onNotices: options.onNotices,
682
- parse: (raw) => {
683
- if (!raw)
684
- return empty();
685
- const parsed = parseEmbeddedJsonResponse(raw);
686
- if (!parsed) {
687
- warn("graph extraction: invalid JSON response from LLM; skipping asset.");
688
- bumpTelemetry(options.telemetry, "failureCount");
689
- return empty("invalid_json", "failed");
690
- }
691
- const extraction = parseGraphExtraction(parsed);
692
- bumpTelemetry(options.telemetry, "filteredGenericEntities", extraction.filteredGenericEntities ?? 0);
693
- bumpTelemetry(options.telemetry, "filteredInvalidRelations", extraction.filteredInvalidRelations ?? 0);
694
- bumpTelemetry(options.telemetry, "filteredLowConfidenceRelations", extraction.filteredLowConfidenceRelations ?? 0);
695
- if (extraction.status === "failed")
696
- bumpTelemetry(options.telemetry, "failureCount");
697
- return extraction;
698
- },
699
- onError: (cls, err) => {
700
- const errMsg = toErrorMessage(err);
701
- if (cls === "context_limit") {
702
- bumpTelemetry(options.telemetry, "failureCount");
703
- warn(`graph extraction: context size exceeded for asset; promptChars=${userPrompt.length}${formatContextHint(llmRunner)}. ` +
704
- `Consider increasing llm.contextLength in config.json.`);
705
- return empty("context_limit", "failed");
706
- }
707
- else if (cls === "html") {
708
- bumpTelemetry(options.telemetry, "htmlErrorCount");
709
- warn(`graph extraction: provider returned HTML instead of JSON for asset; promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
710
- return empty("llm_error", "failed");
711
- }
712
- else {
713
- bumpTelemetry(options.telemetry, "failureCount");
714
- warn(`graph extraction failed for asset; promptChars=${userPrompt.length}${formatContextHint(llmRunner)}: ${errMsg}`);
715
- return empty("llm_error", "failed");
716
- }
717
- },
718
- fallback: empty(),
719
- onFallback,
720
- });
721
- }
722
- // deduplicateGraph lives in src/indexer/graph/graph-dedup.ts (pure utility, no
723
- // LLM calls) — import it from there directly. The re-export that used to live
724
- // here created a value edge back to graph-dedup.ts, which (via its type-only
725
- // import of GraphExtraction/GraphRelation below) formed a 2-file import cycle
726
- // (chunk 9 WI-9.8 KILL 4 sever). No src or test consumer used this re-export
727
- // (graph-extraction.ts and the test suite already import graph-dedup.ts
728
- // directly), so it is deleted outright rather than repointed.