devcouncil 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (365) hide show
  1. package/README.md +46 -30
  2. package/package.json +6 -2
  3. package/packages/codeintel-grammars/hatch_build.py +43 -0
  4. package/packages/codeintel-grammars/pyproject.toml +16 -0
  5. package/packages/codeintel-grammars/src/devcouncil_codeintel_grammars/__init__.py +93 -0
  6. package/pyproject.toml +99 -4
  7. package/src/devcouncil/app/config.py +512 -20
  8. package/src/devcouncil/app/events.py +4 -23
  9. package/src/devcouncil/app/orchestrator.py +5 -0
  10. package/src/devcouncil/app/run_context.py +3 -3
  11. package/src/devcouncil/assets/__init__.py +4 -1
  12. package/src/devcouncil/assets/vendor/force-graph.min.js +5 -0
  13. package/src/devcouncil/campaign/__init__.py +71 -0
  14. package/src/devcouncil/campaign/bloom.py +137 -0
  15. package/src/devcouncil/campaign/dashboard.py +123 -0
  16. package/src/devcouncil/campaign/mailbox.py +305 -0
  17. package/src/devcouncil/campaign/notify.py +91 -0
  18. package/src/devcouncil/campaign/orchestrator.py +592 -0
  19. package/src/devcouncil/campaign/prompts/coordinator.md +29 -0
  20. package/src/devcouncil/campaign/prompts/director.md +21 -0
  21. package/src/devcouncil/campaign/prompts/protocol.md +46 -0
  22. package/src/devcouncil/campaign/prompts/reviewer.md +24 -0
  23. package/src/devcouncil/campaign/prompts/worker.md +24 -0
  24. package/src/devcouncil/campaign/roles.py +202 -0
  25. package/src/devcouncil/campaign/watcher.py +153 -0
  26. package/src/devcouncil/cli/commands/agents.py +24 -17
  27. package/src/devcouncil/cli/commands/artifacts.py +36 -27
  28. package/src/devcouncil/cli/commands/ast.py +12 -3
  29. package/src/devcouncil/cli/commands/baseline.py +21 -12
  30. package/src/devcouncil/cli/commands/boot.py +218 -0
  31. package/src/devcouncil/cli/commands/campaign.py +302 -0
  32. package/src/devcouncil/cli/commands/check.py +225 -12
  33. package/src/devcouncil/cli/commands/config.py +221 -74
  34. package/src/devcouncil/cli/commands/cost.py +137 -28
  35. package/src/devcouncil/cli/commands/dashboard.py +12 -4
  36. package/src/devcouncil/cli/commands/debug_cmd.py +249 -0
  37. package/src/devcouncil/cli/commands/design.py +27 -17
  38. package/src/devcouncil/cli/commands/doctor.py +790 -8
  39. package/src/devcouncil/cli/commands/evidence.py +41 -20
  40. package/src/devcouncil/cli/commands/export.py +73 -0
  41. package/src/devcouncil/cli/commands/gaps.py +175 -0
  42. package/src/devcouncil/cli/commands/gated_write.py +76 -0
  43. package/src/devcouncil/cli/commands/go.py +220 -68
  44. package/src/devcouncil/cli/commands/graph_cmd.py +1192 -0
  45. package/src/devcouncil/cli/commands/handoff.py +45 -34
  46. package/src/devcouncil/cli/commands/hook.py +630 -85
  47. package/src/devcouncil/cli/commands/init.py +89 -30
  48. package/src/devcouncil/cli/commands/integrate.py +296 -1385
  49. package/src/devcouncil/cli/commands/lease.py +120 -0
  50. package/src/devcouncil/cli/commands/logs.py +12 -5
  51. package/src/devcouncil/cli/commands/lsp.py +40 -5
  52. package/src/devcouncil/cli/commands/map.py +317 -74
  53. package/src/devcouncil/cli/commands/mcp_server.py +12 -2
  54. package/src/devcouncil/cli/commands/okf.py +44 -6
  55. package/src/devcouncil/cli/commands/plan.py +184 -69
  56. package/src/devcouncil/cli/commands/prompt.py +26 -17
  57. package/src/devcouncil/cli/commands/provenance.py +79 -0
  58. package/src/devcouncil/cli/commands/repair.py +60 -49
  59. package/src/devcouncil/cli/commands/report.py +148 -40
  60. package/src/devcouncil/cli/commands/requirements.py +104 -0
  61. package/src/devcouncil/cli/commands/reset_demo_state.py +13 -4
  62. package/src/devcouncil/cli/commands/rollback.py +46 -35
  63. package/src/devcouncil/cli/commands/run.py +173 -8
  64. package/src/devcouncil/cli/commands/runs.py +298 -68
  65. package/src/devcouncil/cli/commands/scaffold.py +33 -12
  66. package/src/devcouncil/cli/commands/semantic.py +29 -14
  67. package/src/devcouncil/cli/commands/setup.py +103 -93
  68. package/src/devcouncil/cli/commands/shell.py +51 -42
  69. package/src/devcouncil/cli/commands/show.py +56 -42
  70. package/src/devcouncil/cli/commands/skills.py +29 -20
  71. package/src/devcouncil/cli/commands/status.py +80 -67
  72. package/src/devcouncil/cli/commands/task_gate.py +295 -0
  73. package/src/devcouncil/cli/commands/tasks.py +248 -19
  74. package/src/devcouncil/cli/commands/trace.py +14 -8
  75. package/src/devcouncil/cli/commands/verify.py +33 -8
  76. package/src/devcouncil/cli/commands/version.py +14 -6
  77. package/src/devcouncil/cli/commands/watch.py +49 -30
  78. package/src/devcouncil/cli/commands/watch_fs.py +30 -19
  79. package/src/devcouncil/cli/commands/wiki.py +278 -0
  80. package/src/devcouncil/cli/main.py +58 -1
  81. package/src/devcouncil/codeintel/__init__.py +16 -0
  82. package/src/devcouncil/codeintel/build_control.py +429 -0
  83. package/src/devcouncil/codeintel/build_worker.py +78 -0
  84. package/src/devcouncil/codeintel/debug/__init__.py +17 -0
  85. package/src/devcouncil/codeintel/debug/broker.py +114 -0
  86. package/src/devcouncil/codeintel/debug/broker_client.py +61 -0
  87. package/src/devcouncil/codeintel/debug/consent.py +36 -0
  88. package/src/devcouncil/codeintel/debug/discovery.py +132 -0
  89. package/src/devcouncil/codeintel/debug/fingerprint.py +85 -0
  90. package/src/devcouncil/codeintel/debug/protocol.py +259 -0
  91. package/src/devcouncil/codeintel/debug/python_trace_runner.py +81 -0
  92. package/src/devcouncil/codeintel/debug/session.py +238 -0
  93. package/src/devcouncil/codeintel/debug/tracing.py +201 -0
  94. package/src/devcouncil/codeintel/languages/__init__.py +17 -0
  95. package/src/devcouncil/codeintel/languages/generic_extractor.py +236 -0
  96. package/src/devcouncil/codeintel/languages/registry.py +149 -0
  97. package/src/devcouncil/codeintel/languages/workers.py +245 -0
  98. package/src/devcouncil/codeintel/query/__init__.py +5 -0
  99. package/src/devcouncil/codeintel/query/engine.py +289 -0
  100. package/src/devcouncil/codeintel/resolution/__init__.py +6 -0
  101. package/src/devcouncil/codeintel/resolution/abstract_state.py +301 -0
  102. package/src/devcouncil/codeintel/resolution/frameworks/__init__.py +33 -0
  103. package/src/devcouncil/codeintel/resolution/frameworks/base.py +46 -0
  104. package/src/devcouncil/codeintel/resolution/frameworks/di.py +56 -0
  105. package/src/devcouncil/codeintel/resolution/frameworks/events.py +45 -0
  106. package/src/devcouncil/codeintel/resolution/frameworks/routes.py +88 -0
  107. package/src/devcouncil/codeintel/resolution/semantic.py +887 -0
  108. package/src/devcouncil/codeintel/service.py +104 -0
  109. package/src/devcouncil/codeintel/store/__init__.py +15 -0
  110. package/src/devcouncil/codeintel/store/sqlite.py +1565 -0
  111. package/src/devcouncil/codeintel/sync/__init__.py +19 -0
  112. package/src/devcouncil/codeintel/sync/coordinator.py +430 -0
  113. package/src/devcouncil/codeintel/sync/incremental.py +484 -0
  114. package/src/devcouncil/codeintel/sync/lease.py +96 -0
  115. package/src/devcouncil/codeintel/sync/scope.py +98 -0
  116. package/src/devcouncil/council/__init__.py +4 -0
  117. package/src/devcouncil/council/prompts/__init__.py +4 -0
  118. package/src/devcouncil/domain/checkpoint_refs.py +17 -0
  119. package/src/devcouncil/domain/evidence.py +1 -0
  120. package/src/devcouncil/domain/gap.py +10 -0
  121. package/src/devcouncil/domain/requirement.py +5 -1
  122. package/src/devcouncil/domain/task.py +43 -2
  123. package/src/devcouncil/execution/checkpoints.py +25 -31
  124. package/src/devcouncil/execution/context_builder.py +15 -44
  125. package/src/devcouncil/execution/fs_watcher.py +64 -0
  126. package/src/devcouncil/execution/gated_write.py +203 -0
  127. package/src/devcouncil/execution/handoff.py +2 -1
  128. package/src/devcouncil/execution/hook_policy.py +19 -5
  129. package/src/devcouncil/execution/lease_ops.py +177 -0
  130. package/src/devcouncil/execution/lease_validation.py +71 -0
  131. package/src/devcouncil/execution/patch.py +3 -0
  132. package/src/devcouncil/execution/permissions.py +1 -0
  133. package/src/devcouncil/execution/policy_engine.py +205 -10
  134. package/src/devcouncil/execution/prompt_builder.py +278 -33
  135. package/src/devcouncil/execution/run_trace.py +356 -0
  136. package/src/devcouncil/execution/shell_session.py +46 -5
  137. package/src/devcouncil/execution/stop_gate.py +746 -0
  138. package/src/devcouncil/execution/stop_gate_history.py +113 -0
  139. package/src/devcouncil/execution/stop_gate_state.py +54 -0
  140. package/src/devcouncil/execution/stop_gate_verify_cache.py +69 -0
  141. package/src/devcouncil/execution/task_gate_ops.py +590 -0
  142. package/src/devcouncil/execution/task_runner.py +19 -0
  143. package/src/devcouncil/executors/advisor_tool.py +315 -0
  144. package/src/devcouncil/executors/agent_registry.py +125 -17
  145. package/src/devcouncil/executors/claude_sdk.py +376 -0
  146. package/src/devcouncil/executors/coding_cli.py +724 -25
  147. package/src/devcouncil/executors/mini_swe.py +50 -8
  148. package/src/devcouncil/executors/native/agent.py +224 -19
  149. package/src/devcouncil/executors/openhands.py +50 -8
  150. package/src/devcouncil/executors/transient_retry.py +99 -0
  151. package/src/devcouncil/gating/checks/clean_git.py +5 -2
  152. package/src/devcouncil/gating/checks/planned_files_check.py +38 -11
  153. package/src/devcouncil/gating/checks/secret_scan_check.py +2 -2
  154. package/src/devcouncil/gating/policy.py +46 -2
  155. package/src/devcouncil/indexing/ast_matcher.py +41 -4
  156. package/src/devcouncil/indexing/graph/__init__.py +78 -0
  157. package/src/devcouncil/indexing/graph/api_routes.py +522 -0
  158. package/src/devcouncil/indexing/graph/build.py +862 -0
  159. package/src/devcouncil/indexing/graph/cache.py +329 -0
  160. package/src/devcouncil/indexing/graph/communities.py +28 -0
  161. package/src/devcouncil/indexing/graph/cypher.py +107 -0
  162. package/src/devcouncil/indexing/graph/embeddings.py +194 -0
  163. package/src/devcouncil/indexing/graph/export.py +381 -0
  164. package/src/devcouncil/indexing/graph/export_links.py +81 -0
  165. package/src/devcouncil/indexing/graph/extract_python.py +307 -0
  166. package/src/devcouncil/indexing/graph/extract_ts.py +1205 -0
  167. package/src/devcouncil/indexing/graph/intel.py +668 -0
  168. package/src/devcouncil/indexing/graph/liveness.py +992 -0
  169. package/src/devcouncil/indexing/graph/okf_export.py +65 -0
  170. package/src/devcouncil/indexing/graph/pdg/__init__.py +67 -0
  171. package/src/devcouncil/indexing/graph/pdg/build.py +11 -0
  172. package/src/devcouncil/indexing/graph/pdg/cdg.py +41 -0
  173. package/src/devcouncil/indexing/graph/pdg/cfg.py +199 -0
  174. package/src/devcouncil/indexing/graph/pdg/query.py +21 -0
  175. package/src/devcouncil/indexing/graph/pdg/reaching_def.py +126 -0
  176. package/src/devcouncil/indexing/graph/pdg/schema.py +253 -0
  177. package/src/devcouncil/indexing/graph/pdg/taint.py +154 -0
  178. package/src/devcouncil/indexing/graph/query.py +302 -0
  179. package/src/devcouncil/indexing/graph/resolve.py +1020 -0
  180. package/src/devcouncil/indexing/graph/schema.py +103 -0
  181. package/src/devcouncil/indexing/graph_index.py +20 -29
  182. package/src/devcouncil/indexing/lsp.py +57 -25
  183. package/src/devcouncil/indexing/lsp_client.py +577 -0
  184. package/src/devcouncil/indexing/map_artifacts.py +355 -0
  185. package/src/devcouncil/indexing/map_refresh.py +141 -0
  186. package/src/devcouncil/indexing/repo_mapper.py +1509 -138
  187. package/src/devcouncil/indexing/semantic_index.py +12 -6
  188. package/src/devcouncil/indexing/subsystem_map.py +163 -0
  189. package/src/devcouncil/indexing/ts_imports.py +343 -0
  190. package/src/devcouncil/indexing/viz.py +960 -0
  191. package/src/devcouncil/indexing/walk.py +52 -0
  192. package/src/devcouncil/indexing/wiring.py +1776 -0
  193. package/src/devcouncil/integrations/actions.py +27 -4
  194. package/src/devcouncil/integrations/check.py +211 -16
  195. package/src/devcouncil/integrations/claude_assets.py +209 -12
  196. package/src/devcouncil/integrations/clients/__init__.py +1 -0
  197. package/src/devcouncil/integrations/clients/aider.py +52 -0
  198. package/src/devcouncil/integrations/clients/antigravity.py +87 -0
  199. package/src/devcouncil/integrations/clients/claude.py +339 -0
  200. package/src/devcouncil/integrations/clients/codex.py +39 -0
  201. package/src/devcouncil/integrations/clients/common.py +332 -0
  202. package/src/devcouncil/integrations/clients/cursor.py +164 -0
  203. package/src/devcouncil/integrations/clients/gemini.py +49 -0
  204. package/src/devcouncil/integrations/clients/grok.py +105 -0
  205. package/src/devcouncil/integrations/clients/hooks.py +500 -0
  206. package/src/devcouncil/integrations/clients/opencode.py +96 -0
  207. package/src/devcouncil/integrations/clients/warp.py +75 -0
  208. package/src/devcouncil/integrations/code_review_graph.py +2 -2
  209. package/src/devcouncil/integrations/github.py +73 -7
  210. package/src/devcouncil/integrations/integration_cli.py +197 -0
  211. package/src/devcouncil/integrations/mcp/handlers/__init__.py +1 -0
  212. package/src/devcouncil/integrations/mcp/handlers/ast_lsp.py +77 -0
  213. package/src/devcouncil/integrations/mcp/handlers/checkout.py +50 -0
  214. package/src/devcouncil/integrations/mcp/handlers/cli_gate.py +43 -0
  215. package/src/devcouncil/integrations/mcp/handlers/codeintel.py +182 -0
  216. package/src/devcouncil/integrations/mcp/handlers/debug.py +236 -0
  217. package/src/devcouncil/integrations/mcp/handlers/evidence.py +70 -0
  218. package/src/devcouncil/integrations/mcp/handlers/git.py +99 -0
  219. package/src/devcouncil/integrations/mcp/handlers/graph.py +36 -0
  220. package/src/devcouncil/integrations/mcp/handlers/handoff.py +53 -0
  221. package/src/devcouncil/integrations/mcp/handlers/knowledge.py +30 -0
  222. package/src/devcouncil/integrations/mcp/handlers/lease.py +70 -0
  223. package/src/devcouncil/integrations/mcp/handlers/live.py +108 -0
  224. package/src/devcouncil/integrations/mcp/handlers/map.py +676 -0
  225. package/src/devcouncil/integrations/mcp/handlers/next_task.py +35 -0
  226. package/src/devcouncil/integrations/mcp/handlers/policy.py +80 -0
  227. package/src/devcouncil/integrations/mcp/handlers/prompts.py +168 -0
  228. package/src/devcouncil/integrations/mcp/handlers/provenance.py +101 -0
  229. package/src/devcouncil/integrations/mcp/handlers/read.py +74 -0
  230. package/src/devcouncil/integrations/mcp/handlers/router_cache.py +53 -0
  231. package/src/devcouncil/integrations/mcp/handlers/run.py +53 -0
  232. package/src/devcouncil/integrations/mcp/handlers/runs.py +84 -0
  233. package/src/devcouncil/integrations/mcp/handlers/scope.py +56 -0
  234. package/src/devcouncil/integrations/mcp/handlers/status.py +114 -0
  235. package/src/devcouncil/integrations/mcp/handlers/task.py +100 -0
  236. package/src/devcouncil/integrations/mcp/handlers/tool_specs.py +904 -0
  237. package/src/devcouncil/integrations/mcp/handlers/trace.py +65 -0
  238. package/src/devcouncil/integrations/mcp/handlers/verify.py +45 -0
  239. package/src/devcouncil/integrations/mcp/handlers/wiki.py +60 -0
  240. package/src/devcouncil/integrations/mcp/handlers/write.py +69 -0
  241. package/src/devcouncil/integrations/mcp/server.py +265 -2391
  242. package/src/devcouncil/integrations/mcp/util.py +303 -0
  243. package/src/devcouncil/integrations/setup.py +152 -0
  244. package/src/devcouncil/knowledge/fetch.py +4 -0
  245. package/src/devcouncil/knowledge/knowledge_select.py +38 -0
  246. package/src/devcouncil/knowledge/okf.py +2 -1
  247. package/src/devcouncil/knowledge/resource_discovery.py +40 -0
  248. package/src/devcouncil/knowledge/wiki.py +643 -0
  249. package/src/devcouncil/knowledge/wiki_read.py +87 -0
  250. package/src/devcouncil/live/cards.py +7 -7
  251. package/src/devcouncil/live/models.py +4 -1
  252. package/src/devcouncil/live/reviewer.py +90 -11
  253. package/src/devcouncil/live/signals.py +4 -2
  254. package/src/devcouncil/live/tasks.py +12 -3
  255. package/src/devcouncil/live/transcripts.py +69 -2
  256. package/src/devcouncil/llm/cache.py +5 -6
  257. package/src/devcouncil/llm/model_defaults.yaml +10 -10
  258. package/src/devcouncil/llm/provider.py +647 -73
  259. package/src/devcouncil/llm/router.py +271 -46
  260. package/src/devcouncil/llm/semantic_bridge.py +614 -0
  261. package/src/devcouncil/optimization/gepa_agent.py +6 -4
  262. package/src/devcouncil/optimization/skillopt.py +9 -5
  263. package/src/devcouncil/planning/arbiter_service.py +12 -3
  264. package/src/devcouncil/planning/correction_manifest.py +107 -10
  265. package/src/devcouncil/planning/plan_difficulty.py +69 -0
  266. package/src/devcouncil/planning/plan_service.py +5 -2
  267. package/src/devcouncil/planning/planned_files_reconcile.py +191 -0
  268. package/src/devcouncil/planning/prompt_enhancer_service.py +6 -5
  269. package/src/devcouncil/planning/question_conversion.py +56 -0
  270. package/src/devcouncil/planning/spec_service.py +9 -3
  271. package/src/devcouncil/repo/ci_scaffold.py +197 -1
  272. package/src/devcouncil/repo/gitignore.py +1 -2
  273. package/src/devcouncil/reporting/evidence_export.py +124 -0
  274. package/src/devcouncil/reporting/evidence_html.py +210 -0
  275. package/src/devcouncil/reporting/json_report.py +16 -12
  276. package/src/devcouncil/reporting/markdown_report.py +38 -9
  277. package/src/devcouncil/reporting/mcp_resources.py +142 -0
  278. package/src/devcouncil/reporting/report_builder.py +40 -4
  279. package/src/devcouncil/reporting/task_provenance.py +42 -0
  280. package/src/devcouncil/reporting/verdict.py +75 -0
  281. package/src/devcouncil/skills/library/README.md +1 -0
  282. package/src/devcouncil/skills/library/devcouncil-hero-loop.md +109 -0
  283. package/src/devcouncil/skills/library/devcouncil-verification.md +109 -0
  284. package/src/devcouncil/skills/library/devcouncil.md +93 -0
  285. package/src/devcouncil/skills/registry.py +43 -12
  286. package/src/devcouncil/storage/db.py +57 -11
  287. package/src/devcouncil/storage/models.py +6 -0
  288. package/src/devcouncil/storage/native.py +5 -3
  289. package/src/devcouncil/storage/repositories.py +50 -18
  290. package/src/devcouncil/telemetry/context.py +28 -0
  291. package/src/devcouncil/telemetry/cost.py +4 -5
  292. package/src/devcouncil/telemetry/logging_setup.py +78 -11
  293. package/src/devcouncil/telemetry/model_pricing.yaml +7 -0
  294. package/src/devcouncil/telemetry/stages.py +27 -2
  295. package/src/devcouncil/telemetry/tracker.py +50 -13
  296. package/src/devcouncil/ui/dashboard.py +120 -8
  297. package/src/devcouncil/utils/fsio.py +58 -0
  298. package/src/devcouncil/utils/git_snapshot.py +112 -0
  299. package/src/devcouncil/utils/json_persist.py +53 -0
  300. package/src/devcouncil/utils/proc.py +89 -0
  301. package/src/devcouncil/verification/acceptance_compiler.py +36 -13
  302. package/src/devcouncil/verification/ad_hoc_check.py +95 -3
  303. package/src/devcouncil/verification/checks/__init__.py +41 -0
  304. package/src/devcouncil/verification/checks/acceptance.py +39 -0
  305. package/src/devcouncil/verification/checks/acceptance_corpus.py +194 -0
  306. package/src/devcouncil/verification/checks/acceptance_evidence.py +239 -0
  307. package/src/devcouncil/verification/checks/command_evidence.py +148 -0
  308. package/src/devcouncil/verification/checks/compiled_acceptance.py +179 -0
  309. package/src/devcouncil/verification/checks/corpus_stale.py +124 -0
  310. package/src/devcouncil/verification/checks/corpus_verification.py +9 -0
  311. package/src/devcouncil/verification/checks/dead_symbols.py +360 -0
  312. package/src/devcouncil/verification/checks/diff_coverage_gate.py +101 -0
  313. package/src/devcouncil/verification/checks/doc_code_ref.py +79 -0
  314. package/src/devcouncil/verification/checks/liveness_ratchet.py +336 -0
  315. package/src/devcouncil/verification/checks/orphan_diff.py +104 -0
  316. package/src/devcouncil/verification/checks/planned_files.py +98 -0
  317. package/src/devcouncil/verification/checks/semantic_diff.py +241 -0
  318. package/src/devcouncil/verification/checks/stale_map.py +80 -0
  319. package/src/devcouncil/verification/checks/stub_scan.py +71 -0
  320. package/src/devcouncil/verification/checks/subsystem_boundary.py +103 -0
  321. package/src/devcouncil/verification/checks/wiring.py +216 -0
  322. package/src/devcouncil/verification/claims/__init__.py +23 -0
  323. package/src/devcouncil/verification/claims/checks.py +395 -0
  324. package/src/devcouncil/verification/claims/mapper.py +168 -0
  325. package/src/devcouncil/verification/claims/models.py +39 -0
  326. package/src/devcouncil/verification/claims/transcript.py +92 -0
  327. package/src/devcouncil/verification/claims/verdict.py +88 -0
  328. package/src/devcouncil/verification/command_evidence.py +170 -0
  329. package/src/devcouncil/verification/command_malformation.py +147 -0
  330. package/src/devcouncil/verification/command_runner.py +164 -0
  331. package/src/devcouncil/verification/coverage_measurement.py +292 -0
  332. package/src/devcouncil/verification/diff_coverage.py +151 -0
  333. package/src/devcouncil/verification/difficulty.py +296 -0
  334. package/src/devcouncil/verification/effort_heuristics.py +178 -0
  335. package/src/devcouncil/verification/gap_ids.py +63 -0
  336. package/src/devcouncil/verification/gate_cache.py +194 -0
  337. package/src/devcouncil/verification/gate_selector.py +344 -0
  338. package/src/devcouncil/verification/git_diff_fallback.py +272 -0
  339. package/src/devcouncil/verification/implementation_reviewer.py +13 -0
  340. package/src/devcouncil/verification/incremental_check.py +241 -0
  341. package/src/devcouncil/verification/next_actions.py +60 -1
  342. package/src/devcouncil/verification/rigor_analytics.py +130 -0
  343. package/src/devcouncil/verification/sandbox.py +38 -11
  344. package/src/devcouncil/verification/stub_detector.py +369 -0
  345. package/src/devcouncil/verification/test_resolver.py +67 -1
  346. package/src/devcouncil/verification/verifier.py +137 -1666
  347. package/src/devcouncil/verification/verify_orchestration.py +610 -0
  348. package/src/devcouncil/verification/verify_setup.py +176 -0
  349. package/src/devcouncil/verification/wiki_refresh.py +208 -0
  350. package/src/semantic_layer/__init__.py +58 -0
  351. package/src/semantic_layer/benchmark.py +75 -0
  352. package/src/semantic_layer/cache.py +290 -0
  353. package/src/semantic_layer/compressor.py +137 -0
  354. package/src/semantic_layer/config.py +75 -0
  355. package/src/semantic_layer/embeddings.py +69 -0
  356. package/src/semantic_layer/llm_backends.py +99 -0
  357. package/src/semantic_layer/pipeline.py +111 -0
  358. package/src/semantic_layer/router.py +128 -0
  359. package/src/semantic_layer/tuner.py +72 -0
  360. package/uv.lock +973 -9
  361. package/src/devcouncil/artifacts/migrations.py +0 -20
  362. package/src/devcouncil/artifacts/schemas.py +0 -23
  363. package/src/devcouncil/artifacts/serializer.py +0 -21
  364. package/src/devcouncil/integrations/gitnexus.py +0 -70
  365. package/src/devcouncil/integrations/graphify.py +0 -34
@@ -270,6 +270,157 @@ def measurable_python_changes(changed: Dict[str, Dict[int, str]]) -> Dict[str, D
270
270
  }
271
271
 
272
272
 
273
+ _JS_SUFFIXES = (".js", ".ts", ".jsx", ".tsx", ".mjs", ".cjs")
274
+ _GO_SUFFIX = ".go"
275
+
276
+
277
+ def measurable_js_changes(changed: Dict[str, Dict[int, str]]) -> Dict[str, Dict[int, str]]:
278
+ """Filter to JS/TS source files c8/nyc can instrument."""
279
+ return {
280
+ path: lines
281
+ for path, lines in changed.items()
282
+ if path.endswith(_JS_SUFFIXES) and not is_test_path(path)
283
+ }
284
+
285
+
286
+ def measurable_go_changes(changed: Dict[str, Dict[int, str]]) -> Dict[str, Dict[int, str]]:
287
+ """Filter to Go source files ``go test -coverprofile`` can measure."""
288
+ return {
289
+ path: lines
290
+ for path, lines in changed.items()
291
+ if path.endswith(_GO_SUFFIX) and not is_test_path(path)
292
+ }
293
+
294
+
295
+ def parse_istanbul_json(data: dict, root: Path) -> CoverageData:
296
+ """Parse c8/nyc ``coverage-final.json`` into :class:`CoverageData`."""
297
+ executed: Dict[str, Set[int]] = {}
298
+ executable: Dict[str, Set[int]] = {}
299
+ if not isinstance(data, dict):
300
+ return CoverageData()
301
+ for raw_path, payload in data.items():
302
+ if not isinstance(payload, dict):
303
+ continue
304
+ rel = _relativize(raw_path, root) or _relativize(payload.get("path", raw_path), root)
305
+ if rel is None:
306
+ continue
307
+ stmt_map = payload.get("statementMap") or {}
308
+ hits = payload.get("s") or {}
309
+ run: Set[int] = set()
310
+ all_lines: Set[int] = set()
311
+ for sid, meta in stmt_map.items():
312
+ if not isinstance(meta, dict):
313
+ continue
314
+ start = meta.get("start") or {}
315
+ line = int(start.get("line", 0) or 0)
316
+ if line <= 0:
317
+ continue
318
+ all_lines.add(line)
319
+ if int(hits.get(sid, 0) or 0) > 0:
320
+ run.add(line)
321
+ if all_lines:
322
+ executed[rel] = run
323
+ executable[rel] = all_lines
324
+ return CoverageData(executed=executed, executable=executable)
325
+
326
+
327
+ def parse_go_coverprofile(text: str, root: Path) -> CoverageData:
328
+ """Parse ``go tool cover`` profile text into :class:`CoverageData`."""
329
+ executed: Dict[str, Set[int]] = {}
330
+ executable: Dict[str, Set[int]] = {}
331
+ for raw in text.splitlines():
332
+ if not raw or raw.startswith("mode:"):
333
+ continue
334
+ parts = raw.split()
335
+ if len(parts) < 3:
336
+ continue
337
+ loc, count_str = parts[0], parts[2]
338
+ try:
339
+ hit = int(count_str)
340
+ except ValueError:
341
+ continue
342
+ file_part = loc.split(":")[0]
343
+ rel = _relativize(file_part, root)
344
+ if rel is None:
345
+ # go profiles often use module paths; try basename match later via intersect.
346
+ rel = file_part.replace("\\", "/")
347
+ span = loc.split(":")[-1] if ":" in loc else ""
348
+ if "," not in span:
349
+ continue
350
+ start = span.split(",")[0]
351
+ try:
352
+ line = int(start.split(".")[0])
353
+ except ValueError:
354
+ continue
355
+ executable.setdefault(rel, set()).add(line)
356
+ if hit > 0:
357
+ executed.setdefault(rel, set()).add(line)
358
+ return CoverageData(executed=executed, executable=executable)
359
+
360
+
361
+ def merge_diff_coverage_results(results: List[DiffCoverageResult]) -> DiffCoverageResult:
362
+ """Combine per-language measurements into one result."""
363
+ measured = [r for r in results if r.measured]
364
+ if not measured:
365
+ reason = results[0].reason if results else "no measurable source changes in diff"
366
+ return DiffCoverageResult(measured=False, reason=reason)
367
+ total_exec = sum(r.changed_executable_lines for r in measured)
368
+ total_cov = sum(r.covered_changed_lines for r in measured)
369
+ uncovered: Dict[str, List[int]] = {}
370
+ absent: List[str] = []
371
+ tools: List[str] = []
372
+ for r in measured:
373
+ if r.tool:
374
+ tools.append(r.tool)
375
+ for path, lines in r.uncovered_by_file.items():
376
+ uncovered.setdefault(path, []).extend(lines)
377
+ absent.extend(r.absent_files)
378
+ for path in uncovered:
379
+ uncovered[path] = sorted(set(uncovered[path]))
380
+ return DiffCoverageResult(
381
+ measured=True,
382
+ tool="+".join(dict.fromkeys(tools)) or "coverage",
383
+ changed_executable_lines=total_exec,
384
+ covered_changed_lines=total_cov,
385
+ uncovered_by_file=uncovered,
386
+ absent_files=sorted(set(absent)),
387
+ )
388
+
389
+
390
+ def c8_run_argv(command_argv: List[str], *, reports_dir: str) -> Optional[List[str]]:
391
+ """Wrap a Node/npm test command with c8 instrumentation."""
392
+ if not command_argv:
393
+ return None
394
+ head = Path(command_argv[0]).name.lower()
395
+ if head.endswith(".exe"):
396
+ head = head[:-4]
397
+ # npx c8 ... OR node/node_modules/.bin/c8
398
+ if head in {"npx", "pnpm", "yarn"}:
399
+ return [command_argv[0], "c8", "--reporter=json", f"--reports-dir={reports_dir}", *command_argv[1:]]
400
+ if head in {"npm", "node"}:
401
+ return ["npx", "c8", "--reporter=json", f"--reports-dir={reports_dir}", *command_argv]
402
+ return ["npx", "c8", "--reporter=json", f"--reports-dir={reports_dir}", *command_argv]
403
+
404
+
405
+ def go_cover_run_argv(command_argv: List[str], profile_path: str) -> Optional[List[str]]:
406
+ """Inject ``-coverprofile`` into a ``go test`` invocation."""
407
+ if not command_argv:
408
+ return None
409
+ head = Path(command_argv[0]).name.lower()
410
+ if head.endswith(".exe"):
411
+ head = head[:-4]
412
+ if head != "go":
413
+ return None
414
+ args = list(command_argv)
415
+ if "test" not in args:
416
+ return None
417
+ if "-coverprofile" in args:
418
+ return args
419
+ # Insert after ``go test``.
420
+ idx = args.index("test") + 1
421
+ return [*args[:idx], f"-coverprofile={profile_path}", *args[idx:]]
422
+
423
+
273
424
  def coverage_run_argv(
274
425
  command_argv: List[str],
275
426
  python: str,
@@ -0,0 +1,296 @@
1
+ """Deterministic task-difficulty estimation and the rigor policy derived from it.
2
+
3
+ DevCouncil's anti-laziness gates (stub detection, effort heuristics, coverage
4
+ enforcement) are *advisory everywhere, blocking on hard tasks* by default. This
5
+ module supplies the two pieces that policy needs:
6
+
7
+ - :func:`estimate_difficulty` — a cheap, deterministic (no LLM) classifier of a
8
+ task as ``easy`` / ``normal`` / ``hard`` from its declared scope. A manual
9
+ ``Task.difficulty`` value always wins, so planners and humans can override.
10
+ - :func:`resolve_rigor_policy` — folds the difficulty together with
11
+ ``verification.rigor`` config into a :class:`RigorPolicy` the verifier, prompt
12
+ builder, and repair loop can branch on without re-deriving anything.
13
+
14
+ Everything here must stay side-effect free and never raise: rigor is a layer on
15
+ top of verification, and a bug in it must degrade to "no extra enforcement",
16
+ never to a crashed verify run.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import logging
22
+ import re
23
+ from dataclasses import dataclass, field
24
+ from typing import List, Literal, Optional, cast
25
+
26
+ from devcouncil.domain.requirement import Requirement
27
+ from devcouncil.domain.task import Task
28
+
29
+ logger = logging.getLogger(__name__)
30
+
31
+ Difficulty = Literal["easy", "normal", "hard"]
32
+
33
+ # Keywords whose presence in the task/requirement text signals intrinsically hard
34
+ # work (cross-cutting, stateful, or correctness-critical). Word-ish boundaries so
35
+ # "author" does not match "auth". The whole keyword bucket contributes at most
36
+ # _KEYWORD_CAP points — a wordy description must not outweigh structural signals.
37
+ _HARD_KEYWORD_RE = re.compile(
38
+ r"\b(refactor\w*|migrat\w*|concurren\w*|async\w*|race|deadlock|protocol|parser|"
39
+ r"cache|caching|transaction\w*|auth(?:n|z|entication|orization)?|crypto\w*|"
40
+ r"distributed|backward[- ]compat\w*|thread\w*|lock(?:ing|s)?|schema)\b",
41
+ re.IGNORECASE,
42
+ )
43
+ _KEYWORD_CAP = 1
44
+
45
+ _HARD_THRESHOLD = 4
46
+ _NORMAL_THRESHOLD = 2
47
+
48
+
49
+ def _linked_requirements(task: Task, requirements: Optional[List[Requirement]]) -> List[Requirement]:
50
+ if not requirements:
51
+ return []
52
+ wanted = set(task.requirement_ids)
53
+ return [r for r in requirements if r.id in wanted]
54
+
55
+
56
+ def difficulty_score(task: Task, requirements: Optional[List[Requirement]] = None) -> int:
57
+ """The raw additive score behind :func:`estimate_difficulty` (exposed for tests
58
+ and for surfacing "why was this hard" in logs)."""
59
+ score = 0
60
+ writable = [pf for pf in task.planned_files if pf.allowed_change != "read_only"]
61
+ if len(writable) >= 5:
62
+ score += 2
63
+ elif len(writable) >= 3:
64
+ score += 1
65
+
66
+ ac_count = len(task.acceptance_criterion_ids)
67
+ if ac_count >= 5:
68
+ score += 2
69
+ elif ac_count >= 3:
70
+ score += 1
71
+
72
+ changes = {pf.allowed_change for pf in writable}
73
+ if "create" in changes and "modify" in changes:
74
+ score += 1
75
+
76
+ if len(task.depends_on) >= 2:
77
+ score += 1
78
+
79
+ linked = _linked_requirements(task, requirements)
80
+ text = " ".join(
81
+ [task.title, task.description]
82
+ + [r.title for r in linked]
83
+ + [r.description for r in linked]
84
+ )
85
+ if _HARD_KEYWORD_RE.search(text):
86
+ score += _KEYWORD_CAP
87
+
88
+ if any(r.priority in ("high", "critical") for r in linked):
89
+ score += 1
90
+
91
+ return score
92
+
93
+
94
+ def estimate_difficulty(task: Task, requirements: Optional[List[Requirement]] = None) -> Difficulty:
95
+ """Classify a task's difficulty deterministically.
96
+
97
+ A manual ``Task.difficulty`` (set by a planner or a human) always wins. The
98
+ estimator never raises; any unexpected error degrades to ``"normal"``.
99
+ """
100
+ manual = getattr(task, "difficulty", None)
101
+ if manual in ("easy", "normal", "hard"):
102
+ return cast(Literal["easy", "normal", "hard"], manual)
103
+ try:
104
+ score = difficulty_score(task, requirements)
105
+ except Exception: # pragma: no cover - defensive
106
+ logger.debug("difficulty_score failed for %s; defaulting to normal", getattr(task, "id", "?"), exc_info=True)
107
+ return "normal"
108
+ if score >= _HARD_THRESHOLD:
109
+ return "hard"
110
+ if score >= _NORMAL_THRESHOLD:
111
+ return "normal"
112
+ return "easy"
113
+
114
+
115
+ @dataclass
116
+ class RigorPolicy:
117
+ """Resolved enforcement decisions for one task's verification run.
118
+
119
+ ``*_enabled`` says whether a gate runs at all; ``*_blocking`` says whether its
120
+ findings block. ``applied`` lists the escalations that actually took effect so
121
+ the verification outcome can record "passed under strict gates" vs "passed".
122
+ """
123
+
124
+ difficulty: Difficulty = "normal"
125
+ stub_enabled: bool = True
126
+ stub_blocking: bool = False
127
+ effort_enabled: bool = True
128
+ effort_blocking: bool = False
129
+ coarse_acceptance_enabled: bool = True
130
+ coarse_acceptance_blocking: bool = False
131
+ unwired_enabled: bool = True
132
+ unwired_blocking: bool = False
133
+ dead_symbol_enabled: bool = True
134
+ dead_symbol_blocking: bool = False
135
+ liveness_ratchet_enabled: bool = True
136
+ liveness_ratchet_blocking: bool = False
137
+ stale_map_enabled: bool = True
138
+ stale_map_blocking: bool = False
139
+ corpus_stale_enabled: bool = True
140
+ corpus_stale_blocking: bool = False
141
+ doc_code_ref_enabled: bool = True
142
+ doc_code_ref_blocking: bool = False
143
+ acceptance_corpus_enabled: bool = True
144
+ acceptance_corpus_blocking: bool = False
145
+ enforce_coverage: bool = False
146
+ reviewer_required: bool = False
147
+ extra_repair_attempts: int = 0
148
+ min_added_lines_per_planned_file: int = 5
149
+ min_acceptance_samples: int = 0
150
+ applied: List[str] = field(default_factory=list)
151
+
152
+
153
+ def _mode_flags(mode: str, is_hard: bool) -> tuple[bool, bool]:
154
+ """Map a config mode (``never``/``hard``/``always``) to (enabled, blocking)."""
155
+ normalized = (mode or "hard").strip().lower()
156
+ if normalized == "never":
157
+ return False, False
158
+ if normalized == "always":
159
+ return True, True
160
+ # "hard" (and anything unrecognized, defensively) -> run always, block on hard.
161
+ return True, is_hard
162
+
163
+
164
+ def _soft_mode_flags(mode: str, is_hard: bool) -> tuple[bool, bool]:
165
+ """Map ``never`` / ``soft`` / ``hard`` / ``always`` to (enabled, blocking)."""
166
+ normalized = (mode or "soft").strip().lower()
167
+ if normalized == "never":
168
+ return False, False
169
+ if normalized == "always":
170
+ return True, True
171
+ if normalized == "hard":
172
+ return True, is_hard
173
+ # soft: run on all difficulties, block only on hard tasks
174
+ return True, is_hard
175
+
176
+
177
+ def resolve_rigor_policy(
178
+ task: Task,
179
+ requirements: Optional[List[Requirement]] = None,
180
+ config=None,
181
+ ) -> RigorPolicy:
182
+ """Fold difficulty + ``verification.rigor`` config into a :class:`RigorPolicy`.
183
+
184
+ ``config`` is a loaded ``DevCouncilConfig`` or None (defaults apply). Never
185
+ raises; on any error it returns a policy with no blocking escalations.
186
+ """
187
+ try:
188
+ rigor_cfg = config.verification.rigor if config is not None else None
189
+ except Exception:
190
+ rigor_cfg = None
191
+
192
+ difficulty = estimate_difficulty(task, requirements)
193
+ policy = RigorPolicy(difficulty=difficulty)
194
+
195
+ try:
196
+ enabled = True if rigor_cfg is None else bool(rigor_cfg.enabled)
197
+ if not enabled:
198
+ policy.stub_enabled = False
199
+ policy.effort_enabled = False
200
+ policy.unwired_enabled = False
201
+ policy.dead_symbol_enabled = False
202
+ policy.liveness_ratchet_enabled = False
203
+ policy.stale_map_enabled = False
204
+ policy.corpus_stale_enabled = False
205
+ policy.doc_code_ref_enabled = False
206
+ policy.acceptance_corpus_enabled = False
207
+ return policy
208
+
209
+ is_hard = difficulty == "hard"
210
+ stub_mode = "hard" if rigor_cfg is None else getattr(rigor_cfg, "stub_detection", "hard")
211
+ effort_mode = "hard" if rigor_cfg is None else getattr(rigor_cfg, "effort_heuristics", "hard")
212
+ policy.stub_enabled, policy.stub_blocking = _mode_flags(stub_mode, is_hard)
213
+ policy.effort_enabled, policy.effort_blocking = _mode_flags(effort_mode, is_hard)
214
+ coarse_mode = "hard" if rigor_cfg is None else getattr(
215
+ rigor_cfg, "coarse_acceptance_proof", "hard"
216
+ )
217
+ policy.coarse_acceptance_enabled, policy.coarse_acceptance_blocking = _mode_flags(
218
+ coarse_mode, is_hard
219
+ )
220
+ unwired_mode = "hard" if rigor_cfg is None else getattr(rigor_cfg, "unwired_files", "hard")
221
+ dead_mode = "hard" if rigor_cfg is None else getattr(rigor_cfg, "dead_symbols", "hard")
222
+ ratchet_mode = "hard" if rigor_cfg is None else getattr(
223
+ rigor_cfg, "liveness_ratchet", "hard"
224
+ )
225
+ stale_mode = "hard" if rigor_cfg is None else getattr(rigor_cfg, "stale_map", "hard")
226
+ policy.unwired_enabled, policy.unwired_blocking = _mode_flags(unwired_mode, is_hard)
227
+ policy.dead_symbol_enabled, policy.dead_symbol_blocking = _mode_flags(dead_mode, is_hard)
228
+ policy.liveness_ratchet_enabled, policy.liveness_ratchet_blocking = _mode_flags(
229
+ ratchet_mode, is_hard
230
+ )
231
+ policy.stale_map_enabled, policy.stale_map_blocking = _mode_flags(stale_mode, is_hard)
232
+ corpus_mode = "soft" if rigor_cfg is None else getattr(rigor_cfg, "corpus_stale", "soft")
233
+ doc_ref_mode = "soft" if rigor_cfg is None else getattr(rigor_cfg, "doc_code_ref", "soft")
234
+ policy.corpus_stale_enabled, policy.corpus_stale_blocking = _soft_mode_flags(
235
+ corpus_mode, is_hard
236
+ )
237
+ policy.doc_code_ref_enabled, policy.doc_code_ref_blocking = _soft_mode_flags(
238
+ doc_ref_mode, is_hard
239
+ )
240
+ ac_corpus_mode = "soft" if rigor_cfg is None else getattr(
241
+ rigor_cfg, "acceptance_corpus", "soft"
242
+ )
243
+ policy.acceptance_corpus_enabled, policy.acceptance_corpus_blocking = _soft_mode_flags(
244
+ ac_corpus_mode, is_hard
245
+ )
246
+
247
+ enforce_cov_on_hard = True if rigor_cfg is None else bool(
248
+ getattr(rigor_cfg, "enforce_coverage_on_hard", True)
249
+ )
250
+ policy.enforce_coverage = is_hard and enforce_cov_on_hard
251
+
252
+ reviewer_on_hard = False if rigor_cfg is None else bool(
253
+ getattr(rigor_cfg, "reviewer_required_on_hard", False)
254
+ )
255
+ policy.reviewer_required = is_hard and reviewer_on_hard
256
+
257
+ extra = 1 if rigor_cfg is None else max(
258
+ 0, int(getattr(rigor_cfg, "extra_repair_attempts_on_hard", 1))
259
+ )
260
+ policy.extra_repair_attempts = extra if is_hard else 0
261
+
262
+ policy.min_added_lines_per_planned_file = max(
263
+ 1, int(getattr(rigor_cfg, "min_added_lines_per_planned_file", 5))
264
+ )
265
+
266
+ samples_on_hard = 2 if rigor_cfg is None else max(
267
+ 1, int(getattr(rigor_cfg, "acceptance_samples_on_hard", 2))
268
+ )
269
+ policy.min_acceptance_samples = samples_on_hard if is_hard else 0
270
+
271
+ if policy.stub_blocking:
272
+ policy.applied.append("stub_detection_blocking")
273
+ if policy.effort_blocking:
274
+ policy.applied.append("effort_heuristics_blocking")
275
+ if policy.coarse_acceptance_blocking:
276
+ policy.applied.append("coarse_acceptance_proof_blocking")
277
+ if policy.unwired_blocking:
278
+ policy.applied.append("unwired_files_blocking")
279
+ if policy.dead_symbol_blocking:
280
+ policy.applied.append("dead_symbols_blocking")
281
+ if policy.liveness_ratchet_blocking:
282
+ policy.applied.append("liveness_ratchet_blocking")
283
+ if policy.stale_map_blocking:
284
+ policy.applied.append("stale_map_blocking")
285
+ if policy.enforce_coverage:
286
+ policy.applied.append("coverage_enforced")
287
+ if policy.reviewer_required:
288
+ policy.applied.append("reviewer_required")
289
+ if policy.extra_repair_attempts:
290
+ policy.applied.append(f"extra_repair_attempts:{policy.extra_repair_attempts}")
291
+ if policy.min_acceptance_samples > 1:
292
+ policy.applied.append(f"acceptance_samples:{policy.min_acceptance_samples}")
293
+ except Exception: # pragma: no cover - defensive
294
+ logger.debug("resolve_rigor_policy failed; degrading to advisory-only", exc_info=True)
295
+ return RigorPolicy(difficulty=difficulty)
296
+ return policy
@@ -0,0 +1,178 @@
1
+ """Effort/diff plausibility heuristics — is the diff big enough to be the work?
2
+
3
+ Three deliberately conservative checks, each aimed at a known lazy-agent pattern:
4
+
5
+ - **undersized diff**: the task plans substantial work (several writable files, or
6
+ file creation) but the diff adds only a handful of code lines.
7
+ - **comment-only diff**: files changed, but no added line is actual code, while
8
+ the task has automatable acceptance criteria to prove.
9
+ - **test deletion**: more test lines removed than added — the "make the suite
10
+ pass by deleting the test" move. Always high severity.
11
+
12
+ These are heuristics, so outside hard tasks they surface as advisory
13
+ ``suspicious_effort`` gaps; the caller decides blocking via the rigor policy.
14
+ Never raises.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import logging
20
+ import re
21
+ from dataclasses import dataclass
22
+ from pathlib import Path
23
+ from typing import List, Optional
24
+
25
+ from devcouncil.domain.requirement import Requirement
26
+ from devcouncil.domain.task import Task
27
+ from devcouncil.verification.stub_detector import added_lines_by_file
28
+
29
+ logger = logging.getLogger(__name__)
30
+
31
+ _AUTOMATABLE_METHODS = {"unit_test", "integration_test", "static_check"}
32
+
33
+ _COMMENT_PREFIXES = ("#", "//", "/*", "*", "--", ";", "<!--")
34
+
35
+
36
+ @dataclass
37
+ class EffortFinding:
38
+ reason: str
39
+ detail: str
40
+ severity: str = "medium"
41
+ file: Optional[str] = None
42
+
43
+
44
+ def _is_code_line(text: str) -> bool:
45
+ stripped = text.strip()
46
+ if not stripped:
47
+ return False
48
+ return not stripped.startswith(_COMMENT_PREFIXES)
49
+
50
+
51
+ def _is_test_path(path: str) -> bool:
52
+ lowered = path.lower()
53
+ name = lowered.rsplit("/", 1)[-1]
54
+ return (
55
+ "/tests/" in f"/{lowered}"
56
+ or lowered.startswith("tests/")
57
+ or name.startswith("test_")
58
+ or re.search(r"(_test|\.test|\.spec)\.[a-z]+$", name) is not None
59
+ )
60
+
61
+
62
+ def _removed_lines_by_file(diff_content: str) -> dict:
63
+ """``{old_path: removed_line_count}`` from a unified diff. Tolerant, never raises."""
64
+ out: dict = {}
65
+ current: Optional[str] = None
66
+ for raw in diff_content.splitlines():
67
+ if raw.startswith("--- "):
68
+ target = raw[4:].strip()
69
+ if target == "/dev/null":
70
+ current = None
71
+ else:
72
+ current = target[2:] if target.startswith(("a/", "b/")) else target
73
+ continue
74
+ if current is None:
75
+ continue
76
+ if raw.startswith("-") and not raw.startswith("---"):
77
+ out[current] = out.get(current, 0) + 1
78
+ return out
79
+
80
+
81
+ def _automatable_criteria_present(task: Task, requirements: Optional[List[Requirement]]) -> bool:
82
+ if not task.acceptance_criterion_ids:
83
+ return False
84
+ if not requirements:
85
+ return True # criteria exist; assume provable absent contrary evidence
86
+ wanted = set(task.acceptance_criterion_ids)
87
+ for req in requirements:
88
+ for ac in req.acceptance_criteria:
89
+ if ac.id in wanted and ac.verification_method in _AUTOMATABLE_METHODS:
90
+ return True
91
+ return False
92
+
93
+
94
+ def detect_effort_anomalies(
95
+ task: Task,
96
+ diff_content: str,
97
+ requirements: Optional[List[Requirement]] = None,
98
+ *,
99
+ min_added_lines_per_planned_file: int = 5,
100
+ ) -> List[EffortFinding]:
101
+ """Run all effort heuristics over the task's diff. Never raises."""
102
+ try:
103
+ added = added_lines_by_file(diff_content)
104
+ except Exception: # pragma: no cover - defensive
105
+ logger.debug("effort heuristics: diff parse failed", exc_info=True)
106
+ return []
107
+ if not added:
108
+ return [] # empty diff is the verifier's task_not_implemented gate, not ours
109
+
110
+ findings: List[EffortFinding] = []
111
+ writable = [pf for pf in task.planned_files if pf.allowed_change != "read_only"]
112
+ added_code_lines = sum(
113
+ 1 for lines in added.values() for _, text in lines if _is_code_line(text)
114
+ )
115
+ # Declared-stub lines (devcouncil: allow-stub) do not count toward the
116
+ # undersized-diff threshold — scaffolding must not absorb the whole diff.
117
+ declared_stub_lines = sum(
118
+ 1 for lines in added.values() for _, text in lines if "devcouncil: allow-stub" in text
119
+ )
120
+ effective_code_lines = max(0, added_code_lines - declared_stub_lines)
121
+
122
+ # 1. Undersized diff vs declared scope.
123
+ substantial_scope = len(writable) >= 3 or any(pf.allowed_change == "create" for pf in writable)
124
+ if substantial_scope and writable:
125
+ threshold = min_added_lines_per_planned_file * len(writable)
126
+ if effective_code_lines < threshold:
127
+ findings.append(EffortFinding(
128
+ reason="undersized_diff",
129
+ detail=(
130
+ f"Task plans {len(writable)} writable file(s) but the diff adds only "
131
+ f"{effective_code_lines} substantive code line(s) (< {threshold} expected; "
132
+ f"{declared_stub_lines} declared-stub line(s) excluded). The "
133
+ "implementation may be superficial or incomplete."
134
+ ),
135
+ ))
136
+
137
+ # 2. Comment/whitespace-only diff while automatable criteria exist.
138
+ if added_code_lines == 0 and _automatable_criteria_present(task, requirements):
139
+ findings.append(EffortFinding(
140
+ reason="comment_only_diff",
141
+ detail=(
142
+ "The diff adds no executable code (comments/blank lines only), but the "
143
+ "task has acceptance criteria that require behavioral changes."
144
+ ),
145
+ ))
146
+
147
+ # 3. Test deletion: more test lines removed than added in files referenced by
148
+ # expected_tests — the classic "make the suite pass by deleting the test".
149
+ removed = _removed_lines_by_file(diff_content)
150
+ expected = " ".join(task.expected_tests or [])
151
+ referenced_test_paths = {
152
+ path for path in removed if _is_test_path(path) and path in expected
153
+ }
154
+ if not referenced_test_paths and expected:
155
+ # Fall back: basename match when commands use paths like tests/test_a.py.
156
+ for path in removed:
157
+ if _is_test_path(path) and Path(path).name in expected:
158
+ referenced_test_paths.add(path)
159
+ test_removed = sum(removed[p] for p in referenced_test_paths)
160
+ test_added = sum(
161
+ len(lines) for path, lines in added.items()
162
+ if _is_test_path(path) and path in referenced_test_paths
163
+ )
164
+ if test_removed > 0 and test_removed > test_added:
165
+ worst = max(referenced_test_paths, key=lambda p: removed[p])
166
+ findings.append(EffortFinding(
167
+ reason="test_deletion",
168
+ detail=(
169
+ f"The diff removes {test_removed} line(s) from test file(s) referenced by "
170
+ f"expected_tests but adds only {test_added}. Weakening or deleting tests to "
171
+ "make verification pass is never acceptable; restore the tests or justify "
172
+ "the removal explicitly."
173
+ ),
174
+ severity="high",
175
+ file=worst,
176
+ ))
177
+
178
+ return findings
@@ -0,0 +1,63 @@
1
+ """Deterministic gap IDs and stable ordering for verification runs."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import re
7
+ from typing import Iterable, List
8
+
9
+ from devcouncil.domain.gap import Gap
10
+
11
+ _SEVERITY_ORDER = {"critical": 0, "high": 1, "medium": 2, "low": 3}
12
+
13
+
14
+ def stable_gap_id(task_id: str, kind: str, identity: str = "") -> str:
15
+ """Return a gap id that is stable across verify runs for the same finding."""
16
+ key = f"{task_id}|{kind}|{identity or kind}"
17
+ digest = hashlib.sha256(key.encode("utf-8")).hexdigest()[:10]
18
+ safe_kind = re.sub(r"[^A-Za-z0-9]", "", kind)[:16] or "GAP"
19
+ return f"GAP-{task_id}-{safe_kind}-{digest}"
20
+
21
+
22
+ def gap_identity(gap: Gap) -> str:
23
+ """Stable dedup key for a gap (type + location + criterion + description)."""
24
+ return "|".join(
25
+ (
26
+ gap.gap_type,
27
+ gap.file or "",
28
+ str(gap.line or ""),
29
+ gap.acceptance_criterion_id or "",
30
+ gap.description.strip(),
31
+ )
32
+ )
33
+
34
+
35
+ def normalize_verify_gaps(gaps: Iterable[Gap]) -> List[Gap]:
36
+ """Dedupe and sort gaps so persistence and reconnecting agents stay stable."""
37
+ seen: dict[str, Gap] = {}
38
+ for gap in gaps:
39
+ key = gap_identity(gap)
40
+ existing = seen.get(key)
41
+ if existing is None:
42
+ seen[key] = gap
43
+ continue
44
+ # Prefer blocking / higher severity when duplicates collide on identity.
45
+ if gap.blocking and not existing.blocking:
46
+ seen[key] = gap
47
+ elif gap.blocking == existing.blocking:
48
+ sev_new = _SEVERITY_ORDER.get(gap.severity, 9)
49
+ sev_old = _SEVERITY_ORDER.get(existing.severity, 9)
50
+ if sev_new < sev_old:
51
+ seen[key] = gap
52
+ return sorted(
53
+ seen.values(),
54
+ key=lambda g: (
55
+ not g.blocking,
56
+ _SEVERITY_ORDER.get(g.severity, 9),
57
+ g.gap_type,
58
+ g.file or "",
59
+ g.line or 0,
60
+ g.acceptance_criterion_id or "",
61
+ g.id,
62
+ ),
63
+ )