loopx 0.4.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (811) hide show
  1. loopx/__init__.py +5 -0
  2. loopx/agent_onboarding.py +654 -0
  3. loopx/agent_registry.py +112 -0
  4. loopx/ark_managed_agent_host.py +59 -0
  5. loopx/authority.py +805 -0
  6. loopx/benchmark.py +2875 -0
  7. loopx/benchmark_adapters/__init__.py +1 -0
  8. loopx/benchmark_adapters/agentissue.py +2644 -0
  9. loopx/benchmark_adapters/agents_last_exam.py +3998 -0
  10. loopx/benchmark_adapters/edgebench.py +322 -0
  11. loopx/benchmark_adapters/skillsbench.py +5978 -0
  12. loopx/benchmark_adapters/skillsbench_acp_failure_policy.py +143 -0
  13. loopx/benchmark_adapters/skillsbench_acp_process.py +31 -0
  14. loopx/benchmark_adapters/skillsbench_acp_relay.py +4832 -0
  15. loopx/benchmark_adapters/skillsbench_batch.py +124 -0
  16. loopx/benchmark_adapters/skillsbench_bridge_guard.py +209 -0
  17. loopx/benchmark_adapters/skillsbench_bridge_summary.py +203 -0
  18. loopx/benchmark_adapters/skillsbench_codex_goal_recovery.py +271 -0
  19. loopx/benchmark_adapters/skillsbench_codex_goal_trace.py +81 -0
  20. loopx/benchmark_adapters/skillsbench_codex_runtime.py +339 -0
  21. loopx/benchmark_adapters/skillsbench_dockerfile_runtime.py +467 -0
  22. loopx/benchmark_adapters/skillsbench_failure_signals.py +652 -0
  23. loopx/benchmark_adapters/skillsbench_proxy_runtime.py +327 -0
  24. loopx/benchmark_adapters/skillsbench_remote_bridge.py +402 -0
  25. loopx/benchmark_adapters/skillsbench_result_discovery.py +143 -0
  26. loopx/benchmark_adapters/skillsbench_runner_profile.py +436 -0
  27. loopx/benchmark_adapters/skillsbench_runner_source.py +99 -0
  28. loopx/benchmark_adapters/skillsbench_setup_preflight.py +771 -0
  29. loopx/benchmark_adapters/skillsbench_signals.py +15 -0
  30. loopx/benchmark_adapters/skillsbench_task_source.py +141 -0
  31. loopx/benchmark_adapters/skillsbench_turn_route.py +723 -0
  32. loopx/benchmark_adapters/skillsbench_turn_runtime.py +1069 -0
  33. loopx/benchmark_adapters/skillsbench_typed_repair.py +689 -0
  34. loopx/benchmark_adapters/skillsbench_uv_cache.py +111 -0
  35. loopx/benchmark_adapters/skillsbench_verifier_bootstrap.py +227 -0
  36. loopx/benchmark_adapters/skillsbench_verifier_cache.py +138 -0
  37. loopx/benchmark_adapters/terminal_bench.py +10078 -0
  38. loopx/benchmark_case_analysis.py +1276 -0
  39. loopx/benchmark_case_state.py +1079 -0
  40. loopx/benchmark_core/__init__.py +239 -0
  41. loopx/benchmark_core/adapter.py +84 -0
  42. loopx/benchmark_core/artifacts.py +517 -0
  43. loopx/benchmark_core/attempts.py +199 -0
  44. loopx/benchmark_core/container_exec.py +216 -0
  45. loopx/benchmark_core/io.py +68 -0
  46. loopx/benchmark_core/lifecycle.py +211 -0
  47. loopx/benchmark_core/loop_protocol.py +689 -0
  48. loopx/benchmark_core/observable_handles.py +348 -0
  49. loopx/benchmark_core/parity.py +256 -0
  50. loopx/benchmark_core/remote_closeout.py +482 -0
  51. loopx/benchmark_core/rounds.py +215 -0
  52. loopx/benchmark_core/route_profile.py +509 -0
  53. loopx/benchmark_core/run_permissions.py +206 -0
  54. loopx/benchmark_core/split_control.py +925 -0
  55. loopx/benchmark_core/turn_fidelity.py +326 -0
  56. loopx/benchmark_ledger.py +3793 -0
  57. loopx/benchmark_ledger_countability.py +372 -0
  58. loopx/benchmark_ledger_current.py +724 -0
  59. loopx/benchmark_trajectory.py +405 -0
  60. loopx/benchmarks/__init__.py +1 -0
  61. loopx/benchmarks/qualification/__init__.py +1 -0
  62. loopx/benchmarks/qualification/release_outcome_baseline.py +360 -0
  63. loopx/benchmarks/read_models/__init__.py +1 -0
  64. loopx/benchmarks/read_models/benchmark_attempt_accounting.py +53 -0
  65. loopx/benchmarks/read_models/benchmark_comparison.py +414 -0
  66. loopx/benchmarks/read_models/benchmark_event_timeline.py +113 -0
  67. loopx/benchmarks/read_models/benchmark_experiment_report.py +475 -0
  68. loopx/benchmarks/read_models/benchmark_learning_ledger.py +137 -0
  69. loopx/benchmarks/read_models/benchmark_lifecycle_contracts.py +228 -0
  70. loopx/benchmarks/read_models/benchmark_projection.py +723 -0
  71. loopx/benchmarks/read_models/benchmark_result.py +146 -0
  72. loopx/benchmarks/read_models/benchmark_run_execution_contract.py +116 -0
  73. loopx/benchmarks/read_models/benchmark_run_failure.py +157 -0
  74. loopx/benchmarks/read_models/benchmark_run_metrics.py +213 -0
  75. loopx/benchmarks/read_models/benchmark_run_post_execution.py +635 -0
  76. loopx/benchmarks/read_models/benchmark_run_pre_execution.py +541 -0
  77. loopx/benchmarks/read_models/benchmark_status_compaction.py +1255 -0
  78. loopx/benchmarks/read_models/benchmark_status_runner.py +780 -0
  79. loopx/benchmarks/read_models/goal_start_control_score.py +857 -0
  80. loopx/benchmarks/read_models/skillsbench_post_run_debug.py +746 -0
  81. loopx/benchmarks/read_models/skillsbench_verifier_attribution.py +269 -0
  82. loopx/bootstrap.py +1116 -0
  83. loopx/bootstrap_command_pack.py +2167 -0
  84. loopx/boundary_authority.py +199 -0
  85. loopx/canary/__init__.py +1 -0
  86. loopx/canary/maintainability_ratchet.py +800 -0
  87. loopx/canary/planner.py +1984 -0
  88. loopx/canary/premerge.py +1130 -0
  89. loopx/canary/qualification_profiles.py +309 -0
  90. loopx/canary/quality_surface_catalog.py +838 -0
  91. loopx/canary/release_profiles.py +51 -0
  92. loopx/canary/runner.py +1107 -0
  93. loopx/canary/smoke_health.py +581 -0
  94. loopx/canary/smoke_profiles.py +212 -0
  95. loopx/capabilities/__init__.py +0 -0
  96. loopx/capabilities/agent_turn_recall/__init__.py +17 -0
  97. loopx/capabilities/agent_turn_recall/cli.py +369 -0
  98. loopx/capabilities/agent_turn_recall/core.py +296 -0
  99. loopx/capabilities/auto_research/__init__.py +16 -0
  100. loopx/capabilities/auto_research/bootstrap_contract.py +157 -0
  101. loopx/capabilities/auto_research/cli.py +1468 -0
  102. loopx/capabilities/auto_research/core.py +11 -0
  103. loopx/capabilities/auto_research/defaults.py +79 -0
  104. loopx/capabilities/auto_research/demo_e2e.py +1848 -0
  105. loopx/capabilities/auto_research/demo_supervisor.py +186 -0
  106. loopx/capabilities/auto_research/evidence_packet.py +767 -0
  107. loopx/capabilities/auto_research/human_view.py +794 -0
  108. loopx/capabilities/auto_research/kernel.py +191 -0
  109. loopx/capabilities/auto_research/knn_demo_workspace.py +322 -0
  110. loopx/capabilities/auto_research/live_evidence.py +248 -0
  111. loopx/capabilities/auto_research/preset.py +176 -0
  112. loopx/capabilities/auto_research/research_state.py +1085 -0
  113. loopx/capabilities/auto_research/role_profiles.py +394 -0
  114. loopx/capabilities/auto_research/rollout_append.py +97 -0
  115. loopx/capabilities/auto_research/terminal_result_contract.py +422 -0
  116. loopx/capabilities/auto_research/terminal_result_projection.py +171 -0
  117. loopx/capabilities/auto_research/terminal_result_query.py +233 -0
  118. loopx/capabilities/auto_research/terminal_results.py +349 -0
  119. loopx/capabilities/auto_research/user_contract.py +190 -0
  120. loopx/capabilities/auto_research/worker_loop.py +163 -0
  121. loopx/capabilities/auto_research/worker_runtime.py +777 -0
  122. loopx/capabilities/auto_research/worker_skill/SKILL.md +343 -0
  123. loopx/capabilities/benchmark_toolkit/__init__.py +19 -0
  124. loopx/capabilities/benchmark_toolkit/integrity.py +387 -0
  125. loopx/capabilities/catalog.py +1875 -0
  126. loopx/capabilities/change_quality/__init__.py +19 -0
  127. loopx/capabilities/change_quality/cli.py +171 -0
  128. loopx/capabilities/change_quality/context.py +156 -0
  129. loopx/capabilities/change_quality/oracles.py +269 -0
  130. loopx/capabilities/change_quality/policy.py +34 -0
  131. loopx/capabilities/change_quality/receipt.py +482 -0
  132. loopx/capabilities/change_quality/result.py +493 -0
  133. loopx/capabilities/change_quality/scope.py +171 -0
  134. loopx/capabilities/change_quality/shadow.py +680 -0
  135. loopx/capabilities/content_ops/__init__.py +0 -0
  136. loopx/capabilities/content_ops/cli.py +649 -0
  137. loopx/capabilities/content_ops/connector_packets.py +164 -0
  138. loopx/capabilities/content_ops/item_lifecycle.py +1000 -0
  139. loopx/capabilities/content_ops/layout.py +451 -0
  140. loopx/capabilities/content_ops/markdown.py +456 -0
  141. loopx/capabilities/content_ops/schemas.py +51 -0
  142. loopx/capabilities/content_ops/social_browser_x.py +107 -0
  143. loopx/capabilities/content_ops/surface.py +1956 -0
  144. loopx/capabilities/content_ops/templates/layout-catalog-v0.json +72 -0
  145. loopx/capabilities/context_providers/__init__.py +36 -0
  146. loopx/capabilities/context_providers/base.py +189 -0
  147. loopx/capabilities/context_providers/factory.py +32 -0
  148. loopx/capabilities/context_providers/openviking.py +702 -0
  149. loopx/capabilities/context_providers/service_ownership.py +185 -0
  150. loopx/capabilities/decision_context/__init__.py +129 -0
  151. loopx/capabilities/decision_context/architecture.py +83 -0
  152. loopx/capabilities/decision_context/assembler.py +849 -0
  153. loopx/capabilities/decision_context/catalog_entry.py +195 -0
  154. loopx/capabilities/decision_context/cli.py +310 -0
  155. loopx/capabilities/decision_context/cursor_commit.py +535 -0
  156. loopx/capabilities/decision_context/outcome_feedback.py +352 -0
  157. loopx/capabilities/decision_context/packets.py +654 -0
  158. loopx/capabilities/decision_context/private_state.py +189 -0
  159. loopx/capabilities/decision_context/profile.py +453 -0
  160. loopx/capabilities/decision_context/providers.py +228 -0
  161. loopx/capabilities/decision_context/review_settlement.py +136 -0
  162. loopx/capabilities/decision_context/runtime.py +273 -0
  163. loopx/capabilities/decision_context/sources.py +415 -0
  164. loopx/capabilities/explore/__init__.py +1 -0
  165. loopx/capabilities/explore/activation.py +198 -0
  166. loopx/capabilities/explore/adaptive_replay_planner.py +221 -0
  167. loopx/capabilities/explore/child_replay_runtime.py +463 -0
  168. loopx/capabilities/explore/composition_frontier.py +291 -0
  169. loopx/capabilities/explore/counterfactual_runtime.py +578 -0
  170. loopx/capabilities/explore/episode_runtime.py +647 -0
  171. loopx/capabilities/explore/harness_checkpoint.py +171 -0
  172. loopx/capabilities/explore/harness_gate.py +115 -0
  173. loopx/capabilities/explore/harness_runtime.py +1124 -0
  174. loopx/capabilities/explore/replay_metrics.py +206 -0
  175. loopx/capabilities/explore/replay_runtime.py +1271 -0
  176. loopx/capabilities/explore/resource_portfolio.py +173 -0
  177. loopx/capabilities/explore/result_log.py +974 -0
  178. loopx/capabilities/explore/router_state.py +432 -0
  179. loopx/capabilities/explore/source_history_reconcile.py +255 -0
  180. loopx/capabilities/explore/speculative_scheduler.py +498 -0
  181. loopx/capabilities/explore/todo_branch_plan.py +650 -0
  182. loopx/capabilities/explore/todo_evidence.py +141 -0
  183. loopx/capabilities/explore/trace_runtime.py +284 -0
  184. loopx/capabilities/explore/worker_branch_plan.py +1257 -0
  185. loopx/capabilities/integration_branch/__init__.py +13 -0
  186. loopx/capabilities/integration_branch/cli.py +148 -0
  187. loopx/capabilities/integration_branch/core.py +916 -0
  188. loopx/capabilities/issue_fix/__init__.py +19 -0
  189. loopx/capabilities/issue_fix/acceptance_loop.py +1050 -0
  190. loopx/capabilities/issue_fix/candidate_evidence.py +503 -0
  191. loopx/capabilities/issue_fix/candidate_preflight.py +676 -0
  192. loopx/capabilities/issue_fix/cli.py +1822 -0
  193. loopx/capabilities/issue_fix/cli_input.py +87 -0
  194. loopx/capabilities/issue_fix/content_ops_cli.py +148 -0
  195. loopx/capabilities/issue_fix/discovered_issue_promotion.py +947 -0
  196. loopx/capabilities/issue_fix/explore_projection.py +710 -0
  197. loopx/capabilities/issue_fix/feasibility.py +542 -0
  198. loopx/capabilities/issue_fix/github_public.py +661 -0
  199. loopx/capabilities/issue_fix/intake_surface.py +832 -0
  200. loopx/capabilities/issue_fix/metadata_preview.py +218 -0
  201. loopx/capabilities/issue_fix/metrics_projection.py +1340 -0
  202. loopx/capabilities/issue_fix/metrics_supplement.py +634 -0
  203. loopx/capabilities/issue_fix/metrics_supplement_cli.py +127 -0
  204. loopx/capabilities/issue_fix/outcome_projection.py +1235 -0
  205. loopx/capabilities/issue_fix/periodic_report.py +189 -0
  206. loopx/capabilities/issue_fix/pr_description.py +418 -0
  207. loopx/capabilities/issue_fix/pr_gate_reconcile.py +496 -0
  208. loopx/capabilities/issue_fix/pr_gate_reconcile_cli.py +464 -0
  209. loopx/capabilities/issue_fix/pr_lifecycle.py +1327 -0
  210. loopx/capabilities/issue_fix/pr_lifecycle_rollout.py +85 -0
  211. loopx/capabilities/issue_fix/pr_monitor_materialization.py +257 -0
  212. loopx/capabilities/issue_fix/pr_review_ack.py +439 -0
  213. loopx/capabilities/issue_fix/provider_hooks.py +24 -0
  214. loopx/capabilities/issue_fix/repository_commit_evidence.py +186 -0
  215. loopx/capabilities/issue_fix/repository_context.py +457 -0
  216. loopx/capabilities/issue_fix/repository_memory.py +459 -0
  217. loopx/capabilities/issue_fix/repository_memory_provider.py +1454 -0
  218. loopx/capabilities/issue_fix/repository_snapshot.py +454 -0
  219. loopx/capabilities/issue_fix/reviewer_cli.py +917 -0
  220. loopx/capabilities/issue_fix/reviewer_notification.py +882 -0
  221. loopx/capabilities/issue_fix/reviewer_notification_drain.py +942 -0
  222. loopx/capabilities/issue_fix/reviewer_recommendation.py +1057 -0
  223. loopx/capabilities/issue_fix/reviewer_request.py +1282 -0
  224. loopx/capabilities/issue_fix/reward_memory.py +879 -0
  225. loopx/capabilities/issue_fix/workflow_plan.py +1286 -0
  226. loopx/capabilities/material_lifecycle/__init__.py +161 -0
  227. loopx/capabilities/material_lifecycle/_validation.py +183 -0
  228. loopx/capabilities/material_lifecycle/apply.py +672 -0
  229. loopx/capabilities/material_lifecycle/architecture.py +122 -0
  230. loopx/capabilities/material_lifecycle/cli.py +161 -0
  231. loopx/capabilities/material_lifecycle/decision_planning.py +470 -0
  232. loopx/capabilities/material_lifecycle/explore_execution.py +306 -0
  233. loopx/capabilities/material_lifecycle/intake.py +869 -0
  234. loopx/capabilities/material_lifecycle/inventory.py +147 -0
  235. loopx/capabilities/material_lifecycle/lifecycle.py +98 -0
  236. loopx/capabilities/material_lifecycle/preparation.py +147 -0
  237. loopx/capabilities/material_lifecycle/project_skill.py +83 -0
  238. loopx/capabilities/material_lifecycle/ranking.py +267 -0
  239. loopx/capabilities/material_lifecycle/readable_projection.py +500 -0
  240. loopx/capabilities/material_lifecycle/rebuild.py +480 -0
  241. loopx/capabilities/material_lifecycle/settlement.py +238 -0
  242. loopx/capabilities/periodic_report/__init__.py +71 -0
  243. loopx/capabilities/periodic_report/adapters.py +939 -0
  244. loopx/capabilities/periodic_report/archive.py +422 -0
  245. loopx/capabilities/periodic_report/bindings.py +705 -0
  246. loopx/capabilities/periodic_report/cli.py +277 -0
  247. loopx/capabilities/periodic_report/core.py +691 -0
  248. loopx/capabilities/periodic_report/extension_envelope.py +66 -0
  249. loopx/capabilities/periodic_report/presets.py +103 -0
  250. loopx/capabilities/periodic_report/profile.py +235 -0
  251. loopx/capabilities/periodic_report/project_progress.py +179 -0
  252. loopx/capabilities/periodic_report/triggers.py +452 -0
  253. loopx/capabilities/pr_review_queue/__init__.py +17 -0
  254. loopx/capabilities/pr_review_queue/core.py +506 -0
  255. loopx/capabilities/pr_review_queue/review_contract.py +506 -0
  256. loopx/capabilities/registry.py +192 -0
  257. loopx/capabilities/reward_memory/__init__.py +75 -0
  258. loopx/capabilities/reward_memory/application.py +819 -0
  259. loopx/capabilities/reward_memory/architecture.py +572 -0
  260. loopx/capabilities/reward_memory/candidate_review.py +511 -0
  261. loopx/capabilities/reward_memory/cli.py +469 -0
  262. loopx/capabilities/reward_memory/dogfood.py +574 -0
  263. loopx/capabilities/reward_memory/evaluation.py +296 -0
  264. loopx/capabilities/reward_memory/evaluation_fixtures.py +362 -0
  265. loopx/capabilities/reward_memory/experiment.py +567 -0
  266. loopx/capabilities/reward_memory/health.py +222 -0
  267. loopx/capabilities/reward_memory/ingestion.py +519 -0
  268. loopx/capabilities/reward_memory/registry.py +600 -0
  269. loopx/capabilities/reward_memory/runtime_hooks.py +312 -0
  270. loopx/capabilities/reward_memory/scoped_feedback.py +173 -0
  271. loopx/capabilities/semantic_preference/__init__.py +12 -0
  272. loopx/capabilities/semantic_preference/cli.py +189 -0
  273. loopx/capabilities/semantic_preference/contract.py +592 -0
  274. loopx/capabilities/semantic_preference/reward_memory.py +62 -0
  275. loopx/capabilities/value_connectors/__init__.py +1 -0
  276. loopx/capabilities/value_connectors/cli.py +401 -0
  277. loopx/capabilities/value_connectors/finance_extension_migration.py +108 -0
  278. loopx/capabilities/value_connectors/install_check.py +147 -0
  279. loopx/capabilities/value_connectors/planner.py +733 -0
  280. loopx/capabilities/value_connectors/source_map.py +446 -0
  281. loopx/claude_goal_baseline.py +138 -0
  282. loopx/claude_goal_mode/__init__.py +23 -0
  283. loopx/claude_goal_mode/hooks/goal_policy.py +212 -0
  284. loopx/claude_goal_mode/hooks/goal_state.py +139 -0
  285. loopx/claude_goal_mode/mcp/loopx_mcp.py +167 -0
  286. loopx/claude_goal_mode/scripts/connect.py +103 -0
  287. loopx/claude_goal_mode/scripts/goalmode_cmd.py +241 -0
  288. loopx/claude_goal_mode/scripts/install.py +328 -0
  289. loopx/claude_goal_mode/statusline/goal_status.py +97 -0
  290. loopx/cli.py +836 -0
  291. loopx/cli_commands/__init__.py +334 -0
  292. loopx/cli_commands/_host_thread.py +13 -0
  293. loopx/cli_commands/agentissue_runner_flow.py +447 -0
  294. loopx/cli_commands/agents_last_exam.py +160 -0
  295. loopx/cli_commands/agents_last_exam_baked_input.py +302 -0
  296. loopx/cli_commands/agents_last_exam_host_codex.py +374 -0
  297. loopx/cli_commands/agents_last_exam_launch_dry_run.py +372 -0
  298. loopx/cli_commands/agents_last_exam_local_plan.py +322 -0
  299. loopx/cli_commands/agents_last_exam_runner_source.py +352 -0
  300. loopx/cli_commands/agents_last_exam_task_material.py +335 -0
  301. loopx/cli_commands/agents_last_exam_validation_gate.py +236 -0
  302. loopx/cli_commands/benchmark_boundary.py +499 -0
  303. loopx/cli_commands/benchmark_dispatch.py +161 -0
  304. loopx/cli_commands/benchmark_release_outcome.py +123 -0
  305. loopx/cli_commands/benchmark_review_lifecycle.py +1275 -0
  306. loopx/cli_commands/benchmark_run_ledger.py +763 -0
  307. loopx/cli_commands/benchmark_run_ledger_case_analysis.py +249 -0
  308. loopx/cli_commands/benchmark_run_ledger_classification.py +45 -0
  309. loopx/cli_commands/benchmark_run_ledger_maintenance.py +486 -0
  310. loopx/cli_commands/benchmark_run_ledger_maintenance_registration.py +342 -0
  311. loopx/cli_commands/benchmark_run_ledger_maintenance_rendering.py +233 -0
  312. loopx/cli_commands/benchmark_run_ledger_parity.py +92 -0
  313. loopx/cli_commands/bootstrap_connect.py +238 -0
  314. loopx/cli_commands/canary.py +707 -0
  315. loopx/cli_commands/canary_release_qualification.py +79 -0
  316. loopx/cli_commands/capability.py +96 -0
  317. loopx/cli_commands/doctor.py +43 -0
  318. loopx/cli_commands/dreaming.py +143 -0
  319. loopx/cli_commands/edgebench.py +205 -0
  320. loopx/cli_commands/evidence_log.py +275 -0
  321. loopx/cli_commands/explore.py +989 -0
  322. loopx/cli_commands/explore_planning_commands.py +157 -0
  323. loopx/cli_commands/extension.py +271 -0
  324. loopx/cli_commands/first_run_report.py +73 -0
  325. loopx/cli_commands/goal_channel.py +656 -0
  326. loopx/cli_commands/handoff_mode.py +158 -0
  327. loopx/cli_commands/history.py +622 -0
  328. loopx/cli_commands/host_mode_plan.py +113 -0
  329. loopx/cli_commands/lark_inbox.py +431 -0
  330. loopx/cli_commands/lark_kanban.py +629 -0
  331. loopx/cli_commands/ml_experiment.py +321 -0
  332. loopx/cli_commands/multi_agent.py +211 -0
  333. loopx/cli_commands/opencode2_goal_worker.py +217 -0
  334. loopx/cli_commands/pr_review.py +167 -0
  335. loopx/cli_commands/presentation.py +218 -0
  336. loopx/cli_commands/preset.py +96 -0
  337. loopx/cli_commands/project.py +150 -0
  338. loopx/cli_commands/project_lifecycle.py +915 -0
  339. loopx/cli_commands/quota.py +859 -0
  340. loopx/cli_commands/quota_registration.py +241 -0
  341. loopx/cli_commands/quota_request.py +113 -0
  342. loopx/cli_commands/ready_score.py +110 -0
  343. loopx/cli_commands/registry_admin.py +975 -0
  344. loopx/cli_commands/registry_admin_configure.py +344 -0
  345. loopx/cli_commands/registry_admin_peer.py +84 -0
  346. loopx/cli_commands/registry_authority.py +218 -0
  347. loopx/cli_commands/review_batch.py +146 -0
  348. loopx/cli_commands/slash_commands.py +145 -0
  349. loopx/cli_commands/start_goal.py +251 -0
  350. loopx/cli_commands/starter.py +175 -0
  351. loopx/cli_commands/starter_bootstrap.py +179 -0
  352. loopx/cli_commands/starter_bootstrap_registration.py +198 -0
  353. loopx/cli_commands/starter_runtime_idle.py +107 -0
  354. loopx/cli_commands/starter_scheduler.py +207 -0
  355. loopx/cli_commands/starter_session_runtime.py +152 -0
  356. loopx/cli_commands/starter_visible_common.py +54 -0
  357. loopx/cli_commands/starter_visible_driver.py +161 -0
  358. loopx/cli_commands/starter_visible_pilot.py +278 -0
  359. loopx/cli_commands/status.py +867 -0
  360. loopx/cli_commands/status_registration.py +239 -0
  361. loopx/cli_commands/summary_all.py +222 -0
  362. loopx/cli_commands/support_control.py +809 -0
  363. loopx/cli_commands/support_control_registry.py +68 -0
  364. loopx/cli_commands/support_control_supervisor.py +289 -0
  365. loopx/cli_commands/task_lease.py +306 -0
  366. loopx/cli_commands/terminal_bench_adapter.py +717 -0
  367. loopx/cli_commands/terminal_bench_environment_result.py +1246 -0
  368. loopx/cli_commands/todo.py +940 -0
  369. loopx/cli_commands/todo_argument_validation.py +572 -0
  370. loopx/cli_commands/todo_event.py +114 -0
  371. loopx/cli_commands/turn.py +804 -0
  372. loopx/cli_commands/version.py +46 -0
  373. loopx/cli_commands/worker_bridge.py +659 -0
  374. loopx/cli_rollout.py +314 -0
  375. loopx/codex_cli_goal_tui.py +672 -0
  376. loopx/codex_cli_probe.py +1530 -0
  377. loopx/codex_cli_probe_markdown.py +935 -0
  378. loopx/codex_cli_runtime_probe.py +733 -0
  379. loopx/codex_cli_scheduler.py +564 -0
  380. loopx/codex_goal_baseline.py +620 -0
  381. loopx/configuration_catalog.py +617 -0
  382. loopx/configure_goal.py +1375 -0
  383. loopx/contract.py +996 -0
  384. loopx/control_plane/__init__.py +71 -0
  385. loopx/control_plane/agents/__init__.py +1 -0
  386. loopx/control_plane/agents/agent_lane_recommendation.py +516 -0
  387. loopx/control_plane/agents/agent_scope.py +1578 -0
  388. loopx/control_plane/agents/agent_scope_frontier.py +60 -0
  389. loopx/control_plane/agents/capability_gate.py +531 -0
  390. loopx/control_plane/agents/identity.py +140 -0
  391. loopx/control_plane/agents/legacy_migration.py +169 -0
  392. loopx/control_plane/agents/management_projection.py +658 -0
  393. loopx/control_plane/agents/material_frontier.py +608 -0
  394. loopx/control_plane/agents/material_handoff.py +156 -0
  395. loopx/control_plane/agents/multi_agent/__init__.py +1 -0
  396. loopx/control_plane/agents/multi_agent/codex_executable.py +207 -0
  397. loopx/control_plane/agents/multi_agent/collective_round_ledger.py +387 -0
  398. loopx/control_plane/agents/multi_agent/contract.py +474 -0
  399. loopx/control_plane/agents/multi_agent/recipe.py +110 -0
  400. loopx/control_plane/agents/multi_agent/role_successor.py +297 -0
  401. loopx/control_plane/agents/multi_agent/runtime_scripts.py +426 -0
  402. loopx/control_plane/agents/multi_agent/visible_launch_policy.py +149 -0
  403. loopx/control_plane/agents/multi_agent/visible_wake_scheduler.py +392 -0
  404. loopx/control_plane/agents/profile.py +216 -0
  405. loopx/control_plane/agents/runtime_model.py +73 -0
  406. loopx/control_plane/agents/subagent_activity.py +164 -0
  407. loopx/control_plane/agents/supervisor.py +544 -0
  408. loopx/control_plane/agents/supervisor_events.py +462 -0
  409. loopx/control_plane/agents/supervisor_inject.py +204 -0
  410. loopx/control_plane/agents/work_mode.py +56 -0
  411. loopx/control_plane/agents/workspace_guard.py +364 -0
  412. loopx/control_plane/effect_program.py +644 -0
  413. loopx/control_plane/goals/__init__.py +1 -0
  414. loopx/control_plane/goals/active_state_event_projection.py +103 -0
  415. loopx/control_plane/goals/active_state_metadata.py +47 -0
  416. loopx/control_plane/goals/active_state_sections.py +58 -0
  417. loopx/control_plane/goals/configure_goal_service.py +354 -0
  418. loopx/control_plane/goals/contract_health.py +132 -0
  419. loopx/control_plane/goals/dreaming.py +152 -0
  420. loopx/control_plane/goals/global_registry_health.py +199 -0
  421. loopx/control_plane/goals/global_registry_shadow.py +33 -0
  422. loopx/control_plane/goals/goal_channel.py +34 -0
  423. loopx/control_plane/goals/goal_channel_projection.py +560 -0
  424. loopx/control_plane/goals/goal_frontier/__init__.py +1917 -0
  425. loopx/control_plane/goals/goal_frontier/ack_policy.py +149 -0
  426. loopx/control_plane/goals/goal_frontier/outcome_continuity.py +437 -0
  427. loopx/control_plane/goals/goal_frontier/replan_rules.py +210 -0
  428. loopx/control_plane/goals/goal_frontier/semantic_history.py +314 -0
  429. loopx/control_plane/goals/goal_frontier/terminal.py +180 -0
  430. loopx/control_plane/goals/goal_vision.py +443 -0
  431. loopx/control_plane/goals/goal_vision_policy.py +36 -0
  432. loopx/control_plane/goals/goal_vision_state.py +62 -0
  433. loopx/control_plane/goals/goal_vision_wait.py +290 -0
  434. loopx/control_plane/goals/path_resolution.py +20 -0
  435. loopx/control_plane/goals/start_contract.py +206 -0
  436. loopx/control_plane/goals/vision_checkpoint.py +92 -0
  437. loopx/control_plane/handoff/__init__.py +1 -0
  438. loopx/control_plane/handoff/cross_runtime_impl_review.py +311 -0
  439. loopx/control_plane/handoff/delivery_contract.py +161 -0
  440. loopx/control_plane/handoff/handoff_runs.py +71 -0
  441. loopx/control_plane/handoff/project_handoff.py +155 -0
  442. loopx/control_plane/handoff/review_batch.py +463 -0
  443. loopx/control_plane/handoff/review_packet_context.py +216 -0
  444. loopx/control_plane/heartbeat/agent.py +173 -0
  445. loopx/control_plane/heartbeat/budget.py +66 -0
  446. loopx/control_plane/heartbeat/builder.py +501 -0
  447. loopx/control_plane/heartbeat/host.py +64 -0
  448. loopx/control_plane/heartbeat/rules.py +68 -0
  449. loopx/control_plane/heartbeat/task_body.py +759 -0
  450. loopx/control_plane/heartbeat/visible_goal.py +86 -0
  451. loopx/control_plane/projects/__init__.py +1 -0
  452. loopx/control_plane/projects/contract.py +25 -0
  453. loopx/control_plane/projects/registry.py +663 -0
  454. loopx/control_plane/quota/__init__.py +1 -0
  455. loopx/control_plane/quota/cli_projection.py +704 -0
  456. loopx/control_plane/quota/decision_summary.py +431 -0
  457. loopx/control_plane/quota/effect_program.py +152 -0
  458. loopx/control_plane/quota/error_codes.py +19 -0
  459. loopx/control_plane/quota/goal_boundary.py +464 -0
  460. loopx/control_plane/quota/heartbeat_receipt.py +277 -0
  461. loopx/control_plane/quota/heartbeat_recommendation.py +718 -0
  462. loopx/control_plane/quota/host_poll_receipts.py +162 -0
  463. loopx/control_plane/quota/live_decision.py +142 -0
  464. loopx/control_plane/quota/monitor_poll.py +786 -0
  465. loopx/control_plane/quota/policy_constants.py +40 -0
  466. loopx/control_plane/quota/projection_repair.py +262 -0
  467. loopx/control_plane/quota/recent_runs.py +210 -0
  468. loopx/control_plane/quota/scheduler_ack.py +490 -0
  469. loopx/control_plane/quota/selected_todo_projection.py +139 -0
  470. loopx/control_plane/quota/settlement.py +437 -0
  471. loopx/control_plane/quota/settlement_cli.py +246 -0
  472. loopx/control_plane/quota/settlement_validation.py +64 -0
  473. loopx/control_plane/quota/settlement_workspace_causality.py +180 -0
  474. loopx/control_plane/quota/should_run.py +249 -0
  475. loopx/control_plane/quota/should_run_packet.py +1165 -0
  476. loopx/control_plane/quota/should_run_prepare.py +675 -0
  477. loopx/control_plane/quota/slot_accounting.py +1123 -0
  478. loopx/control_plane/quota/spend_sources.py +11 -0
  479. loopx/control_plane/quota/stall_repair.py +397 -0
  480. loopx/control_plane/quota/states.py +29 -0
  481. loopx/control_plane/quota/task_orchestration.py +448 -0
  482. loopx/control_plane/quota/task_orchestration_admission.py +497 -0
  483. loopx/control_plane/quota/turn_envelope.py +889 -0
  484. loopx/control_plane/quota/usage_summary.py +140 -0
  485. loopx/control_plane/reward_memory.py +43 -0
  486. loopx/control_plane/runtime/__init__.py +2 -0
  487. loopx/control_plane/runtime/active_user_assisted_pilot.py +275 -0
  488. loopx/control_plane/runtime/agent_scoped_evidence_log.py +435 -0
  489. loopx/control_plane/runtime/decision_freshness.py +203 -0
  490. loopx/control_plane/runtime/event_ledger.py +197 -0
  491. loopx/control_plane/runtime/event_store_migration_bridge.py +196 -0
  492. loopx/control_plane/runtime/goal_project_route.py +70 -0
  493. loopx/control_plane/runtime/local_state_write_correctness.py +242 -0
  494. loopx/control_plane/runtime/promotion_readiness.py +152 -0
  495. loopx/control_plane/runtime/public_safety.py +120 -0
  496. loopx/control_plane/runtime/run_artifacts.py +78 -0
  497. loopx/control_plane/runtime/run_compaction.py +397 -0
  498. loopx/control_plane/runtime/run_context_retention.py +241 -0
  499. loopx/control_plane/runtime/run_history.py +132 -0
  500. loopx/control_plane/runtime/run_index_duplicates.py +205 -0
  501. loopx/control_plane/runtime/run_index_rebuild.py +263 -0
  502. loopx/control_plane/runtime/run_ingest_health.py +336 -0
  503. loopx/control_plane/runtime/runtime_projection_route.py +624 -0
  504. loopx/control_plane/runtime/runtime_projection_writer.py +98 -0
  505. loopx/control_plane/runtime/session_runtime.py +339 -0
  506. loopx/control_plane/runtime/shared_runtime_material_projection.py +332 -0
  507. loopx/control_plane/runtime/shared_runtime_refresh_projection.py +183 -0
  508. loopx/control_plane/runtime/stale_latest_run.py +90 -0
  509. loopx/control_plane/runtime/status_classifications.py +49 -0
  510. loopx/control_plane/runtime/status_projection_cache.py +235 -0
  511. loopx/control_plane/runtime/stride_observation.py +144 -0
  512. loopx/control_plane/runtime/time.py +39 -0
  513. loopx/control_plane/runtime/trajectory_hygiene.py +149 -0
  514. loopx/control_plane/runtime/validation_command.py +69 -0
  515. loopx/control_plane/scheduler/__init__.py +1 -0
  516. loopx/control_plane/scheduler/ack.py +329 -0
  517. loopx/control_plane/scheduler/arbitration.py +188 -0
  518. loopx/control_plane/scheduler/automation_liveness.py +183 -0
  519. loopx/control_plane/scheduler/execution_context.py +555 -0
  520. loopx/control_plane/scheduler/external_evidence_observation.py +428 -0
  521. loopx/control_plane/scheduler/monitor_display.py +143 -0
  522. loopx/control_plane/scheduler/monitor_poll_policy.py +161 -0
  523. loopx/control_plane/scheduler/monitor_poll_writeback.py +351 -0
  524. loopx/control_plane/scheduler/monitor_target.py +64 -0
  525. loopx/control_plane/scheduler/monitor_todo.py +146 -0
  526. loopx/control_plane/scheduler/monitor_wait.py +237 -0
  527. loopx/control_plane/scheduler/scheduler_hint.py +1284 -0
  528. loopx/control_plane/scheduler/state.py +354 -0
  529. loopx/control_plane/scheduler/state_transition_rules.py +179 -0
  530. loopx/control_plane/scheduler/time.py +10 -0
  531. loopx/control_plane/settlement_driver.py +293 -0
  532. loopx/control_plane/status/__init__.py +6 -0
  533. loopx/control_plane/status/active_state_projection.py +105 -0
  534. loopx/control_plane/status/agent_lane_projection.py +375 -0
  535. loopx/control_plane/status/attention_projection.py +74 -0
  536. loopx/control_plane/status/autonomous_replan_projection.py +103 -0
  537. loopx/control_plane/status/collection.py +140 -0
  538. loopx/control_plane/status/contract_projection.py +31 -0
  539. loopx/control_plane/status/dreaming_projection.py +52 -0
  540. loopx/control_plane/status/goal_attention_projection.py +157 -0
  541. loopx/control_plane/status/lifecycle_projection.py +110 -0
  542. loopx/control_plane/status/monitor_display_projection.py +69 -0
  543. loopx/control_plane/status/registry_health_projection.py +75 -0
  544. loopx/control_plane/status/run_projection.py +70 -0
  545. loopx/control_plane/status/runtime_summaries.py +161 -0
  546. loopx/control_plane/testing/__init__.py +1 -0
  547. loopx/control_plane/testing/actual_default_model_behavior_portfolio.py +1371 -0
  548. loopx/control_plane/testing/canary_harness.py +182 -0
  549. loopx/control_plane/testing/capability_monitor_repair_tool_behavior.py +674 -0
  550. loopx/control_plane/testing/cli_output_budget.py +807 -0
  551. loopx/control_plane/testing/cli_output_differential.py +250 -0
  552. loopx/control_plane/testing/cli_output_semantics.py +87 -0
  553. loopx/control_plane/testing/control_plane_composition_scenarios.py +225 -0
  554. loopx/control_plane/testing/decision_replay.py +268 -0
  555. loopx/control_plane/testing/doubao_model_behavior_actor.py +559 -0
  556. loopx/control_plane/testing/model_behavior_corpus.py +344 -0
  557. loopx/control_plane/testing/model_behavior_qualification.py +769 -0
  558. loopx/control_plane/testing/model_behavior_retained_cases.py +235 -0
  559. loopx/control_plane/testing/model_tool_behavior.py +536 -0
  560. loopx/control_plane/testing/onboarding_model_behavior_qualification.py +642 -0
  561. loopx/control_plane/testing/quota_fixtures.py +208 -0
  562. loopx/control_plane/testing/quota_should_run_parity.py +57 -0
  563. loopx/control_plane/testing/release_commit_qualification.py +671 -0
  564. loopx/control_plane/testing/replan_semantic_action_behavior.py +1302 -0
  565. loopx/control_plane/testing/scoped_gate_successor_tool_behavior.py +527 -0
  566. loopx/control_plane/testing/selected_todo_tool_behavior.py +1002 -0
  567. loopx/control_plane/testing/terminal_settlement_tool_behavior.py +656 -0
  568. loopx/control_plane/todos/__init__.py +1 -0
  569. loopx/control_plane/todos/active_state_editing.py +296 -0
  570. loopx/control_plane/todos/active_state_todo_parser.py +138 -0
  571. loopx/control_plane/todos/active_state_todos.py +175 -0
  572. loopx/control_plane/todos/addition.py +103 -0
  573. loopx/control_plane/todos/claim_visibility.py +253 -0
  574. loopx/control_plane/todos/completed_archive.py +139 -0
  575. loopx/control_plane/todos/completion_fence.py +49 -0
  576. loopx/control_plane/todos/completion_policy.py +153 -0
  577. loopx/control_plane/todos/completion_validation.py +248 -0
  578. loopx/control_plane/todos/completion_validation_accountability.py +27 -0
  579. loopx/control_plane/todos/completion_validation_projection.py +57 -0
  580. loopx/control_plane/todos/contract.py +1476 -0
  581. loopx/control_plane/todos/decision_scope.py +554 -0
  582. loopx/control_plane/todos/deferred_resume.py +546 -0
  583. loopx/control_plane/todos/durable_completion.py +201 -0
  584. loopx/control_plane/todos/event_writeback.py +484 -0
  585. loopx/control_plane/todos/frontier_deadline.py +132 -0
  586. loopx/control_plane/todos/handoff_gate.py +283 -0
  587. loopx/control_plane/todos/handoff_mode.py +444 -0
  588. loopx/control_plane/todos/handoff_note.py +202 -0
  589. loopx/control_plane/todos/line_update.py +361 -0
  590. loopx/control_plane/todos/list_projection.py +205 -0
  591. loopx/control_plane/todos/markdown.py +199 -0
  592. loopx/control_plane/todos/monitor_metadata.py +88 -0
  593. loopx/control_plane/todos/mutation_authority.py +299 -0
  594. loopx/control_plane/todos/projection.py +655 -0
  595. loopx/control_plane/todos/quota_summary.py +1138 -0
  596. loopx/control_plane/todos/route_continuation.py +267 -0
  597. loopx/control_plane/todos/succession_warning.py +174 -0
  598. loopx/control_plane/todos/summary_item.py +223 -0
  599. loopx/control_plane/todos/text.py +30 -0
  600. loopx/control_plane/todos/todo_index.py +226 -0
  601. loopx/control_plane/todos/todo_summary.py +1458 -0
  602. loopx/control_plane/todos/unblock_resume.py +326 -0
  603. loopx/control_plane/todos/user_gate.py +263 -0
  604. loopx/control_plane/todos/write_hint.py +63 -0
  605. loopx/control_plane/todos/write_policy.py +135 -0
  606. loopx/control_plane/turn_driver/__init__.py +85 -0
  607. loopx/control_plane/turn_driver/codex_cli.py +502 -0
  608. loopx/control_plane/turn_driver/driver.py +355 -0
  609. loopx/control_plane/turn_driver/executor.py +1468 -0
  610. loopx/control_plane/turn_driver/loop_controller.py +669 -0
  611. loopx/control_plane/turn_driver/settlement.py +318 -0
  612. loopx/control_plane/turn_driver/transaction.py +375 -0
  613. loopx/control_plane/work_items/__init__.py +1 -0
  614. loopx/control_plane/work_items/attention_fields.py +56 -0
  615. loopx/control_plane/work_items/attention_item.py +77 -0
  616. loopx/control_plane/work_items/attention_queue.py +322 -0
  617. loopx/control_plane/work_items/attention_routing.py +213 -0
  618. loopx/control_plane/work_items/autonomous_candidates.py +135 -0
  619. loopx/control_plane/work_items/autonomous_replan_ack.py +276 -0
  620. loopx/control_plane/work_items/autonomous_replan_obligation.py +786 -0
  621. loopx/control_plane/work_items/backlog_hygiene.py +59 -0
  622. loopx/control_plane/work_items/capability_monitor_fallback.py +221 -0
  623. loopx/control_plane/work_items/delivery_batch_scale.py +66 -0
  624. loopx/control_plane/work_items/delivery_outcome.py +152 -0
  625. loopx/control_plane/work_items/delivery_signals.py +113 -0
  626. loopx/control_plane/work_items/execution_obligation.py +235 -0
  627. loopx/control_plane/work_items/goal_route_hint.py +320 -0
  628. loopx/control_plane/work_items/interaction_contract.py +1540 -0
  629. loopx/control_plane/work_items/issue_meta_surface.py +159 -0
  630. loopx/control_plane/work_items/lifecycle.py +139 -0
  631. loopx/control_plane/work_items/operator_inbox.py +266 -0
  632. loopx/control_plane/work_items/outcome_followthrough.py +69 -0
  633. loopx/control_plane/work_items/primary_action.py +326 -0
  634. loopx/control_plane/work_items/progress_observation.py +630 -0
  635. loopx/control_plane/work_items/project_asset.py +675 -0
  636. loopx/control_plane/work_items/repair_delta.py +693 -0
  637. loopx/control_plane/work_items/runtime_capability_reentry.py +168 -0
  638. loopx/control_plane/work_items/semantic_replan_writeback.py +177 -0
  639. loopx/control_plane/work_items/status_contract.py +49 -0
  640. loopx/control_plane/work_items/task_graph.py +1046 -0
  641. loopx/control_plane/work_items/task_lease.py +1254 -0
  642. loopx/control_plane/work_items/task_lease_settlement.py +422 -0
  643. loopx/control_plane/work_items/work_lane.py +510 -0
  644. loopx/control_plane/work_items/work_lane_context.py +161 -0
  645. loopx/demo.py +247 -0
  646. loopx/diagnose.py +633 -0
  647. loopx/doctor.py +1251 -0
  648. loopx/domain_packs/__init__.py +1 -0
  649. loopx/domain_packs/issue_fix.py +571 -0
  650. loopx/domain_packs/ml_experiment.py +854 -0
  651. loopx/domain_state.py +137 -0
  652. loopx/dreaming.py +706 -0
  653. loopx/entrypoint.py +16 -0
  654. loopx/event_sourced_state.py +981 -0
  655. loopx/execution_profile.py +286 -0
  656. loopx/experiments/__init__.py +1 -0
  657. loopx/experiments/planner_worker/__init__.py +1 -0
  658. loopx/experiments/planner_worker/contract.py +523 -0
  659. loopx/experiments/planner_worker/runtime.py +391 -0
  660. loopx/experiments/planner_worker/traex.py +461 -0
  661. loopx/explore_graph.py +11 -0
  662. loopx/extensions/__init__.py +1 -0
  663. loopx/extensions/bundled.py +28 -0
  664. loopx/extensions/execution_envelope.py +126 -0
  665. loopx/extensions/lark/__init__.py +11 -0
  666. loopx/extensions/lark/event_collector.py +478 -0
  667. loopx/extensions/lark/event_collector_runtime.py +506 -0
  668. loopx/extensions/lark/event_inbox.py +454 -0
  669. loopx/extensions/lark/extension.toml +88 -0
  670. loopx/extensions/lark/goal_channel.py +44 -0
  671. loopx/extensions/lark/goal_channel_contracts.py +388 -0
  672. loopx/extensions/lark/goal_channel_lifecycle.py +218 -0
  673. loopx/extensions/lark/goal_channel_runtime.py +792 -0
  674. loopx/extensions/lark/goal_channel_setup.py +805 -0
  675. loopx/extensions/lark/goal_channel_targets.py +215 -0
  676. loopx/extensions/lark/goal_channel_transport.py +281 -0
  677. loopx/extensions/lark/inbox_reactions.py +650 -0
  678. loopx/extensions/lark/inbox_reply.py +430 -0
  679. loopx/extensions/lark/presentation/__init__.py +11 -0
  680. loopx/extensions/lark/presentation/explore_results.py +2276 -0
  681. loopx/extensions/lark/presentation/explore_singleflight.py +127 -0
  682. loopx/extensions/lark/presentation/explore_source_guard.py +121 -0
  683. loopx/extensions/lark/presentation/explore_stage_document.py +703 -0
  684. loopx/extensions/lark/presentation/explore_visual_integrity.py +122 -0
  685. loopx/extensions/lark/presentation/explore_visual_readback.py +452 -0
  686. loopx/extensions/lark/presentation/explore_visual_styles.py +156 -0
  687. loopx/extensions/lark/presentation/issue_fix_surface.py +612 -0
  688. loopx/extensions/lark/presentation/kanban.py +2791 -0
  689. loopx/extensions/lark/presentation/message_card.py +112 -0
  690. loopx/extensions/lark/presentation/periodic_report.py +261 -0
  691. loopx/extensions/lark/presentation/projection_rows.py +600 -0
  692. loopx/extensions/lark/presentation/record_io.py +95 -0
  693. loopx/extensions/lark/presentation/sync_receipt.py +145 -0
  694. loopx/extensions/lark/private_json.py +40 -0
  695. loopx/extensions/lark/provider.py +86 -0
  696. loopx/extensions/lark/reviewer_notification.py +604 -0
  697. loopx/extensions/manifest.py +385 -0
  698. loopx/extensions/openviking_periodic_report/__init__.py +17 -0
  699. loopx/extensions/openviking_periodic_report/activation.py +173 -0
  700. loopx/extensions/openviking_periodic_report/extension.toml +17 -0
  701. loopx/extensions/openviking_periodic_report/provider.py +355 -0
  702. loopx/extensions/openviking_periodic_report/sink.py +117 -0
  703. loopx/extensions/openviking_semantic_preference/__init__.py +5 -0
  704. loopx/extensions/openviking_semantic_preference/extension.toml +16 -0
  705. loopx/extensions/openviking_semantic_preference/history_export.py +484 -0
  706. loopx/extensions/openviking_semantic_preference/project_peer.py +68 -0
  707. loopx/extensions/openviking_semantic_preference/provider.py +312 -0
  708. loopx/extensions/presentation.py +979 -0
  709. loopx/extensions/process_runtime.py +204 -0
  710. loopx/extensions/readiness.py +168 -0
  711. loopx/extensions/runtime.py +931 -0
  712. loopx/extensions/scaffold.py +335 -0
  713. loopx/feedback.py +581 -0
  714. loopx/file_lock.py +382 -0
  715. loopx/global_registry.py +842 -0
  716. loopx/global_risks.py +970 -0
  717. loopx/global_todos.py +568 -0
  718. loopx/handoff_budget.py +28 -0
  719. loopx/heartbeat_prequota.py +80 -0
  720. loopx/heartbeat_prompt.py +159 -0
  721. loopx/help_surface.py +516 -0
  722. loopx/history.py +1507 -0
  723. loopx/host_loop_activation.py +1311 -0
  724. loopx/host_mode_planner.py +991 -0
  725. loopx/install_contract.py +1 -0
  726. loopx/interface_budget.py +196 -0
  727. loopx/long_task_cadence.py +208 -0
  728. loopx/materials.py +185 -0
  729. loopx/ml_experiment.py +3 -0
  730. loopx/onboarding.py +214 -0
  731. loopx/opencode2_goal_mode/README.md +81 -0
  732. loopx/opencode2_goal_mode/__init__.py +9 -0
  733. loopx/opencode2_goal_mode/opencode2-goal-worker.mjs +1018 -0
  734. loopx/opencode_goal_mode/README.md +99 -0
  735. loopx/opencode_goal_mode/__init__.py +13 -0
  736. loopx/opencode_goal_mode/goal-bridge-runtime.mjs +858 -0
  737. loopx/opencode_goal_mode/loopx-goal.js +8 -0
  738. loopx/operator_gate.py +420 -0
  739. loopx/orchestration.py +127 -0
  740. loopx/paths.py +59 -0
  741. loopx/pi_goal_mode/README.md +67 -0
  742. loopx/pi_goal_mode/__init__.py +13 -0
  743. loopx/pi_goal_mode/loopx-goal.ts +254 -0
  744. loopx/pi_goal_mode/pi-goal-loop-runtime.mjs +574 -0
  745. loopx/pr_review.py +1206 -0
  746. loopx/presentation/__init__.py +1 -0
  747. loopx/presentation/explore_views.py +1334 -0
  748. loopx/presentation/markdown.py +61 -0
  749. loopx/presentation/projection_source_reconcile.py +140 -0
  750. loopx/presentation/public_safety.py +42 -0
  751. loopx/presentation/renderers/__init__.py +17 -0
  752. loopx/presentation/renderers/goal_channel_html.py +269 -0
  753. loopx/presentation/renderers/periodic_report_html.py +786 -0
  754. loopx/presentation/renderers/periodic_report_markdown.py +184 -0
  755. loopx/presentation/renderers/quota_event_markdown.py +116 -0
  756. loopx/presentation/renderers/quota_markdown.py +1112 -0
  757. loopx/presentation/renderers/status_markdown.py +1570 -0
  758. loopx/presentation/renderers/trajectory_hygiene_markdown.py +39 -0
  759. loopx/presentation/renderers/turn_envelope_markdown.py +33 -0
  760. loopx/presentation/sinks/__init__.py +5 -0
  761. loopx/presentation/sinks/openviking_periodic_report.py +7 -0
  762. loopx/presentation/static_site.py +691 -0
  763. loopx/presets.py +369 -0
  764. loopx/project_alias.py +217 -0
  765. loopx/project_map.py +589 -0
  766. loopx/project_prompt.py +1153 -0
  767. loopx/project_skill_cli.py +125 -0
  768. loopx/project_skill_delivery.py +470 -0
  769. loopx/project_uninstall.py +462 -0
  770. loopx/promotion_gate.py +197 -0
  771. loopx/quota.py +1197 -0
  772. loopx/ready_score.py +413 -0
  773. loopx/registry.py +621 -0
  774. loopx/registry_writability.py +64 -0
  775. loopx/release_candidate.py +148 -0
  776. loopx/release_manifest.py +316 -0
  777. loopx/repository_identity.py +100 -0
  778. loopx/review_packet.py +1024 -0
  779. loopx/rollout_event_log.py +505 -0
  780. loopx/runtime.py +112 -0
  781. loopx/self_update.py +750 -0
  782. loopx/session_runtime.py +418 -0
  783. loopx/skill_install_readback.py +500 -0
  784. loopx/slash_command_install.py +1393 -0
  785. loopx/slash_commands.py +264 -0
  786. loopx/state_backup.py +573 -0
  787. loopx/state_migration.py +350 -0
  788. loopx/state_projection.py +809 -0
  789. loopx/state_refresh.py +1416 -0
  790. loopx/status.py +1383 -0
  791. loopx/status_server.py +935 -0
  792. loopx/summary_all.py +725 -0
  793. loopx/terminal_bench_agent.py +2056 -0
  794. loopx/thread_agent_binding.py +408 -0
  795. loopx/todo_followups.py +168 -0
  796. loopx/todo_suggestion_prompt.py +204 -0
  797. loopx/todos.py +2229 -0
  798. loopx/turn_identity.py +17 -0
  799. loopx/upgrade.py +1083 -0
  800. loopx/visible_governance.py +667 -0
  801. loopx/visible_multi_agent_launcher.py +1253 -0
  802. loopx/visible_multi_agent_tmux.py +429 -0
  803. loopx/worker_bridge.py +1574 -0
  804. loopx-0.4.8.dist-info/METADATA +708 -0
  805. loopx-0.4.8.dist-info/RECORD +811 -0
  806. loopx-0.4.8.dist-info/WHEEL +5 -0
  807. loopx-0.4.8.dist-info/entry_points.txt +5 -0
  808. loopx-0.4.8.dist-info/licenses/LICENSE +202 -0
  809. loopx-0.4.8.dist-info/licenses/LICENSE-MIT +21 -0
  810. loopx-0.4.8.dist-info/licenses/NOTICE +6 -0
  811. loopx-0.4.8.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1275 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from collections.abc import Callable
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ from ..benchmark import (
11
+ build_benchmark_adapter_kwarg_absorption_review,
12
+ build_benchmark_attempt_learning_gate,
13
+ build_benchmark_baseline_failure_gate_comparison,
14
+ build_benchmark_claim_review,
15
+ build_benchmark_learning_ledger,
16
+ build_benchmark_lifecycle_state,
17
+ build_benchmark_runner_invariant_review,
18
+ build_benchmark_verifier_attribution_review,
19
+ benchmark_result_from_benchmark_run_for_baseline_gate,
20
+ )
21
+ from ..benchmark_adapters.terminal_bench import (
22
+ TERMINAL_BENCH_MANAGED_CODEX_LOOPX_KWARGS,
23
+ agent_kwargs_from_invocation,
24
+ )
25
+ from ..benchmarks.read_models.benchmark_comparison import (
26
+ compact_benchmark_comparison,
27
+ )
28
+ from ..benchmarks.read_models.benchmark_learning_ledger import (
29
+ compact_benchmark_learning_ledger,
30
+ )
31
+ from ..benchmarks.read_models.benchmark_result import compact_benchmark_result
32
+ from ..control_plane.work_items.delivery_batch_scale import DELIVERY_BATCH_SCALE_CHOICES
33
+ from ..control_plane.work_items.delivery_outcome import DELIVERY_OUTCOME_CHOICES
34
+ from ..global_registry import sync_project_registry_to_global
35
+ from ..history import append_benchmark_comparison
36
+ from ..status import compact_benchmark_run
37
+
38
+
39
+ PrintPayload = Callable[
40
+ [dict[str, object], str, Callable[[dict[str, object]], str]],
41
+ None,
42
+ ]
43
+ OutputFormat = Callable[[argparse.Namespace], str]
44
+
45
+ BENCHMARK_REVIEW_LIFECYCLE_COMMANDS = {
46
+ "review-claim",
47
+ "learning-ledger",
48
+ "attempt-learning-gate",
49
+ "baseline-failure-gate",
50
+ "review-adapter-kwargs",
51
+ "lifecycle-state",
52
+ "review-runner-invariants",
53
+ "review-verifier-attribution",
54
+ }
55
+
56
+
57
+ def render_benchmark_claim_review_markdown(payload: dict[str, object]) -> str:
58
+ decision = payload.get("decision") if isinstance(payload.get("decision"), dict) else {}
59
+ treatment = (
60
+ payload.get("treatment_worker_evidence")
61
+ if isinstance(payload.get("treatment_worker_evidence"), dict)
62
+ else {}
63
+ )
64
+ read_boundary = (
65
+ payload.get("read_boundary")
66
+ if isinstance(payload.get("read_boundary"), dict)
67
+ else {}
68
+ )
69
+ lines = [
70
+ "# Benchmark Claim Review",
71
+ "",
72
+ f"- Schema: `{payload.get('schema_version')}`",
73
+ f"- Task: `{payload.get('task_id')}`",
74
+ f"- Comparison: `{payload.get('comparison_id')}`",
75
+ f"- Official delta: `{payload.get('official_task_score_delta')}`",
76
+ f"- Claim strength: `{decision.get('claim_strength')}`",
77
+ f"- Validation candidate: `{decision.get('validation_enhancement_candidate')}`",
78
+ f"- Clean validation: `{decision.get('clean_validation_enhancement')}`",
79
+ f"- Blockers: `{decision.get('blockers')}`",
80
+ f"- Next action: {decision.get('next_action')}",
81
+ f"- Treatment worker GH calls: `{treatment.get('worker_loopx_cli_call_total')}`",
82
+ f"- Baseline attribution: `{payload.get('baseline_score_failure_attribution')}`",
83
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
84
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
85
+ ]
86
+ return "\n".join(lines) + "\n"
87
+
88
+
89
+ def render_benchmark_learning_ledger_markdown(payload: dict[str, object]) -> str:
90
+ lifecycle_gate = (
91
+ payload.get("lifecycle_gate")
92
+ if isinstance(payload.get("lifecycle_gate"), dict)
93
+ else {}
94
+ )
95
+ learning_quota_gate = (
96
+ payload.get("learning_quota_gate")
97
+ if isinstance(payload.get("learning_quota_gate"), dict)
98
+ else {}
99
+ )
100
+ routing = (
101
+ payload.get("routing") if isinstance(payload.get("routing"), dict) else {}
102
+ )
103
+ overhead = (
104
+ payload.get("overhead") if isinstance(payload.get("overhead"), dict) else {}
105
+ )
106
+ read_boundary = (
107
+ payload.get("read_boundary")
108
+ if isinstance(payload.get("read_boundary"), dict)
109
+ else {}
110
+ )
111
+ lines = [
112
+ "# Benchmark Learning Ledger",
113
+ "",
114
+ f"- Schema: `{payload.get('schema_version')}`",
115
+ f"- Task: `{payload.get('task_id')}`",
116
+ f"- Comparison: `{payload.get('comparison_id')}`",
117
+ f"- Official delta: `{payload.get('official_task_score_delta')}`",
118
+ f"- Learning status: `{payload.get('learning_status')}`",
119
+ f"- Claim strength: `{payload.get('claim_strength')}`",
120
+ f"- Repair candidates: `{payload.get('repair_candidates')}`",
121
+ f"- Claim blockers: `{payload.get('claim_blockers')}`",
122
+ f"- Budget count allowed: `{lifecycle_gate.get('budget_count_allowed')}`",
123
+ f"- Learning spend allowed: `{learning_quota_gate.get('spend_allowed')}`",
124
+ f"- Actionable reasons: `{learning_quota_gate.get('actionable_reasons')}`",
125
+ f"- Overhead label: `{overhead.get('label')}`",
126
+ f"- Repeat allowed: `{routing.get('repeat_allowed')}`",
127
+ f"- New candidate allowed: `{routing.get('new_candidate_allowed')}`",
128
+ f"- Next action: {routing.get('next_allowed_action')}",
129
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
130
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
131
+ ]
132
+ return "\n".join(lines) + "\n"
133
+
134
+
135
+ def render_benchmark_attempt_learning_gate_markdown(
136
+ payload: dict[str, object],
137
+ ) -> str:
138
+ read_boundary = (
139
+ payload.get("read_boundary")
140
+ if isinstance(payload.get("read_boundary"), dict)
141
+ else {}
142
+ )
143
+ lines = [
144
+ "# Benchmark Attempt Learning Gate",
145
+ "",
146
+ f"- Schema: `{payload.get('schema_version')}`",
147
+ f"- Benchmark: `{payload.get('benchmark_id')}`",
148
+ f"- Mode: `{payload.get('mode')}`",
149
+ f"- Classification: `{payload.get('classification')}`",
150
+ f"- Countable attempt: `{payload.get('countable_attempt')}`",
151
+ f"- Learning row present: `{payload.get('learning_row_present')}`",
152
+ f"- Learning row actionable: `{payload.get('learning_row_actionable')}`",
153
+ f"- Budget count allowed: `{payload.get('budget_count_allowed')}`",
154
+ f"- Repair candidates: `{payload.get('repair_candidates')}`",
155
+ f"- Next action: {payload.get('next_required_action')}",
156
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
157
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
158
+ ]
159
+ return "\n".join(lines) + "\n"
160
+
161
+
162
+ def render_benchmark_adapter_kwarg_absorption_review_markdown(
163
+ payload: dict[str, object],
164
+ ) -> str:
165
+ read_boundary = (
166
+ payload.get("read_boundary")
167
+ if isinstance(payload.get("read_boundary"), dict)
168
+ else {}
169
+ )
170
+ lines = [
171
+ "# Benchmark Adapter Kwarg Absorption Review",
172
+ "",
173
+ f"- Schema: `{payload.get('schema_version')}`",
174
+ f"- Adapter: `{payload.get('adapter_label')}`",
175
+ f"- Classification: `{payload.get('classification')}`",
176
+ f"- Clean: `{payload.get('clean')}`",
177
+ f"- Generated GH kwargs: `{payload.get('generated_loopx_kwarg_count')}`",
178
+ f"- Absorbed GH kwargs: `{payload.get('absorbed_loopx_kwarg_count')}`",
179
+ f"- Leaked GH kwargs: `{payload.get('leaked_loopx_kwarg_count')}`",
180
+ f"- Leaked keys: `{payload.get('leaked_loopx_kwarg_keys')}`",
181
+ f"- Next action: {payload.get('next_required_action')}",
182
+ f"- Kwarg values recorded: `{(payload.get('claim_boundary') or {}).get('kwarg_values_recorded') if isinstance(payload.get('claim_boundary'), dict) else None}`",
183
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
184
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
185
+ ]
186
+ return "\n".join(lines) + "\n"
187
+
188
+
189
+ def render_benchmark_lifecycle_state_markdown(payload: dict[str, object]) -> str:
190
+ gates = payload.get("gates") if isinstance(payload.get("gates"), dict) else {}
191
+ setup = (
192
+ payload.get("environment_setup_readiness_preflight")
193
+ if isinstance(payload.get("environment_setup_readiness_preflight"), dict)
194
+ else {}
195
+ )
196
+ read_boundary = (
197
+ payload.get("read_boundary")
198
+ if isinstance(payload.get("read_boundary"), dict)
199
+ else {}
200
+ )
201
+ return "\n".join(
202
+ [
203
+ "# Benchmark Lifecycle State",
204
+ "",
205
+ f"- Schema: `{payload.get('schema_version')}`",
206
+ f"- Current phase: `{payload.get('current_phase')}`",
207
+ f"- First blocker: `{payload.get('first_blocker')}`",
208
+ f"- Next transition: `{payload.get('next_required_transition')}`",
209
+ f"- Launch state countable: `{gates.get('launch_state_countable')}`",
210
+ f"- Compact ingest allowed: `{gates.get('compact_result_ingest_allowed')}`",
211
+ f"- Budget count allowed: `{gates.get('budget_count_allowed')}`",
212
+ f"- New candidate allowed: `{gates.get('new_candidate_allowed')}`",
213
+ f"- Repeat allowed: `{gates.get('repeat_allowed')}`",
214
+ "- Environment setup repeat allowed: "
215
+ f"`{gates.get('environment_setup_repeat_allowed')}`",
216
+ "- Environment setup next action: "
217
+ f"{setup.get('next_allowed_action') or ''}",
218
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
219
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
220
+ ]
221
+ ) + "\n"
222
+
223
+
224
+ def render_benchmark_verifier_attribution_review_markdown(
225
+ payload: dict[str, object],
226
+ ) -> str:
227
+ decision = (
228
+ payload.get("decision") if isinstance(payload.get("decision"), dict) else {}
229
+ )
230
+ routing = (
231
+ payload.get("routing") if isinstance(payload.get("routing"), dict) else {}
232
+ )
233
+ read_boundary = (
234
+ payload.get("read_boundary")
235
+ if isinstance(payload.get("read_boundary"), dict)
236
+ else {}
237
+ )
238
+ lines = [
239
+ "# Benchmark Verifier Attribution Review",
240
+ "",
241
+ f"- Schema: `{payload.get('schema_version')}`",
242
+ f"- Reviewed runs: `{payload.get('reviewed_run_count')}`",
243
+ f"- Baseline index: `{payload.get('baseline_run_index')}`",
244
+ "- Baseline caveat resolved: "
245
+ f"`{decision.get('baseline_claim_caveat_resolved')}`",
246
+ f"- Clean model attribution: `{decision.get('clean_model_failure_attribution')}`",
247
+ f"- Blockers: `{decision.get('blockers')}`",
248
+ f"- Next action: {decision.get('next_action')}",
249
+ f"- Treatment eligible: `{routing.get('treatment_eligible')}`",
250
+ f"- Repeat allowed: `{routing.get('repeat_allowed')}`",
251
+ f"- New candidate allowed: `{routing.get('new_candidate_allowed')}`",
252
+ f"- Routing action: {routing.get('next_allowed_action')}",
253
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
254
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
255
+ ]
256
+ return "\n".join(lines) + "\n"
257
+
258
+
259
+ def render_benchmark_runner_invariant_review_markdown(
260
+ payload: dict[str, object],
261
+ ) -> str:
262
+ read_boundary = (
263
+ payload.get("read_boundary")
264
+ if isinstance(payload.get("read_boundary"), dict)
265
+ else {}
266
+ )
267
+ lines = [
268
+ "# Benchmark Runner Invariant Review",
269
+ "",
270
+ f"- Schema: `{payload.get('schema_version')}`",
271
+ f"- Benchmark: `{payload.get('benchmark_id')}`",
272
+ f"- Mode: `{payload.get('mode')}`",
273
+ f"- Runner label: `{payload.get('runner_label')}`",
274
+ f"- Classification: `{payload.get('classification')}`",
275
+ f"- Clean: `{payload.get('clean')}`",
276
+ f"- Mismatches: `{payload.get('mismatch_count')}`",
277
+ f"- Missing fields: `{payload.get('missing_field_count')}`",
278
+ f"- Repair: {payload.get('repair_recommendation')}",
279
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
280
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
281
+ ]
282
+ return "\n".join(lines) + "\n"
283
+
284
+
285
+ def render_benchmark_baseline_failure_gate_markdown(payload: dict[str, object]) -> str:
286
+ comparison = (
287
+ payload.get("benchmark_comparison")
288
+ if isinstance(payload.get("benchmark_comparison"), dict)
289
+ else {}
290
+ )
291
+ gate = (
292
+ comparison.get("baseline_failure_gate")
293
+ if isinstance(comparison.get("baseline_failure_gate"), dict)
294
+ else {}
295
+ )
296
+ lines = [
297
+ "# Benchmark Baseline Failure Gate",
298
+ "",
299
+ f"- ok: `{payload.get('ok')}`",
300
+ f"- dry_run: `{payload.get('dry_run')}`",
301
+ f"- appended: `{payload.get('appended')}`",
302
+ f"- goal_id: `{payload.get('goal_id')}`",
303
+ f"- benchmark_id: `{comparison.get('benchmark_id')}`",
304
+ f"- task_id: `{comparison.get('task_id')}`",
305
+ f"- baseline_failed: `{gate.get('baseline_failed')}`",
306
+ f"- control_plane_addressable: `{gate.get('control_plane_addressable')}`",
307
+ f"- treatment_eligible: `{gate.get('treatment_eligible')}`",
308
+ f"- failure_class: `{gate.get('failure_class')}`",
309
+ ]
310
+ if gate.get("minimum_next_evidence"):
311
+ lines.append(f"- minimum_next_evidence: {gate.get('minimum_next_evidence')}")
312
+ if gate.get("negative_selection_reason"):
313
+ lines.append(
314
+ f"- negative_selection_reason: {gate.get('negative_selection_reason')}"
315
+ )
316
+ if payload.get("error"):
317
+ lines.append(f"- error: {payload.get('error')}")
318
+ return "\n".join(lines) + "\n"
319
+
320
+
321
+
322
+ def register_benchmark_review_lifecycle_commands(
323
+ benchmark_subparsers: argparse._SubParsersAction,
324
+ add_subcommand_format: Callable[[argparse.ArgumentParser], None],
325
+ ) -> None:
326
+ benchmark_claim_review_parser = benchmark_subparsers.add_parser(
327
+ "review-claim",
328
+ help=(
329
+ "Review compact benchmark comparison/run JSON and classify claim strength. "
330
+ "This reads only compact JSON inputs, not raw task text, logs, traces, "
331
+ "Harbor job directories, Docker, model APIs, or uploads."
332
+ ),
333
+ )
334
+ add_subcommand_format(benchmark_claim_review_parser)
335
+ benchmark_claim_review_parser.add_argument(
336
+ "--benchmark-comparison-json",
337
+ required=True,
338
+ help="Path to a compact benchmark_comparison_v0 JSON object.",
339
+ )
340
+ benchmark_claim_review_parser.add_argument(
341
+ "--benchmark-run-json",
342
+ action="append",
343
+ default=[],
344
+ help=(
345
+ "Path to a compact benchmark_run_v0 JSON object. Repeat for baseline "
346
+ "and treatment compact run files."
347
+ ),
348
+ )
349
+
350
+ benchmark_learning_ledger_parser = benchmark_subparsers.add_parser(
351
+ "learning-ledger",
352
+ help=(
353
+ "Build a compact benchmark learning ledger row from comparison/run "
354
+ "JSON. This turns paired outcomes into repair-vs-repeat guidance "
355
+ "without opening raw task text, logs, traces, Harbor job directories, "
356
+ "Docker, model APIs, or uploads."
357
+ ),
358
+ )
359
+ add_subcommand_format(benchmark_learning_ledger_parser)
360
+ benchmark_learning_ledger_parser.add_argument(
361
+ "--benchmark-comparison-json",
362
+ required=True,
363
+ help="Path to a compact benchmark_comparison_v0 JSON object.",
364
+ )
365
+ benchmark_learning_ledger_parser.add_argument(
366
+ "--benchmark-run-json",
367
+ action="append",
368
+ default=[],
369
+ help=(
370
+ "Path to a compact benchmark_run_v0 JSON object. Repeat for baseline "
371
+ "and treatment compact run files."
372
+ ),
373
+ )
374
+ benchmark_learning_ledger_parser.add_argument(
375
+ "--require-actionable-learning",
376
+ action="store_true",
377
+ help=(
378
+ "Return non-zero unless the compact ledger contains an actionable "
379
+ "LoopX learning signal, such as a repair candidate or clean "
380
+ "score-recovery evidence."
381
+ ),
382
+ )
383
+
384
+ benchmark_attempt_learning_gate_parser = benchmark_subparsers.add_parser(
385
+ "attempt-learning-gate",
386
+ help=(
387
+ "Gate benchmark budget counting and follow-up routing on a durable "
388
+ "compact learning ledger row. This reads only compact JSON, not raw "
389
+ "task text, logs, traces, Harbor job directories, Docker, model APIs, "
390
+ "uploads, screenshots, or credentials."
391
+ ),
392
+ )
393
+ add_subcommand_format(benchmark_attempt_learning_gate_parser)
394
+ benchmark_attempt_learning_gate_parser.add_argument(
395
+ "--benchmark-run-json",
396
+ required=True,
397
+ help="Path to a compact benchmark_run_v0 JSON object.",
398
+ )
399
+ benchmark_attempt_learning_gate_parser.add_argument(
400
+ "--benchmark-learning-ledger-json",
401
+ help="Optional path to a compact benchmark_learning_ledger_v0 JSON object.",
402
+ )
403
+ benchmark_attempt_learning_gate_parser.add_argument(
404
+ "--require-budget-count-allowed",
405
+ action="store_true",
406
+ help=(
407
+ "Return non-zero unless the compact attempt has an actionable "
408
+ "learning row and may be counted."
409
+ ),
410
+ )
411
+
412
+ benchmark_adapter_kwarg_review_parser = benchmark_subparsers.add_parser(
413
+ "review-adapter-kwargs",
414
+ help=(
415
+ "Review generated benchmark adapter kwargs and flag loopx_* "
416
+ "keys that are not absorbed by the adapter contract. Values are not "
417
+ "recorded. This does not start workers, Docker, model APIs, uploads, "
418
+ "or read task material."
419
+ ),
420
+ )
421
+ add_subcommand_format(benchmark_adapter_kwarg_review_parser)
422
+ benchmark_adapter_kwarg_review_parser.add_argument(
423
+ "--adapter-label",
424
+ default="benchmark-adapter",
425
+ help="Public-safe adapter label.",
426
+ )
427
+ benchmark_adapter_kwarg_review_parser.add_argument(
428
+ "--agent-kwarg",
429
+ action="append",
430
+ default=[],
431
+ help="Generated agent kwarg in KEY=VALUE form. Repeat as needed.",
432
+ )
433
+ benchmark_adapter_kwarg_review_parser.add_argument(
434
+ "--command-json",
435
+ help=(
436
+ "Optional JSON file containing a command argv list from which "
437
+ "--agent-kwarg entries will be extracted. The path is not recorded."
438
+ ),
439
+ )
440
+ benchmark_adapter_kwarg_review_parser.add_argument(
441
+ "--accepted-loopx-kwarg",
442
+ action="append",
443
+ default=[],
444
+ help=(
445
+ "LoopX kwarg key explicitly consumed by the adapter. Repeat "
446
+ "as needed unless --terminal-bench-managed-codex is used."
447
+ ),
448
+ )
449
+ benchmark_adapter_kwarg_review_parser.add_argument(
450
+ "--allowed-base-passthrough",
451
+ action="append",
452
+ default=[],
453
+ help="Optional loopx_* kwarg key allowed to pass to the base constructor.",
454
+ )
455
+ benchmark_adapter_kwarg_review_parser.add_argument(
456
+ "--terminal-bench-managed-codex",
457
+ action="store_true",
458
+ help="Use the built-in GoalHarnessManagedCodex accepted kwarg contract.",
459
+ )
460
+ benchmark_adapter_kwarg_review_parser.add_argument(
461
+ "--require-clean",
462
+ action="store_true",
463
+ help="Return non-zero unless all generated loopx_* kwargs are absorbed.",
464
+ )
465
+
466
+ benchmark_lifecycle_state_parser = benchmark_subparsers.add_parser(
467
+ "lifecycle-state",
468
+ help=(
469
+ "Reduce compact benchmark preflight/launch/materialization/result/"
470
+ "comparison/ledger JSON into an explicit lifecycle state without "
471
+ "opening raw task text, logs, traces, job directories, Docker, model "
472
+ "APIs, or uploads."
473
+ ),
474
+ )
475
+ add_subcommand_format(benchmark_lifecycle_state_parser)
476
+ benchmark_lifecycle_state_parser.add_argument(
477
+ "--preflight-json",
478
+ help="Path to compact benchmark preflight JSON.",
479
+ )
480
+ benchmark_lifecycle_state_parser.add_argument(
481
+ "--launch-json",
482
+ help="Path to compact launch summary JSON.",
483
+ )
484
+ benchmark_lifecycle_state_parser.add_argument(
485
+ "--post-launch-json",
486
+ help="Path to compact post-launch materialization JSON.",
487
+ )
488
+ benchmark_lifecycle_state_parser.add_argument(
489
+ "--benchmark-run-json",
490
+ help="Path to compact benchmark_run_v0 JSON.",
491
+ )
492
+ benchmark_lifecycle_state_parser.add_argument(
493
+ "--benchmark-comparison-json",
494
+ help="Path to compact benchmark_comparison_v0 JSON.",
495
+ )
496
+ benchmark_lifecycle_state_parser.add_argument(
497
+ "--claim-review-json",
498
+ help="Path to compact benchmark_claim_review_v0 JSON.",
499
+ )
500
+ benchmark_lifecycle_state_parser.add_argument(
501
+ "--benchmark-learning-ledger-json",
502
+ help="Path to compact benchmark_learning_ledger_v0 JSON.",
503
+ )
504
+ benchmark_lifecycle_state_parser.add_argument(
505
+ "--require-budget-count-allowed",
506
+ action="store_true",
507
+ help="Return non-zero unless the lifecycle state allows budget counting.",
508
+ )
509
+
510
+ benchmark_baseline_gate_parser = benchmark_subparsers.add_parser(
511
+ "baseline-failure-gate",
512
+ help=(
513
+ "Reduce a compact goal-mode baseline benchmark_result_v0 or "
514
+ "benchmark_run_v0 into a benchmark_comparison_v0 baseline-failure "
515
+ "gate. This reads only compact JSON, not raw task text, logs, "
516
+ "traces, Harbor job directories, Docker, model APIs, uploads, "
517
+ "screenshots, or credentials."
518
+ ),
519
+ )
520
+ add_subcommand_format(benchmark_baseline_gate_parser)
521
+ benchmark_baseline_gate_parser.add_argument(
522
+ "--goal-id",
523
+ help="Goal id for optional append context. Required with --execute.",
524
+ )
525
+ benchmark_baseline_gate_parser.add_argument(
526
+ "--benchmark-id",
527
+ required=True,
528
+ help="Public-safe benchmark id for the comparison row.",
529
+ )
530
+ benchmark_baseline_gate_parser.add_argument(
531
+ "--baseline-result-json",
532
+ required=True,
533
+ help=(
534
+ "Path to a compact benchmark_result_v0 or benchmark_run_v0 JSON object. "
535
+ "Use '-' to read stdin."
536
+ ),
537
+ )
538
+ benchmark_baseline_gate_parser.add_argument(
539
+ "--baseline-mode",
540
+ default="codex_cli_goal_mode",
541
+ help="Public-safe baseline mode label.",
542
+ )
543
+ benchmark_baseline_gate_parser.add_argument(
544
+ "--treatment-scenario-id",
545
+ default="codex_loopx",
546
+ help="Public-safe treatment scenario id planned after the gate.",
547
+ )
548
+ benchmark_baseline_gate_parser.add_argument("--comparison-id")
549
+ benchmark_baseline_gate_parser.add_argument("--failure-phase")
550
+ benchmark_baseline_gate_parser.add_argument("--failure-class")
551
+ benchmark_baseline_gate_parser.add_argument(
552
+ "--failure-attribution-label",
553
+ action="append",
554
+ default=[],
555
+ help="Public-safe failure attribution label. Repeat as needed.",
556
+ )
557
+ benchmark_baseline_gate_parser.add_argument(
558
+ "--control-plane-addressable",
559
+ action="store_true",
560
+ help=(
561
+ "Mark the baseline failure as plausibly fixable by LoopX "
562
+ "control-plane intervention."
563
+ ),
564
+ )
565
+ benchmark_baseline_gate_parser.add_argument(
566
+ "--same-task-semantics",
567
+ action="store_true",
568
+ help="Confirm treatment will use the same benchmark task semantics.",
569
+ )
570
+ benchmark_baseline_gate_parser.add_argument(
571
+ "--same-runner-protocol",
572
+ action="store_true",
573
+ help="Confirm treatment will use the same runner protocol.",
574
+ )
575
+ benchmark_baseline_gate_parser.add_argument(
576
+ "--trace-publicness-verified",
577
+ action="store_true",
578
+ help="Confirm the compact result excludes private raw trace material.",
579
+ )
580
+ benchmark_baseline_gate_parser.add_argument(
581
+ "--baseline-attempt-count",
582
+ type=int,
583
+ default=1,
584
+ help="Number of baseline attempts represented by this gate.",
585
+ )
586
+ benchmark_baseline_gate_parser.add_argument("--minimum-next-evidence")
587
+ benchmark_baseline_gate_parser.add_argument("--negative-selection-reason")
588
+ benchmark_baseline_gate_parser.add_argument("--next-action")
589
+ benchmark_baseline_gate_parser.add_argument(
590
+ "--evidence-ref",
591
+ action="append",
592
+ default=[],
593
+ help="Public-safe compact evidence reference. Repeat as needed.",
594
+ )
595
+ benchmark_baseline_gate_parser.add_argument("--classification")
596
+ benchmark_baseline_gate_parser.add_argument("--recommended-action")
597
+ benchmark_baseline_gate_parser.add_argument(
598
+ "--delivery-batch-scale",
599
+ choices=DELIVERY_BATCH_SCALE_CHOICES,
600
+ help="Optional delivery scale label for the run index.",
601
+ )
602
+ benchmark_baseline_gate_parser.add_argument(
603
+ "--delivery-outcome",
604
+ choices=DELIVERY_OUTCOME_CHOICES,
605
+ help="Optional delivery outcome label for the run index.",
606
+ )
607
+ benchmark_baseline_gate_parser.add_argument(
608
+ "--dry-run",
609
+ action="store_true",
610
+ help="Preview without writing. This is the default.",
611
+ )
612
+ benchmark_baseline_gate_parser.add_argument(
613
+ "--execute",
614
+ action="store_true",
615
+ help="Append the compact baseline gate comparison.",
616
+ )
617
+ benchmark_baseline_gate_parser.add_argument(
618
+ "--no-global-sync",
619
+ action="store_true",
620
+ help="Skip global registry sync after append.",
621
+ )
622
+
623
+ benchmark_verifier_attribution_parser = benchmark_subparsers.add_parser(
624
+ "review-verifier-attribution",
625
+ help=(
626
+ "Review compact benchmark_run_v0 verifier attribution and decide "
627
+ "whether a score-failure caveat is resolved without opening raw logs, "
628
+ "task text, traces, Harbor job directories, Docker, model APIs, or uploads."
629
+ ),
630
+ )
631
+ add_subcommand_format(benchmark_verifier_attribution_parser)
632
+ benchmark_verifier_attribution_parser.add_argument(
633
+ "--benchmark-run-json",
634
+ action="append",
635
+ required=True,
636
+ help=(
637
+ "Path to a compact benchmark_run_v0 JSON object. Repeat for baseline "
638
+ "and treatment compact run files."
639
+ ),
640
+ )
641
+
642
+ benchmark_runner_invariant_parser = benchmark_subparsers.add_parser(
643
+ "review-runner-invariants",
644
+ help=(
645
+ "Review compact benchmark_run_v0 runner-owned boundary invariants "
646
+ "before trusting worker writeback. This reads only compact JSON, not "
647
+ "raw task text, logs, traces, Harbor job directories, Docker, model "
648
+ "APIs, uploads, or screenshots."
649
+ ),
650
+ )
651
+ add_subcommand_format(benchmark_runner_invariant_parser)
652
+ benchmark_runner_invariant_parser.add_argument(
653
+ "--benchmark-run-json",
654
+ required=True,
655
+ help="Path to a compact benchmark_run_v0 JSON object.",
656
+ )
657
+ benchmark_runner_invariant_parser.add_argument(
658
+ "--runner-label",
659
+ help="Public-safe runner label to include in the review payload.",
660
+ )
661
+ benchmark_runner_invariant_parser.add_argument(
662
+ "--expect-submit-eligible",
663
+ choices=["true", "false"],
664
+ default="false",
665
+ help="Expected runner-owned submit_eligible value. Defaults to false.",
666
+ )
667
+ benchmark_runner_invariant_parser.add_argument(
668
+ "--expect-leaderboard-evidence",
669
+ choices=["true", "false"],
670
+ default="false",
671
+ help="Expected runner-owned leaderboard_evidence value. Defaults to false.",
672
+ )
673
+ benchmark_runner_invariant_parser.add_argument(
674
+ "--expect-compact-only",
675
+ choices=["true", "false"],
676
+ default="true",
677
+ help="Expected compact read boundary. Defaults to true.",
678
+ )
679
+ benchmark_runner_invariant_parser.add_argument(
680
+ "--expect-raw-artifacts-read",
681
+ choices=["true", "false"],
682
+ default="false",
683
+ help="Expected raw artifact read boundary. Defaults to false.",
684
+ )
685
+ benchmark_runner_invariant_parser.add_argument(
686
+ "--expect-task-text-read",
687
+ choices=["true", "false"],
688
+ default="false",
689
+ help="Expected task text read boundary. Defaults to false.",
690
+ )
691
+ benchmark_runner_invariant_parser.add_argument(
692
+ "--expect-local-paths-recorded",
693
+ choices=["true", "false"],
694
+ default="false",
695
+ help="Expected local path recording boundary. Defaults to false.",
696
+ )
697
+ benchmark_runner_invariant_parser.add_argument(
698
+ "--require-clean",
699
+ action="store_true",
700
+ help="Return non-zero unless all runner-owned invariant fields match.",
701
+ )
702
+
703
+
704
+
705
+ def handle_benchmark_review_lifecycle_command(
706
+ args: argparse.Namespace,
707
+ *,
708
+ registry_path: Path | None = None,
709
+ print_payload: PrintPayload,
710
+ output_format: OutputFormat,
711
+ ) -> int | None:
712
+ if args.benchmark_command not in BENCHMARK_REVIEW_LIFECYCLE_COMMANDS:
713
+ return None
714
+
715
+ if args.benchmark_command == "review-claim":
716
+ try:
717
+ comparison_input = json.loads(
718
+ Path(args.benchmark_comparison_json).expanduser().read_text(encoding="utf-8")
719
+ )
720
+ if not isinstance(comparison_input, dict):
721
+ raise ValueError("--benchmark-comparison-json must contain a JSON object")
722
+ comparison = compact_benchmark_comparison(comparison_input)
723
+ if not comparison:
724
+ raise ValueError(
725
+ "--benchmark-comparison-json did not contain a compactable benchmark_comparison_v0 object"
726
+ )
727
+ runs = []
728
+ for run_json in args.benchmark_run_json:
729
+ run_input = json.loads(
730
+ Path(run_json).expanduser().read_text(encoding="utf-8")
731
+ )
732
+ if not isinstance(run_input, dict):
733
+ raise ValueError("--benchmark-run-json must contain JSON objects")
734
+ run = compact_benchmark_run(run_input)
735
+ if not run:
736
+ raise ValueError(
737
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
738
+ )
739
+ runs.append(run)
740
+ payload = build_benchmark_claim_review(
741
+ comparison,
742
+ benchmark_runs=runs,
743
+ )
744
+ except Exception as exc:
745
+ payload = {
746
+ "ok": False,
747
+ "schema_version": "benchmark_claim_review_v0",
748
+ "error": str(exc),
749
+ "read_boundary": {
750
+ "compact_only": True,
751
+ "raw_artifacts_read": False,
752
+ "task_text_read": False,
753
+ "local_paths_recorded": False,
754
+ },
755
+ }
756
+ else:
757
+ payload["ok"] = True
758
+ print_payload(
759
+ payload,
760
+ output_format(args),
761
+ render_benchmark_claim_review_markdown,
762
+ )
763
+ return 0 if payload.get("ok") else 1
764
+ if args.benchmark_command == "learning-ledger":
765
+ try:
766
+ comparison_input = json.loads(
767
+ Path(args.benchmark_comparison_json).expanduser().read_text(encoding="utf-8")
768
+ )
769
+ if not isinstance(comparison_input, dict):
770
+ raise ValueError("--benchmark-comparison-json must contain a JSON object")
771
+ comparison = compact_benchmark_comparison(comparison_input)
772
+ if not comparison:
773
+ raise ValueError(
774
+ "--benchmark-comparison-json did not contain a compactable benchmark_comparison_v0 object"
775
+ )
776
+ runs = []
777
+ for run_json in args.benchmark_run_json:
778
+ run_input = json.loads(
779
+ Path(run_json).expanduser().read_text(encoding="utf-8")
780
+ )
781
+ if not isinstance(run_input, dict):
782
+ raise ValueError("--benchmark-run-json must contain JSON objects")
783
+ run = compact_benchmark_run(run_input)
784
+ if not run:
785
+ raise ValueError(
786
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
787
+ )
788
+ runs.append(run)
789
+ payload = build_benchmark_learning_ledger(
790
+ comparison,
791
+ benchmark_runs=runs,
792
+ )
793
+ except Exception as exc:
794
+ payload = {
795
+ "ok": False,
796
+ "schema_version": "benchmark_learning_ledger_v0",
797
+ "error": str(exc),
798
+ "read_boundary": {
799
+ "compact_only": True,
800
+ "raw_artifacts_read": False,
801
+ "task_text_read": False,
802
+ "local_paths_recorded": False,
803
+ },
804
+ }
805
+ else:
806
+ payload["ok"] = True
807
+ learning_gate = (
808
+ payload.get("learning_quota_gate")
809
+ if isinstance(payload.get("learning_quota_gate"), dict)
810
+ else {}
811
+ )
812
+ if (
813
+ args.require_actionable_learning
814
+ and learning_gate.get("spend_allowed") is not True
815
+ ):
816
+ payload["ok"] = False
817
+ payload["error"] = (
818
+ learning_gate.get("blocked_reason")
819
+ or "missing_actionable_loopx_learning_signal"
820
+ )
821
+ print_payload(
822
+ payload,
823
+ output_format(args),
824
+ render_benchmark_learning_ledger_markdown,
825
+ )
826
+ return 0 if payload.get("ok") else 1
827
+ if args.benchmark_command == "attempt-learning-gate":
828
+ try:
829
+ run_input = json.loads(
830
+ Path(args.benchmark_run_json).expanduser().read_text(encoding="utf-8")
831
+ )
832
+ if not isinstance(run_input, dict):
833
+ raise ValueError("--benchmark-run-json must contain a JSON object")
834
+ run = compact_benchmark_run(run_input)
835
+ if not run:
836
+ raise ValueError(
837
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
838
+ )
839
+
840
+ learning_ledger = None
841
+ if args.benchmark_learning_ledger_json:
842
+ ledger_input = json.loads(
843
+ Path(args.benchmark_learning_ledger_json)
844
+ .expanduser()
845
+ .read_text(encoding="utf-8")
846
+ )
847
+ if not isinstance(ledger_input, dict):
848
+ raise ValueError(
849
+ "--benchmark-learning-ledger-json must contain a JSON object"
850
+ )
851
+ learning_ledger = compact_benchmark_learning_ledger(ledger_input)
852
+ if not learning_ledger:
853
+ raise ValueError(
854
+ "--benchmark-learning-ledger-json did not contain a compactable benchmark_learning_ledger_v0 object"
855
+ )
856
+
857
+ payload = build_benchmark_attempt_learning_gate(
858
+ run,
859
+ benchmark_learning_ledger=learning_ledger,
860
+ )
861
+ payload["ok"] = True
862
+ if (
863
+ args.require_budget_count_allowed
864
+ and payload.get("budget_count_allowed") is not True
865
+ ):
866
+ payload["ok"] = False
867
+ payload["error"] = (
868
+ payload.get("classification")
869
+ or "benchmark_attempt_learning_gate_not_ready"
870
+ )
871
+ payload["require_budget_count_allowed"] = bool(
872
+ args.require_budget_count_allowed
873
+ )
874
+ except Exception as exc:
875
+ payload = {
876
+ "ok": False,
877
+ "schema_version": "benchmark_attempt_learning_gate_v0",
878
+ "error": str(exc),
879
+ "read_boundary": {
880
+ "compact_only": True,
881
+ "raw_artifacts_read": False,
882
+ "task_text_read": False,
883
+ "local_paths_recorded": False,
884
+ },
885
+ }
886
+ print_payload(
887
+ payload,
888
+ output_format(args),
889
+ render_benchmark_attempt_learning_gate_markdown,
890
+ )
891
+ return 0 if payload.get("ok") else 1
892
+ if args.benchmark_command == "review-adapter-kwargs":
893
+ try:
894
+ agent_kwargs: dict[str, Any] = {}
895
+ if args.command_json:
896
+ command_input = json.loads(
897
+ Path(args.command_json).expanduser().read_text(encoding="utf-8")
898
+ )
899
+ if not isinstance(command_input, list):
900
+ raise ValueError("--command-json must contain a JSON argv list")
901
+ agent_kwargs.update(agent_kwargs_from_invocation(command_input))
902
+ for raw_kwarg in args.agent_kwarg:
903
+ key, separator, value = str(raw_kwarg).partition("=")
904
+ key = key.strip()
905
+ if not separator or not key:
906
+ raise ValueError("--agent-kwarg values must use KEY=VALUE form")
907
+ agent_kwargs[key] = value
908
+ accepted = list(args.accepted_loopx_kwarg)
909
+ if args.terminal_bench_managed_codex:
910
+ accepted.extend(TERMINAL_BENCH_MANAGED_CODEX_LOOPX_KWARGS)
911
+ payload = build_benchmark_adapter_kwarg_absorption_review(
912
+ adapter_label=args.adapter_label,
913
+ agent_kwargs=agent_kwargs,
914
+ accepted_loopx_kwargs=accepted,
915
+ allowed_base_passthrough=args.allowed_base_passthrough,
916
+ )
917
+ payload["ok"] = True
918
+ if args.require_clean and payload.get("clean") is not True:
919
+ payload["ok"] = False
920
+ payload["error"] = (
921
+ payload.get("classification")
922
+ or "benchmark_adapter_kwarg_absorption_not_clean"
923
+ )
924
+ payload["require_clean"] = bool(args.require_clean)
925
+ except Exception as exc:
926
+ payload = {
927
+ "ok": False,
928
+ "schema_version": "benchmark_adapter_kwarg_absorption_review_v0",
929
+ "error": str(exc),
930
+ "read_boundary": {
931
+ "compact_only": True,
932
+ "raw_artifacts_read": False,
933
+ "task_text_read": False,
934
+ "local_paths_recorded": False,
935
+ "docker_invoked": False,
936
+ "model_api_invoked": False,
937
+ "upload_invoked": False,
938
+ },
939
+ }
940
+ print_payload(
941
+ payload,
942
+ output_format(args),
943
+ render_benchmark_adapter_kwarg_absorption_review_markdown,
944
+ )
945
+ return 0 if payload.get("ok") else 1
946
+ if args.benchmark_command == "lifecycle-state":
947
+ def read_optional_json(path_text: str | None) -> dict[str, object] | None:
948
+ if not path_text:
949
+ return None
950
+ payload = json.loads(Path(path_text).expanduser().read_text(encoding="utf-8"))
951
+ if not isinstance(payload, dict):
952
+ raise ValueError("lifecycle input JSON must contain an object")
953
+ return payload
954
+
955
+ try:
956
+ preflight = read_optional_json(args.preflight_json)
957
+ launch = read_optional_json(args.launch_json)
958
+ post_launch = read_optional_json(args.post_launch_json)
959
+
960
+ benchmark_run = None
961
+ run_input = read_optional_json(args.benchmark_run_json)
962
+ if run_input is not None:
963
+ benchmark_run = compact_benchmark_run(run_input)
964
+ if not benchmark_run:
965
+ raise ValueError(
966
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
967
+ )
968
+
969
+ benchmark_comparison = None
970
+ comparison_input = read_optional_json(args.benchmark_comparison_json)
971
+ if comparison_input is not None:
972
+ benchmark_comparison = compact_benchmark_comparison(comparison_input)
973
+ if not benchmark_comparison:
974
+ raise ValueError(
975
+ "--benchmark-comparison-json did not contain a compactable benchmark_comparison_v0 object"
976
+ )
977
+
978
+ claim_review = read_optional_json(args.claim_review_json)
979
+ if (
980
+ claim_review is not None
981
+ and claim_review.get("schema_version") != "benchmark_claim_review_v0"
982
+ ):
983
+ raise ValueError("--claim-review-json must contain benchmark_claim_review_v0")
984
+
985
+ learning_ledger = None
986
+ ledger_input = read_optional_json(args.benchmark_learning_ledger_json)
987
+ if ledger_input is not None:
988
+ learning_ledger = compact_benchmark_learning_ledger(ledger_input)
989
+ if not learning_ledger:
990
+ raise ValueError(
991
+ "--benchmark-learning-ledger-json did not contain a compactable benchmark_learning_ledger_v0 object"
992
+ )
993
+
994
+ payload = build_benchmark_lifecycle_state(
995
+ preflight=preflight,
996
+ launch=launch,
997
+ post_launch_materialization=post_launch,
998
+ benchmark_run=benchmark_run,
999
+ benchmark_comparison=benchmark_comparison,
1000
+ claim_review=claim_review,
1001
+ learning_ledger=learning_ledger,
1002
+ )
1003
+ payload["ok"] = True
1004
+ gates = payload.get("gates") if isinstance(payload.get("gates"), dict) else {}
1005
+ if (
1006
+ args.require_budget_count_allowed
1007
+ and gates.get("budget_count_allowed") is not True
1008
+ ):
1009
+ payload["ok"] = False
1010
+ payload["error"] = (
1011
+ payload.get("first_blocker")
1012
+ or "benchmark_lifecycle_budget_count_not_allowed"
1013
+ )
1014
+ payload["require_budget_count_allowed"] = bool(
1015
+ args.require_budget_count_allowed
1016
+ )
1017
+ except Exception as exc:
1018
+ payload = {
1019
+ "ok": False,
1020
+ "schema_version": "benchmark_lifecycle_state_v0",
1021
+ "error": str(exc),
1022
+ "read_boundary": {
1023
+ "compact_only": True,
1024
+ "raw_artifacts_read": False,
1025
+ "task_text_read": False,
1026
+ "trajectory_read": False,
1027
+ "local_paths_recorded": False,
1028
+ "docker_invoked": False,
1029
+ "model_api_invoked": False,
1030
+ "upload_invoked": False,
1031
+ },
1032
+ }
1033
+ print_payload(
1034
+ payload,
1035
+ output_format(args),
1036
+ render_benchmark_lifecycle_state_markdown,
1037
+ )
1038
+ return 0 if payload.get("ok") else 1
1039
+ if args.benchmark_command == "review-verifier-attribution":
1040
+ try:
1041
+ runs = []
1042
+ for run_json in args.benchmark_run_json:
1043
+ run_input = json.loads(
1044
+ Path(run_json).expanduser().read_text(encoding="utf-8")
1045
+ )
1046
+ if not isinstance(run_input, dict):
1047
+ raise ValueError("--benchmark-run-json must contain JSON objects")
1048
+ run = compact_benchmark_run(run_input)
1049
+ if not run:
1050
+ raise ValueError(
1051
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
1052
+ )
1053
+ runs.append(run)
1054
+ payload = build_benchmark_verifier_attribution_review(
1055
+ benchmark_runs=runs,
1056
+ )
1057
+ except Exception as exc:
1058
+ payload = {
1059
+ "ok": False,
1060
+ "schema_version": "benchmark_verifier_attribution_review_v0",
1061
+ "error": str(exc),
1062
+ "read_boundary": {
1063
+ "compact_only": True,
1064
+ "raw_artifacts_read": False,
1065
+ "task_text_read": False,
1066
+ "local_paths_recorded": False,
1067
+ },
1068
+ }
1069
+ else:
1070
+ payload["ok"] = True
1071
+ print_payload(
1072
+ payload,
1073
+ output_format(args),
1074
+ render_benchmark_verifier_attribution_review_markdown,
1075
+ )
1076
+ return 0 if payload.get("ok") else 1
1077
+ if args.benchmark_command == "review-runner-invariants":
1078
+ try:
1079
+ run_input = json.loads(
1080
+ Path(args.benchmark_run_json).expanduser().read_text(encoding="utf-8")
1081
+ )
1082
+ if not isinstance(run_input, dict):
1083
+ raise ValueError("--benchmark-run-json must contain a JSON object")
1084
+ run = compact_benchmark_run(run_input)
1085
+ if not run:
1086
+ raise ValueError(
1087
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
1088
+ )
1089
+ payload = build_benchmark_runner_invariant_review(
1090
+ run,
1091
+ expected_flags={
1092
+ "submit_eligible": args.expect_submit_eligible == "true",
1093
+ "leaderboard_evidence": args.expect_leaderboard_evidence
1094
+ == "true",
1095
+ },
1096
+ expected_read_boundary={
1097
+ "compact_only": args.expect_compact_only == "true",
1098
+ "raw_artifacts_read": args.expect_raw_artifacts_read == "true",
1099
+ "task_text_read": args.expect_task_text_read == "true",
1100
+ "local_paths_recorded": args.expect_local_paths_recorded
1101
+ == "true",
1102
+ },
1103
+ runner_label=args.runner_label,
1104
+ )
1105
+ payload["ok"] = True
1106
+ if args.require_clean and payload.get("clean") is not True:
1107
+ payload["ok"] = False
1108
+ payload["error"] = payload.get("classification") or (
1109
+ "benchmark_runner_invariant_review_not_clean"
1110
+ )
1111
+ payload["require_clean"] = bool(args.require_clean)
1112
+ except Exception as exc:
1113
+ payload = {
1114
+ "ok": False,
1115
+ "schema_version": "benchmark_runner_invariant_review_v0",
1116
+ "error": str(exc),
1117
+ "read_boundary": {
1118
+ "compact_only": True,
1119
+ "raw_artifacts_read": False,
1120
+ "task_text_read": False,
1121
+ "local_paths_recorded": False,
1122
+ },
1123
+ }
1124
+ print_payload(
1125
+ payload,
1126
+ output_format(args),
1127
+ render_benchmark_runner_invariant_review_markdown,
1128
+ )
1129
+ return 0 if payload.get("ok") else 1
1130
+ if args.benchmark_command == "baseline-failure-gate":
1131
+ try:
1132
+ if args.dry_run and args.execute:
1133
+ raise ValueError(
1134
+ "benchmark baseline-failure-gate accepts either --dry-run or --execute, not both"
1135
+ )
1136
+ if args.execute and not args.goal_id:
1137
+ raise ValueError(
1138
+ "benchmark baseline-failure-gate requires --goal-id with --execute"
1139
+ )
1140
+ if args.execute and registry_path is None:
1141
+ raise ValueError(
1142
+ "benchmark baseline-failure-gate requires registry_path with --execute"
1143
+ )
1144
+ if args.baseline_result_json == "-":
1145
+ baseline_result_input = json.loads(sys.stdin.read())
1146
+ else:
1147
+ baseline_result_input = json.loads(
1148
+ Path(args.baseline_result_json)
1149
+ .expanduser()
1150
+ .read_text(encoding="utf-8")
1151
+ )
1152
+ if not isinstance(baseline_result_input, dict):
1153
+ raise ValueError("--baseline-result-json must contain a JSON object")
1154
+ baseline_result = compact_benchmark_result(baseline_result_input)
1155
+ baseline_gate_source = "compact_benchmark_result_v0"
1156
+ if not baseline_result:
1157
+ benchmark_run = compact_benchmark_run(baseline_result_input)
1158
+ if benchmark_run:
1159
+ baseline_result = benchmark_result_from_benchmark_run_for_baseline_gate(
1160
+ benchmark_run
1161
+ )
1162
+ baseline_gate_source = "compact_benchmark_run_v0"
1163
+ else:
1164
+ raise ValueError(
1165
+ "--baseline-result-json did not contain a compactable benchmark_result_v0 or benchmark_run_v0 object"
1166
+ )
1167
+ comparison_input = build_benchmark_baseline_failure_gate_comparison(
1168
+ baseline_result=baseline_result,
1169
+ benchmark_id=args.benchmark_id,
1170
+ baseline_mode=args.baseline_mode,
1171
+ treatment_scenario_id=args.treatment_scenario_id,
1172
+ comparison_id=args.comparison_id,
1173
+ failure_phase=args.failure_phase,
1174
+ failure_class=args.failure_class,
1175
+ failure_attribution_labels=args.failure_attribution_label,
1176
+ control_plane_addressable=bool(args.control_plane_addressable),
1177
+ same_task_semantics=bool(args.same_task_semantics),
1178
+ same_runner_protocol=bool(args.same_runner_protocol),
1179
+ trace_publicness_verified=bool(args.trace_publicness_verified),
1180
+ baseline_attempt_count=args.baseline_attempt_count,
1181
+ minimum_next_evidence=args.minimum_next_evidence,
1182
+ negative_selection_reason=args.negative_selection_reason,
1183
+ next_action=args.next_action,
1184
+ evidence_refs=args.evidence_ref,
1185
+ )
1186
+ comparison = compact_benchmark_comparison(comparison_input)
1187
+ if not comparison:
1188
+ raise ValueError(
1189
+ "baseline gate reducer did not produce a compactable benchmark_comparison_v0 object"
1190
+ )
1191
+ dry_run = not bool(args.execute)
1192
+ if args.execute:
1193
+ payload = append_benchmark_comparison(
1194
+ registry_path=registry_path,
1195
+ runtime_root_override=args.runtime_root,
1196
+ goal_id=args.goal_id,
1197
+ benchmark_comparison=comparison,
1198
+ classification=args.classification or "benchmark_comparison_v0",
1199
+ recommended_action=args.recommended_action
1200
+ or (
1201
+ comparison.get("next_action")
1202
+ if isinstance(comparison.get("next_action"), str)
1203
+ else None
1204
+ )
1205
+ or "route the baseline failure gate before any treatment run",
1206
+ delivery_batch_scale=args.delivery_batch_scale,
1207
+ delivery_outcome=args.delivery_outcome,
1208
+ dry_run=False,
1209
+ )
1210
+ if args.no_global_sync:
1211
+ payload["global_sync"] = {
1212
+ "ok": True,
1213
+ "dry_run": False,
1214
+ "skipped": True,
1215
+ "reason": "disabled by --no-global-sync",
1216
+ }
1217
+ else:
1218
+ payload["global_sync"] = sync_project_registry_to_global(
1219
+ registry_path=registry_path,
1220
+ runtime_root_override=args.runtime_root,
1221
+ goal_id=args.goal_id,
1222
+ dry_run=False,
1223
+ )
1224
+ else:
1225
+ payload = {
1226
+ "ok": True,
1227
+ "dry_run": dry_run,
1228
+ "appended": False,
1229
+ "goal_id": args.goal_id,
1230
+ "classification": args.classification
1231
+ or "benchmark_comparison_v0",
1232
+ "benchmark_comparison": comparison,
1233
+ }
1234
+ payload["baseline_gate_cli"] = {
1235
+ "source": baseline_gate_source,
1236
+ "accepted_schemas": [
1237
+ "benchmark_result_v0",
1238
+ "benchmark_run_v0",
1239
+ ],
1240
+ "raw_artifacts_read": False,
1241
+ "task_text_read": False,
1242
+ "local_paths_recorded": False,
1243
+ "docker_invoked": False,
1244
+ "model_api_invoked": False,
1245
+ "upload_invoked": False,
1246
+ }
1247
+ except Exception as exc:
1248
+ payload = {
1249
+ "ok": False,
1250
+ "dry_run": not bool(getattr(args, "execute", False)),
1251
+ "appended": False,
1252
+ "goal_id": getattr(args, "goal_id", None),
1253
+ "classification": getattr(args, "classification", None)
1254
+ or "benchmark_comparison_v0",
1255
+ "error": str(exc),
1256
+ "baseline_gate_cli": {
1257
+ "source": "compact_benchmark_result_v0_or_benchmark_run_v0",
1258
+ "accepted_schemas": [
1259
+ "benchmark_result_v0",
1260
+ "benchmark_run_v0",
1261
+ ],
1262
+ "raw_artifacts_read": False,
1263
+ "task_text_read": False,
1264
+ "local_paths_recorded": False,
1265
+ "docker_invoked": False,
1266
+ "model_api_invoked": False,
1267
+ "upload_invoked": False,
1268
+ },
1269
+ }
1270
+ print_payload(
1271
+ payload,
1272
+ output_format(args),
1273
+ render_benchmark_baseline_failure_gate_markdown,
1274
+ )
1275
+ return 0 if payload.get("ok") else 1