loopx 0.4.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (811) hide show
  1. loopx/__init__.py +5 -0
  2. loopx/agent_onboarding.py +654 -0
  3. loopx/agent_registry.py +112 -0
  4. loopx/ark_managed_agent_host.py +59 -0
  5. loopx/authority.py +805 -0
  6. loopx/benchmark.py +2875 -0
  7. loopx/benchmark_adapters/__init__.py +1 -0
  8. loopx/benchmark_adapters/agentissue.py +2644 -0
  9. loopx/benchmark_adapters/agents_last_exam.py +3998 -0
  10. loopx/benchmark_adapters/edgebench.py +322 -0
  11. loopx/benchmark_adapters/skillsbench.py +5978 -0
  12. loopx/benchmark_adapters/skillsbench_acp_failure_policy.py +143 -0
  13. loopx/benchmark_adapters/skillsbench_acp_process.py +31 -0
  14. loopx/benchmark_adapters/skillsbench_acp_relay.py +4832 -0
  15. loopx/benchmark_adapters/skillsbench_batch.py +124 -0
  16. loopx/benchmark_adapters/skillsbench_bridge_guard.py +209 -0
  17. loopx/benchmark_adapters/skillsbench_bridge_summary.py +203 -0
  18. loopx/benchmark_adapters/skillsbench_codex_goal_recovery.py +271 -0
  19. loopx/benchmark_adapters/skillsbench_codex_goal_trace.py +81 -0
  20. loopx/benchmark_adapters/skillsbench_codex_runtime.py +339 -0
  21. loopx/benchmark_adapters/skillsbench_dockerfile_runtime.py +467 -0
  22. loopx/benchmark_adapters/skillsbench_failure_signals.py +652 -0
  23. loopx/benchmark_adapters/skillsbench_proxy_runtime.py +327 -0
  24. loopx/benchmark_adapters/skillsbench_remote_bridge.py +402 -0
  25. loopx/benchmark_adapters/skillsbench_result_discovery.py +143 -0
  26. loopx/benchmark_adapters/skillsbench_runner_profile.py +436 -0
  27. loopx/benchmark_adapters/skillsbench_runner_source.py +99 -0
  28. loopx/benchmark_adapters/skillsbench_setup_preflight.py +771 -0
  29. loopx/benchmark_adapters/skillsbench_signals.py +15 -0
  30. loopx/benchmark_adapters/skillsbench_task_source.py +141 -0
  31. loopx/benchmark_adapters/skillsbench_turn_route.py +723 -0
  32. loopx/benchmark_adapters/skillsbench_turn_runtime.py +1069 -0
  33. loopx/benchmark_adapters/skillsbench_typed_repair.py +689 -0
  34. loopx/benchmark_adapters/skillsbench_uv_cache.py +111 -0
  35. loopx/benchmark_adapters/skillsbench_verifier_bootstrap.py +227 -0
  36. loopx/benchmark_adapters/skillsbench_verifier_cache.py +138 -0
  37. loopx/benchmark_adapters/terminal_bench.py +10078 -0
  38. loopx/benchmark_case_analysis.py +1276 -0
  39. loopx/benchmark_case_state.py +1079 -0
  40. loopx/benchmark_core/__init__.py +239 -0
  41. loopx/benchmark_core/adapter.py +84 -0
  42. loopx/benchmark_core/artifacts.py +517 -0
  43. loopx/benchmark_core/attempts.py +199 -0
  44. loopx/benchmark_core/container_exec.py +216 -0
  45. loopx/benchmark_core/io.py +68 -0
  46. loopx/benchmark_core/lifecycle.py +211 -0
  47. loopx/benchmark_core/loop_protocol.py +689 -0
  48. loopx/benchmark_core/observable_handles.py +348 -0
  49. loopx/benchmark_core/parity.py +256 -0
  50. loopx/benchmark_core/remote_closeout.py +482 -0
  51. loopx/benchmark_core/rounds.py +215 -0
  52. loopx/benchmark_core/route_profile.py +509 -0
  53. loopx/benchmark_core/run_permissions.py +206 -0
  54. loopx/benchmark_core/split_control.py +925 -0
  55. loopx/benchmark_core/turn_fidelity.py +326 -0
  56. loopx/benchmark_ledger.py +3793 -0
  57. loopx/benchmark_ledger_countability.py +372 -0
  58. loopx/benchmark_ledger_current.py +724 -0
  59. loopx/benchmark_trajectory.py +405 -0
  60. loopx/benchmarks/__init__.py +1 -0
  61. loopx/benchmarks/qualification/__init__.py +1 -0
  62. loopx/benchmarks/qualification/release_outcome_baseline.py +360 -0
  63. loopx/benchmarks/read_models/__init__.py +1 -0
  64. loopx/benchmarks/read_models/benchmark_attempt_accounting.py +53 -0
  65. loopx/benchmarks/read_models/benchmark_comparison.py +414 -0
  66. loopx/benchmarks/read_models/benchmark_event_timeline.py +113 -0
  67. loopx/benchmarks/read_models/benchmark_experiment_report.py +475 -0
  68. loopx/benchmarks/read_models/benchmark_learning_ledger.py +137 -0
  69. loopx/benchmarks/read_models/benchmark_lifecycle_contracts.py +228 -0
  70. loopx/benchmarks/read_models/benchmark_projection.py +723 -0
  71. loopx/benchmarks/read_models/benchmark_result.py +146 -0
  72. loopx/benchmarks/read_models/benchmark_run_execution_contract.py +116 -0
  73. loopx/benchmarks/read_models/benchmark_run_failure.py +157 -0
  74. loopx/benchmarks/read_models/benchmark_run_metrics.py +213 -0
  75. loopx/benchmarks/read_models/benchmark_run_post_execution.py +635 -0
  76. loopx/benchmarks/read_models/benchmark_run_pre_execution.py +541 -0
  77. loopx/benchmarks/read_models/benchmark_status_compaction.py +1255 -0
  78. loopx/benchmarks/read_models/benchmark_status_runner.py +780 -0
  79. loopx/benchmarks/read_models/goal_start_control_score.py +857 -0
  80. loopx/benchmarks/read_models/skillsbench_post_run_debug.py +746 -0
  81. loopx/benchmarks/read_models/skillsbench_verifier_attribution.py +269 -0
  82. loopx/bootstrap.py +1116 -0
  83. loopx/bootstrap_command_pack.py +2167 -0
  84. loopx/boundary_authority.py +199 -0
  85. loopx/canary/__init__.py +1 -0
  86. loopx/canary/maintainability_ratchet.py +800 -0
  87. loopx/canary/planner.py +1984 -0
  88. loopx/canary/premerge.py +1130 -0
  89. loopx/canary/qualification_profiles.py +309 -0
  90. loopx/canary/quality_surface_catalog.py +838 -0
  91. loopx/canary/release_profiles.py +51 -0
  92. loopx/canary/runner.py +1107 -0
  93. loopx/canary/smoke_health.py +581 -0
  94. loopx/canary/smoke_profiles.py +212 -0
  95. loopx/capabilities/__init__.py +0 -0
  96. loopx/capabilities/agent_turn_recall/__init__.py +17 -0
  97. loopx/capabilities/agent_turn_recall/cli.py +369 -0
  98. loopx/capabilities/agent_turn_recall/core.py +296 -0
  99. loopx/capabilities/auto_research/__init__.py +16 -0
  100. loopx/capabilities/auto_research/bootstrap_contract.py +157 -0
  101. loopx/capabilities/auto_research/cli.py +1468 -0
  102. loopx/capabilities/auto_research/core.py +11 -0
  103. loopx/capabilities/auto_research/defaults.py +79 -0
  104. loopx/capabilities/auto_research/demo_e2e.py +1848 -0
  105. loopx/capabilities/auto_research/demo_supervisor.py +186 -0
  106. loopx/capabilities/auto_research/evidence_packet.py +767 -0
  107. loopx/capabilities/auto_research/human_view.py +794 -0
  108. loopx/capabilities/auto_research/kernel.py +191 -0
  109. loopx/capabilities/auto_research/knn_demo_workspace.py +322 -0
  110. loopx/capabilities/auto_research/live_evidence.py +248 -0
  111. loopx/capabilities/auto_research/preset.py +176 -0
  112. loopx/capabilities/auto_research/research_state.py +1085 -0
  113. loopx/capabilities/auto_research/role_profiles.py +394 -0
  114. loopx/capabilities/auto_research/rollout_append.py +97 -0
  115. loopx/capabilities/auto_research/terminal_result_contract.py +422 -0
  116. loopx/capabilities/auto_research/terminal_result_projection.py +171 -0
  117. loopx/capabilities/auto_research/terminal_result_query.py +233 -0
  118. loopx/capabilities/auto_research/terminal_results.py +349 -0
  119. loopx/capabilities/auto_research/user_contract.py +190 -0
  120. loopx/capabilities/auto_research/worker_loop.py +163 -0
  121. loopx/capabilities/auto_research/worker_runtime.py +777 -0
  122. loopx/capabilities/auto_research/worker_skill/SKILL.md +343 -0
  123. loopx/capabilities/benchmark_toolkit/__init__.py +19 -0
  124. loopx/capabilities/benchmark_toolkit/integrity.py +387 -0
  125. loopx/capabilities/catalog.py +1875 -0
  126. loopx/capabilities/change_quality/__init__.py +19 -0
  127. loopx/capabilities/change_quality/cli.py +171 -0
  128. loopx/capabilities/change_quality/context.py +156 -0
  129. loopx/capabilities/change_quality/oracles.py +269 -0
  130. loopx/capabilities/change_quality/policy.py +34 -0
  131. loopx/capabilities/change_quality/receipt.py +482 -0
  132. loopx/capabilities/change_quality/result.py +493 -0
  133. loopx/capabilities/change_quality/scope.py +171 -0
  134. loopx/capabilities/change_quality/shadow.py +680 -0
  135. loopx/capabilities/content_ops/__init__.py +0 -0
  136. loopx/capabilities/content_ops/cli.py +649 -0
  137. loopx/capabilities/content_ops/connector_packets.py +164 -0
  138. loopx/capabilities/content_ops/item_lifecycle.py +1000 -0
  139. loopx/capabilities/content_ops/layout.py +451 -0
  140. loopx/capabilities/content_ops/markdown.py +456 -0
  141. loopx/capabilities/content_ops/schemas.py +51 -0
  142. loopx/capabilities/content_ops/social_browser_x.py +107 -0
  143. loopx/capabilities/content_ops/surface.py +1956 -0
  144. loopx/capabilities/content_ops/templates/layout-catalog-v0.json +72 -0
  145. loopx/capabilities/context_providers/__init__.py +36 -0
  146. loopx/capabilities/context_providers/base.py +189 -0
  147. loopx/capabilities/context_providers/factory.py +32 -0
  148. loopx/capabilities/context_providers/openviking.py +702 -0
  149. loopx/capabilities/context_providers/service_ownership.py +185 -0
  150. loopx/capabilities/decision_context/__init__.py +129 -0
  151. loopx/capabilities/decision_context/architecture.py +83 -0
  152. loopx/capabilities/decision_context/assembler.py +849 -0
  153. loopx/capabilities/decision_context/catalog_entry.py +195 -0
  154. loopx/capabilities/decision_context/cli.py +310 -0
  155. loopx/capabilities/decision_context/cursor_commit.py +535 -0
  156. loopx/capabilities/decision_context/outcome_feedback.py +352 -0
  157. loopx/capabilities/decision_context/packets.py +654 -0
  158. loopx/capabilities/decision_context/private_state.py +189 -0
  159. loopx/capabilities/decision_context/profile.py +453 -0
  160. loopx/capabilities/decision_context/providers.py +228 -0
  161. loopx/capabilities/decision_context/review_settlement.py +136 -0
  162. loopx/capabilities/decision_context/runtime.py +273 -0
  163. loopx/capabilities/decision_context/sources.py +415 -0
  164. loopx/capabilities/explore/__init__.py +1 -0
  165. loopx/capabilities/explore/activation.py +198 -0
  166. loopx/capabilities/explore/adaptive_replay_planner.py +221 -0
  167. loopx/capabilities/explore/child_replay_runtime.py +463 -0
  168. loopx/capabilities/explore/composition_frontier.py +291 -0
  169. loopx/capabilities/explore/counterfactual_runtime.py +578 -0
  170. loopx/capabilities/explore/episode_runtime.py +647 -0
  171. loopx/capabilities/explore/harness_checkpoint.py +171 -0
  172. loopx/capabilities/explore/harness_gate.py +115 -0
  173. loopx/capabilities/explore/harness_runtime.py +1124 -0
  174. loopx/capabilities/explore/replay_metrics.py +206 -0
  175. loopx/capabilities/explore/replay_runtime.py +1271 -0
  176. loopx/capabilities/explore/resource_portfolio.py +173 -0
  177. loopx/capabilities/explore/result_log.py +974 -0
  178. loopx/capabilities/explore/router_state.py +432 -0
  179. loopx/capabilities/explore/source_history_reconcile.py +255 -0
  180. loopx/capabilities/explore/speculative_scheduler.py +498 -0
  181. loopx/capabilities/explore/todo_branch_plan.py +650 -0
  182. loopx/capabilities/explore/todo_evidence.py +141 -0
  183. loopx/capabilities/explore/trace_runtime.py +284 -0
  184. loopx/capabilities/explore/worker_branch_plan.py +1257 -0
  185. loopx/capabilities/integration_branch/__init__.py +13 -0
  186. loopx/capabilities/integration_branch/cli.py +148 -0
  187. loopx/capabilities/integration_branch/core.py +916 -0
  188. loopx/capabilities/issue_fix/__init__.py +19 -0
  189. loopx/capabilities/issue_fix/acceptance_loop.py +1050 -0
  190. loopx/capabilities/issue_fix/candidate_evidence.py +503 -0
  191. loopx/capabilities/issue_fix/candidate_preflight.py +676 -0
  192. loopx/capabilities/issue_fix/cli.py +1822 -0
  193. loopx/capabilities/issue_fix/cli_input.py +87 -0
  194. loopx/capabilities/issue_fix/content_ops_cli.py +148 -0
  195. loopx/capabilities/issue_fix/discovered_issue_promotion.py +947 -0
  196. loopx/capabilities/issue_fix/explore_projection.py +710 -0
  197. loopx/capabilities/issue_fix/feasibility.py +542 -0
  198. loopx/capabilities/issue_fix/github_public.py +661 -0
  199. loopx/capabilities/issue_fix/intake_surface.py +832 -0
  200. loopx/capabilities/issue_fix/metadata_preview.py +218 -0
  201. loopx/capabilities/issue_fix/metrics_projection.py +1340 -0
  202. loopx/capabilities/issue_fix/metrics_supplement.py +634 -0
  203. loopx/capabilities/issue_fix/metrics_supplement_cli.py +127 -0
  204. loopx/capabilities/issue_fix/outcome_projection.py +1235 -0
  205. loopx/capabilities/issue_fix/periodic_report.py +189 -0
  206. loopx/capabilities/issue_fix/pr_description.py +418 -0
  207. loopx/capabilities/issue_fix/pr_gate_reconcile.py +496 -0
  208. loopx/capabilities/issue_fix/pr_gate_reconcile_cli.py +464 -0
  209. loopx/capabilities/issue_fix/pr_lifecycle.py +1327 -0
  210. loopx/capabilities/issue_fix/pr_lifecycle_rollout.py +85 -0
  211. loopx/capabilities/issue_fix/pr_monitor_materialization.py +257 -0
  212. loopx/capabilities/issue_fix/pr_review_ack.py +439 -0
  213. loopx/capabilities/issue_fix/provider_hooks.py +24 -0
  214. loopx/capabilities/issue_fix/repository_commit_evidence.py +186 -0
  215. loopx/capabilities/issue_fix/repository_context.py +457 -0
  216. loopx/capabilities/issue_fix/repository_memory.py +459 -0
  217. loopx/capabilities/issue_fix/repository_memory_provider.py +1454 -0
  218. loopx/capabilities/issue_fix/repository_snapshot.py +454 -0
  219. loopx/capabilities/issue_fix/reviewer_cli.py +917 -0
  220. loopx/capabilities/issue_fix/reviewer_notification.py +882 -0
  221. loopx/capabilities/issue_fix/reviewer_notification_drain.py +942 -0
  222. loopx/capabilities/issue_fix/reviewer_recommendation.py +1057 -0
  223. loopx/capabilities/issue_fix/reviewer_request.py +1282 -0
  224. loopx/capabilities/issue_fix/reward_memory.py +879 -0
  225. loopx/capabilities/issue_fix/workflow_plan.py +1286 -0
  226. loopx/capabilities/material_lifecycle/__init__.py +161 -0
  227. loopx/capabilities/material_lifecycle/_validation.py +183 -0
  228. loopx/capabilities/material_lifecycle/apply.py +672 -0
  229. loopx/capabilities/material_lifecycle/architecture.py +122 -0
  230. loopx/capabilities/material_lifecycle/cli.py +161 -0
  231. loopx/capabilities/material_lifecycle/decision_planning.py +470 -0
  232. loopx/capabilities/material_lifecycle/explore_execution.py +306 -0
  233. loopx/capabilities/material_lifecycle/intake.py +869 -0
  234. loopx/capabilities/material_lifecycle/inventory.py +147 -0
  235. loopx/capabilities/material_lifecycle/lifecycle.py +98 -0
  236. loopx/capabilities/material_lifecycle/preparation.py +147 -0
  237. loopx/capabilities/material_lifecycle/project_skill.py +83 -0
  238. loopx/capabilities/material_lifecycle/ranking.py +267 -0
  239. loopx/capabilities/material_lifecycle/readable_projection.py +500 -0
  240. loopx/capabilities/material_lifecycle/rebuild.py +480 -0
  241. loopx/capabilities/material_lifecycle/settlement.py +238 -0
  242. loopx/capabilities/periodic_report/__init__.py +71 -0
  243. loopx/capabilities/periodic_report/adapters.py +939 -0
  244. loopx/capabilities/periodic_report/archive.py +422 -0
  245. loopx/capabilities/periodic_report/bindings.py +705 -0
  246. loopx/capabilities/periodic_report/cli.py +277 -0
  247. loopx/capabilities/periodic_report/core.py +691 -0
  248. loopx/capabilities/periodic_report/extension_envelope.py +66 -0
  249. loopx/capabilities/periodic_report/presets.py +103 -0
  250. loopx/capabilities/periodic_report/profile.py +235 -0
  251. loopx/capabilities/periodic_report/project_progress.py +179 -0
  252. loopx/capabilities/periodic_report/triggers.py +452 -0
  253. loopx/capabilities/pr_review_queue/__init__.py +17 -0
  254. loopx/capabilities/pr_review_queue/core.py +506 -0
  255. loopx/capabilities/pr_review_queue/review_contract.py +506 -0
  256. loopx/capabilities/registry.py +192 -0
  257. loopx/capabilities/reward_memory/__init__.py +75 -0
  258. loopx/capabilities/reward_memory/application.py +819 -0
  259. loopx/capabilities/reward_memory/architecture.py +572 -0
  260. loopx/capabilities/reward_memory/candidate_review.py +511 -0
  261. loopx/capabilities/reward_memory/cli.py +469 -0
  262. loopx/capabilities/reward_memory/dogfood.py +574 -0
  263. loopx/capabilities/reward_memory/evaluation.py +296 -0
  264. loopx/capabilities/reward_memory/evaluation_fixtures.py +362 -0
  265. loopx/capabilities/reward_memory/experiment.py +567 -0
  266. loopx/capabilities/reward_memory/health.py +222 -0
  267. loopx/capabilities/reward_memory/ingestion.py +519 -0
  268. loopx/capabilities/reward_memory/registry.py +600 -0
  269. loopx/capabilities/reward_memory/runtime_hooks.py +312 -0
  270. loopx/capabilities/reward_memory/scoped_feedback.py +173 -0
  271. loopx/capabilities/semantic_preference/__init__.py +12 -0
  272. loopx/capabilities/semantic_preference/cli.py +189 -0
  273. loopx/capabilities/semantic_preference/contract.py +592 -0
  274. loopx/capabilities/semantic_preference/reward_memory.py +62 -0
  275. loopx/capabilities/value_connectors/__init__.py +1 -0
  276. loopx/capabilities/value_connectors/cli.py +401 -0
  277. loopx/capabilities/value_connectors/finance_extension_migration.py +108 -0
  278. loopx/capabilities/value_connectors/install_check.py +147 -0
  279. loopx/capabilities/value_connectors/planner.py +733 -0
  280. loopx/capabilities/value_connectors/source_map.py +446 -0
  281. loopx/claude_goal_baseline.py +138 -0
  282. loopx/claude_goal_mode/__init__.py +23 -0
  283. loopx/claude_goal_mode/hooks/goal_policy.py +212 -0
  284. loopx/claude_goal_mode/hooks/goal_state.py +139 -0
  285. loopx/claude_goal_mode/mcp/loopx_mcp.py +167 -0
  286. loopx/claude_goal_mode/scripts/connect.py +103 -0
  287. loopx/claude_goal_mode/scripts/goalmode_cmd.py +241 -0
  288. loopx/claude_goal_mode/scripts/install.py +328 -0
  289. loopx/claude_goal_mode/statusline/goal_status.py +97 -0
  290. loopx/cli.py +836 -0
  291. loopx/cli_commands/__init__.py +334 -0
  292. loopx/cli_commands/_host_thread.py +13 -0
  293. loopx/cli_commands/agentissue_runner_flow.py +447 -0
  294. loopx/cli_commands/agents_last_exam.py +160 -0
  295. loopx/cli_commands/agents_last_exam_baked_input.py +302 -0
  296. loopx/cli_commands/agents_last_exam_host_codex.py +374 -0
  297. loopx/cli_commands/agents_last_exam_launch_dry_run.py +372 -0
  298. loopx/cli_commands/agents_last_exam_local_plan.py +322 -0
  299. loopx/cli_commands/agents_last_exam_runner_source.py +352 -0
  300. loopx/cli_commands/agents_last_exam_task_material.py +335 -0
  301. loopx/cli_commands/agents_last_exam_validation_gate.py +236 -0
  302. loopx/cli_commands/benchmark_boundary.py +499 -0
  303. loopx/cli_commands/benchmark_dispatch.py +161 -0
  304. loopx/cli_commands/benchmark_release_outcome.py +123 -0
  305. loopx/cli_commands/benchmark_review_lifecycle.py +1275 -0
  306. loopx/cli_commands/benchmark_run_ledger.py +763 -0
  307. loopx/cli_commands/benchmark_run_ledger_case_analysis.py +249 -0
  308. loopx/cli_commands/benchmark_run_ledger_classification.py +45 -0
  309. loopx/cli_commands/benchmark_run_ledger_maintenance.py +486 -0
  310. loopx/cli_commands/benchmark_run_ledger_maintenance_registration.py +342 -0
  311. loopx/cli_commands/benchmark_run_ledger_maintenance_rendering.py +233 -0
  312. loopx/cli_commands/benchmark_run_ledger_parity.py +92 -0
  313. loopx/cli_commands/bootstrap_connect.py +238 -0
  314. loopx/cli_commands/canary.py +707 -0
  315. loopx/cli_commands/canary_release_qualification.py +79 -0
  316. loopx/cli_commands/capability.py +96 -0
  317. loopx/cli_commands/doctor.py +43 -0
  318. loopx/cli_commands/dreaming.py +143 -0
  319. loopx/cli_commands/edgebench.py +205 -0
  320. loopx/cli_commands/evidence_log.py +275 -0
  321. loopx/cli_commands/explore.py +989 -0
  322. loopx/cli_commands/explore_planning_commands.py +157 -0
  323. loopx/cli_commands/extension.py +271 -0
  324. loopx/cli_commands/first_run_report.py +73 -0
  325. loopx/cli_commands/goal_channel.py +656 -0
  326. loopx/cli_commands/handoff_mode.py +158 -0
  327. loopx/cli_commands/history.py +622 -0
  328. loopx/cli_commands/host_mode_plan.py +113 -0
  329. loopx/cli_commands/lark_inbox.py +431 -0
  330. loopx/cli_commands/lark_kanban.py +629 -0
  331. loopx/cli_commands/ml_experiment.py +321 -0
  332. loopx/cli_commands/multi_agent.py +211 -0
  333. loopx/cli_commands/opencode2_goal_worker.py +217 -0
  334. loopx/cli_commands/pr_review.py +167 -0
  335. loopx/cli_commands/presentation.py +218 -0
  336. loopx/cli_commands/preset.py +96 -0
  337. loopx/cli_commands/project.py +150 -0
  338. loopx/cli_commands/project_lifecycle.py +915 -0
  339. loopx/cli_commands/quota.py +859 -0
  340. loopx/cli_commands/quota_registration.py +241 -0
  341. loopx/cli_commands/quota_request.py +113 -0
  342. loopx/cli_commands/ready_score.py +110 -0
  343. loopx/cli_commands/registry_admin.py +975 -0
  344. loopx/cli_commands/registry_admin_configure.py +344 -0
  345. loopx/cli_commands/registry_admin_peer.py +84 -0
  346. loopx/cli_commands/registry_authority.py +218 -0
  347. loopx/cli_commands/review_batch.py +146 -0
  348. loopx/cli_commands/slash_commands.py +145 -0
  349. loopx/cli_commands/start_goal.py +251 -0
  350. loopx/cli_commands/starter.py +175 -0
  351. loopx/cli_commands/starter_bootstrap.py +179 -0
  352. loopx/cli_commands/starter_bootstrap_registration.py +198 -0
  353. loopx/cli_commands/starter_runtime_idle.py +107 -0
  354. loopx/cli_commands/starter_scheduler.py +207 -0
  355. loopx/cli_commands/starter_session_runtime.py +152 -0
  356. loopx/cli_commands/starter_visible_common.py +54 -0
  357. loopx/cli_commands/starter_visible_driver.py +161 -0
  358. loopx/cli_commands/starter_visible_pilot.py +278 -0
  359. loopx/cli_commands/status.py +867 -0
  360. loopx/cli_commands/status_registration.py +239 -0
  361. loopx/cli_commands/summary_all.py +222 -0
  362. loopx/cli_commands/support_control.py +809 -0
  363. loopx/cli_commands/support_control_registry.py +68 -0
  364. loopx/cli_commands/support_control_supervisor.py +289 -0
  365. loopx/cli_commands/task_lease.py +306 -0
  366. loopx/cli_commands/terminal_bench_adapter.py +717 -0
  367. loopx/cli_commands/terminal_bench_environment_result.py +1246 -0
  368. loopx/cli_commands/todo.py +940 -0
  369. loopx/cli_commands/todo_argument_validation.py +572 -0
  370. loopx/cli_commands/todo_event.py +114 -0
  371. loopx/cli_commands/turn.py +804 -0
  372. loopx/cli_commands/version.py +46 -0
  373. loopx/cli_commands/worker_bridge.py +659 -0
  374. loopx/cli_rollout.py +314 -0
  375. loopx/codex_cli_goal_tui.py +672 -0
  376. loopx/codex_cli_probe.py +1530 -0
  377. loopx/codex_cli_probe_markdown.py +935 -0
  378. loopx/codex_cli_runtime_probe.py +733 -0
  379. loopx/codex_cli_scheduler.py +564 -0
  380. loopx/codex_goal_baseline.py +620 -0
  381. loopx/configuration_catalog.py +617 -0
  382. loopx/configure_goal.py +1375 -0
  383. loopx/contract.py +996 -0
  384. loopx/control_plane/__init__.py +71 -0
  385. loopx/control_plane/agents/__init__.py +1 -0
  386. loopx/control_plane/agents/agent_lane_recommendation.py +516 -0
  387. loopx/control_plane/agents/agent_scope.py +1578 -0
  388. loopx/control_plane/agents/agent_scope_frontier.py +60 -0
  389. loopx/control_plane/agents/capability_gate.py +531 -0
  390. loopx/control_plane/agents/identity.py +140 -0
  391. loopx/control_plane/agents/legacy_migration.py +169 -0
  392. loopx/control_plane/agents/management_projection.py +658 -0
  393. loopx/control_plane/agents/material_frontier.py +608 -0
  394. loopx/control_plane/agents/material_handoff.py +156 -0
  395. loopx/control_plane/agents/multi_agent/__init__.py +1 -0
  396. loopx/control_plane/agents/multi_agent/codex_executable.py +207 -0
  397. loopx/control_plane/agents/multi_agent/collective_round_ledger.py +387 -0
  398. loopx/control_plane/agents/multi_agent/contract.py +474 -0
  399. loopx/control_plane/agents/multi_agent/recipe.py +110 -0
  400. loopx/control_plane/agents/multi_agent/role_successor.py +297 -0
  401. loopx/control_plane/agents/multi_agent/runtime_scripts.py +426 -0
  402. loopx/control_plane/agents/multi_agent/visible_launch_policy.py +149 -0
  403. loopx/control_plane/agents/multi_agent/visible_wake_scheduler.py +392 -0
  404. loopx/control_plane/agents/profile.py +216 -0
  405. loopx/control_plane/agents/runtime_model.py +73 -0
  406. loopx/control_plane/agents/subagent_activity.py +164 -0
  407. loopx/control_plane/agents/supervisor.py +544 -0
  408. loopx/control_plane/agents/supervisor_events.py +462 -0
  409. loopx/control_plane/agents/supervisor_inject.py +204 -0
  410. loopx/control_plane/agents/work_mode.py +56 -0
  411. loopx/control_plane/agents/workspace_guard.py +364 -0
  412. loopx/control_plane/effect_program.py +644 -0
  413. loopx/control_plane/goals/__init__.py +1 -0
  414. loopx/control_plane/goals/active_state_event_projection.py +103 -0
  415. loopx/control_plane/goals/active_state_metadata.py +47 -0
  416. loopx/control_plane/goals/active_state_sections.py +58 -0
  417. loopx/control_plane/goals/configure_goal_service.py +354 -0
  418. loopx/control_plane/goals/contract_health.py +132 -0
  419. loopx/control_plane/goals/dreaming.py +152 -0
  420. loopx/control_plane/goals/global_registry_health.py +199 -0
  421. loopx/control_plane/goals/global_registry_shadow.py +33 -0
  422. loopx/control_plane/goals/goal_channel.py +34 -0
  423. loopx/control_plane/goals/goal_channel_projection.py +560 -0
  424. loopx/control_plane/goals/goal_frontier/__init__.py +1917 -0
  425. loopx/control_plane/goals/goal_frontier/ack_policy.py +149 -0
  426. loopx/control_plane/goals/goal_frontier/outcome_continuity.py +437 -0
  427. loopx/control_plane/goals/goal_frontier/replan_rules.py +210 -0
  428. loopx/control_plane/goals/goal_frontier/semantic_history.py +314 -0
  429. loopx/control_plane/goals/goal_frontier/terminal.py +180 -0
  430. loopx/control_plane/goals/goal_vision.py +443 -0
  431. loopx/control_plane/goals/goal_vision_policy.py +36 -0
  432. loopx/control_plane/goals/goal_vision_state.py +62 -0
  433. loopx/control_plane/goals/goal_vision_wait.py +290 -0
  434. loopx/control_plane/goals/path_resolution.py +20 -0
  435. loopx/control_plane/goals/start_contract.py +206 -0
  436. loopx/control_plane/goals/vision_checkpoint.py +92 -0
  437. loopx/control_plane/handoff/__init__.py +1 -0
  438. loopx/control_plane/handoff/cross_runtime_impl_review.py +311 -0
  439. loopx/control_plane/handoff/delivery_contract.py +161 -0
  440. loopx/control_plane/handoff/handoff_runs.py +71 -0
  441. loopx/control_plane/handoff/project_handoff.py +155 -0
  442. loopx/control_plane/handoff/review_batch.py +463 -0
  443. loopx/control_plane/handoff/review_packet_context.py +216 -0
  444. loopx/control_plane/heartbeat/agent.py +173 -0
  445. loopx/control_plane/heartbeat/budget.py +66 -0
  446. loopx/control_plane/heartbeat/builder.py +501 -0
  447. loopx/control_plane/heartbeat/host.py +64 -0
  448. loopx/control_plane/heartbeat/rules.py +68 -0
  449. loopx/control_plane/heartbeat/task_body.py +759 -0
  450. loopx/control_plane/heartbeat/visible_goal.py +86 -0
  451. loopx/control_plane/projects/__init__.py +1 -0
  452. loopx/control_plane/projects/contract.py +25 -0
  453. loopx/control_plane/projects/registry.py +663 -0
  454. loopx/control_plane/quota/__init__.py +1 -0
  455. loopx/control_plane/quota/cli_projection.py +704 -0
  456. loopx/control_plane/quota/decision_summary.py +431 -0
  457. loopx/control_plane/quota/effect_program.py +152 -0
  458. loopx/control_plane/quota/error_codes.py +19 -0
  459. loopx/control_plane/quota/goal_boundary.py +464 -0
  460. loopx/control_plane/quota/heartbeat_receipt.py +277 -0
  461. loopx/control_plane/quota/heartbeat_recommendation.py +718 -0
  462. loopx/control_plane/quota/host_poll_receipts.py +162 -0
  463. loopx/control_plane/quota/live_decision.py +142 -0
  464. loopx/control_plane/quota/monitor_poll.py +786 -0
  465. loopx/control_plane/quota/policy_constants.py +40 -0
  466. loopx/control_plane/quota/projection_repair.py +262 -0
  467. loopx/control_plane/quota/recent_runs.py +210 -0
  468. loopx/control_plane/quota/scheduler_ack.py +490 -0
  469. loopx/control_plane/quota/selected_todo_projection.py +139 -0
  470. loopx/control_plane/quota/settlement.py +437 -0
  471. loopx/control_plane/quota/settlement_cli.py +246 -0
  472. loopx/control_plane/quota/settlement_validation.py +64 -0
  473. loopx/control_plane/quota/settlement_workspace_causality.py +180 -0
  474. loopx/control_plane/quota/should_run.py +249 -0
  475. loopx/control_plane/quota/should_run_packet.py +1165 -0
  476. loopx/control_plane/quota/should_run_prepare.py +675 -0
  477. loopx/control_plane/quota/slot_accounting.py +1123 -0
  478. loopx/control_plane/quota/spend_sources.py +11 -0
  479. loopx/control_plane/quota/stall_repair.py +397 -0
  480. loopx/control_plane/quota/states.py +29 -0
  481. loopx/control_plane/quota/task_orchestration.py +448 -0
  482. loopx/control_plane/quota/task_orchestration_admission.py +497 -0
  483. loopx/control_plane/quota/turn_envelope.py +889 -0
  484. loopx/control_plane/quota/usage_summary.py +140 -0
  485. loopx/control_plane/reward_memory.py +43 -0
  486. loopx/control_plane/runtime/__init__.py +2 -0
  487. loopx/control_plane/runtime/active_user_assisted_pilot.py +275 -0
  488. loopx/control_plane/runtime/agent_scoped_evidence_log.py +435 -0
  489. loopx/control_plane/runtime/decision_freshness.py +203 -0
  490. loopx/control_plane/runtime/event_ledger.py +197 -0
  491. loopx/control_plane/runtime/event_store_migration_bridge.py +196 -0
  492. loopx/control_plane/runtime/goal_project_route.py +70 -0
  493. loopx/control_plane/runtime/local_state_write_correctness.py +242 -0
  494. loopx/control_plane/runtime/promotion_readiness.py +152 -0
  495. loopx/control_plane/runtime/public_safety.py +120 -0
  496. loopx/control_plane/runtime/run_artifacts.py +78 -0
  497. loopx/control_plane/runtime/run_compaction.py +397 -0
  498. loopx/control_plane/runtime/run_context_retention.py +241 -0
  499. loopx/control_plane/runtime/run_history.py +132 -0
  500. loopx/control_plane/runtime/run_index_duplicates.py +205 -0
  501. loopx/control_plane/runtime/run_index_rebuild.py +263 -0
  502. loopx/control_plane/runtime/run_ingest_health.py +336 -0
  503. loopx/control_plane/runtime/runtime_projection_route.py +624 -0
  504. loopx/control_plane/runtime/runtime_projection_writer.py +98 -0
  505. loopx/control_plane/runtime/session_runtime.py +339 -0
  506. loopx/control_plane/runtime/shared_runtime_material_projection.py +332 -0
  507. loopx/control_plane/runtime/shared_runtime_refresh_projection.py +183 -0
  508. loopx/control_plane/runtime/stale_latest_run.py +90 -0
  509. loopx/control_plane/runtime/status_classifications.py +49 -0
  510. loopx/control_plane/runtime/status_projection_cache.py +235 -0
  511. loopx/control_plane/runtime/stride_observation.py +144 -0
  512. loopx/control_plane/runtime/time.py +39 -0
  513. loopx/control_plane/runtime/trajectory_hygiene.py +149 -0
  514. loopx/control_plane/runtime/validation_command.py +69 -0
  515. loopx/control_plane/scheduler/__init__.py +1 -0
  516. loopx/control_plane/scheduler/ack.py +329 -0
  517. loopx/control_plane/scheduler/arbitration.py +188 -0
  518. loopx/control_plane/scheduler/automation_liveness.py +183 -0
  519. loopx/control_plane/scheduler/execution_context.py +555 -0
  520. loopx/control_plane/scheduler/external_evidence_observation.py +428 -0
  521. loopx/control_plane/scheduler/monitor_display.py +143 -0
  522. loopx/control_plane/scheduler/monitor_poll_policy.py +161 -0
  523. loopx/control_plane/scheduler/monitor_poll_writeback.py +351 -0
  524. loopx/control_plane/scheduler/monitor_target.py +64 -0
  525. loopx/control_plane/scheduler/monitor_todo.py +146 -0
  526. loopx/control_plane/scheduler/monitor_wait.py +237 -0
  527. loopx/control_plane/scheduler/scheduler_hint.py +1284 -0
  528. loopx/control_plane/scheduler/state.py +354 -0
  529. loopx/control_plane/scheduler/state_transition_rules.py +179 -0
  530. loopx/control_plane/scheduler/time.py +10 -0
  531. loopx/control_plane/settlement_driver.py +293 -0
  532. loopx/control_plane/status/__init__.py +6 -0
  533. loopx/control_plane/status/active_state_projection.py +105 -0
  534. loopx/control_plane/status/agent_lane_projection.py +375 -0
  535. loopx/control_plane/status/attention_projection.py +74 -0
  536. loopx/control_plane/status/autonomous_replan_projection.py +103 -0
  537. loopx/control_plane/status/collection.py +140 -0
  538. loopx/control_plane/status/contract_projection.py +31 -0
  539. loopx/control_plane/status/dreaming_projection.py +52 -0
  540. loopx/control_plane/status/goal_attention_projection.py +157 -0
  541. loopx/control_plane/status/lifecycle_projection.py +110 -0
  542. loopx/control_plane/status/monitor_display_projection.py +69 -0
  543. loopx/control_plane/status/registry_health_projection.py +75 -0
  544. loopx/control_plane/status/run_projection.py +70 -0
  545. loopx/control_plane/status/runtime_summaries.py +161 -0
  546. loopx/control_plane/testing/__init__.py +1 -0
  547. loopx/control_plane/testing/actual_default_model_behavior_portfolio.py +1371 -0
  548. loopx/control_plane/testing/canary_harness.py +182 -0
  549. loopx/control_plane/testing/capability_monitor_repair_tool_behavior.py +674 -0
  550. loopx/control_plane/testing/cli_output_budget.py +807 -0
  551. loopx/control_plane/testing/cli_output_differential.py +250 -0
  552. loopx/control_plane/testing/cli_output_semantics.py +87 -0
  553. loopx/control_plane/testing/control_plane_composition_scenarios.py +225 -0
  554. loopx/control_plane/testing/decision_replay.py +268 -0
  555. loopx/control_plane/testing/doubao_model_behavior_actor.py +559 -0
  556. loopx/control_plane/testing/model_behavior_corpus.py +344 -0
  557. loopx/control_plane/testing/model_behavior_qualification.py +769 -0
  558. loopx/control_plane/testing/model_behavior_retained_cases.py +235 -0
  559. loopx/control_plane/testing/model_tool_behavior.py +536 -0
  560. loopx/control_plane/testing/onboarding_model_behavior_qualification.py +642 -0
  561. loopx/control_plane/testing/quota_fixtures.py +208 -0
  562. loopx/control_plane/testing/quota_should_run_parity.py +57 -0
  563. loopx/control_plane/testing/release_commit_qualification.py +671 -0
  564. loopx/control_plane/testing/replan_semantic_action_behavior.py +1302 -0
  565. loopx/control_plane/testing/scoped_gate_successor_tool_behavior.py +527 -0
  566. loopx/control_plane/testing/selected_todo_tool_behavior.py +1002 -0
  567. loopx/control_plane/testing/terminal_settlement_tool_behavior.py +656 -0
  568. loopx/control_plane/todos/__init__.py +1 -0
  569. loopx/control_plane/todos/active_state_editing.py +296 -0
  570. loopx/control_plane/todos/active_state_todo_parser.py +138 -0
  571. loopx/control_plane/todos/active_state_todos.py +175 -0
  572. loopx/control_plane/todos/addition.py +103 -0
  573. loopx/control_plane/todos/claim_visibility.py +253 -0
  574. loopx/control_plane/todos/completed_archive.py +139 -0
  575. loopx/control_plane/todos/completion_fence.py +49 -0
  576. loopx/control_plane/todos/completion_policy.py +153 -0
  577. loopx/control_plane/todos/completion_validation.py +248 -0
  578. loopx/control_plane/todos/completion_validation_accountability.py +27 -0
  579. loopx/control_plane/todos/completion_validation_projection.py +57 -0
  580. loopx/control_plane/todos/contract.py +1476 -0
  581. loopx/control_plane/todos/decision_scope.py +554 -0
  582. loopx/control_plane/todos/deferred_resume.py +546 -0
  583. loopx/control_plane/todos/durable_completion.py +201 -0
  584. loopx/control_plane/todos/event_writeback.py +484 -0
  585. loopx/control_plane/todos/frontier_deadline.py +132 -0
  586. loopx/control_plane/todos/handoff_gate.py +283 -0
  587. loopx/control_plane/todos/handoff_mode.py +444 -0
  588. loopx/control_plane/todos/handoff_note.py +202 -0
  589. loopx/control_plane/todos/line_update.py +361 -0
  590. loopx/control_plane/todos/list_projection.py +205 -0
  591. loopx/control_plane/todos/markdown.py +199 -0
  592. loopx/control_plane/todos/monitor_metadata.py +88 -0
  593. loopx/control_plane/todos/mutation_authority.py +299 -0
  594. loopx/control_plane/todos/projection.py +655 -0
  595. loopx/control_plane/todos/quota_summary.py +1138 -0
  596. loopx/control_plane/todos/route_continuation.py +267 -0
  597. loopx/control_plane/todos/succession_warning.py +174 -0
  598. loopx/control_plane/todos/summary_item.py +223 -0
  599. loopx/control_plane/todos/text.py +30 -0
  600. loopx/control_plane/todos/todo_index.py +226 -0
  601. loopx/control_plane/todos/todo_summary.py +1458 -0
  602. loopx/control_plane/todos/unblock_resume.py +326 -0
  603. loopx/control_plane/todos/user_gate.py +263 -0
  604. loopx/control_plane/todos/write_hint.py +63 -0
  605. loopx/control_plane/todos/write_policy.py +135 -0
  606. loopx/control_plane/turn_driver/__init__.py +85 -0
  607. loopx/control_plane/turn_driver/codex_cli.py +502 -0
  608. loopx/control_plane/turn_driver/driver.py +355 -0
  609. loopx/control_plane/turn_driver/executor.py +1468 -0
  610. loopx/control_plane/turn_driver/loop_controller.py +669 -0
  611. loopx/control_plane/turn_driver/settlement.py +318 -0
  612. loopx/control_plane/turn_driver/transaction.py +375 -0
  613. loopx/control_plane/work_items/__init__.py +1 -0
  614. loopx/control_plane/work_items/attention_fields.py +56 -0
  615. loopx/control_plane/work_items/attention_item.py +77 -0
  616. loopx/control_plane/work_items/attention_queue.py +322 -0
  617. loopx/control_plane/work_items/attention_routing.py +213 -0
  618. loopx/control_plane/work_items/autonomous_candidates.py +135 -0
  619. loopx/control_plane/work_items/autonomous_replan_ack.py +276 -0
  620. loopx/control_plane/work_items/autonomous_replan_obligation.py +786 -0
  621. loopx/control_plane/work_items/backlog_hygiene.py +59 -0
  622. loopx/control_plane/work_items/capability_monitor_fallback.py +221 -0
  623. loopx/control_plane/work_items/delivery_batch_scale.py +66 -0
  624. loopx/control_plane/work_items/delivery_outcome.py +152 -0
  625. loopx/control_plane/work_items/delivery_signals.py +113 -0
  626. loopx/control_plane/work_items/execution_obligation.py +235 -0
  627. loopx/control_plane/work_items/goal_route_hint.py +320 -0
  628. loopx/control_plane/work_items/interaction_contract.py +1540 -0
  629. loopx/control_plane/work_items/issue_meta_surface.py +159 -0
  630. loopx/control_plane/work_items/lifecycle.py +139 -0
  631. loopx/control_plane/work_items/operator_inbox.py +266 -0
  632. loopx/control_plane/work_items/outcome_followthrough.py +69 -0
  633. loopx/control_plane/work_items/primary_action.py +326 -0
  634. loopx/control_plane/work_items/progress_observation.py +630 -0
  635. loopx/control_plane/work_items/project_asset.py +675 -0
  636. loopx/control_plane/work_items/repair_delta.py +693 -0
  637. loopx/control_plane/work_items/runtime_capability_reentry.py +168 -0
  638. loopx/control_plane/work_items/semantic_replan_writeback.py +177 -0
  639. loopx/control_plane/work_items/status_contract.py +49 -0
  640. loopx/control_plane/work_items/task_graph.py +1046 -0
  641. loopx/control_plane/work_items/task_lease.py +1254 -0
  642. loopx/control_plane/work_items/task_lease_settlement.py +422 -0
  643. loopx/control_plane/work_items/work_lane.py +510 -0
  644. loopx/control_plane/work_items/work_lane_context.py +161 -0
  645. loopx/demo.py +247 -0
  646. loopx/diagnose.py +633 -0
  647. loopx/doctor.py +1251 -0
  648. loopx/domain_packs/__init__.py +1 -0
  649. loopx/domain_packs/issue_fix.py +571 -0
  650. loopx/domain_packs/ml_experiment.py +854 -0
  651. loopx/domain_state.py +137 -0
  652. loopx/dreaming.py +706 -0
  653. loopx/entrypoint.py +16 -0
  654. loopx/event_sourced_state.py +981 -0
  655. loopx/execution_profile.py +286 -0
  656. loopx/experiments/__init__.py +1 -0
  657. loopx/experiments/planner_worker/__init__.py +1 -0
  658. loopx/experiments/planner_worker/contract.py +523 -0
  659. loopx/experiments/planner_worker/runtime.py +391 -0
  660. loopx/experiments/planner_worker/traex.py +461 -0
  661. loopx/explore_graph.py +11 -0
  662. loopx/extensions/__init__.py +1 -0
  663. loopx/extensions/bundled.py +28 -0
  664. loopx/extensions/execution_envelope.py +126 -0
  665. loopx/extensions/lark/__init__.py +11 -0
  666. loopx/extensions/lark/event_collector.py +478 -0
  667. loopx/extensions/lark/event_collector_runtime.py +506 -0
  668. loopx/extensions/lark/event_inbox.py +454 -0
  669. loopx/extensions/lark/extension.toml +88 -0
  670. loopx/extensions/lark/goal_channel.py +44 -0
  671. loopx/extensions/lark/goal_channel_contracts.py +388 -0
  672. loopx/extensions/lark/goal_channel_lifecycle.py +218 -0
  673. loopx/extensions/lark/goal_channel_runtime.py +792 -0
  674. loopx/extensions/lark/goal_channel_setup.py +805 -0
  675. loopx/extensions/lark/goal_channel_targets.py +215 -0
  676. loopx/extensions/lark/goal_channel_transport.py +281 -0
  677. loopx/extensions/lark/inbox_reactions.py +650 -0
  678. loopx/extensions/lark/inbox_reply.py +430 -0
  679. loopx/extensions/lark/presentation/__init__.py +11 -0
  680. loopx/extensions/lark/presentation/explore_results.py +2276 -0
  681. loopx/extensions/lark/presentation/explore_singleflight.py +127 -0
  682. loopx/extensions/lark/presentation/explore_source_guard.py +121 -0
  683. loopx/extensions/lark/presentation/explore_stage_document.py +703 -0
  684. loopx/extensions/lark/presentation/explore_visual_integrity.py +122 -0
  685. loopx/extensions/lark/presentation/explore_visual_readback.py +452 -0
  686. loopx/extensions/lark/presentation/explore_visual_styles.py +156 -0
  687. loopx/extensions/lark/presentation/issue_fix_surface.py +612 -0
  688. loopx/extensions/lark/presentation/kanban.py +2791 -0
  689. loopx/extensions/lark/presentation/message_card.py +112 -0
  690. loopx/extensions/lark/presentation/periodic_report.py +261 -0
  691. loopx/extensions/lark/presentation/projection_rows.py +600 -0
  692. loopx/extensions/lark/presentation/record_io.py +95 -0
  693. loopx/extensions/lark/presentation/sync_receipt.py +145 -0
  694. loopx/extensions/lark/private_json.py +40 -0
  695. loopx/extensions/lark/provider.py +86 -0
  696. loopx/extensions/lark/reviewer_notification.py +604 -0
  697. loopx/extensions/manifest.py +385 -0
  698. loopx/extensions/openviking_periodic_report/__init__.py +17 -0
  699. loopx/extensions/openviking_periodic_report/activation.py +173 -0
  700. loopx/extensions/openviking_periodic_report/extension.toml +17 -0
  701. loopx/extensions/openviking_periodic_report/provider.py +355 -0
  702. loopx/extensions/openviking_periodic_report/sink.py +117 -0
  703. loopx/extensions/openviking_semantic_preference/__init__.py +5 -0
  704. loopx/extensions/openviking_semantic_preference/extension.toml +16 -0
  705. loopx/extensions/openviking_semantic_preference/history_export.py +484 -0
  706. loopx/extensions/openviking_semantic_preference/project_peer.py +68 -0
  707. loopx/extensions/openviking_semantic_preference/provider.py +312 -0
  708. loopx/extensions/presentation.py +979 -0
  709. loopx/extensions/process_runtime.py +204 -0
  710. loopx/extensions/readiness.py +168 -0
  711. loopx/extensions/runtime.py +931 -0
  712. loopx/extensions/scaffold.py +335 -0
  713. loopx/feedback.py +581 -0
  714. loopx/file_lock.py +382 -0
  715. loopx/global_registry.py +842 -0
  716. loopx/global_risks.py +970 -0
  717. loopx/global_todos.py +568 -0
  718. loopx/handoff_budget.py +28 -0
  719. loopx/heartbeat_prequota.py +80 -0
  720. loopx/heartbeat_prompt.py +159 -0
  721. loopx/help_surface.py +516 -0
  722. loopx/history.py +1507 -0
  723. loopx/host_loop_activation.py +1311 -0
  724. loopx/host_mode_planner.py +991 -0
  725. loopx/install_contract.py +1 -0
  726. loopx/interface_budget.py +196 -0
  727. loopx/long_task_cadence.py +208 -0
  728. loopx/materials.py +185 -0
  729. loopx/ml_experiment.py +3 -0
  730. loopx/onboarding.py +214 -0
  731. loopx/opencode2_goal_mode/README.md +81 -0
  732. loopx/opencode2_goal_mode/__init__.py +9 -0
  733. loopx/opencode2_goal_mode/opencode2-goal-worker.mjs +1018 -0
  734. loopx/opencode_goal_mode/README.md +99 -0
  735. loopx/opencode_goal_mode/__init__.py +13 -0
  736. loopx/opencode_goal_mode/goal-bridge-runtime.mjs +858 -0
  737. loopx/opencode_goal_mode/loopx-goal.js +8 -0
  738. loopx/operator_gate.py +420 -0
  739. loopx/orchestration.py +127 -0
  740. loopx/paths.py +59 -0
  741. loopx/pi_goal_mode/README.md +67 -0
  742. loopx/pi_goal_mode/__init__.py +13 -0
  743. loopx/pi_goal_mode/loopx-goal.ts +254 -0
  744. loopx/pi_goal_mode/pi-goal-loop-runtime.mjs +574 -0
  745. loopx/pr_review.py +1206 -0
  746. loopx/presentation/__init__.py +1 -0
  747. loopx/presentation/explore_views.py +1334 -0
  748. loopx/presentation/markdown.py +61 -0
  749. loopx/presentation/projection_source_reconcile.py +140 -0
  750. loopx/presentation/public_safety.py +42 -0
  751. loopx/presentation/renderers/__init__.py +17 -0
  752. loopx/presentation/renderers/goal_channel_html.py +269 -0
  753. loopx/presentation/renderers/periodic_report_html.py +786 -0
  754. loopx/presentation/renderers/periodic_report_markdown.py +184 -0
  755. loopx/presentation/renderers/quota_event_markdown.py +116 -0
  756. loopx/presentation/renderers/quota_markdown.py +1112 -0
  757. loopx/presentation/renderers/status_markdown.py +1570 -0
  758. loopx/presentation/renderers/trajectory_hygiene_markdown.py +39 -0
  759. loopx/presentation/renderers/turn_envelope_markdown.py +33 -0
  760. loopx/presentation/sinks/__init__.py +5 -0
  761. loopx/presentation/sinks/openviking_periodic_report.py +7 -0
  762. loopx/presentation/static_site.py +691 -0
  763. loopx/presets.py +369 -0
  764. loopx/project_alias.py +217 -0
  765. loopx/project_map.py +589 -0
  766. loopx/project_prompt.py +1153 -0
  767. loopx/project_skill_cli.py +125 -0
  768. loopx/project_skill_delivery.py +470 -0
  769. loopx/project_uninstall.py +462 -0
  770. loopx/promotion_gate.py +197 -0
  771. loopx/quota.py +1197 -0
  772. loopx/ready_score.py +413 -0
  773. loopx/registry.py +621 -0
  774. loopx/registry_writability.py +64 -0
  775. loopx/release_candidate.py +148 -0
  776. loopx/release_manifest.py +316 -0
  777. loopx/repository_identity.py +100 -0
  778. loopx/review_packet.py +1024 -0
  779. loopx/rollout_event_log.py +505 -0
  780. loopx/runtime.py +112 -0
  781. loopx/self_update.py +750 -0
  782. loopx/session_runtime.py +418 -0
  783. loopx/skill_install_readback.py +500 -0
  784. loopx/slash_command_install.py +1393 -0
  785. loopx/slash_commands.py +264 -0
  786. loopx/state_backup.py +573 -0
  787. loopx/state_migration.py +350 -0
  788. loopx/state_projection.py +809 -0
  789. loopx/state_refresh.py +1416 -0
  790. loopx/status.py +1383 -0
  791. loopx/status_server.py +935 -0
  792. loopx/summary_all.py +725 -0
  793. loopx/terminal_bench_agent.py +2056 -0
  794. loopx/thread_agent_binding.py +408 -0
  795. loopx/todo_followups.py +168 -0
  796. loopx/todo_suggestion_prompt.py +204 -0
  797. loopx/todos.py +2229 -0
  798. loopx/turn_identity.py +17 -0
  799. loopx/upgrade.py +1083 -0
  800. loopx/visible_governance.py +667 -0
  801. loopx/visible_multi_agent_launcher.py +1253 -0
  802. loopx/visible_multi_agent_tmux.py +429 -0
  803. loopx/worker_bridge.py +1574 -0
  804. loopx-0.4.8.dist-info/METADATA +708 -0
  805. loopx-0.4.8.dist-info/RECORD +811 -0
  806. loopx-0.4.8.dist-info/WHEEL +5 -0
  807. loopx-0.4.8.dist-info/entry_points.txt +5 -0
  808. loopx-0.4.8.dist-info/licenses/LICENSE +202 -0
  809. loopx-0.4.8.dist-info/licenses/LICENSE-MIT +21 -0
  810. loopx-0.4.8.dist-info/licenses/NOTICE +6 -0
  811. loopx-0.4.8.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1246 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from collections.abc import Callable
7
+ from pathlib import Path
8
+
9
+ from ..benchmark_adapters.terminal_bench import (
10
+ TERMINAL_BENCH_CODEX_INSTALL_STRATEGIES,
11
+ TERMINAL_BENCH_CODEX_INSTALL_STRATEGY_RUNTIME_INSTALL_IF_MISSING,
12
+ TERMINAL_BENCH_DEFAULT_DATASET,
13
+ TERMINAL_BENCH_DEFAULT_MODEL,
14
+ TERMINAL_BENCH_DEFAULT_TASK,
15
+ TERMINAL_BENCH_WORKER_CODEX_MATERIALIZATION_STRATEGIES,
16
+ TERMINAL_BENCH_WORKER_CODEX_MATERIALIZATION_STRATEGY_WORKER_PATH,
17
+ build_terminal_bench_environment_setup_probe_gate,
18
+ build_terminal_bench_result_finalization_gate,
19
+ launch_terminal_bench_case_run,
20
+ launch_terminal_bench_environment_setup_probe,
21
+ launch_terminal_bench_worker_materialization_probe,
22
+ poll_terminal_bench_worker_materialization_probe,
23
+ resume_terminal_bench_materialized_job,
24
+ summarize_terminal_bench_post_launch_materialization,
25
+ )
26
+ from ..status import compact_benchmark_run
27
+
28
+
29
+ PrintPayload = Callable[
30
+ [dict[str, object], str, Callable[[dict[str, object]], str]],
31
+ None,
32
+ ]
33
+ OutputFormat = Callable[[argparse.Namespace], str]
34
+
35
+ TERMINAL_BENCH_ENVIRONMENT_RESULT_COMMANDS = {
36
+ "environment-setup-gate",
37
+ "launch-environment-setup-probe",
38
+ "launch-worker-materialization-probe",
39
+ "launch-terminal-bench-run",
40
+ "poll-worker-materialization-probe",
41
+ "result-finalization-gate",
42
+ "resume-terminal-bench-job",
43
+ "summarize-post-launch",
44
+ }
45
+
46
+
47
+ def render_terminal_bench_post_launch_materialization_markdown(
48
+ payload: dict[str, object],
49
+ ) -> str:
50
+ lines = [
51
+ "# Terminal-Bench Post-Launch Materialization",
52
+ "",
53
+ f"- Schema: `{payload.get('schema_version')}`",
54
+ f"- Checked: `{payload.get('checked')}`",
55
+ f"- Ready for launch state: `{payload.get('ready_for_launch_state')}`",
56
+ "- Ready for compact result ingest: "
57
+ f"`{payload.get('ready_for_compact_result_ingest')}`",
58
+ "- Ready for compact failure marker: "
59
+ f"`{payload.get('ready_for_compact_failure_marker')}`",
60
+ f"- First blocker: `{payload.get('first_blocker')}`",
61
+ f"- Job name: `{payload.get('job_name')}`",
62
+ f"- Jobs dir present: `{payload.get('jobs_dir_present')}`",
63
+ f"- Job root present: `{payload.get('job_root_present')}`",
64
+ f"- Job lock present: `{payload.get('job_lock_present')}`",
65
+ f"- Job result present: `{payload.get('job_result_present')}`",
66
+ f"- Trial results: `{payload.get('trial_result_present_count')}`",
67
+ f"- Raw paths recorded: `{payload.get('raw_paths_recorded')}`",
68
+ f"- Raw logs read: `{payload.get('raw_logs_read')}`",
69
+ f"- Task text read: `{payload.get('raw_task_text_read')}`",
70
+ f"- Trajectory read: `{payload.get('trajectory_read')}`",
71
+ f"- External handle kind: `{payload.get('external_handle_kind')}`",
72
+ f"- External handle state: `{payload.get('external_handle_state')}`",
73
+ f"- External handle terminal: `{payload.get('external_handle_terminal')}`",
74
+ f"- Compact monitor class: `{payload.get('compact_monitor_class')}`",
75
+ "- Stale active reconcile requested: "
76
+ f"`{payload.get('stale_active_reconcile_requested')}`",
77
+ f"- Compact failure class: `{payload.get('compact_failure_class')}`",
78
+ ]
79
+ if payload.get("error"):
80
+ lines.append(f"- Error: {payload.get('error')}")
81
+ return "\n".join(lines) + "\n"
82
+
83
+
84
+ def render_terminal_bench_result_finalization_gate_markdown(
85
+ payload: dict[str, object],
86
+ ) -> str:
87
+ conditions = (
88
+ payload.get("gate_conditions")
89
+ if isinstance(payload.get("gate_conditions"), dict)
90
+ else {}
91
+ )
92
+ constraints = (
93
+ payload.get("rerun_constraints")
94
+ if isinstance(payload.get("rerun_constraints"), dict)
95
+ else {}
96
+ )
97
+ read_boundary = (
98
+ payload.get("read_boundary")
99
+ if isinstance(payload.get("read_boundary"), dict)
100
+ else {}
101
+ )
102
+ lines = [
103
+ "# Terminal-Bench Result Finalization Gate",
104
+ "",
105
+ f"- Schema: `{payload.get('schema_version')}`",
106
+ f"- Decision: `{payload.get('decision')}`",
107
+ f"- Failure class: `{payload.get('failure_class')}`",
108
+ f"- Root cause: `{payload.get('root_cause')}`",
109
+ f"- First blocker: `{payload.get('first_blocker')}`",
110
+ f"- Repair class: `{payload.get('repair_class')}`",
111
+ "- Result finalization repair required: "
112
+ f"`{payload.get('result_finalization_repair_required')}`",
113
+ "- Repaired baseline rerun allowed: "
114
+ f"`{payload.get('repaired_baseline_rerun_allowed')}`",
115
+ f"- Next action: {payload.get('next_allowed_action')}",
116
+ f"- Launch state countable: `{conditions.get('launch_state_countable')}`",
117
+ f"- External handle terminal: `{conditions.get('external_handle_terminal')}`",
118
+ f"- No trial result: `{conditions.get('no_trial_result')}`",
119
+ f"- Baseline only: `{constraints.get('baseline_only')}`",
120
+ f"- Max reruns: `{constraints.get('max_reruns')}`",
121
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
122
+ f"- Raw artifacts read: `{read_boundary.get('raw_artifacts_read')}`",
123
+ ]
124
+ if payload.get("error"):
125
+ lines.append(f"- Error: {payload.get('error')}")
126
+ return "\n".join(lines) + "\n"
127
+
128
+
129
+
130
+ def render_terminal_bench_environment_setup_gate_markdown(
131
+ payload: dict[str, object],
132
+ ) -> str:
133
+ capability = (
134
+ payload.get("harbor_run_help_capability")
135
+ if isinstance(payload.get("harbor_run_help_capability"), dict)
136
+ else {}
137
+ )
138
+ contract = (
139
+ payload.get("probe_contract")
140
+ if isinstance(payload.get("probe_contract"), dict)
141
+ else {}
142
+ )
143
+ read_boundary = (
144
+ payload.get("read_boundary")
145
+ if isinstance(payload.get("read_boundary"), dict)
146
+ else {}
147
+ )
148
+ lines = [
149
+ "# Terminal-Bench Environment Setup Gate",
150
+ "",
151
+ f"- Schema: `{payload.get('schema_version')}`",
152
+ f"- Benchmark: `{payload.get('benchmark_id')}`",
153
+ f"- Task: `{payload.get('task_id')}`",
154
+ f"- Preflight ready: `{payload.get('preflight_ready')}`",
155
+ "- Previous setup failure: "
156
+ f"`{payload.get('previous_environment_setup_failure_present')}`",
157
+ f"- Help probe ok: `{capability.get('probe_ok')}`",
158
+ f"- Direct setup-only route: `{payload.get('direct_setup_only_route_allowed')}`",
159
+ "- NOP disable-verification route: "
160
+ f"`{payload.get('nop_disable_verification_probe_allowed')}`",
161
+ "- Environment setup probe allowed: "
162
+ f"`{payload.get('environment_setup_probe_allowed')}`",
163
+ f"- Same-task repeat allowed: `{payload.get('same_task_repeat_allowed')}`",
164
+ f"- First blocker: `{payload.get('first_blocker')}`",
165
+ f"- Next action: {payload.get('next_allowed_action')}",
166
+ f"- Probe agent: `{contract.get('agent')}`",
167
+ f"- No upload / submit eligible: `{contract.get('no_upload')}` / `{contract.get('submit_eligible')}`",
168
+ f"- Codex invoked: `{contract.get('codex_invoked')}`",
169
+ f"- Compact only: `{read_boundary.get('compact_only')}`",
170
+ f"- Raw help recorded: `{read_boundary.get('raw_help_recorded')}`",
171
+ ]
172
+ return "\n".join(lines) + "\n"
173
+
174
+
175
+ def render_terminal_bench_environment_setup_probe_launch_markdown(
176
+ payload: dict[str, object],
177
+ ) -> str:
178
+ post_launch = (
179
+ payload.get("post_launch_materialization")
180
+ if isinstance(payload.get("post_launch_materialization"), dict)
181
+ else {}
182
+ )
183
+ boundary = payload.get("boundary") if isinstance(payload.get("boundary"), dict) else {}
184
+ lines = [
185
+ "# Terminal-Bench Environment Setup Probe Launch",
186
+ "",
187
+ f"- Schema: `{payload.get('schema_version')}`",
188
+ f"- Dry run: `{payload.get('dry_run')}`",
189
+ f"- Run: `{payload.get('run_basename')}`",
190
+ f"- Job: `{payload.get('job_name')}`",
191
+ f"- Process started: `{payload.get('process_started')}`",
192
+ f"- Process state: `{payload.get('process_state')}`",
193
+ f"- Return code: `{payload.get('returncode')}`",
194
+ f"- Timed out: `{payload.get('process_timed_out')}`",
195
+ f"- Materialization wait seconds: `{payload.get('materialization_wait_seconds')}`",
196
+ f"- Materialization wait timed out: `{payload.get('materialization_wait_timed_out')}`",
197
+ f"- First blocker: `{payload.get('first_blocker')}`",
198
+ f"- Compact failure: `{payload.get('compact_failure_class')}`",
199
+ f"- Ready for launch state: `{payload.get('ready_for_launch_state')}`",
200
+ f"- Ready for compact ingest: `{payload.get('ready_for_compact_result_ingest')}`",
201
+ f"- Ready for failure marker: `{payload.get('ready_for_compact_failure_marker')}`",
202
+ f"- Post-launch blocker: `{post_launch.get('first_blocker')}`",
203
+ f"- No upload / submit eligible: `{boundary.get('no_upload')}` / `{boundary.get('submit_eligible')}`",
204
+ f"- Raw logs read: `{boundary.get('raw_logs_read')}`",
205
+ f"- Task text read: `{boundary.get('task_text_read')}`",
206
+ ]
207
+ return "\n".join(lines) + "\n"
208
+
209
+
210
+ def render_terminal_bench_worker_materialization_probe_launch_markdown(
211
+ payload: dict[str, object],
212
+ ) -> str:
213
+ post_launch = (
214
+ payload.get("post_launch_materialization")
215
+ if isinstance(payload.get("post_launch_materialization"), dict)
216
+ else {}
217
+ )
218
+ boundary = payload.get("boundary") if isinstance(payload.get("boundary"), dict) else {}
219
+ command_shape = (
220
+ payload.get("command_shape")
221
+ if isinstance(payload.get("command_shape"), dict)
222
+ else {}
223
+ )
224
+ lines = [
225
+ "# Terminal-Bench Worker Materialization Probe Launch",
226
+ "",
227
+ f"- Schema: `{payload.get('schema_version')}`",
228
+ f"- Dry run: `{payload.get('dry_run')}`",
229
+ f"- Run: `{payload.get('run_basename')}`",
230
+ f"- Job: `{payload.get('job_name')}`",
231
+ f"- Process started: `{payload.get('process_started')}`",
232
+ f"- Process state: `{payload.get('process_state')}`",
233
+ f"- Return code: `{payload.get('returncode')}`",
234
+ f"- Timed out: `{payload.get('process_timed_out')}`",
235
+ f"- Resume after materialization: `{payload.get('resume_after_materialization')}`",
236
+ f"- Resume attempted: `{payload.get('resume_after_materialization_attempted')}`",
237
+ f"- First blocker: `{payload.get('first_blocker')}`",
238
+ f"- Compact failure: `{payload.get('compact_failure_class')}`",
239
+ f"- Ready for launch state: `{payload.get('ready_for_launch_state')}`",
240
+ f"- Ready for compact ingest: `{payload.get('ready_for_compact_result_ingest')}`",
241
+ f"- Post-launch blocker: `{post_launch.get('first_blocker')}`",
242
+ "- Probe-only kwarg: "
243
+ f"`{command_shape.get('worker_materialization_probe_only')}`",
244
+ f"- No upload / submit eligible: `{boundary.get('no_upload')}` / `{boundary.get('submit_eligible')}`",
245
+ f"- Task solver invoked by probe: `{boundary.get('task_solver_invoked_by_probe')}`",
246
+ f"- Raw logs read: `{boundary.get('raw_logs_read')}`",
247
+ f"- Task text read: `{boundary.get('task_text_read')}`",
248
+ ]
249
+ return "\n".join(lines) + "\n"
250
+
251
+
252
+ def render_terminal_bench_case_run_launch_markdown(
253
+ payload: dict[str, object],
254
+ ) -> str:
255
+ post_launch = (
256
+ payload.get("post_launch_materialization")
257
+ if isinstance(payload.get("post_launch_materialization"), dict)
258
+ else {}
259
+ )
260
+ boundary = payload.get("boundary") if isinstance(payload.get("boundary"), dict) else {}
261
+ command_shape = (
262
+ payload.get("command_shape")
263
+ if isinstance(payload.get("command_shape"), dict)
264
+ else {}
265
+ )
266
+ lines = [
267
+ "# Terminal-Bench Case Run Launch",
268
+ "",
269
+ f"- Schema: `{payload.get('schema_version')}`",
270
+ f"- Dry run: `{payload.get('dry_run')}`",
271
+ f"- Run: `{payload.get('run_basename')}`",
272
+ f"- Job: `{payload.get('job_name')}`",
273
+ f"- Process started: `{payload.get('process_started')}`",
274
+ f"- Process state: `{payload.get('process_state')}`",
275
+ f"- Return code: `{payload.get('returncode')}`",
276
+ f"- Timed out: `{payload.get('process_timed_out')}`",
277
+ f"- First blocker: `{payload.get('first_blocker')}`",
278
+ f"- Compact failure: `{payload.get('compact_failure_class')}`",
279
+ f"- Ready for launch state: `{payload.get('ready_for_launch_state')}`",
280
+ f"- Ready for compact ingest: `{payload.get('ready_for_compact_result_ingest')}`",
281
+ f"- Post-launch blocker: `{post_launch.get('first_blocker')}`",
282
+ "- Probe-only kwarg: "
283
+ f"`{command_shape.get('worker_materialization_probe_only')}`",
284
+ f"- No upload / submit eligible: `{boundary.get('no_upload')}` / `{boundary.get('submit_eligible')}`",
285
+ f"- Task solver invoked: `{boundary.get('task_solver_invoked')}`",
286
+ f"- Model API expected: `{boundary.get('model_api_expected')}`",
287
+ f"- Raw logs read: `{boundary.get('raw_logs_read')}`",
288
+ f"- Task text read: `{boundary.get('task_text_read')}`",
289
+ ]
290
+ return "\n".join(lines) + "\n"
291
+
292
+
293
+ def render_terminal_bench_worker_materialization_probe_poll_markdown(
294
+ payload: dict[str, object],
295
+ ) -> str:
296
+ post_launch = (
297
+ payload.get("post_launch_materialization")
298
+ if isinstance(payload.get("post_launch_materialization"), dict)
299
+ else {}
300
+ )
301
+ boundary = payload.get("boundary") if isinstance(payload.get("boundary"), dict) else {}
302
+ pid_state = (
303
+ payload.get("pid_state") if isinstance(payload.get("pid_state"), dict) else {}
304
+ )
305
+ lines = [
306
+ "# Terminal-Bench Worker Materialization Probe Poll",
307
+ "",
308
+ f"- Schema: `{payload.get('schema_version')}`",
309
+ f"- Run: `{payload.get('run_basename')}`",
310
+ f"- Job: `{payload.get('job_name')}`",
311
+ f"- Process state: `{payload.get('process_state')}`",
312
+ f"- PID file present/parsed: `{pid_state.get('pid_file_present')}`/`{pid_state.get('pid_parse_ok')}`",
313
+ f"- First blocker: `{payload.get('first_blocker')}`",
314
+ f"- Compact failure: `{payload.get('compact_failure_class')}`",
315
+ f"- Ready for launch state: `{payload.get('ready_for_launch_state')}`",
316
+ f"- Ready for compact ingest: `{payload.get('ready_for_compact_result_ingest')}`",
317
+ f"- Ready for failure marker: `{payload.get('ready_for_compact_failure_marker')}`",
318
+ f"- Post-launch blocker: `{post_launch.get('first_blocker')}`",
319
+ f"- No upload / submit eligible: `{boundary.get('no_upload')}` / `{boundary.get('submit_eligible')}`",
320
+ f"- Raw logs read: `{boundary.get('raw_logs_read')}`",
321
+ f"- Task text read: `{boundary.get('task_text_read')}`",
322
+ f"- Command line read: `{boundary.get('command_line_read')}`",
323
+ ]
324
+ return "\n".join(lines) + "\n"
325
+
326
+
327
+ def render_terminal_bench_resume_observation_markdown(
328
+ payload: dict[str, object],
329
+ ) -> str:
330
+ post_launch = (
331
+ payload.get("post_launch_materialization")
332
+ if isinstance(payload.get("post_launch_materialization"), dict)
333
+ else {}
334
+ )
335
+ boundary = payload.get("boundary") if isinstance(payload.get("boundary"), dict) else {}
336
+ command_shape = (
337
+ payload.get("command_shape")
338
+ if isinstance(payload.get("command_shape"), dict)
339
+ else {}
340
+ )
341
+ lines = [
342
+ "# Terminal-Bench Job Resume",
343
+ "",
344
+ f"- Schema: `{payload.get('schema_version')}`",
345
+ f"- Dry run: `{payload.get('dry_run')}`",
346
+ f"- Run: `{payload.get('run_basename')}`",
347
+ f"- Job: `{payload.get('job_name')}`",
348
+ f"- Process started: `{payload.get('process_started')}`",
349
+ f"- Process state: `{payload.get('process_state')}`",
350
+ f"- Return code: `{payload.get('returncode')}`",
351
+ f"- Timed out: `{payload.get('process_timed_out')}`",
352
+ f"- First blocker: `{payload.get('first_blocker')}`",
353
+ f"- Compact failure: `{payload.get('compact_failure_class')}`",
354
+ f"- Ready for launch state: `{payload.get('ready_for_launch_state')}`",
355
+ f"- Ready for compact ingest: `{payload.get('ready_for_compact_result_ingest')}`",
356
+ f"- Ready for failure marker: `{payload.get('ready_for_compact_failure_marker')}`",
357
+ f"- Post-launch blocker: `{post_launch.get('first_blocker')}`",
358
+ f"- Uses Harbor job resume: `{command_shape.get('uses_harbor_job_resume')}`",
359
+ f"- No upload / submit eligible: `{boundary.get('no_upload')}` / `{boundary.get('submit_eligible')}`",
360
+ f"- Resume invoked: `{boundary.get('resume_invoked')}`",
361
+ f"- Raw logs read: `{boundary.get('raw_logs_read')}`",
362
+ f"- Task text read: `{boundary.get('task_text_read')}`",
363
+ ]
364
+ return "\n".join(lines) + "\n"
365
+
366
+
367
+
368
+
369
+ def register_terminal_bench_environment_result_commands(
370
+ benchmark_subparsers: argparse._SubParsersAction,
371
+ add_subcommand_format: Callable[[argparse.ArgumentParser], None],
372
+ ) -> None:
373
+ benchmark_post_launch_parser = benchmark_subparsers.add_parser(
374
+ "summarize-post-launch",
375
+ help=(
376
+ "Summarize whether a Terminal-Bench Harbor launch materialized a "
377
+ "pollable job directory. This records booleans, counts, and job "
378
+ "basenames only; it does not read logs, task text, trajectories, "
379
+ "Docker, model APIs, or uploads."
380
+ ),
381
+ )
382
+ add_subcommand_format(benchmark_post_launch_parser)
383
+ benchmark_post_launch_parser.add_argument(
384
+ "benchmark_name",
385
+ choices=["terminal-bench"],
386
+ help="Benchmark family.",
387
+ )
388
+ benchmark_post_launch_parser.add_argument(
389
+ "--jobs-dir",
390
+ required=True,
391
+ help=(
392
+ "Private Harbor jobs directory to check. The value is used only for "
393
+ "local filesystem probing and is not echoed in output."
394
+ ),
395
+ )
396
+ benchmark_post_launch_parser.add_argument(
397
+ "--job-name",
398
+ help="Expected Harbor job directory basename.",
399
+ )
400
+ benchmark_post_launch_parser.add_argument(
401
+ "--detached-process-state",
402
+ choices=["unknown", "running", "ended"],
403
+ default="unknown",
404
+ help=(
405
+ "Optional public-safe state of the detached worker process observed "
406
+ "by an external handle. When ended and no compact result exists, the "
407
+ "summary emits a compact failure marker instead of an open-ended "
408
+ "polling blocker."
409
+ ),
410
+ )
411
+ benchmark_post_launch_parser.add_argument(
412
+ "--require-ready-for-launch-state",
413
+ action="store_true",
414
+ help=(
415
+ "Return non-zero unless the job root and lock.json are present. "
416
+ "Use this before declaring a private launch state durable."
417
+ ),
418
+ )
419
+ benchmark_post_launch_parser.add_argument(
420
+ "--reconcile-stale-active",
421
+ action="store_true",
422
+ help=(
423
+ "When an externally ended worker still has a stale active Harbor "
424
+ "job with no trial result, emit a compact failure marker instead "
425
+ "of leaving the state as polling. This does not read logs, task "
426
+ "text, trajectories, Docker, model APIs, or uploads."
427
+ ),
428
+ )
429
+
430
+ benchmark_result_finalization_gate_parser = benchmark_subparsers.add_parser(
431
+ "result-finalization-gate",
432
+ help=(
433
+ "Reduce compact Terminal-Bench post-launch evidence into a "
434
+ "result-finalization repair and repaired-baseline rerun gate. "
435
+ "This reads only compact JSON, not logs, task text, trajectories, "
436
+ "Docker, model APIs, uploads, or local paths."
437
+ ),
438
+ )
439
+ add_subcommand_format(benchmark_result_finalization_gate_parser)
440
+ benchmark_result_finalization_gate_parser.add_argument(
441
+ "benchmark_name",
442
+ choices=["terminal-bench"],
443
+ help="Benchmark family.",
444
+ )
445
+ benchmark_result_finalization_gate_parser.add_argument(
446
+ "--post-launch-json",
447
+ required=True,
448
+ help=(
449
+ "Path to compact terminal_bench_post_launch_materialization_v0 JSON. "
450
+ "Use '-' to read stdin."
451
+ ),
452
+ )
453
+ benchmark_result_finalization_gate_parser.add_argument(
454
+ "--max-repaired-baseline-reruns",
455
+ type=int,
456
+ default=1,
457
+ help="Maximum repaired baseline reruns this gate may authorize.",
458
+ )
459
+ benchmark_result_finalization_gate_parser.add_argument(
460
+ "--require-rerun-allowed",
461
+ action="store_true",
462
+ help="Return non-zero unless the gate allows exactly one repaired baseline rerun.",
463
+ )
464
+
465
+
466
+ benchmark_environment_setup_gate_parser = benchmark_subparsers.add_parser(
467
+ "environment-setup-gate",
468
+ help=(
469
+ "Gate a Terminal-Bench same-task environment setup probe after a "
470
+ "compact environment_setup failure. Reads compact JSON and optional "
471
+ "Harbor help only; it does not start Docker, Codex, model APIs, "
472
+ "uploads, or benchmark tasks."
473
+ ),
474
+ )
475
+ add_subcommand_format(benchmark_environment_setup_gate_parser)
476
+ benchmark_environment_setup_gate_parser.add_argument(
477
+ "benchmark_name",
478
+ choices=["terminal-bench"],
479
+ help="Benchmark family. Only terminal-bench is supported.",
480
+ )
481
+ benchmark_environment_setup_gate_parser.add_argument(
482
+ "--dataset",
483
+ default=TERMINAL_BENCH_DEFAULT_DATASET,
484
+ )
485
+ benchmark_environment_setup_gate_parser.add_argument(
486
+ "--include-task-name",
487
+ default=TERMINAL_BENCH_DEFAULT_TASK,
488
+ )
489
+ benchmark_environment_setup_gate_parser.add_argument(
490
+ "--preflight-json",
491
+ help="Path to compact preflight or benchmark-run append JSON.",
492
+ )
493
+ benchmark_environment_setup_gate_parser.add_argument(
494
+ "--benchmark-run-json",
495
+ required=True,
496
+ help="Path to the compact benchmark_run_v0 with the prior environment_setup failure.",
497
+ )
498
+ benchmark_environment_setup_gate_parser.add_argument(
499
+ "--probe-runner-help",
500
+ action="store_true",
501
+ help=(
502
+ "Probe `harbor run --help` via uvx and store only compact capability "
503
+ "booleans. This does not run Docker, Codex, model APIs, uploads, or tasks."
504
+ ),
505
+ )
506
+ benchmark_environment_setup_gate_parser.add_argument(
507
+ "--harbor-run-help-text",
508
+ help=(
509
+ "Fixture help text for deterministic tests. The raw text is consumed "
510
+ "only to derive capability booleans and is not emitted."
511
+ ),
512
+ )
513
+ benchmark_environment_setup_gate_parser.add_argument(
514
+ "--require-probe-allowed",
515
+ action="store_true",
516
+ help="Return non-zero unless a no-upload environment setup probe route is allowed.",
517
+ )
518
+
519
+ benchmark_environment_setup_probe_launch_parser = benchmark_subparsers.add_parser(
520
+ "launch-environment-setup-probe",
521
+ help=(
522
+ "Launch a gated Terminal-Bench no-upload NOP/disable-verification "
523
+ "environment setup probe and emit only compact process/materialization "
524
+ "signals. Stdout/stderr stay in a private log and are not read."
525
+ ),
526
+ )
527
+ add_subcommand_format(benchmark_environment_setup_probe_launch_parser)
528
+ benchmark_environment_setup_probe_launch_parser.add_argument(
529
+ "benchmark_name",
530
+ choices=["terminal-bench"],
531
+ help="Benchmark family. Only terminal-bench is supported.",
532
+ )
533
+ benchmark_environment_setup_probe_launch_parser.add_argument(
534
+ "--gate-json",
535
+ required=True,
536
+ help="Path to terminal_bench_environment_setup_probe_gate_v0 JSON.",
537
+ )
538
+ benchmark_environment_setup_probe_launch_parser.add_argument(
539
+ "--run-root",
540
+ required=True,
541
+ help=(
542
+ "Private run root for launcher artifacts. The value is used locally "
543
+ "and only its basename is emitted."
544
+ ),
545
+ )
546
+ benchmark_environment_setup_probe_launch_parser.add_argument(
547
+ "--jobs-dir",
548
+ required=True,
549
+ help=(
550
+ "Private Harbor jobs directory. The value is used locally and is not "
551
+ "echoed in output."
552
+ ),
553
+ )
554
+ benchmark_environment_setup_probe_launch_parser.add_argument(
555
+ "--wait-seconds",
556
+ type=int,
557
+ default=20,
558
+ help="Seconds to wait for an immediate launcher exit before returning running state.",
559
+ )
560
+ benchmark_environment_setup_probe_launch_parser.add_argument(
561
+ "--execute",
562
+ action="store_true",
563
+ help="Actually start the local no-upload setup probe. Without this flag, dry-run only.",
564
+ )
565
+
566
+ benchmark_worker_materialization_probe_launch_parser = benchmark_subparsers.add_parser(
567
+ "launch-worker-materialization-probe",
568
+ help=(
569
+ "Launch a Terminal-Bench no-upload Codex worker materialization "
570
+ "probe that stops after Codex setup/preflight and emits compact "
571
+ "process/materialization signals. Stdout/stderr stay in a private "
572
+ "log and are not read."
573
+ ),
574
+ )
575
+ add_subcommand_format(benchmark_worker_materialization_probe_launch_parser)
576
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
577
+ "benchmark_name",
578
+ choices=["terminal-bench"],
579
+ help="Benchmark family. Only terminal-bench is supported.",
580
+ )
581
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
582
+ "--mode",
583
+ choices=["codex-goal-mode", "hardened-codex"],
584
+ default="codex-goal-mode",
585
+ help="Baseline worker surface to materialize without solving the task.",
586
+ )
587
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
588
+ "--dataset",
589
+ default=TERMINAL_BENCH_DEFAULT_DATASET,
590
+ )
591
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
592
+ "--include-task-name",
593
+ default=TERMINAL_BENCH_DEFAULT_TASK,
594
+ )
595
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
596
+ "--model",
597
+ default=TERMINAL_BENCH_DEFAULT_MODEL,
598
+ )
599
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
600
+ "--job-name",
601
+ help="Optional public-safe Harbor job basename.",
602
+ )
603
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
604
+ "--worker-codex-materialization-strategy",
605
+ choices=TERMINAL_BENCH_WORKER_CODEX_MATERIALIZATION_STRATEGIES,
606
+ default=TERMINAL_BENCH_WORKER_CODEX_MATERIALIZATION_STRATEGY_WORKER_PATH,
607
+ help=(
608
+ "Worker Codex materialization route to probe before task solving. "
609
+ "Defaults to the fail-fast worker PATH probe."
610
+ ),
611
+ )
612
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
613
+ "--run-root",
614
+ required=True,
615
+ help=(
616
+ "Private run root for launcher artifacts. The value is used locally "
617
+ "and only its basename is emitted."
618
+ ),
619
+ )
620
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
621
+ "--jobs-dir",
622
+ required=True,
623
+ help=(
624
+ "Private Harbor jobs directory. The value is used locally and is not "
625
+ "echoed in output."
626
+ ),
627
+ )
628
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
629
+ "--wait-seconds",
630
+ type=int,
631
+ default=20,
632
+ help="Seconds to wait for an immediate launcher exit before returning running state.",
633
+ )
634
+ benchmark_worker_materialization_probe_launch_parser.add_argument(
635
+ "--execute",
636
+ action="store_true",
637
+ help=(
638
+ "Actually start the local no-upload worker materialization probe. "
639
+ "Without this flag, dry-run only."
640
+ ),
641
+ )
642
+
643
+ benchmark_case_run_launch_parser = benchmark_subparsers.add_parser(
644
+ "launch-terminal-bench-run",
645
+ help=(
646
+ "Launch one Terminal-Bench no-upload case run with compact "
647
+ "process/materialization reporting. Stdout/stderr stay in a "
648
+ "private log and are not read."
649
+ ),
650
+ )
651
+ add_subcommand_format(benchmark_case_run_launch_parser)
652
+ benchmark_case_run_launch_parser.add_argument(
653
+ "benchmark_name",
654
+ choices=["terminal-bench"],
655
+ help="Benchmark family. Only terminal-bench is supported.",
656
+ )
657
+ benchmark_case_run_launch_parser.add_argument(
658
+ "--mode",
659
+ choices=[
660
+ "codex-goal-mode",
661
+ "codex-app-server-goal",
662
+ "hardened-codex",
663
+ "codex-loopx",
664
+ "loopx-managed-codex",
665
+ ],
666
+ default="codex-goal-mode",
667
+ help="Terminal-Bench worker surface to run.",
668
+ )
669
+ benchmark_case_run_launch_parser.add_argument(
670
+ "--dataset",
671
+ default=TERMINAL_BENCH_DEFAULT_DATASET,
672
+ )
673
+ benchmark_case_run_launch_parser.add_argument(
674
+ "--include-task-name",
675
+ default=TERMINAL_BENCH_DEFAULT_TASK,
676
+ )
677
+ benchmark_case_run_launch_parser.add_argument(
678
+ "--model",
679
+ default=TERMINAL_BENCH_DEFAULT_MODEL,
680
+ )
681
+ benchmark_case_run_launch_parser.add_argument(
682
+ "--job-name",
683
+ help="Optional public-safe Harbor job basename.",
684
+ )
685
+ benchmark_case_run_launch_parser.add_argument(
686
+ "--run-root",
687
+ required=True,
688
+ help=(
689
+ "Private run root for launcher artifacts. The value is used locally "
690
+ "and only its basename is emitted."
691
+ ),
692
+ )
693
+ benchmark_case_run_launch_parser.add_argument(
694
+ "--jobs-dir",
695
+ required=True,
696
+ help=(
697
+ "Private Harbor jobs directory. The value is used locally and is not "
698
+ "echoed in output."
699
+ ),
700
+ )
701
+ benchmark_case_run_launch_parser.add_argument(
702
+ "--wait-seconds",
703
+ type=int,
704
+ default=20,
705
+ help="Seconds to wait for an immediate launcher exit before returning running state.",
706
+ )
707
+ benchmark_case_run_launch_parser.add_argument(
708
+ "--materialization-wait-seconds",
709
+ type=int,
710
+ default=0,
711
+ help=(
712
+ "Seconds to wait for the Harbor job root or a compact startup "
713
+ "failure marker after launching. This observes only process state "
714
+ "and compact job materialization signals."
715
+ ),
716
+ )
717
+ benchmark_case_run_launch_parser.add_argument(
718
+ "--resume-after-materialization",
719
+ action="store_true",
720
+ help=(
721
+ "If the launch driver exits after a Harbor job materializes with "
722
+ "active pending/running trials but no trial result, run one "
723
+ "no-upload `harbor job resume` driver and report compact state."
724
+ ),
725
+ )
726
+ benchmark_case_run_launch_parser.add_argument("--timeout-multiplier", type=float)
727
+ benchmark_case_run_launch_parser.add_argument("--agent-timeout-multiplier", type=float)
728
+ benchmark_case_run_launch_parser.add_argument("--verifier-timeout-multiplier", type=float)
729
+ benchmark_case_run_launch_parser.add_argument("--agent-setup-timeout-multiplier", type=float)
730
+ benchmark_case_run_launch_parser.add_argument(
731
+ "--environment-build-timeout-multiplier",
732
+ type=float,
733
+ )
734
+ benchmark_case_run_launch_parser.add_argument(
735
+ "--codex-install-strategy",
736
+ choices=TERMINAL_BENCH_CODEX_INSTALL_STRATEGIES,
737
+ default=TERMINAL_BENCH_CODEX_INSTALL_STRATEGY_RUNTIME_INSTALL_IF_MISSING,
738
+ )
739
+ benchmark_case_run_launch_parser.add_argument(
740
+ "--codex-preflight-timeout-sec",
741
+ type=int,
742
+ )
743
+ benchmark_case_run_launch_parser.add_argument(
744
+ "--worker-codex-materialization-strategy",
745
+ choices=TERMINAL_BENCH_WORKER_CODEX_MATERIALIZATION_STRATEGIES,
746
+ )
747
+ benchmark_case_run_launch_parser.add_argument(
748
+ "--setup-timeout-repair-profile",
749
+ action="store_true",
750
+ help=(
751
+ "Apply the generic setup-timeout repair launch profile before "
752
+ "starting the case run."
753
+ ),
754
+ )
755
+ benchmark_case_run_launch_parser.add_argument(
756
+ "--execute",
757
+ action="store_true",
758
+ help=(
759
+ "Actually start the local no-upload Terminal-Bench case run. "
760
+ "Without this flag, dry-run only."
761
+ ),
762
+ )
763
+
764
+ benchmark_resume_terminal_bench_job_parser = benchmark_subparsers.add_parser(
765
+ "resume-terminal-bench-job",
766
+ help=(
767
+ "Run one no-upload Harbor job resume for a materialized "
768
+ "Terminal-Bench job and emit compact process/result-finalization "
769
+ "state. Stdout/stderr stay private and are not read."
770
+ ),
771
+ )
772
+ add_subcommand_format(benchmark_resume_terminal_bench_job_parser)
773
+ benchmark_resume_terminal_bench_job_parser.add_argument(
774
+ "benchmark_name",
775
+ choices=["terminal-bench"],
776
+ help="Benchmark family. Only terminal-bench is supported.",
777
+ )
778
+ benchmark_resume_terminal_bench_job_parser.add_argument(
779
+ "--run-root",
780
+ required=True,
781
+ help=(
782
+ "Private run root for resume artifacts. The value is used locally "
783
+ "and only its basename is emitted."
784
+ ),
785
+ )
786
+ benchmark_resume_terminal_bench_job_parser.add_argument(
787
+ "--jobs-dir",
788
+ required=True,
789
+ help=(
790
+ "Private Harbor jobs directory. The value is used locally and is not "
791
+ "echoed in output."
792
+ ),
793
+ )
794
+ benchmark_resume_terminal_bench_job_parser.add_argument(
795
+ "--job-name",
796
+ required=True,
797
+ help="Public-safe Harbor job basename to resume.",
798
+ )
799
+ benchmark_resume_terminal_bench_job_parser.add_argument(
800
+ "--wait-seconds",
801
+ type=int,
802
+ default=120,
803
+ help="Seconds to wait for the resume process before returning running state.",
804
+ )
805
+ benchmark_resume_terminal_bench_job_parser.add_argument(
806
+ "--execute",
807
+ action="store_true",
808
+ help=(
809
+ "Actually start the local no-upload Harbor resume. Without this "
810
+ "flag, dry-run only."
811
+ ),
812
+ )
813
+
814
+ benchmark_worker_materialization_probe_poll_parser = benchmark_subparsers.add_parser(
815
+ "poll-worker-materialization-probe",
816
+ help=(
817
+ "Poll a Terminal-Bench worker materialization probe by private pid "
818
+ "state plus compact Harbor materialization signals. This does not "
819
+ "read stdout/stderr logs, task text, trajectories, argv, Docker, "
820
+ "model APIs, or uploads."
821
+ ),
822
+ )
823
+ add_subcommand_format(benchmark_worker_materialization_probe_poll_parser)
824
+ benchmark_worker_materialization_probe_poll_parser.add_argument(
825
+ "benchmark_name",
826
+ choices=["terminal-bench"],
827
+ help="Benchmark family. Only terminal-bench is supported.",
828
+ )
829
+ benchmark_worker_materialization_probe_poll_parser.add_argument(
830
+ "--run-root",
831
+ required=True,
832
+ help=(
833
+ "Private run root containing the probe pid file. The value is used "
834
+ "locally and only its basename is emitted."
835
+ ),
836
+ )
837
+ benchmark_worker_materialization_probe_poll_parser.add_argument(
838
+ "--jobs-dir",
839
+ required=True,
840
+ help=(
841
+ "Private Harbor jobs directory. The value is used locally and is not "
842
+ "echoed in output."
843
+ ),
844
+ )
845
+ benchmark_worker_materialization_probe_poll_parser.add_argument(
846
+ "--job-name",
847
+ required=True,
848
+ help="Public-safe Harbor job basename to summarize.",
849
+ )
850
+
851
+
852
+
853
+ def handle_terminal_bench_environment_result_command(
854
+ args: argparse.Namespace,
855
+ *,
856
+ print_payload: PrintPayload,
857
+ output_format: OutputFormat,
858
+ ) -> int | None:
859
+ if args.benchmark_command not in TERMINAL_BENCH_ENVIRONMENT_RESULT_COMMANDS:
860
+ return None
861
+
862
+ if args.benchmark_command == "environment-setup-gate":
863
+ def read_optional_json(path_text: str | None) -> dict[str, object] | None:
864
+ if not path_text:
865
+ return None
866
+ payload = json.loads(Path(path_text).expanduser().read_text(encoding="utf-8"))
867
+ if not isinstance(payload, dict):
868
+ raise ValueError("environment setup gate input JSON must contain an object")
869
+ return payload
870
+
871
+ try:
872
+ if args.benchmark_name != "terminal-bench":
873
+ raise ValueError("only terminal-bench is supported")
874
+ preflight = read_optional_json(args.preflight_json)
875
+ run_input = read_optional_json(args.benchmark_run_json)
876
+ if run_input is None:
877
+ raise ValueError("--benchmark-run-json is required")
878
+ benchmark_run = compact_benchmark_run(run_input)
879
+ if not benchmark_run:
880
+ raise ValueError(
881
+ "--benchmark-run-json did not contain a compactable benchmark_run_v0 object"
882
+ )
883
+ payload = build_terminal_bench_environment_setup_probe_gate(
884
+ dataset=args.dataset,
885
+ task_id=args.include_task_name,
886
+ preflight=preflight,
887
+ previous_benchmark_run=benchmark_run,
888
+ harbor_run_help_text=args.harbor_run_help_text,
889
+ probe_runner_help=bool(args.probe_runner_help),
890
+ )
891
+ payload["ok"] = True
892
+ if (
893
+ args.require_probe_allowed
894
+ and payload.get("environment_setup_probe_allowed") is not True
895
+ ):
896
+ payload["ok"] = False
897
+ payload["error"] = (
898
+ payload.get("first_blocker")
899
+ or "environment_setup_probe_not_allowed"
900
+ )
901
+ payload["require_probe_allowed"] = bool(args.require_probe_allowed)
902
+ except Exception as exc:
903
+ payload = {
904
+ "ok": False,
905
+ "schema_version": "terminal_bench_environment_setup_probe_gate_v0",
906
+ "error": str(exc),
907
+ "read_boundary": {
908
+ "compact_only": True,
909
+ "raw_help_recorded": False,
910
+ "raw_artifacts_read": False,
911
+ "raw_logs_read": False,
912
+ "task_text_read": False,
913
+ "trajectory_read": False,
914
+ "local_paths_recorded": False,
915
+ "credential_values_recorded": False,
916
+ "codex_invoked": False,
917
+ "model_api_invoked": False,
918
+ "upload_invoked": False,
919
+ },
920
+ }
921
+ print_payload(
922
+ payload,
923
+ output_format(args),
924
+ render_terminal_bench_environment_setup_gate_markdown,
925
+ )
926
+ return 0 if payload.get("ok") else 1
927
+ if args.benchmark_command == "launch-environment-setup-probe":
928
+ try:
929
+ if args.benchmark_name != "terminal-bench":
930
+ raise ValueError("only terminal-bench is supported")
931
+ gate = json.loads(Path(args.gate_json).expanduser().read_text(encoding="utf-8"))
932
+ if not isinstance(gate, dict):
933
+ raise ValueError("--gate-json must contain a JSON object")
934
+ payload = launch_terminal_bench_environment_setup_probe(
935
+ gate=gate,
936
+ jobs_dir=args.jobs_dir,
937
+ run_root=args.run_root,
938
+ wait_seconds=args.wait_seconds,
939
+ execute=bool(args.execute),
940
+ )
941
+ payload["ok"] = True
942
+ except Exception as exc:
943
+ payload = {
944
+ "ok": False,
945
+ "schema_version": "terminal_bench_environment_setup_probe_launch_v0",
946
+ "dry_run": not bool(args.execute),
947
+ "error": str(exc),
948
+ "boundary": {
949
+ "raw_logs_read": False,
950
+ "task_text_read": False,
951
+ "trajectory_read": False,
952
+ "local_paths_recorded": False,
953
+ "command_argv_recorded": False,
954
+ "codex_invoked": False,
955
+ "model_api_invoked": False,
956
+ "upload_invoked": False,
957
+ },
958
+ }
959
+ print_payload(
960
+ payload,
961
+ output_format(args),
962
+ render_terminal_bench_environment_setup_probe_launch_markdown,
963
+ )
964
+ return 0 if payload.get("ok") else 1
965
+ if args.benchmark_command == "launch-worker-materialization-probe":
966
+ try:
967
+ if args.benchmark_name != "terminal-bench":
968
+ raise ValueError("only terminal-bench is supported")
969
+ payload = launch_terminal_bench_worker_materialization_probe(
970
+ jobs_dir=args.jobs_dir,
971
+ run_root=args.run_root,
972
+ dataset=args.dataset,
973
+ task_id=args.include_task_name,
974
+ model=args.model,
975
+ mode=args.mode,
976
+ job_name=args.job_name,
977
+ worker_codex_materialization_strategy=(
978
+ args.worker_codex_materialization_strategy
979
+ ),
980
+ wait_seconds=args.wait_seconds,
981
+ execute=bool(args.execute),
982
+ )
983
+ payload["ok"] = True
984
+ except Exception as exc:
985
+ payload = {
986
+ "ok": False,
987
+ "schema_version": "terminal_bench_worker_materialization_probe_launch_v0",
988
+ "dry_run": not bool(args.execute),
989
+ "error": str(exc),
990
+ "boundary": {
991
+ "raw_logs_read": False,
992
+ "task_text_read": False,
993
+ "trajectory_read": False,
994
+ "local_paths_recorded": False,
995
+ "command_argv_recorded": False,
996
+ "task_solver_invoked_by_probe": False,
997
+ "model_api_expected": False,
998
+ "upload_invoked": False,
999
+ },
1000
+ }
1001
+ print_payload(
1002
+ payload,
1003
+ output_format(args),
1004
+ render_terminal_bench_worker_materialization_probe_launch_markdown,
1005
+ )
1006
+ return 0 if payload.get("ok") else 1
1007
+ if args.benchmark_command == "launch-terminal-bench-run":
1008
+ try:
1009
+ if args.benchmark_name != "terminal-bench":
1010
+ raise ValueError("only terminal-bench is supported")
1011
+ payload = launch_terminal_bench_case_run(
1012
+ jobs_dir=args.jobs_dir,
1013
+ run_root=args.run_root,
1014
+ dataset=args.dataset,
1015
+ task_id=args.include_task_name,
1016
+ model=args.model,
1017
+ mode=args.mode,
1018
+ job_name=args.job_name,
1019
+ wait_seconds=args.wait_seconds,
1020
+ materialization_wait_seconds=(
1021
+ args.materialization_wait_seconds
1022
+ ),
1023
+ resume_after_materialization=bool(
1024
+ args.resume_after_materialization
1025
+ ),
1026
+ execute=bool(args.execute),
1027
+ timeout_multiplier=args.timeout_multiplier,
1028
+ agent_timeout_multiplier=args.agent_timeout_multiplier,
1029
+ verifier_timeout_multiplier=args.verifier_timeout_multiplier,
1030
+ agent_setup_timeout_multiplier=(
1031
+ args.agent_setup_timeout_multiplier
1032
+ ),
1033
+ environment_build_timeout_multiplier=(
1034
+ args.environment_build_timeout_multiplier
1035
+ ),
1036
+ codex_install_strategy=args.codex_install_strategy,
1037
+ codex_preflight_timeout_sec=args.codex_preflight_timeout_sec,
1038
+ worker_codex_materialization_strategy=(
1039
+ args.worker_codex_materialization_strategy
1040
+ ),
1041
+ setup_timeout_repair_profile=bool(
1042
+ args.setup_timeout_repair_profile
1043
+ ),
1044
+ )
1045
+ payload["ok"] = True
1046
+ except Exception as exc:
1047
+ payload = {
1048
+ "ok": False,
1049
+ "schema_version": "terminal_bench_case_run_launch_v0",
1050
+ "dry_run": not bool(args.execute),
1051
+ "error": str(exc),
1052
+ "boundary": {
1053
+ "raw_logs_read": False,
1054
+ "task_text_read": False,
1055
+ "trajectory_read": False,
1056
+ "local_paths_recorded": False,
1057
+ "command_argv_recorded": False,
1058
+ "task_solver_invoked": False,
1059
+ "model_api_expected": False,
1060
+ "upload_invoked": False,
1061
+ },
1062
+ }
1063
+ print_payload(
1064
+ payload,
1065
+ output_format(args),
1066
+ render_terminal_bench_case_run_launch_markdown,
1067
+ )
1068
+ return 0 if payload.get("ok") else 1
1069
+ if args.benchmark_command == "resume-terminal-bench-job":
1070
+ try:
1071
+ if args.benchmark_name != "terminal-bench":
1072
+ raise ValueError("only terminal-bench is supported")
1073
+ payload = resume_terminal_bench_materialized_job(
1074
+ jobs_dir=args.jobs_dir,
1075
+ run_root=args.run_root,
1076
+ job_name=args.job_name,
1077
+ wait_seconds=args.wait_seconds,
1078
+ execute=bool(args.execute),
1079
+ )
1080
+ payload["ok"] = True
1081
+ except Exception as exc:
1082
+ payload = {
1083
+ "ok": False,
1084
+ "schema_version": "terminal_bench_harbor_resume_observation_v0",
1085
+ "dry_run": not bool(args.execute),
1086
+ "error": str(exc),
1087
+ "boundary": {
1088
+ "raw_logs_read": False,
1089
+ "task_text_read": False,
1090
+ "trajectory_read": False,
1091
+ "local_paths_recorded": False,
1092
+ "command_argv_recorded": False,
1093
+ "resume_invoked": False,
1094
+ "model_api_expected": False,
1095
+ "upload_invoked": False,
1096
+ },
1097
+ }
1098
+ print_payload(
1099
+ payload,
1100
+ output_format(args),
1101
+ render_terminal_bench_resume_observation_markdown,
1102
+ )
1103
+ return 0 if payload.get("ok") else 1
1104
+ if args.benchmark_command == "poll-worker-materialization-probe":
1105
+ try:
1106
+ if args.benchmark_name != "terminal-bench":
1107
+ raise ValueError("only terminal-bench is supported")
1108
+ payload = poll_terminal_bench_worker_materialization_probe(
1109
+ jobs_dir=args.jobs_dir,
1110
+ run_root=args.run_root,
1111
+ job_name=args.job_name,
1112
+ )
1113
+ payload["ok"] = True
1114
+ except Exception as exc:
1115
+ payload = {
1116
+ "ok": False,
1117
+ "schema_version": "terminal_bench_worker_materialization_probe_poll_v0",
1118
+ "error": str(exc),
1119
+ "boundary": {
1120
+ "raw_logs_read": False,
1121
+ "task_text_read": False,
1122
+ "trajectory_read": False,
1123
+ "local_paths_recorded": False,
1124
+ "command_argv_recorded": False,
1125
+ "command_line_read": False,
1126
+ "docker_invoked": False,
1127
+ "model_api_invoked": False,
1128
+ "upload_invoked": False,
1129
+ },
1130
+ }
1131
+ print_payload(
1132
+ payload,
1133
+ output_format(args),
1134
+ render_terminal_bench_worker_materialization_probe_poll_markdown,
1135
+ )
1136
+ return 0 if payload.get("ok") else 1
1137
+
1138
+ if args.benchmark_command == "summarize-post-launch":
1139
+ try:
1140
+ if args.benchmark_name != "terminal-bench":
1141
+ raise ValueError("only terminal-bench is supported")
1142
+ payload = summarize_terminal_bench_post_launch_materialization(
1143
+ args.jobs_dir,
1144
+ job_name=args.job_name,
1145
+ detached_process_state=args.detached_process_state,
1146
+ reconcile_stale_active=args.reconcile_stale_active,
1147
+ )
1148
+ ready = payload.get("ready_for_launch_state") is True
1149
+ payload["ok"] = (
1150
+ ready if args.require_ready_for_launch_state else True
1151
+ )
1152
+ payload["require_ready_for_launch_state"] = bool(
1153
+ args.require_ready_for_launch_state
1154
+ )
1155
+ payload["read_boundary"] = {
1156
+ "raw_paths_recorded": False,
1157
+ "raw_logs_read": False,
1158
+ "task_text_read": False,
1159
+ "trajectory_read": False,
1160
+ "docker_invoked": False,
1161
+ "model_api_invoked": False,
1162
+ "upload_invoked": False,
1163
+ }
1164
+ if args.require_ready_for_launch_state and not ready:
1165
+ payload["error"] = (
1166
+ "post-launch materialization is not ready for launch state"
1167
+ )
1168
+ except Exception as exc:
1169
+ payload = {
1170
+ "ok": False,
1171
+ "schema_version": "terminal_bench_post_launch_materialization_v0",
1172
+ "error": str(exc),
1173
+ "read_boundary": {
1174
+ "raw_paths_recorded": False,
1175
+ "raw_logs_read": False,
1176
+ "task_text_read": False,
1177
+ "trajectory_read": False,
1178
+ "docker_invoked": False,
1179
+ "model_api_invoked": False,
1180
+ "upload_invoked": False,
1181
+ },
1182
+ }
1183
+ print_payload(
1184
+ payload,
1185
+ output_format(args),
1186
+ render_terminal_bench_post_launch_materialization_markdown,
1187
+ )
1188
+ return 0 if payload.get("ok") else 1
1189
+ if args.benchmark_command == "result-finalization-gate":
1190
+ try:
1191
+ if args.benchmark_name != "terminal-bench":
1192
+ raise ValueError("only terminal-bench is supported")
1193
+ if args.post_launch_json == "-":
1194
+ post_launch = json.loads(sys.stdin.read())
1195
+ else:
1196
+ post_launch = json.loads(
1197
+ Path(args.post_launch_json)
1198
+ .expanduser()
1199
+ .read_text(encoding="utf-8")
1200
+ )
1201
+ if not isinstance(post_launch, dict):
1202
+ raise ValueError("--post-launch-json must contain a JSON object")
1203
+ payload = build_terminal_bench_result_finalization_gate(
1204
+ post_launch,
1205
+ max_repaired_baseline_reruns=(
1206
+ args.max_repaired_baseline_reruns
1207
+ ),
1208
+ )
1209
+ payload["require_rerun_allowed"] = bool(
1210
+ args.require_rerun_allowed
1211
+ )
1212
+ if (
1213
+ args.require_rerun_allowed
1214
+ and payload.get("repaired_baseline_rerun_allowed") is not True
1215
+ ):
1216
+ payload["ok"] = False
1217
+ payload["error"] = (
1218
+ payload.get("first_blocker")
1219
+ or "result_finalization_gate_rerun_not_allowed"
1220
+ )
1221
+ except Exception as exc:
1222
+ payload = {
1223
+ "ok": False,
1224
+ "schema_version": "terminal_bench_result_finalization_gate_v0",
1225
+ "error": str(exc),
1226
+ "read_boundary": {
1227
+ "compact_only": True,
1228
+ "raw_artifacts_read": False,
1229
+ "raw_paths_recorded": False,
1230
+ "raw_logs_read": False,
1231
+ "task_text_read": False,
1232
+ "trajectory_read": False,
1233
+ "docker_invoked": False,
1234
+ "model_api_invoked": False,
1235
+ "upload_invoked": False,
1236
+ "raw_external_handle_payload_recorded": False,
1237
+ },
1238
+ }
1239
+ print_payload(
1240
+ payload,
1241
+ output_format(args),
1242
+ render_terminal_bench_result_finalization_gate_markdown,
1243
+ )
1244
+ return 0 if payload.get("ok") else 1
1245
+
1246
+ return None