algo-cli-runtime 0.14.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. algo_cli/__init__.py +3 -0
  2. algo_cli/__main__.py +7 -0
  3. algo_cli/_internal/__init__.py +12 -0
  4. algo_cli/_internal/policy_chain.py +259 -0
  5. algo_cli/action_registry.py +1047 -0
  6. algo_cli/agent_blocks.py +550 -0
  7. algo_cli/agent_pipeline.py +1457 -0
  8. algo_cli/agent_threads.py +308 -0
  9. algo_cli/animations.py +316 -0
  10. algo_cli/cache_admission.py +209 -0
  11. algo_cli/capability_mask.py +66 -0
  12. algo_cli/chat_protocol.py +116 -0
  13. algo_cli/chatgpt_auth.py +510 -0
  14. algo_cli/chatgpt_client.py +657 -0
  15. algo_cli/code_rag.py +479 -0
  16. algo_cli/config.py +651 -0
  17. algo_cli/context_budget.py +679 -0
  18. algo_cli/credential_helpers.py +315 -0
  19. algo_cli/deliberation.py +29 -0
  20. algo_cli/display.py +1470 -0
  21. algo_cli/evals/__init__.py +21 -0
  22. algo_cli/evals/algorithm_effectiveness.py +560 -0
  23. algo_cli/evals/competitive_harness_rating.py +702 -0
  24. algo_cli/evals/cot_quality.py +220 -0
  25. algo_cli/evals/harness_retrieval_benchmark.py +401 -0
  26. algo_cli/evals/performance_regression.py +136 -0
  27. algo_cli/evals/scorecard_grading.py +308 -0
  28. algo_cli/evals/session_distribution.py +84 -0
  29. algo_cli/execution_guardrails.py +806 -0
  30. algo_cli/extensions_manifest.py +84 -0
  31. algo_cli/git_evidence.py +227 -0
  32. algo_cli/google_workspace.py +407 -0
  33. algo_cli/google_workspace_auth.py +523 -0
  34. algo_cli/harness.py +2587 -0
  35. algo_cli/identity.py +557 -0
  36. algo_cli/index_compute_lab.py +228 -0
  37. algo_cli/inference_harness.py +70 -0
  38. algo_cli/intelligence/__init__.py +1103 -0
  39. algo_cli/intelligence/acrobat_config.py +307 -0
  40. algo_cli/intelligence/acrobat_manifests.py +338 -0
  41. algo_cli/intelligence/acrobat_models.py +195 -0
  42. algo_cli/intelligence/acrobat_pipeline.py +295 -0
  43. algo_cli/intelligence/acrobat_runtime.py +302 -0
  44. algo_cli/intelligence/acrobat_security.py +261 -0
  45. algo_cli/intelligence/acrobat_workflows.py +226 -0
  46. algo_cli/intelligence/actionability.py +165 -0
  47. algo_cli/intelligence/adversarial_audit.py +136 -0
  48. algo_cli/intelligence/agent_arena.py +92 -0
  49. algo_cli/intelligence/agent_benchmark.py +236 -0
  50. algo_cli/intelligence/agent_runtime.py +171 -0
  51. algo_cli/intelligence/agents_as_tools.py +70 -0
  52. algo_cli/intelligence/artifact_binding.py +80 -0
  53. algo_cli/intelligence/autonomous_engineer.py +1976 -0
  54. algo_cli/intelligence/backpressure.py +99 -0
  55. algo_cli/intelligence/bloom_filter.py +186 -0
  56. algo_cli/intelligence/bonferroni.py +66 -0
  57. algo_cli/intelligence/boundary_compaction.py +98 -0
  58. algo_cli/intelligence/catalog_verifier.py +172 -0
  59. algo_cli/intelligence/cavecrew.py +118 -0
  60. algo_cli/intelligence/changelog.py +176 -0
  61. algo_cli/intelligence/checkpoint_resume.py +92 -0
  62. algo_cli/intelligence/circuit_breaker.py +88 -0
  63. algo_cli/intelligence/clarification_gate.py +101 -0
  64. algo_cli/intelligence/code_graph.py +180 -0
  65. algo_cli/intelligence/coderank.py +97 -0
  66. algo_cli/intelligence/consistent_hash.py +150 -0
  67. algo_cli/intelligence/consortium_synthesis.py +139 -0
  68. algo_cli/intelligence/construction/__init__.py +241 -0
  69. algo_cli/intelligence/construction/common.py +273 -0
  70. algo_cli/intelligence/construction/documents.py +496 -0
  71. algo_cli/intelligence/construction/labor_units.py +1395 -0
  72. algo_cli/intelligence/construction/payments.py +470 -0
  73. algo_cli/intelligence/construction/risk.py +784 -0
  74. algo_cli/intelligence/content_extractor.py +132 -0
  75. algo_cli/intelligence/context_adaptive.py +102 -0
  76. algo_cli/intelligence/context_ops.py +95 -0
  77. algo_cli/intelligence/count_min.py +145 -0
  78. algo_cli/intelligence/cow_state.py +103 -0
  79. algo_cli/intelligence/critic_loop.py +119 -0
  80. algo_cli/intelligence/cross_source.py +113 -0
  81. algo_cli/intelligence/daemon_mode.py +99 -0
  82. algo_cli/intelligence/dag_orchestration.py +151 -0
  83. algo_cli/intelligence/deep_research.py +155 -0
  84. algo_cli/intelligence/degenerate_detector.py +78 -0
  85. algo_cli/intelligence/delta_report.py +92 -0
  86. algo_cli/intelligence/discovery_event_log.py +92 -0
  87. algo_cli/intelligence/document_ingest.py +298 -0
  88. algo_cli/intelligence/dual_layer_validate.py +151 -0
  89. algo_cli/intelligence/echo_fidelity.py +73 -0
  90. algo_cli/intelligence/ema_tuning.py +104 -0
  91. algo_cli/intelligence/event_log.py +92 -0
  92. algo_cli/intelligence/evidence_graph.py +114 -0
  93. algo_cli/intelligence/extension_host.py +162 -0
  94. algo_cli/intelligence/extension_manifest.py +115 -0
  95. algo_cli/intelligence/falsification_suite.py +178 -0
  96. algo_cli/intelligence/finance/__init__.py +169 -0
  97. algo_cli/intelligence/finance/anomalies.py +135 -0
  98. algo_cli/intelligence/finance/ap_ar.py +351 -0
  99. algo_cli/intelligence/finance/cash.py +162 -0
  100. algo_cli/intelligence/finance/close.py +332 -0
  101. algo_cli/intelligence/finance/common.py +244 -0
  102. algo_cli/intelligence/finance/construction.py +135 -0
  103. algo_cli/intelligence/finance/controls.py +172 -0
  104. algo_cli/intelligence/finance/evidence.py +119 -0
  105. algo_cli/intelligence/finance/exceptions.py +157 -0
  106. algo_cli/intelligence/finance/reconciliations.py +254 -0
  107. algo_cli/intelligence/finance/revenue.py +109 -0
  108. algo_cli/intelligence/finance/tax.py +74 -0
  109. algo_cli/intelligence/finance/workpapers.py +111 -0
  110. algo_cli/intelligence/finding_record.py +120 -0
  111. algo_cli/intelligence/flow_dag.py +267 -0
  112. algo_cli/intelligence/gatherer.py +223 -0
  113. algo_cli/intelligence/golden_master.py +98 -0
  114. algo_cli/intelligence/graph_rag.py +195 -0
  115. algo_cli/intelligence/group_chat.py +143 -0
  116. algo_cli/intelligence/hash_dedup.py +145 -0
  117. algo_cli/intelligence/hyperloglog.py +128 -0
  118. algo_cli/intelligence/incremental_index.py +316 -0
  119. algo_cli/intelligence/index_store.py +16 -0
  120. algo_cli/intelligence/iteration_plan.py +133 -0
  121. algo_cli/intelligence/kernel_plugins.py +167 -0
  122. algo_cli/intelligence/lesson_catalog.py +135 -0
  123. algo_cli/intelligence/llm_fallback.py +169 -0
  124. algo_cli/intelligence/log2_histogram.py +267 -0
  125. algo_cli/intelligence/lsp_integration.py +147 -0
  126. algo_cli/intelligence/memory_evolution.py +117 -0
  127. algo_cli/intelligence/minhash_lsh.py +182 -0
  128. algo_cli/intelligence/multi_model_score.py +174 -0
  129. algo_cli/intelligence/multi_tier_grade.py +211 -0
  130. algo_cli/intelligence/negative_controls.py +113 -0
  131. algo_cli/intelligence/numeric_clamp.py +63 -0
  132. algo_cli/intelligence/occ_editor.py +66 -0
  133. algo_cli/intelligence/output_normalize.py +112 -0
  134. algo_cli/intelligence/parallel_delegation.py +98 -0
  135. algo_cli/intelligence/parallel_fanout.py +104 -0
  136. algo_cli/intelligence/permission_modes.py +105 -0
  137. algo_cli/intelligence/pre_push_gate.py +68 -0
  138. algo_cli/intelligence/prefetch.py +171 -0
  139. algo_cli/intelligence/process_framework.py +217 -0
  140. algo_cli/intelligence/project_graph.py +387 -0
  141. algo_cli/intelligence/query_expansion.py +146 -0
  142. algo_cli/intelligence/ralph_loop.py +117 -0
  143. algo_cli/intelligence/rate_limiter.py +153 -0
  144. algo_cli/intelligence/refactor_transaction.py +94 -0
  145. algo_cli/intelligence/research_workspace.py +108 -0
  146. algo_cli/intelligence/retraction_ledger.py +72 -0
  147. algo_cli/intelligence/saga_pattern.py +88 -0
  148. algo_cli/intelligence/session_fork.py +100 -0
  149. algo_cli/intelligence/shadow_editor.py +67 -0
  150. algo_cli/intelligence/shell_session.py +213 -0
  151. algo_cli/intelligence/source_registry.py +143 -0
  152. algo_cli/intelligence/spawn_scales.py +99 -0
  153. algo_cli/intelligence/stat_stability.py +104 -0
  154. algo_cli/intelligence/structural_validator.py +148 -0
  155. algo_cli/intelligence/subagent_spawner.py +111 -0
  156. algo_cli/intelligence/symmetric_verify.py +70 -0
  157. algo_cli/intelligence/task_classifier.py +129 -0
  158. algo_cli/intelligence/team_execution.py +122 -0
  159. algo_cli/intelligence/tiered_access.py +121 -0
  160. algo_cli/intelligence/utility_registry.py +159 -0
  161. algo_cli/intuition_engine.py +560 -0
  162. algo_cli/intuition_injector.py +82 -0
  163. algo_cli/kernels/__init__.py +5 -0
  164. algo_cli/kernels/manifest.py +763 -0
  165. algo_cli/main.py +3903 -0
  166. algo_cli/memory_candidates.py +541 -0
  167. algo_cli/memory_echo_veil.py +394 -0
  168. algo_cli/memory_runtime.py +112 -0
  169. algo_cli/model_info.py +548 -0
  170. algo_cli/model_profile.py +160 -0
  171. algo_cli/model_routing.py +74 -0
  172. algo_cli/oneshot.py +331 -0
  173. algo_cli/perf_telemetry.py +389 -0
  174. algo_cli/plugins.py +245 -0
  175. algo_cli/private_event_store.py +654 -0
  176. algo_cli/quantization/__init__.py +24 -0
  177. algo_cli/quantization/lloyd_max.py +98 -0
  178. algo_cli/quantization/turbo_quant.py +308 -0
  179. algo_cli/reasoning/__init__.py +46 -0
  180. algo_cli/reasoning/combinatorial.py +356 -0
  181. algo_cli/reasoning/graph_of_thought.py +297 -0
  182. algo_cli/reasoning/mcts.py +220 -0
  183. algo_cli/reasoning/neuro_symbolic.py +250 -0
  184. algo_cli/reasoning/react.py +246 -0
  185. algo_cli/reasoning/reflexion.py +225 -0
  186. algo_cli/reasoning/tree_of_thought.py +241 -0
  187. algo_cli/reasoning_bridge.py +150 -0
  188. algo_cli/reconciliation.py +284 -0
  189. algo_cli/reflex.py +385 -0
  190. algo_cli/resources/docs/ALGO.md +13958 -0
  191. algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
  192. algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
  193. algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
  194. algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
  195. algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
  196. algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
  197. algo_cli/resources/docs/main-split-map.md +35 -0
  198. algo_cli/resources/docs/privacy-and-context.md +48 -0
  199. algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
  200. algo_cli/resources/skills/README.md +26 -0
  201. algo_cli/resources/skills/algo-cli.md +59 -0
  202. algo_cli/resources/skills/edit-file-precision.md +49 -0
  203. algo_cli/resources/skills/harness-search-first.md +47 -0
  204. algo_cli/resources/skills/memory-recall-ritual.md +51 -0
  205. algo_cli/resources/skills/qol-algorithms.md +224 -0
  206. algo_cli/resources/skills/smart-error-recovery.md +56 -0
  207. algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
  208. algo_cli/retrieval_algorithms.py +127 -0
  209. algo_cli/runtime_qos.py +236 -0
  210. algo_cli/runtime_services.py +320 -0
  211. algo_cli/session_commands.py +95 -0
  212. algo_cli/session_mode.py +113 -0
  213. algo_cli/skills.py +430 -0
  214. algo_cli/slash_dispatch.py +1265 -0
  215. algo_cli/small_context.py +206 -0
  216. algo_cli/spawn_budget.py +89 -0
  217. algo_cli/task_ledger.py +84 -0
  218. algo_cli/task_router.py +197 -0
  219. algo_cli/tool_context.py +94 -0
  220. algo_cli/tool_contract.py +99 -0
  221. algo_cli/tool_policy.py +357 -0
  222. algo_cli/tool_runtime.py +647 -0
  223. algo_cli/tools.py +3056 -0
  224. algo_cli/url_scheme.py +174 -0
  225. algo_cli/verify.py +154 -0
  226. algo_cli/version_manifest.py +178 -0
  227. algo_cli/vision_screenshot_verify.py +76 -0
  228. algo_cli/workspace_resolver.py +68 -0
  229. algo_cli/x_account.py +209 -0
  230. algo_cli/xai_auth.py +374 -0
  231. algo_cli/xai_client.py +600 -0
  232. algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
  233. algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
  234. algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
  235. algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
  236. algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
  237. ollama_cli/__init__.py +67 -0
@@ -0,0 +1,226 @@
1
+ """B157-B158: Acrobat-derived workflow and policy profile patterns.
2
+
3
+ - B157: Declarative Workflow Sequence Engine
4
+ - B158: Named Policy Profile Packs
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass, field
10
+ from typing import Any, Callable
11
+
12
+
13
+ # ── B157: Declarative Workflow Sequence Engine ────────────────────────
14
+
15
+
16
+ @dataclass
17
+ class WorkflowItem:
18
+ """A typed parameter for a workflow command."""
19
+ name: str
20
+ item_type: str # "boolean", "integer", "text"
21
+ value: Any = None
22
+
23
+
24
+ @dataclass
25
+ class WorkflowInstruction:
26
+ """An instruction step in a workflow."""
27
+ label: str
28
+
29
+
30
+ @dataclass
31
+ class WorkflowCommand:
32
+ """A command step in a workflow."""
33
+ name: str
34
+ prompt_user: bool = False
35
+ pause_before: bool = False
36
+ items: list[WorkflowItem] = field(default_factory=list)
37
+
38
+
39
+ @dataclass
40
+ class WorkflowGroup:
41
+ """A group of steps in a workflow."""
42
+ label: str
43
+ steps: list[WorkflowInstruction | WorkflowCommand | "WorkflowSeparator"] = field(default_factory=list)
44
+
45
+
46
+ @dataclass
47
+ class WorkflowSeparator:
48
+ """A visual separator in a workflow."""
49
+ pass
50
+
51
+
52
+ @dataclass
53
+ class WorkflowDefinition:
54
+ """A declarative workflow definition (B157)."""
55
+ title: str
56
+ description: str = ""
57
+ major_version: int = 1
58
+ minor_version: int = 0
59
+ groups: list[WorkflowGroup] = field(default_factory=list)
60
+
61
+ def all_command_names(self) -> list[str]:
62
+ """Extract all command names from the workflow."""
63
+ names: list[str] = []
64
+ for group in self.groups:
65
+ for step in group.steps:
66
+ if isinstance(step, WorkflowCommand):
67
+ names.append(step.name)
68
+ return names
69
+
70
+ def validate(self, command_registry: set[str]) -> list[str]:
71
+ """Validate workflow against a command registry. Returns errors."""
72
+ errors: list[str] = []
73
+ for name in self.all_command_names():
74
+ if name not in command_registry:
75
+ errors.append(f"unknown command: {name}")
76
+ # Validate typed items
77
+ for group in self.groups:
78
+ for step in group.steps:
79
+ if isinstance(step, WorkflowCommand):
80
+ for item in step.items:
81
+ if item.item_type == "boolean" and not isinstance(item.value, bool | type(None)):
82
+ if item.value is not None:
83
+ errors.append(f"bad boolean value for {item.name}: {item.value}")
84
+ elif item.item_type == "integer" and item.value is not None:
85
+ if not isinstance(item.value, int):
86
+ errors.append(f"bad integer value for {item.name}: {item.value}")
87
+ return errors
88
+
89
+
90
+ @dataclass
91
+ class WorkflowStepResult:
92
+ """Result of executing a single workflow step."""
93
+ step_name: str
94
+ success: bool = True
95
+ output: Any = None
96
+ skipped: bool = False
97
+ error: str = ""
98
+
99
+
100
+ @dataclass
101
+ class WorkflowExecutionResult:
102
+ """Result of executing an entire workflow."""
103
+ title: str
104
+ steps: list[WorkflowStepResult] = field(default_factory=list)
105
+ completed: bool = False
106
+ paused_at: str = ""
107
+
108
+ @property
109
+ def failed_steps(self) -> list[WorkflowStepResult]:
110
+ return [s for s in self.steps if not s.success and not s.skipped]
111
+
112
+
113
+ class WorkflowExecutor:
114
+ """Executes declarative workflows (B157)."""
115
+
116
+ def __init__(self) -> None:
117
+ self._handlers: dict[str, Callable[[dict], Any]] = {}
118
+
119
+ def register_command(self, name: str, handler: Callable[[dict], Any]) -> None:
120
+ self._handlers[name] = handler
121
+
122
+ def validate(self, workflow: WorkflowDefinition) -> list[str]:
123
+ """Validate a workflow before execution."""
124
+ return workflow.validate(set(self._handlers.keys()))
125
+
126
+ def execute(self, workflow: WorkflowDefinition, context: dict | None = None) -> WorkflowExecutionResult:
127
+ """Execute a workflow step by step."""
128
+ ctx = context or {}
129
+ result = WorkflowExecutionResult(title=workflow.title)
130
+ errors = self.validate(workflow)
131
+ if errors:
132
+ result.steps.append(WorkflowStepResult(
133
+ step_name="validation",
134
+ success=False,
135
+ error="; ".join(errors),
136
+ ))
137
+ return result
138
+
139
+ for group in workflow.groups:
140
+ for step in group.steps:
141
+ if isinstance(step, WorkflowInstruction | WorkflowSeparator):
142
+ continue
143
+ if isinstance(step, WorkflowCommand):
144
+ if step.pause_before:
145
+ result.paused_at = step.name
146
+ return result
147
+ handler = self._handlers.get(step.name)
148
+ if not handler:
149
+ result.steps.append(WorkflowStepResult(
150
+ step_name=step.name,
151
+ success=False,
152
+ error=f"no handler: {step.name}",
153
+ ))
154
+ return result
155
+ try:
156
+ item_dict = {item.name: item.value for item in step.items}
157
+ output = handler({**ctx, **item_dict})
158
+ result.steps.append(WorkflowStepResult(
159
+ step_name=step.name,
160
+ success=True,
161
+ output=output,
162
+ ))
163
+ except Exception as e:
164
+ result.steps.append(WorkflowStepResult(
165
+ step_name=step.name,
166
+ success=False,
167
+ error=str(e),
168
+ ))
169
+ return result
170
+ result.completed = True
171
+ return result
172
+
173
+
174
+ # ── B158: Named Policy Profile Packs ──────────────────────────────────
175
+
176
+
177
+ @dataclass
178
+ class PolicyProfile:
179
+ """A named policy profile (B158)."""
180
+ name: str
181
+ description: str = ""
182
+ parameters: dict[str, Any] = field(default_factory=dict)
183
+ overrides: dict[str, Any] = field(default_factory=dict)
184
+ source: str = "shipped" # "shipped" or "user"
185
+
186
+ def resolve(self, defaults: dict[str, Any]) -> dict[str, Any]:
187
+ """Resolve profile against defaults, applying overrides."""
188
+ result = dict(defaults)
189
+ result.update(self.parameters)
190
+ result.update(self.overrides)
191
+ return result
192
+
193
+
194
+ class PolicyProfilePack:
195
+ """A collection of named policy profiles (B158)."""
196
+
197
+ def __init__(self) -> None:
198
+ self._profiles: dict[str, PolicyProfile] = {}
199
+ self._defaults: dict[str, Any] = {}
200
+
201
+ def set_defaults(self, defaults: dict[str, Any]) -> None:
202
+ self._defaults = dict(defaults)
203
+
204
+ def register(self, profile: PolicyProfile) -> None:
205
+ self._profiles[profile.name] = profile
206
+
207
+ def get(self, name: str) -> PolicyProfile | None:
208
+ return self._profiles.get(name)
209
+
210
+ def resolve(self, name: str) -> dict[str, Any] | None:
211
+ """Resolve a profile by name."""
212
+ profile = self._profiles.get(name)
213
+ if not profile:
214
+ return None
215
+ return profile.resolve(self._defaults)
216
+
217
+ def available_profiles(self) -> list[str]:
218
+ return list(self._profiles.keys())
219
+
220
+ def add_user_override(self, profile_name: str, key: str, value: Any) -> bool:
221
+ """Add a user override to a profile."""
222
+ profile = self._profiles.get(profile_name)
223
+ if not profile:
224
+ return False
225
+ profile.overrides[key] = value
226
+ return True
@@ -0,0 +1,165 @@
1
+ """B43. Auto-Wait Actionability Checks (Playwright Pattern).
2
+
3
+ Before performing an action (click, type, navigate), check that the target
4
+ is ready: exists, visible, stable, enabled, and receives input. No blind
5
+ sleeps — poll with timeout until conditions are met or fail fast.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import time
11
+ from dataclasses import dataclass, field
12
+ from typing import Any, Callable
13
+
14
+
15
+ class CheckResult:
16
+ PASSED = "passed"
17
+ FAILED = "failed"
18
+ TIMEOUT = "timeout"
19
+
20
+
21
+ @dataclass
22
+ class ActionabilityCheck:
23
+ name: str
24
+ check_fn: Callable[[], bool]
25
+ timeout_ms: int = 5000
26
+ poll_interval_ms: int = 100
27
+
28
+
29
+ @dataclass
30
+ class ActionabilityReport:
31
+ checks: dict[str, str] = field(default_factory=dict) # name -> result
32
+ all_passed: bool = False
33
+ duration_ms: float = 0.0
34
+ failed_checks: list[str] = field(default_factory=list)
35
+
36
+
37
+ def auto_wait(checks: list[ActionabilityCheck]) -> ActionabilityReport:
38
+ """Run all actionability checks with polling and timeout.
39
+
40
+ Each check is polled at its interval until it passes or times out.
41
+ All checks must pass for the report to be all_passed=True.
42
+ """
43
+ start = time.monotonic()
44
+ results: dict[str, str] = {}
45
+ failed: list[str] = []
46
+
47
+ for check in checks:
48
+ deadline = time.monotonic() + check.timeout_ms / 1000
49
+ passed = False
50
+ while time.monotonic() < deadline:
51
+ try:
52
+ if check.check_fn():
53
+ passed = True
54
+ break
55
+ except Exception:
56
+ pass
57
+ time.sleep(check.poll_interval_ms / 1000)
58
+
59
+ if passed:
60
+ results[check.name] = CheckResult.PASSED
61
+ elif time.monotonic() >= deadline:
62
+ results[check.name] = CheckResult.TIMEOUT
63
+ failed.append(check.name)
64
+ else:
65
+ results[check.name] = CheckResult.FAILED
66
+ failed.append(check.name)
67
+
68
+ duration = (time.monotonic() - start) * 1000
69
+ return ActionabilityReport(
70
+ checks=results,
71
+ all_passed=len(failed) == 0,
72
+ duration_ms=duration,
73
+ failed_checks=failed,
74
+ )
75
+
76
+
77
+ # ── standard checks ───────────────────────────────────────────────────
78
+
79
+
80
+ def check_exists(predicate: Callable[[], Any]) -> ActionabilityCheck:
81
+ """Target exists (not None)."""
82
+ return ActionabilityCheck(
83
+ name="exists",
84
+ check_fn=lambda: predicate() is not None,
85
+ )
86
+
87
+
88
+ def check_visible(predicate: Callable[[], bool]) -> ActionabilityCheck:
89
+ """Target is visible."""
90
+ return ActionabilityCheck(
91
+ name="visible",
92
+ check_fn=predicate,
93
+ )
94
+
95
+
96
+ def check_stable(predicate: Callable[[], bool], samples: int = 3, interval: float = 0.05) -> ActionabilityCheck:
97
+ """Target has not changed over N consecutive samples."""
98
+ last_values: list[Any] = []
99
+
100
+ def is_stable() -> bool:
101
+ val = predicate()
102
+ last_values.append(val)
103
+ if len(last_values) > samples:
104
+ last_values.pop(0)
105
+ if len(last_values) < samples:
106
+ return False
107
+ return all(v == last_values[0] for v in last_values)
108
+
109
+ return ActionabilityCheck(name="stable", check_fn=is_stable)
110
+
111
+
112
+ def check_enabled(predicate: Callable[[], bool]) -> ActionabilityCheck:
113
+ """Target is enabled (not disabled)."""
114
+ return ActionabilityCheck(
115
+ name="enabled",
116
+ check_fn=predicate,
117
+ )
118
+
119
+
120
+ def check_receives_input(predicate: Callable[[], bool]) -> ActionabilityCheck:
121
+ """Target can receive input (focusable)."""
122
+ return ActionabilityCheck(
123
+ name="receives_input",
124
+ check_fn=predicate,
125
+ )
126
+
127
+
128
+ # ── trace recorder ────────────────────────────────────────────────────
129
+
130
+
131
+ @dataclass
132
+ class ActionTrace:
133
+ action: str
134
+ target: str
135
+ report: ActionabilityReport
136
+ timestamp: float = field(default_factory=time.time)
137
+ screenshot_path: str = ""
138
+ html_snapshot: str = ""
139
+ console_logs: list[str] = field(default_factory=list)
140
+ network_summary: dict[str, Any] = field(default_factory=dict)
141
+
142
+
143
+ class TraceRecorder:
144
+ """Records action traces for debugging and replay."""
145
+
146
+ def __init__(self) -> None:
147
+ self.traces: list[ActionTrace] = []
148
+
149
+ def record(self, action: str, target: str, report: ActionabilityReport, **kwargs: Any) -> None:
150
+ self.traces.append(ActionTrace(
151
+ action=action, target=target, report=report, **kwargs,
152
+ ))
153
+
154
+ def failed_actions(self) -> list[ActionTrace]:
155
+ return [t for t in self.traces if not t.report.all_passed]
156
+
157
+ def summary(self) -> dict[str, Any]:
158
+ total = len(self.traces)
159
+ passed = sum(1 for t in self.traces if t.report.all_passed)
160
+ return {
161
+ "total_actions": total,
162
+ "passed": passed,
163
+ "failed": total - passed,
164
+ "avg_wait_ms": sum(t.report.duration_ms for t in self.traces) / max(total, 1),
165
+ }
@@ -0,0 +1,136 @@
1
+ """H8 — Adversarial Self-Audit.
2
+
3
+ Catches fabricated or exaggerated claims by comparing claimed behavior
4
+ against actual test results.
5
+ Mined from GLOSSOPETRAE §6.5 falsify_workflow.mjs.
6
+
7
+ LLM integration: optionally uses an LLM to analyze discrepancies.
8
+ Falls back to rule-based comparison when no model is available.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from dataclasses import dataclass, field
13
+ from typing import Any
14
+
15
+
16
+ @dataclass
17
+ class Claim:
18
+ """A claimed behavior to audit."""
19
+
20
+ entry_id: str
21
+ claimed_status: str # "implemented", "partial", "planned"
22
+ claimed_tests: list[str] = field(default_factory=list)
23
+ description: str = ""
24
+
25
+
26
+ @dataclass
27
+ class AuditResult:
28
+ """Result of auditing a single claim."""
29
+
30
+ entry_id: str
31
+ passed: bool
32
+ discrepancies: list[str] = field(default_factory=list)
33
+ actual_status: str = ""
34
+ verified_tests: list[str] = field(default_factory=list)
35
+
36
+
37
+ def audit_claim(
38
+ claim: Claim,
39
+ actual_tests: list[str],
40
+ actual_status: str,
41
+ model_client: Any | None = None,
42
+ ) -> AuditResult:
43
+ """Audit a single claim against actual results.
44
+
45
+ Args:
46
+ claim: The claimed behavior.
47
+ actual_tests: List of actual test names that passed.
48
+ actual_status: The actual status string.
49
+ model_client: Optional LLM for deeper analysis.
50
+
51
+ Returns:
52
+ AuditResult with discrepancies found.
53
+ """
54
+ discrepancies: list[str] = []
55
+
56
+ # Check status mismatch
57
+ if claim.claimed_status != actual_status:
58
+ discrepancies.append(
59
+ f"Status mismatch: claimed '{claim.claimed_status}', actual '{actual_status}'"
60
+ )
61
+
62
+ # Check missing tests
63
+ claimed_set = set(claim.claimed_tests)
64
+ actual_set = set(actual_tests)
65
+ missing = claimed_set - actual_set
66
+ if missing:
67
+ discrepancies.append(f"Missing tests: {sorted(missing)}")
68
+
69
+ # Check extra tests (not necessarily bad, but worth noting)
70
+ extra = actual_set - claimed_set
71
+ if extra and claim.claimed_tests:
72
+ discrepancies.append(f"Unclaimed tests: {sorted(extra)}")
73
+
74
+ # If claimed implemented but no tests, that's fabrication
75
+ if claim.claimed_status == "implemented" and not actual_tests:
76
+ discrepancies.append("Claimed 'implemented' but no passing tests found")
77
+
78
+ passed = len(discrepancies) == 0
79
+ return AuditResult(
80
+ entry_id=claim.entry_id,
81
+ passed=passed,
82
+ discrepancies=discrepancies,
83
+ actual_status=actual_status,
84
+ verified_tests=list(actual_set),
85
+ )
86
+
87
+
88
+ def audit_batch(
89
+ claims: list[Claim],
90
+ actual_results: dict[str, tuple[list[str], str]],
91
+ model_client: Any | None = None,
92
+ ) -> list[AuditResult]:
93
+ """Audit multiple claims against actual results.
94
+
95
+ Args:
96
+ claims: List of claimed behaviors.
97
+ actual_results: Dict mapping entry_id to (test_names, status).
98
+ model_client: Optional LLM for deeper analysis.
99
+
100
+ Returns:
101
+ List of AuditResults.
102
+ """
103
+ results: list[AuditResult] = []
104
+ for claim in claims:
105
+ if claim.entry_id in actual_results:
106
+ tests, status = actual_results[claim.entry_id]
107
+ results.append(audit_claim(claim, tests, status, model_client))
108
+ else:
109
+ results.append(
110
+ AuditResult(
111
+ entry_id=claim.entry_id,
112
+ passed=False,
113
+ discrepancies=[f"No actual results found for entry '{claim.entry_id}'"],
114
+ actual_status="missing",
115
+ verified_tests=[],
116
+ )
117
+ )
118
+ return results
119
+
120
+
121
+ def summarize_audit(results: list[AuditResult]) -> dict[str, Any]:
122
+ """Summarize audit results."""
123
+ total = len(results)
124
+ passed = sum(1 for r in results if r.passed)
125
+ failed = total - passed
126
+ all_discrepancies: list[str] = []
127
+ for r in results:
128
+ all_discrepancies.extend(r.discrepancies)
129
+ return {
130
+ "total": total,
131
+ "passed": passed,
132
+ "failed": failed,
133
+ "pass_rate": passed / total if total > 0 else 0.0,
134
+ "discrepancies": all_discrepancies,
135
+ "discrepancy_count": len(all_discrepancies),
136
+ }
@@ -0,0 +1,92 @@
1
+ """B58. Agent Arena: Multi-Model Head-to-Head.
2
+
3
+ Run the same task across multiple models and compare outputs.
4
+ Source: qwen-code pattern.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import time
9
+ from dataclasses import dataclass, field
10
+ from enum import Enum, auto
11
+ from typing import Callable
12
+
13
+
14
+ class ArenaOutcome(Enum):
15
+ WINNER = auto()
16
+ TIE = auto()
17
+ ALL_FAILED = auto()
18
+
19
+
20
+ @dataclass
21
+ class ArenaEntry:
22
+ model: str
23
+ output: str = ""
24
+ duration_s: float = 0.0
25
+ token_count: int = 0
26
+ error: str | None = None
27
+ success: bool = True
28
+
29
+
30
+ @dataclass
31
+ class ArenaResult:
32
+ task: str
33
+ entries: list[ArenaEntry] = field(default_factory=list)
34
+ winner: str | None = None
35
+ outcome: ArenaOutcome = ArenaOutcome.TIE
36
+ comparison: str = ""
37
+
38
+
39
+ class AgentArena:
40
+ """Run the same task across multiple models and compare."""
41
+
42
+ def __init__(self, judge_fn: Callable[[str, list[ArenaEntry]], str] | None = None) -> None:
43
+ self._judge = judge_fn or self._default_judge
44
+
45
+ @staticmethod
46
+ def _default_judge(task: str, entries: list[ArenaEntry]) -> str:
47
+ """Default judge: pick the longest successful output."""
48
+ successful = [e for e in entries if e.success and e.output]
49
+ if not successful:
50
+ return ""
51
+ return max(successful, key=lambda e: len(e.output)).model
52
+
53
+ def run(
54
+ self,
55
+ task: str,
56
+ models: dict[str, Callable[[str], str]],
57
+ timeout: float = 30.0,
58
+ ) -> ArenaResult:
59
+ """Run task on all models, return comparison."""
60
+ result = ArenaResult(task=task)
61
+ for model_name, run_fn in models.items():
62
+ entry = ArenaEntry(model=model_name)
63
+ start = time.time()
64
+ try:
65
+ entry.output = run_fn(task)
66
+ entry.token_count = len(entry.output) // 4
67
+ except Exception as e:
68
+ entry.error = str(e)
69
+ entry.success = False
70
+ entry.duration_s = time.time() - start
71
+ result.entries.append(entry)
72
+
73
+ # Judge
74
+ winner = self._judge(task, result.entries)
75
+ if winner:
76
+ result.winner = winner
77
+ result.outcome = ArenaOutcome.WINNER
78
+ elif any(e.success for e in result.entries):
79
+ result.outcome = ArenaOutcome.TIE
80
+ else:
81
+ result.outcome = ArenaOutcome.ALL_FAILED
82
+
83
+ result.comparison = self._format_comparison(result)
84
+ return result
85
+
86
+ @staticmethod
87
+ def _format_comparison(result: ArenaResult) -> str:
88
+ lines = [f"Task: {result.task}", f"Winner: {result.winner or 'N/A'}", ""]
89
+ for e in result.entries:
90
+ status = "OK" if e.success else f"FAIL: {e.error}"
91
+ lines.append(f" {e.model}: {e.duration_s:.1f}s, {e.token_count}t — {status}")
92
+ return "\n".join(lines)