algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""B157-B158: Acrobat-derived workflow and policy profile patterns.
|
|
2
|
+
|
|
3
|
+
- B157: Declarative Workflow Sequence Engine
|
|
4
|
+
- B158: Named Policy Profile Packs
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Any, Callable
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# ── B157: Declarative Workflow Sequence Engine ────────────────────────
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class WorkflowItem:
|
|
18
|
+
"""A typed parameter for a workflow command."""
|
|
19
|
+
name: str
|
|
20
|
+
item_type: str # "boolean", "integer", "text"
|
|
21
|
+
value: Any = None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class WorkflowInstruction:
|
|
26
|
+
"""An instruction step in a workflow."""
|
|
27
|
+
label: str
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class WorkflowCommand:
|
|
32
|
+
"""A command step in a workflow."""
|
|
33
|
+
name: str
|
|
34
|
+
prompt_user: bool = False
|
|
35
|
+
pause_before: bool = False
|
|
36
|
+
items: list[WorkflowItem] = field(default_factory=list)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class WorkflowGroup:
|
|
41
|
+
"""A group of steps in a workflow."""
|
|
42
|
+
label: str
|
|
43
|
+
steps: list[WorkflowInstruction | WorkflowCommand | "WorkflowSeparator"] = field(default_factory=list)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class WorkflowSeparator:
|
|
48
|
+
"""A visual separator in a workflow."""
|
|
49
|
+
pass
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class WorkflowDefinition:
|
|
54
|
+
"""A declarative workflow definition (B157)."""
|
|
55
|
+
title: str
|
|
56
|
+
description: str = ""
|
|
57
|
+
major_version: int = 1
|
|
58
|
+
minor_version: int = 0
|
|
59
|
+
groups: list[WorkflowGroup] = field(default_factory=list)
|
|
60
|
+
|
|
61
|
+
def all_command_names(self) -> list[str]:
|
|
62
|
+
"""Extract all command names from the workflow."""
|
|
63
|
+
names: list[str] = []
|
|
64
|
+
for group in self.groups:
|
|
65
|
+
for step in group.steps:
|
|
66
|
+
if isinstance(step, WorkflowCommand):
|
|
67
|
+
names.append(step.name)
|
|
68
|
+
return names
|
|
69
|
+
|
|
70
|
+
def validate(self, command_registry: set[str]) -> list[str]:
|
|
71
|
+
"""Validate workflow against a command registry. Returns errors."""
|
|
72
|
+
errors: list[str] = []
|
|
73
|
+
for name in self.all_command_names():
|
|
74
|
+
if name not in command_registry:
|
|
75
|
+
errors.append(f"unknown command: {name}")
|
|
76
|
+
# Validate typed items
|
|
77
|
+
for group in self.groups:
|
|
78
|
+
for step in group.steps:
|
|
79
|
+
if isinstance(step, WorkflowCommand):
|
|
80
|
+
for item in step.items:
|
|
81
|
+
if item.item_type == "boolean" and not isinstance(item.value, bool | type(None)):
|
|
82
|
+
if item.value is not None:
|
|
83
|
+
errors.append(f"bad boolean value for {item.name}: {item.value}")
|
|
84
|
+
elif item.item_type == "integer" and item.value is not None:
|
|
85
|
+
if not isinstance(item.value, int):
|
|
86
|
+
errors.append(f"bad integer value for {item.name}: {item.value}")
|
|
87
|
+
return errors
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass
|
|
91
|
+
class WorkflowStepResult:
|
|
92
|
+
"""Result of executing a single workflow step."""
|
|
93
|
+
step_name: str
|
|
94
|
+
success: bool = True
|
|
95
|
+
output: Any = None
|
|
96
|
+
skipped: bool = False
|
|
97
|
+
error: str = ""
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass
|
|
101
|
+
class WorkflowExecutionResult:
|
|
102
|
+
"""Result of executing an entire workflow."""
|
|
103
|
+
title: str
|
|
104
|
+
steps: list[WorkflowStepResult] = field(default_factory=list)
|
|
105
|
+
completed: bool = False
|
|
106
|
+
paused_at: str = ""
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def failed_steps(self) -> list[WorkflowStepResult]:
|
|
110
|
+
return [s for s in self.steps if not s.success and not s.skipped]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class WorkflowExecutor:
|
|
114
|
+
"""Executes declarative workflows (B157)."""
|
|
115
|
+
|
|
116
|
+
def __init__(self) -> None:
|
|
117
|
+
self._handlers: dict[str, Callable[[dict], Any]] = {}
|
|
118
|
+
|
|
119
|
+
def register_command(self, name: str, handler: Callable[[dict], Any]) -> None:
|
|
120
|
+
self._handlers[name] = handler
|
|
121
|
+
|
|
122
|
+
def validate(self, workflow: WorkflowDefinition) -> list[str]:
|
|
123
|
+
"""Validate a workflow before execution."""
|
|
124
|
+
return workflow.validate(set(self._handlers.keys()))
|
|
125
|
+
|
|
126
|
+
def execute(self, workflow: WorkflowDefinition, context: dict | None = None) -> WorkflowExecutionResult:
|
|
127
|
+
"""Execute a workflow step by step."""
|
|
128
|
+
ctx = context or {}
|
|
129
|
+
result = WorkflowExecutionResult(title=workflow.title)
|
|
130
|
+
errors = self.validate(workflow)
|
|
131
|
+
if errors:
|
|
132
|
+
result.steps.append(WorkflowStepResult(
|
|
133
|
+
step_name="validation",
|
|
134
|
+
success=False,
|
|
135
|
+
error="; ".join(errors),
|
|
136
|
+
))
|
|
137
|
+
return result
|
|
138
|
+
|
|
139
|
+
for group in workflow.groups:
|
|
140
|
+
for step in group.steps:
|
|
141
|
+
if isinstance(step, WorkflowInstruction | WorkflowSeparator):
|
|
142
|
+
continue
|
|
143
|
+
if isinstance(step, WorkflowCommand):
|
|
144
|
+
if step.pause_before:
|
|
145
|
+
result.paused_at = step.name
|
|
146
|
+
return result
|
|
147
|
+
handler = self._handlers.get(step.name)
|
|
148
|
+
if not handler:
|
|
149
|
+
result.steps.append(WorkflowStepResult(
|
|
150
|
+
step_name=step.name,
|
|
151
|
+
success=False,
|
|
152
|
+
error=f"no handler: {step.name}",
|
|
153
|
+
))
|
|
154
|
+
return result
|
|
155
|
+
try:
|
|
156
|
+
item_dict = {item.name: item.value for item in step.items}
|
|
157
|
+
output = handler({**ctx, **item_dict})
|
|
158
|
+
result.steps.append(WorkflowStepResult(
|
|
159
|
+
step_name=step.name,
|
|
160
|
+
success=True,
|
|
161
|
+
output=output,
|
|
162
|
+
))
|
|
163
|
+
except Exception as e:
|
|
164
|
+
result.steps.append(WorkflowStepResult(
|
|
165
|
+
step_name=step.name,
|
|
166
|
+
success=False,
|
|
167
|
+
error=str(e),
|
|
168
|
+
))
|
|
169
|
+
return result
|
|
170
|
+
result.completed = True
|
|
171
|
+
return result
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# ── B158: Named Policy Profile Packs ──────────────────────────────────
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@dataclass
|
|
178
|
+
class PolicyProfile:
|
|
179
|
+
"""A named policy profile (B158)."""
|
|
180
|
+
name: str
|
|
181
|
+
description: str = ""
|
|
182
|
+
parameters: dict[str, Any] = field(default_factory=dict)
|
|
183
|
+
overrides: dict[str, Any] = field(default_factory=dict)
|
|
184
|
+
source: str = "shipped" # "shipped" or "user"
|
|
185
|
+
|
|
186
|
+
def resolve(self, defaults: dict[str, Any]) -> dict[str, Any]:
|
|
187
|
+
"""Resolve profile against defaults, applying overrides."""
|
|
188
|
+
result = dict(defaults)
|
|
189
|
+
result.update(self.parameters)
|
|
190
|
+
result.update(self.overrides)
|
|
191
|
+
return result
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
class PolicyProfilePack:
|
|
195
|
+
"""A collection of named policy profiles (B158)."""
|
|
196
|
+
|
|
197
|
+
def __init__(self) -> None:
|
|
198
|
+
self._profiles: dict[str, PolicyProfile] = {}
|
|
199
|
+
self._defaults: dict[str, Any] = {}
|
|
200
|
+
|
|
201
|
+
def set_defaults(self, defaults: dict[str, Any]) -> None:
|
|
202
|
+
self._defaults = dict(defaults)
|
|
203
|
+
|
|
204
|
+
def register(self, profile: PolicyProfile) -> None:
|
|
205
|
+
self._profiles[profile.name] = profile
|
|
206
|
+
|
|
207
|
+
def get(self, name: str) -> PolicyProfile | None:
|
|
208
|
+
return self._profiles.get(name)
|
|
209
|
+
|
|
210
|
+
def resolve(self, name: str) -> dict[str, Any] | None:
|
|
211
|
+
"""Resolve a profile by name."""
|
|
212
|
+
profile = self._profiles.get(name)
|
|
213
|
+
if not profile:
|
|
214
|
+
return None
|
|
215
|
+
return profile.resolve(self._defaults)
|
|
216
|
+
|
|
217
|
+
def available_profiles(self) -> list[str]:
|
|
218
|
+
return list(self._profiles.keys())
|
|
219
|
+
|
|
220
|
+
def add_user_override(self, profile_name: str, key: str, value: Any) -> bool:
|
|
221
|
+
"""Add a user override to a profile."""
|
|
222
|
+
profile = self._profiles.get(profile_name)
|
|
223
|
+
if not profile:
|
|
224
|
+
return False
|
|
225
|
+
profile.overrides[key] = value
|
|
226
|
+
return True
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""B43. Auto-Wait Actionability Checks (Playwright Pattern).
|
|
2
|
+
|
|
3
|
+
Before performing an action (click, type, navigate), check that the target
|
|
4
|
+
is ready: exists, visible, stable, enabled, and receives input. No blind
|
|
5
|
+
sleeps — poll with timeout until conditions are met or fail fast.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Callable
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class CheckResult:
|
|
16
|
+
PASSED = "passed"
|
|
17
|
+
FAILED = "failed"
|
|
18
|
+
TIMEOUT = "timeout"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ActionabilityCheck:
|
|
23
|
+
name: str
|
|
24
|
+
check_fn: Callable[[], bool]
|
|
25
|
+
timeout_ms: int = 5000
|
|
26
|
+
poll_interval_ms: int = 100
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class ActionabilityReport:
|
|
31
|
+
checks: dict[str, str] = field(default_factory=dict) # name -> result
|
|
32
|
+
all_passed: bool = False
|
|
33
|
+
duration_ms: float = 0.0
|
|
34
|
+
failed_checks: list[str] = field(default_factory=list)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def auto_wait(checks: list[ActionabilityCheck]) -> ActionabilityReport:
|
|
38
|
+
"""Run all actionability checks with polling and timeout.
|
|
39
|
+
|
|
40
|
+
Each check is polled at its interval until it passes or times out.
|
|
41
|
+
All checks must pass for the report to be all_passed=True.
|
|
42
|
+
"""
|
|
43
|
+
start = time.monotonic()
|
|
44
|
+
results: dict[str, str] = {}
|
|
45
|
+
failed: list[str] = []
|
|
46
|
+
|
|
47
|
+
for check in checks:
|
|
48
|
+
deadline = time.monotonic() + check.timeout_ms / 1000
|
|
49
|
+
passed = False
|
|
50
|
+
while time.monotonic() < deadline:
|
|
51
|
+
try:
|
|
52
|
+
if check.check_fn():
|
|
53
|
+
passed = True
|
|
54
|
+
break
|
|
55
|
+
except Exception:
|
|
56
|
+
pass
|
|
57
|
+
time.sleep(check.poll_interval_ms / 1000)
|
|
58
|
+
|
|
59
|
+
if passed:
|
|
60
|
+
results[check.name] = CheckResult.PASSED
|
|
61
|
+
elif time.monotonic() >= deadline:
|
|
62
|
+
results[check.name] = CheckResult.TIMEOUT
|
|
63
|
+
failed.append(check.name)
|
|
64
|
+
else:
|
|
65
|
+
results[check.name] = CheckResult.FAILED
|
|
66
|
+
failed.append(check.name)
|
|
67
|
+
|
|
68
|
+
duration = (time.monotonic() - start) * 1000
|
|
69
|
+
return ActionabilityReport(
|
|
70
|
+
checks=results,
|
|
71
|
+
all_passed=len(failed) == 0,
|
|
72
|
+
duration_ms=duration,
|
|
73
|
+
failed_checks=failed,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ── standard checks ───────────────────────────────────────────────────
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def check_exists(predicate: Callable[[], Any]) -> ActionabilityCheck:
|
|
81
|
+
"""Target exists (not None)."""
|
|
82
|
+
return ActionabilityCheck(
|
|
83
|
+
name="exists",
|
|
84
|
+
check_fn=lambda: predicate() is not None,
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def check_visible(predicate: Callable[[], bool]) -> ActionabilityCheck:
|
|
89
|
+
"""Target is visible."""
|
|
90
|
+
return ActionabilityCheck(
|
|
91
|
+
name="visible",
|
|
92
|
+
check_fn=predicate,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def check_stable(predicate: Callable[[], bool], samples: int = 3, interval: float = 0.05) -> ActionabilityCheck:
|
|
97
|
+
"""Target has not changed over N consecutive samples."""
|
|
98
|
+
last_values: list[Any] = []
|
|
99
|
+
|
|
100
|
+
def is_stable() -> bool:
|
|
101
|
+
val = predicate()
|
|
102
|
+
last_values.append(val)
|
|
103
|
+
if len(last_values) > samples:
|
|
104
|
+
last_values.pop(0)
|
|
105
|
+
if len(last_values) < samples:
|
|
106
|
+
return False
|
|
107
|
+
return all(v == last_values[0] for v in last_values)
|
|
108
|
+
|
|
109
|
+
return ActionabilityCheck(name="stable", check_fn=is_stable)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def check_enabled(predicate: Callable[[], bool]) -> ActionabilityCheck:
|
|
113
|
+
"""Target is enabled (not disabled)."""
|
|
114
|
+
return ActionabilityCheck(
|
|
115
|
+
name="enabled",
|
|
116
|
+
check_fn=predicate,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def check_receives_input(predicate: Callable[[], bool]) -> ActionabilityCheck:
|
|
121
|
+
"""Target can receive input (focusable)."""
|
|
122
|
+
return ActionabilityCheck(
|
|
123
|
+
name="receives_input",
|
|
124
|
+
check_fn=predicate,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
# ── trace recorder ────────────────────────────────────────────────────
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class ActionTrace:
|
|
133
|
+
action: str
|
|
134
|
+
target: str
|
|
135
|
+
report: ActionabilityReport
|
|
136
|
+
timestamp: float = field(default_factory=time.time)
|
|
137
|
+
screenshot_path: str = ""
|
|
138
|
+
html_snapshot: str = ""
|
|
139
|
+
console_logs: list[str] = field(default_factory=list)
|
|
140
|
+
network_summary: dict[str, Any] = field(default_factory=dict)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class TraceRecorder:
|
|
144
|
+
"""Records action traces for debugging and replay."""
|
|
145
|
+
|
|
146
|
+
def __init__(self) -> None:
|
|
147
|
+
self.traces: list[ActionTrace] = []
|
|
148
|
+
|
|
149
|
+
def record(self, action: str, target: str, report: ActionabilityReport, **kwargs: Any) -> None:
|
|
150
|
+
self.traces.append(ActionTrace(
|
|
151
|
+
action=action, target=target, report=report, **kwargs,
|
|
152
|
+
))
|
|
153
|
+
|
|
154
|
+
def failed_actions(self) -> list[ActionTrace]:
|
|
155
|
+
return [t for t in self.traces if not t.report.all_passed]
|
|
156
|
+
|
|
157
|
+
def summary(self) -> dict[str, Any]:
|
|
158
|
+
total = len(self.traces)
|
|
159
|
+
passed = sum(1 for t in self.traces if t.report.all_passed)
|
|
160
|
+
return {
|
|
161
|
+
"total_actions": total,
|
|
162
|
+
"passed": passed,
|
|
163
|
+
"failed": total - passed,
|
|
164
|
+
"avg_wait_ms": sum(t.report.duration_ms for t in self.traces) / max(total, 1),
|
|
165
|
+
}
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""H8 — Adversarial Self-Audit.
|
|
2
|
+
|
|
3
|
+
Catches fabricated or exaggerated claims by comparing claimed behavior
|
|
4
|
+
against actual test results.
|
|
5
|
+
Mined from GLOSSOPETRAE §6.5 falsify_workflow.mjs.
|
|
6
|
+
|
|
7
|
+
LLM integration: optionally uses an LLM to analyze discrepancies.
|
|
8
|
+
Falls back to rule-based comparison when no model is available.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class Claim:
|
|
18
|
+
"""A claimed behavior to audit."""
|
|
19
|
+
|
|
20
|
+
entry_id: str
|
|
21
|
+
claimed_status: str # "implemented", "partial", "planned"
|
|
22
|
+
claimed_tests: list[str] = field(default_factory=list)
|
|
23
|
+
description: str = ""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class AuditResult:
|
|
28
|
+
"""Result of auditing a single claim."""
|
|
29
|
+
|
|
30
|
+
entry_id: str
|
|
31
|
+
passed: bool
|
|
32
|
+
discrepancies: list[str] = field(default_factory=list)
|
|
33
|
+
actual_status: str = ""
|
|
34
|
+
verified_tests: list[str] = field(default_factory=list)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def audit_claim(
|
|
38
|
+
claim: Claim,
|
|
39
|
+
actual_tests: list[str],
|
|
40
|
+
actual_status: str,
|
|
41
|
+
model_client: Any | None = None,
|
|
42
|
+
) -> AuditResult:
|
|
43
|
+
"""Audit a single claim against actual results.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
claim: The claimed behavior.
|
|
47
|
+
actual_tests: List of actual test names that passed.
|
|
48
|
+
actual_status: The actual status string.
|
|
49
|
+
model_client: Optional LLM for deeper analysis.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
AuditResult with discrepancies found.
|
|
53
|
+
"""
|
|
54
|
+
discrepancies: list[str] = []
|
|
55
|
+
|
|
56
|
+
# Check status mismatch
|
|
57
|
+
if claim.claimed_status != actual_status:
|
|
58
|
+
discrepancies.append(
|
|
59
|
+
f"Status mismatch: claimed '{claim.claimed_status}', actual '{actual_status}'"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
# Check missing tests
|
|
63
|
+
claimed_set = set(claim.claimed_tests)
|
|
64
|
+
actual_set = set(actual_tests)
|
|
65
|
+
missing = claimed_set - actual_set
|
|
66
|
+
if missing:
|
|
67
|
+
discrepancies.append(f"Missing tests: {sorted(missing)}")
|
|
68
|
+
|
|
69
|
+
# Check extra tests (not necessarily bad, but worth noting)
|
|
70
|
+
extra = actual_set - claimed_set
|
|
71
|
+
if extra and claim.claimed_tests:
|
|
72
|
+
discrepancies.append(f"Unclaimed tests: {sorted(extra)}")
|
|
73
|
+
|
|
74
|
+
# If claimed implemented but no tests, that's fabrication
|
|
75
|
+
if claim.claimed_status == "implemented" and not actual_tests:
|
|
76
|
+
discrepancies.append("Claimed 'implemented' but no passing tests found")
|
|
77
|
+
|
|
78
|
+
passed = len(discrepancies) == 0
|
|
79
|
+
return AuditResult(
|
|
80
|
+
entry_id=claim.entry_id,
|
|
81
|
+
passed=passed,
|
|
82
|
+
discrepancies=discrepancies,
|
|
83
|
+
actual_status=actual_status,
|
|
84
|
+
verified_tests=list(actual_set),
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def audit_batch(
|
|
89
|
+
claims: list[Claim],
|
|
90
|
+
actual_results: dict[str, tuple[list[str], str]],
|
|
91
|
+
model_client: Any | None = None,
|
|
92
|
+
) -> list[AuditResult]:
|
|
93
|
+
"""Audit multiple claims against actual results.
|
|
94
|
+
|
|
95
|
+
Args:
|
|
96
|
+
claims: List of claimed behaviors.
|
|
97
|
+
actual_results: Dict mapping entry_id to (test_names, status).
|
|
98
|
+
model_client: Optional LLM for deeper analysis.
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
List of AuditResults.
|
|
102
|
+
"""
|
|
103
|
+
results: list[AuditResult] = []
|
|
104
|
+
for claim in claims:
|
|
105
|
+
if claim.entry_id in actual_results:
|
|
106
|
+
tests, status = actual_results[claim.entry_id]
|
|
107
|
+
results.append(audit_claim(claim, tests, status, model_client))
|
|
108
|
+
else:
|
|
109
|
+
results.append(
|
|
110
|
+
AuditResult(
|
|
111
|
+
entry_id=claim.entry_id,
|
|
112
|
+
passed=False,
|
|
113
|
+
discrepancies=[f"No actual results found for entry '{claim.entry_id}'"],
|
|
114
|
+
actual_status="missing",
|
|
115
|
+
verified_tests=[],
|
|
116
|
+
)
|
|
117
|
+
)
|
|
118
|
+
return results
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def summarize_audit(results: list[AuditResult]) -> dict[str, Any]:
|
|
122
|
+
"""Summarize audit results."""
|
|
123
|
+
total = len(results)
|
|
124
|
+
passed = sum(1 for r in results if r.passed)
|
|
125
|
+
failed = total - passed
|
|
126
|
+
all_discrepancies: list[str] = []
|
|
127
|
+
for r in results:
|
|
128
|
+
all_discrepancies.extend(r.discrepancies)
|
|
129
|
+
return {
|
|
130
|
+
"total": total,
|
|
131
|
+
"passed": passed,
|
|
132
|
+
"failed": failed,
|
|
133
|
+
"pass_rate": passed / total if total > 0 else 0.0,
|
|
134
|
+
"discrepancies": all_discrepancies,
|
|
135
|
+
"discrepancy_count": len(all_discrepancies),
|
|
136
|
+
}
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""B58. Agent Arena: Multi-Model Head-to-Head.
|
|
2
|
+
|
|
3
|
+
Run the same task across multiple models and compare outputs.
|
|
4
|
+
Source: qwen-code pattern.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from enum import Enum, auto
|
|
11
|
+
from typing import Callable
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ArenaOutcome(Enum):
|
|
15
|
+
WINNER = auto()
|
|
16
|
+
TIE = auto()
|
|
17
|
+
ALL_FAILED = auto()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class ArenaEntry:
|
|
22
|
+
model: str
|
|
23
|
+
output: str = ""
|
|
24
|
+
duration_s: float = 0.0
|
|
25
|
+
token_count: int = 0
|
|
26
|
+
error: str | None = None
|
|
27
|
+
success: bool = True
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class ArenaResult:
|
|
32
|
+
task: str
|
|
33
|
+
entries: list[ArenaEntry] = field(default_factory=list)
|
|
34
|
+
winner: str | None = None
|
|
35
|
+
outcome: ArenaOutcome = ArenaOutcome.TIE
|
|
36
|
+
comparison: str = ""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class AgentArena:
|
|
40
|
+
"""Run the same task across multiple models and compare."""
|
|
41
|
+
|
|
42
|
+
def __init__(self, judge_fn: Callable[[str, list[ArenaEntry]], str] | None = None) -> None:
|
|
43
|
+
self._judge = judge_fn or self._default_judge
|
|
44
|
+
|
|
45
|
+
@staticmethod
|
|
46
|
+
def _default_judge(task: str, entries: list[ArenaEntry]) -> str:
|
|
47
|
+
"""Default judge: pick the longest successful output."""
|
|
48
|
+
successful = [e for e in entries if e.success and e.output]
|
|
49
|
+
if not successful:
|
|
50
|
+
return ""
|
|
51
|
+
return max(successful, key=lambda e: len(e.output)).model
|
|
52
|
+
|
|
53
|
+
def run(
|
|
54
|
+
self,
|
|
55
|
+
task: str,
|
|
56
|
+
models: dict[str, Callable[[str], str]],
|
|
57
|
+
timeout: float = 30.0,
|
|
58
|
+
) -> ArenaResult:
|
|
59
|
+
"""Run task on all models, return comparison."""
|
|
60
|
+
result = ArenaResult(task=task)
|
|
61
|
+
for model_name, run_fn in models.items():
|
|
62
|
+
entry = ArenaEntry(model=model_name)
|
|
63
|
+
start = time.time()
|
|
64
|
+
try:
|
|
65
|
+
entry.output = run_fn(task)
|
|
66
|
+
entry.token_count = len(entry.output) // 4
|
|
67
|
+
except Exception as e:
|
|
68
|
+
entry.error = str(e)
|
|
69
|
+
entry.success = False
|
|
70
|
+
entry.duration_s = time.time() - start
|
|
71
|
+
result.entries.append(entry)
|
|
72
|
+
|
|
73
|
+
# Judge
|
|
74
|
+
winner = self._judge(task, result.entries)
|
|
75
|
+
if winner:
|
|
76
|
+
result.winner = winner
|
|
77
|
+
result.outcome = ArenaOutcome.WINNER
|
|
78
|
+
elif any(e.success for e in result.entries):
|
|
79
|
+
result.outcome = ArenaOutcome.TIE
|
|
80
|
+
else:
|
|
81
|
+
result.outcome = ArenaOutcome.ALL_FAILED
|
|
82
|
+
|
|
83
|
+
result.comparison = self._format_comparison(result)
|
|
84
|
+
return result
|
|
85
|
+
|
|
86
|
+
@staticmethod
|
|
87
|
+
def _format_comparison(result: ArenaResult) -> str:
|
|
88
|
+
lines = [f"Task: {result.task}", f"Winner: {result.winner or 'N/A'}", ""]
|
|
89
|
+
for e in result.entries:
|
|
90
|
+
status = "OK" if e.success else f"FAIL: {e.error}"
|
|
91
|
+
lines.append(f" {e.model}: {e.duration_s:.1f}s, {e.token_count}t — {status}")
|
|
92
|
+
return "\n".join(lines)
|