algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""B47. Gym-like Agent Evaluation Environment (TextWorld Pattern).
|
|
2
|
+
|
|
3
|
+
Benchmark Algo CLI agent performance with reproducible episodes, score
|
|
4
|
+
tracking, and step limits. Uses a Gym-like API: reset() → (obs, infos),
|
|
5
|
+
step(action) → (obs, score, done, infos).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Callable
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class EpisodeResult:
|
|
17
|
+
task: str
|
|
18
|
+
steps: int
|
|
19
|
+
max_steps: int
|
|
20
|
+
score: float
|
|
21
|
+
done: bool
|
|
22
|
+
success: bool
|
|
23
|
+
tool_calls: list[dict] = field(default_factory=list)
|
|
24
|
+
errors: list[str] = field(default_factory=list)
|
|
25
|
+
trace: list[dict] = field(default_factory=list)
|
|
26
|
+
duration_ms: float = 0.0
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class BenchmarkScenario:
|
|
31
|
+
"""A single benchmark scenario definition."""
|
|
32
|
+
name: str
|
|
33
|
+
task: str
|
|
34
|
+
seed: int = 42
|
|
35
|
+
max_steps: int = 50
|
|
36
|
+
setup_fn: Callable[[], dict] | None = None
|
|
37
|
+
verify_fn: Callable[[str, dict], tuple[bool, float]] | None = None
|
|
38
|
+
# verify_fn returns (success, score)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class AgentEnv:
|
|
42
|
+
"""Gym-like environment for a single benchmark episode."""
|
|
43
|
+
|
|
44
|
+
def __init__(self, scenario: BenchmarkScenario):
|
|
45
|
+
self.scenario = scenario
|
|
46
|
+
self.task = scenario.task
|
|
47
|
+
self.seed = scenario.seed
|
|
48
|
+
self.max_steps = scenario.max_steps
|
|
49
|
+
self._step = 0
|
|
50
|
+
self._done = False
|
|
51
|
+
self._score = 0.0
|
|
52
|
+
self._tool_calls: list[dict] = []
|
|
53
|
+
self._errors: list[str] = []
|
|
54
|
+
self._trace: list[dict] = []
|
|
55
|
+
self._state: dict[str, Any] = {}
|
|
56
|
+
|
|
57
|
+
def reset(self) -> tuple[str, dict]:
|
|
58
|
+
"""Reset environment to initial state."""
|
|
59
|
+
self._step = 0
|
|
60
|
+
self._done = False
|
|
61
|
+
self._score = 0.0
|
|
62
|
+
self._tool_calls = []
|
|
63
|
+
self._errors = []
|
|
64
|
+
self._trace = []
|
|
65
|
+
if self.scenario.setup_fn:
|
|
66
|
+
self._state = self.scenario.setup_fn()
|
|
67
|
+
else:
|
|
68
|
+
self._state = {}
|
|
69
|
+
return self.task, {"step": 0, "state": self._state}
|
|
70
|
+
|
|
71
|
+
def step(self, action: dict) -> tuple[str, float, bool, dict]:
|
|
72
|
+
"""Execute one agent action.
|
|
73
|
+
|
|
74
|
+
action is a dict with:
|
|
75
|
+
- "type": "tool_call", "response", "done"
|
|
76
|
+
- "tool": tool name (for tool_call)
|
|
77
|
+
- "params": tool params
|
|
78
|
+
- "content": response text
|
|
79
|
+
"""
|
|
80
|
+
self._step += 1
|
|
81
|
+
action_type = action.get("type", "response")
|
|
82
|
+
reward = 0.0
|
|
83
|
+
|
|
84
|
+
if action_type == "tool_call":
|
|
85
|
+
self._tool_calls.append({
|
|
86
|
+
"tool": action.get("tool", ""),
|
|
87
|
+
"params": action.get("params", {}),
|
|
88
|
+
"step": self._step,
|
|
89
|
+
})
|
|
90
|
+
self._trace.append({"step": self._step, "type": "tool_call", "tool": action.get("tool", "")})
|
|
91
|
+
# Small penalty for each tool call (encourages efficiency)
|
|
92
|
+
reward = -0.1
|
|
93
|
+
|
|
94
|
+
elif action_type == "response":
|
|
95
|
+
content = action.get("content", "")
|
|
96
|
+
self._trace.append({"step": self._step, "type": "response", "content": content[:200]})
|
|
97
|
+
# Check if the response is correct
|
|
98
|
+
if self.scenario.verify_fn:
|
|
99
|
+
success, score = self.scenario.verify_fn(content, self._state)
|
|
100
|
+
reward = score
|
|
101
|
+
if success:
|
|
102
|
+
self._done = True
|
|
103
|
+
else:
|
|
104
|
+
reward = 0.0
|
|
105
|
+
|
|
106
|
+
elif action_type == "done":
|
|
107
|
+
self._done = True
|
|
108
|
+
|
|
109
|
+
elif action_type == "error":
|
|
110
|
+
self._errors.append(action.get("content", "unknown error"))
|
|
111
|
+
reward = -0.5
|
|
112
|
+
|
|
113
|
+
self._score += reward
|
|
114
|
+
|
|
115
|
+
if self._step >= self.max_steps:
|
|
116
|
+
self._done = True
|
|
117
|
+
|
|
118
|
+
infos = {
|
|
119
|
+
"step": self._step,
|
|
120
|
+
"score": self._score,
|
|
121
|
+
"tool_calls": list(self._tool_calls),
|
|
122
|
+
"errors": list(self._errors),
|
|
123
|
+
"trace": list(self._trace),
|
|
124
|
+
"success": self._done and self._score > 0,
|
|
125
|
+
}
|
|
126
|
+
return "", reward, self._done, infos
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class AgentBenchmark:
|
|
130
|
+
"""Runs benchmark episodes and aggregates results."""
|
|
131
|
+
|
|
132
|
+
def __init__(self, max_steps: int = 50):
|
|
133
|
+
self.max_steps = max_steps
|
|
134
|
+
self.scenarios: dict[str, BenchmarkScenario] = {}
|
|
135
|
+
self.results: list[EpisodeResult] = []
|
|
136
|
+
|
|
137
|
+
def add_scenario(self, scenario: BenchmarkScenario) -> None:
|
|
138
|
+
self.scenarios[scenario.name] = scenario
|
|
139
|
+
|
|
140
|
+
def run_episode(self, scenario_name: str, agent_fn: Callable[[str, dict], dict]) -> EpisodeResult:
|
|
141
|
+
"""Run a single episode with the given agent function."""
|
|
142
|
+
scenario = self.scenarios.get(scenario_name)
|
|
143
|
+
if not scenario:
|
|
144
|
+
return EpisodeResult(
|
|
145
|
+
task="", steps=0, max_steps=0, score=0,
|
|
146
|
+
done=True, success=False, errors=["scenario not found"],
|
|
147
|
+
)
|
|
148
|
+
env = AgentEnv(scenario)
|
|
149
|
+
obs, infos = env.reset()
|
|
150
|
+
start = time.monotonic()
|
|
151
|
+
while not env._done:
|
|
152
|
+
action = agent_fn(obs, infos)
|
|
153
|
+
obs, reward, done, infos = env.step(action)
|
|
154
|
+
if done:
|
|
155
|
+
break
|
|
156
|
+
duration = (time.monotonic() - start) * 1000
|
|
157
|
+
result = EpisodeResult(
|
|
158
|
+
task=scenario.task,
|
|
159
|
+
steps=env._step,
|
|
160
|
+
max_steps=scenario.max_steps,
|
|
161
|
+
score=env._score,
|
|
162
|
+
done=env._done,
|
|
163
|
+
success=infos.get("success", False),
|
|
164
|
+
tool_calls=env._tool_calls,
|
|
165
|
+
errors=env._errors,
|
|
166
|
+
trace=env._trace,
|
|
167
|
+
duration_ms=duration,
|
|
168
|
+
)
|
|
169
|
+
self.results.append(result)
|
|
170
|
+
return result
|
|
171
|
+
|
|
172
|
+
def report(self) -> dict[str, Any]:
|
|
173
|
+
"""Aggregate results across all episodes."""
|
|
174
|
+
if not self.results:
|
|
175
|
+
return {"total": 0}
|
|
176
|
+
scores = [r.score for r in self.results]
|
|
177
|
+
steps = [r.steps for r in self.results]
|
|
178
|
+
successes = sum(1 for r in self.results if r.success)
|
|
179
|
+
sorted_scores = sorted(scores)
|
|
180
|
+
n = len(sorted_scores)
|
|
181
|
+
p50 = sorted_scores[n // 2]
|
|
182
|
+
p90 = sorted_scores[int(n * 0.9)] if n > 1 else sorted_scores[0]
|
|
183
|
+
return {
|
|
184
|
+
"total": len(self.results),
|
|
185
|
+
"successes": successes,
|
|
186
|
+
"success_rate": successes / len(self.results),
|
|
187
|
+
"avg_score": sum(scores) / len(scores),
|
|
188
|
+
"p50_score": p50,
|
|
189
|
+
"p90_score": p90,
|
|
190
|
+
"avg_steps": sum(steps) / len(steps),
|
|
191
|
+
"avg_duration_ms": sum(r.duration_ms for r in self.results) / len(self.results),
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
# ── built-in scenarios ────────────────────────────────────────────────
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def make_find_file_scenario() -> BenchmarkScenario:
|
|
199
|
+
"""Scenario: find a file and extract a value."""
|
|
200
|
+
state = {"target_file": "EST-2026-002_200A.html", "target_value": "8291.60"}
|
|
201
|
+
|
|
202
|
+
def setup() -> dict:
|
|
203
|
+
return state
|
|
204
|
+
|
|
205
|
+
def verify(content: str, st: dict) -> tuple[bool, float]:
|
|
206
|
+
if st["target_value"] in content:
|
|
207
|
+
return True, 1.0
|
|
208
|
+
return False, 0.0
|
|
209
|
+
|
|
210
|
+
return BenchmarkScenario(
|
|
211
|
+
name="find_file",
|
|
212
|
+
task=f"Find the file {state['target_file']} and extract the total price.",
|
|
213
|
+
setup_fn=setup,
|
|
214
|
+
verify_fn=verify,
|
|
215
|
+
max_steps=10,
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def make_pdf_extract_scenario() -> BenchmarkScenario:
|
|
220
|
+
"""Scenario: extract line items from a PDF."""
|
|
221
|
+
|
|
222
|
+
def setup() -> dict:
|
|
223
|
+
return {"expected_items": ["panel", "breakers", "conductors", "meter base", "grounding"]}
|
|
224
|
+
|
|
225
|
+
def verify(content: str, st: dict) -> tuple[bool, float]:
|
|
226
|
+
found = sum(1 for item in st["expected_items"] if item.lower() in content.lower())
|
|
227
|
+
score = found / len(st["expected_items"])
|
|
228
|
+
return (score >= 0.8, score)
|
|
229
|
+
|
|
230
|
+
return BenchmarkScenario(
|
|
231
|
+
name="pdf_extract",
|
|
232
|
+
task="Read the attached PDF and list all line items.",
|
|
233
|
+
setup_fn=setup,
|
|
234
|
+
verify_fn=verify,
|
|
235
|
+
max_steps=15,
|
|
236
|
+
)
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""B36. Layered Agent Runtime + Workbench Boundary (AutoGen Pattern).
|
|
2
|
+
|
|
3
|
+
Separates agent message passing (runtime), agent logic (agent API), and
|
|
4
|
+
tool execution (workbench) into distinct layers with a trust boundary.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import time
|
|
10
|
+
import uuid
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Callable, Protocol
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# ── message layer ─────────────────────────────────────────────────────
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class Message:
|
|
20
|
+
id: str
|
|
21
|
+
role: str # "user", "assistant", "tool", "system"
|
|
22
|
+
content: str
|
|
23
|
+
agent_name: str = ""
|
|
24
|
+
tool_call_id: str = ""
|
|
25
|
+
timestamp: float = field(default_factory=time.time)
|
|
26
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class TraceEvent:
|
|
31
|
+
type: str # "message", "tool_call", "tool_result", "cancel", "handoff"
|
|
32
|
+
data: dict[str, Any] = field(default_factory=dict)
|
|
33
|
+
timestamp: float = field(default_factory=time.time)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ── runtime layer ─────────────────────────────────────────────────────
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class AgentRuntime:
|
|
40
|
+
"""Message bus + trace recorder + cancellation support."""
|
|
41
|
+
|
|
42
|
+
def __init__(self) -> None:
|
|
43
|
+
self.messages: list[Message] = []
|
|
44
|
+
self.traces: list[TraceEvent] = []
|
|
45
|
+
self._cancelled = False
|
|
46
|
+
self._trace_id = str(uuid.uuid4())
|
|
47
|
+
|
|
48
|
+
@property
|
|
49
|
+
def trace_id(self) -> str:
|
|
50
|
+
return self._trace_id
|
|
51
|
+
|
|
52
|
+
def post(self, msg: Message) -> None:
|
|
53
|
+
self.messages.append(msg)
|
|
54
|
+
self.traces.append(TraceEvent(type="message", data={"role": msg.role, "agent": msg.agent_name}))
|
|
55
|
+
|
|
56
|
+
def cancel(self) -> None:
|
|
57
|
+
self._cancelled = True
|
|
58
|
+
self.traces.append(TraceEvent(type="cancel"))
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def cancelled(self) -> bool:
|
|
62
|
+
return self._cancelled
|
|
63
|
+
|
|
64
|
+
def get_history(self, agent_name: str | None = None) -> list[Message]:
|
|
65
|
+
if agent_name:
|
|
66
|
+
return [m for m in self.messages if m.agent_name == agent_name]
|
|
67
|
+
return list(self.messages)
|
|
68
|
+
|
|
69
|
+
def handoff(self, from_agent: str, to_agent: str, context: str = "") -> None:
|
|
70
|
+
self.traces.append(TraceEvent(
|
|
71
|
+
type="handoff",
|
|
72
|
+
data={"from": from_agent, "to": to_agent, "context": context},
|
|
73
|
+
))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# ── agent API layer ───────────────────────────────────────────────────
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Agent(Protocol):
|
|
80
|
+
"""Protocol for agent implementations."""
|
|
81
|
+
|
|
82
|
+
name: str
|
|
83
|
+
role: str
|
|
84
|
+
|
|
85
|
+
def respond(self, runtime: AgentRuntime, task: str) -> str: ...
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class SimpleAgent:
|
|
89
|
+
"""A basic agent that uses a callable to generate responses."""
|
|
90
|
+
|
|
91
|
+
def __init__(self, name: str, role: str, fn: Callable[[str, list[Message]], str]):
|
|
92
|
+
self.name = name
|
|
93
|
+
self.role = role
|
|
94
|
+
self._fn = fn
|
|
95
|
+
|
|
96
|
+
def respond(self, runtime: AgentRuntime, task: str) -> str:
|
|
97
|
+
history = runtime.get_history()
|
|
98
|
+
result = self._fn(task, history)
|
|
99
|
+
runtime.post(Message(
|
|
100
|
+
id=str(uuid.uuid4()),
|
|
101
|
+
role="assistant",
|
|
102
|
+
content=result,
|
|
103
|
+
agent_name=self.name,
|
|
104
|
+
))
|
|
105
|
+
return result
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
# ── workbench layer (tool execution boundary) ────────────────────────
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@dataclass
|
|
112
|
+
class ToolAction:
|
|
113
|
+
tool: str
|
|
114
|
+
params: dict[str, Any]
|
|
115
|
+
requires_approval: bool = False
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
@dataclass
|
|
119
|
+
class ToolResult:
|
|
120
|
+
tool: str
|
|
121
|
+
success: bool
|
|
122
|
+
output: str
|
|
123
|
+
error: str | None = None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class Workbench:
|
|
127
|
+
"""Untrusted tool execution boundary.
|
|
128
|
+
|
|
129
|
+
Tools cannot directly mutate files or send externally — they return
|
|
130
|
+
action proposals that the host gates.
|
|
131
|
+
"""
|
|
132
|
+
|
|
133
|
+
def __init__(self, approval_fn: Callable[[ToolAction], bool] | None = None):
|
|
134
|
+
self._tools: dict[str, Callable[[dict], str]] = {}
|
|
135
|
+
self._approval_fn = approval_fn or (lambda a: not a.requires_approval)
|
|
136
|
+
self._allowlisted: set[str] = set()
|
|
137
|
+
|
|
138
|
+
def register_tool(self, name: str, fn: Callable[[dict], str], allowlisted: bool = False) -> None:
|
|
139
|
+
self._tools[name] = fn
|
|
140
|
+
if allowlisted:
|
|
141
|
+
self._allowlisted.add(name)
|
|
142
|
+
|
|
143
|
+
def execute(self, action: ToolAction) -> ToolResult:
|
|
144
|
+
if action.tool not in self._tools:
|
|
145
|
+
return ToolResult(tool=action.tool, success=False, output="", error="unknown tool")
|
|
146
|
+
if action.requires_approval and not self._approval_fn(action):
|
|
147
|
+
return ToolResult(tool=action.tool, success=False, output="", error="approval denied")
|
|
148
|
+
try:
|
|
149
|
+
output = self._tools[action.tool](action.params)
|
|
150
|
+
return ToolResult(tool=action.tool, success=True, output=output)
|
|
151
|
+
except Exception as e:
|
|
152
|
+
return ToolResult(tool=action.tool, success=False, output="", error=str(e))
|
|
153
|
+
|
|
154
|
+
@property
|
|
155
|
+
def allowlisted_tools(self) -> set[str]:
|
|
156
|
+
return set(self._allowlisted)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# ── agent-as-tool wrapper ─────────────────────────────────────────────
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class AgentAsTool:
|
|
163
|
+
"""Wraps an agent so it can be invoked as a tool by another agent."""
|
|
164
|
+
|
|
165
|
+
def __init__(self, agent: Agent, runtime: AgentRuntime):
|
|
166
|
+
self._agent = agent
|
|
167
|
+
self._runtime = runtime
|
|
168
|
+
|
|
169
|
+
def __call__(self, params: dict) -> str:
|
|
170
|
+
task = params.get("task", "")
|
|
171
|
+
return self._agent.respond(self._runtime, task)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""B73. Agents-as-Tools + Handoffs.
|
|
2
|
+
|
|
3
|
+
Delegate to other agents via tool calls or handoffs.
|
|
4
|
+
Source: OpenAI Agents SDK pattern.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from enum import Enum, auto
|
|
10
|
+
from typing import Callable
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class DelegationMode(Enum):
|
|
14
|
+
AS_TOOL = auto() # parent calls agent, gets result, continues
|
|
15
|
+
HANDOFF = auto() # parent transfers control entirely
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class AgentTool:
|
|
20
|
+
name: str
|
|
21
|
+
description: str
|
|
22
|
+
agent_id: str
|
|
23
|
+
run_fn: Callable[[str], str]
|
|
24
|
+
mode: DelegationMode = DelegationMode.AS_TOOL
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class HandoffResult:
|
|
29
|
+
transferred_to: str
|
|
30
|
+
reason: str = ""
|
|
31
|
+
context_summary: str = ""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class AgentsAsTools:
|
|
35
|
+
"""Register agents as callable tools or handoff targets."""
|
|
36
|
+
|
|
37
|
+
def __init__(self) -> None:
|
|
38
|
+
self._tools: dict[str, AgentTool] = {}
|
|
39
|
+
self._handoff_targets: dict[str, AgentTool] = {}
|
|
40
|
+
|
|
41
|
+
def register_as_tool(self, tool: AgentTool) -> None:
|
|
42
|
+
tool.mode = DelegationMode.AS_TOOL
|
|
43
|
+
self._tools[tool.name] = tool
|
|
44
|
+
|
|
45
|
+
def register_as_handoff(self, tool: AgentTool) -> None:
|
|
46
|
+
tool.mode = DelegationMode.HANDOFF
|
|
47
|
+
self._handoff_targets[tool.agent_id] = tool
|
|
48
|
+
|
|
49
|
+
def call_tool(self, tool_name: str, input: str) -> str:
|
|
50
|
+
"""Call an agent as a tool — parent continues after."""
|
|
51
|
+
tool = self._tools.get(tool_name)
|
|
52
|
+
if not tool:
|
|
53
|
+
raise KeyError(f"Tool '{tool_name}' not found")
|
|
54
|
+
return tool.run_fn(input)
|
|
55
|
+
|
|
56
|
+
def handoff(self, agent_id: str, context: str, reason: str = "") -> HandoffResult:
|
|
57
|
+
"""Transfer control to another agent entirely."""
|
|
58
|
+
tool = self._handoff_targets.get(agent_id)
|
|
59
|
+
if not tool:
|
|
60
|
+
raise KeyError(f"Handoff target '{agent_id}' not found")
|
|
61
|
+
result = HandoffResult(transferred_to=agent_id, reason=reason, context_summary=context[:500])
|
|
62
|
+
tool.run_fn(context)
|
|
63
|
+
return result
|
|
64
|
+
|
|
65
|
+
def list_tools(self) -> list[dict[str, str]]:
|
|
66
|
+
return [{"name": t.name, "description": t.description, "agent": t.agent_id}
|
|
67
|
+
for t in self._tools.values()]
|
|
68
|
+
|
|
69
|
+
def list_handoff_targets(self) -> list[str]:
|
|
70
|
+
return list(self._handoff_targets.keys())
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""H9 — Ground-Truth Artifact Binding.
|
|
2
|
+
|
|
3
|
+
Every claim binds to a raw artifact with a content hash.
|
|
4
|
+
Mined from GLOSSOPETRAE 78 raw JSONs + T3MP3ST bench/.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import hashlib
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _compute_hash(content: bytes | str) -> str:
|
|
14
|
+
if isinstance(content, str):
|
|
15
|
+
content = content.encode("utf-8")
|
|
16
|
+
return hashlib.sha256(content).hexdigest()
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class ArtifactBinding:
|
|
21
|
+
"""A binding between a claim and a raw artifact."""
|
|
22
|
+
|
|
23
|
+
claim_id: str
|
|
24
|
+
artifact_path: str
|
|
25
|
+
content_hash: str
|
|
26
|
+
artifact_type: str = "json"
|
|
27
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
28
|
+
|
|
29
|
+
def to_dict(self) -> dict[str, Any]:
|
|
30
|
+
return {
|
|
31
|
+
"claim_id": self.claim_id,
|
|
32
|
+
"artifact_path": self.artifact_path,
|
|
33
|
+
"content_hash": self.content_hash,
|
|
34
|
+
"artifact_type": self.artifact_type,
|
|
35
|
+
"metadata": dict(self.metadata),
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class ArtifactBinder:
|
|
40
|
+
"""Bind claims to raw artifacts with content hashes."""
|
|
41
|
+
|
|
42
|
+
def __init__(self) -> None:
|
|
43
|
+
self._bindings: dict[str, ArtifactBinding] = {}
|
|
44
|
+
|
|
45
|
+
def bind(
|
|
46
|
+
self,
|
|
47
|
+
claim_id: str,
|
|
48
|
+
artifact_path: str,
|
|
49
|
+
content: bytes | str,
|
|
50
|
+
artifact_type: str = "json",
|
|
51
|
+
metadata: dict[str, Any] | None = None,
|
|
52
|
+
) -> ArtifactBinding:
|
|
53
|
+
content_hash = _compute_hash(content)
|
|
54
|
+
binding = ArtifactBinding(
|
|
55
|
+
claim_id=claim_id,
|
|
56
|
+
artifact_path=artifact_path,
|
|
57
|
+
content_hash=content_hash,
|
|
58
|
+
artifact_type=artifact_type,
|
|
59
|
+
metadata=metadata or {},
|
|
60
|
+
)
|
|
61
|
+
self._bindings[claim_id] = binding
|
|
62
|
+
return binding
|
|
63
|
+
|
|
64
|
+
def verify(self, claim_id: str, content: bytes | str) -> bool:
|
|
65
|
+
binding = self._bindings.get(claim_id)
|
|
66
|
+
if binding is None:
|
|
67
|
+
return False
|
|
68
|
+
return _compute_hash(content) == binding.content_hash
|
|
69
|
+
|
|
70
|
+
def get(self, claim_id: str) -> ArtifactBinding | None:
|
|
71
|
+
return self._bindings.get(claim_id)
|
|
72
|
+
|
|
73
|
+
def all(self) -> list[ArtifactBinding]:
|
|
74
|
+
return list(self._bindings.values())
|
|
75
|
+
|
|
76
|
+
def count(self) -> int:
|
|
77
|
+
return len(self._bindings)
|
|
78
|
+
|
|
79
|
+
def remove(self, claim_id: str) -> bool:
|
|
80
|
+
return self._bindings.pop(claim_id, None) is not None
|