algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""ReAct+ Enhanced Reasoning-Action Loop.
|
|
2
|
+
|
|
3
|
+
Interleaves structured Thought, Action, and Observation steps with:
|
|
4
|
+
- Typed action parsing (tool calls, sub-questions, assertions)
|
|
5
|
+
- Observation summarization to prevent context bloat
|
|
6
|
+
- Automatic thought-chain compaction for long episodes
|
|
7
|
+
- Loop detection with strategy broadening
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import re
|
|
14
|
+
import time
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import Any, Callable
|
|
17
|
+
|
|
18
|
+
from ..chat_protocol import get_attr
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ReactStep:
|
|
23
|
+
"""One step in a ReAct episode."""
|
|
24
|
+
thought: str
|
|
25
|
+
action: str
|
|
26
|
+
action_input: dict[str, Any] | str
|
|
27
|
+
observation: str
|
|
28
|
+
timestamp: float = field(default_factory=time.time)
|
|
29
|
+
|
|
30
|
+
def to_message(self) -> dict[str, Any]:
|
|
31
|
+
return {
|
|
32
|
+
"role": "assistant",
|
|
33
|
+
"content": f"Thought: {self.thought}\nAction: {self.action}\nAction Input: {self._format_input()}",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
def observation_message(self) -> dict[str, Any]:
|
|
37
|
+
return {
|
|
38
|
+
"role": "user",
|
|
39
|
+
"content": f"Observation: {self.observation}",
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
def _format_input(self) -> str:
|
|
43
|
+
if isinstance(self.action_input, dict):
|
|
44
|
+
return json.dumps(self.action_input)
|
|
45
|
+
return str(self.action_input)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
THOUGHT_RE = re.compile(r"Thought:\s*(.+?)(?=\nAction:|$)", re.DOTALL)
|
|
49
|
+
ACTION_RE = re.compile(r"Action:\s*(.+?)(?=\nAction Input:|$)", re.DOTALL)
|
|
50
|
+
ACTION_INPUT_RE = re.compile(r"Action Input:\s*(.+?)$", re.DOTALL)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def parse_react_output(text: str) -> tuple[str, str, dict[str, Any] | str]:
|
|
54
|
+
"""Parse a ReAct-formatted response into (thought, action, action_input)."""
|
|
55
|
+
thought_m = THOUGHT_RE.search(text)
|
|
56
|
+
action_m = ACTION_RE.search(text)
|
|
57
|
+
input_m = ACTION_INPUT_RE.search(text)
|
|
58
|
+
|
|
59
|
+
thought = thought_m.group(1).strip() if thought_m else ""
|
|
60
|
+
action = action_m.group(1).strip() if action_m else ""
|
|
61
|
+
raw_input = input_m.group(1).strip() if input_m else ""
|
|
62
|
+
|
|
63
|
+
# Try parsing action input as JSON; fall back to string
|
|
64
|
+
action_input: dict[str, Any] | str = raw_input
|
|
65
|
+
if raw_input:
|
|
66
|
+
try:
|
|
67
|
+
parsed = json.loads(raw_input)
|
|
68
|
+
if isinstance(parsed, dict):
|
|
69
|
+
action_input = parsed
|
|
70
|
+
except (json.JSONDecodeError, ValueError):
|
|
71
|
+
pass
|
|
72
|
+
|
|
73
|
+
return thought, action, action_input
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def compact_observations(steps: list[ReactStep], max_chars: int = 4000) -> str:
|
|
77
|
+
"""Compact a ReAct episode into a summary for context injection."""
|
|
78
|
+
if not steps:
|
|
79
|
+
return ""
|
|
80
|
+
lines: list[str] = []
|
|
81
|
+
total = 0
|
|
82
|
+
for i, step in enumerate(steps):
|
|
83
|
+
entry = f"[{i+1}] Thought: {step.thought[:200]}\n Action: {step.action} -> Obs: {step.observation[:300]}"
|
|
84
|
+
if total + len(entry) > max_chars:
|
|
85
|
+
remaining = len(steps) - i
|
|
86
|
+
lines.append(f"... ({remaining} earlier steps compacted)")
|
|
87
|
+
break
|
|
88
|
+
lines.append(entry)
|
|
89
|
+
total += len(entry)
|
|
90
|
+
return "\n".join(lines)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass
|
|
94
|
+
class ReactLoop:
|
|
95
|
+
"""Stateful ReAct+ loop for agent harness integration."""
|
|
96
|
+
max_steps: int = 10
|
|
97
|
+
observation_limit: int = 2000
|
|
98
|
+
loop_detection_window: int = 3
|
|
99
|
+
|
|
100
|
+
steps: list[ReactStep] = field(default_factory=list)
|
|
101
|
+
_action_history: list[str] = field(default_factory=list)
|
|
102
|
+
|
|
103
|
+
def detect_loop(self) -> bool:
|
|
104
|
+
"""Detect if the last N actions are identical (stuck loop)."""
|
|
105
|
+
if len(self._action_history) < self.loop_detection_window:
|
|
106
|
+
return False
|
|
107
|
+
window = self._action_history[-self.loop_detection_window:]
|
|
108
|
+
return len(set(window)) == 1
|
|
109
|
+
|
|
110
|
+
def add_step(self, step: ReactStep) -> None:
|
|
111
|
+
self.steps.append(step)
|
|
112
|
+
self._action_history.append(f"{step.action}:{str(step.action_input)[:80]}")
|
|
113
|
+
|
|
114
|
+
def truncate_observation(self, obs: str) -> str:
|
|
115
|
+
if len(obs) > self.observation_limit:
|
|
116
|
+
return obs[:self.observation_limit - 20] + "\n...[truncated]"
|
|
117
|
+
return obs
|
|
118
|
+
|
|
119
|
+
def build_context(self, task: str, system: str) -> list[dict[str, Any]]:
|
|
120
|
+
"""Build the message list for the next LLM call."""
|
|
121
|
+
messages: list[dict[str, Any]] = [
|
|
122
|
+
{"role": "system", "content": system},
|
|
123
|
+
{"role": "user", "content": task},
|
|
124
|
+
]
|
|
125
|
+
for step in self.steps:
|
|
126
|
+
messages.append(step.to_message())
|
|
127
|
+
messages.append(step.observation_message())
|
|
128
|
+
# Final prompt to continue reasoning
|
|
129
|
+
messages.append({
|
|
130
|
+
"role": "user",
|
|
131
|
+
"content": "Continue with your next Thought and Action. If you have enough information to answer, respond with just the final answer.",
|
|
132
|
+
})
|
|
133
|
+
return messages
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def run_react_loop(
|
|
137
|
+
*,
|
|
138
|
+
task: str,
|
|
139
|
+
client: Any,
|
|
140
|
+
model: str,
|
|
141
|
+
tools: list[Any] | None = None,
|
|
142
|
+
system: str = "You are a reasoning agent. Use the ReAct format:\nThought: <your reasoning>\nAction: <tool name or 'finish'>\nAction Input: <JSON args or final answer>",
|
|
143
|
+
max_steps: int = 10,
|
|
144
|
+
tool_map: dict[str, Callable] | None = None,
|
|
145
|
+
observation_limit: int = 2000,
|
|
146
|
+
) -> list[ReactStep]:
|
|
147
|
+
"""Run a complete ReAct+ episode.
|
|
148
|
+
|
|
149
|
+
Args:
|
|
150
|
+
task: The task to solve.
|
|
151
|
+
client: Ollama client instance.
|
|
152
|
+
model: Model name.
|
|
153
|
+
tools: Optional list of tool functions for the model.
|
|
154
|
+
system: System prompt (ReAct format instructions).
|
|
155
|
+
max_steps: Maximum reasoning steps.
|
|
156
|
+
tool_map: Optional mapping of action names to callables for executing actions.
|
|
157
|
+
observation_limit: Max chars per observation.
|
|
158
|
+
|
|
159
|
+
Returns:
|
|
160
|
+
List of ReactStep records.
|
|
161
|
+
"""
|
|
162
|
+
loop = ReactLoop(max_steps=max_steps, observation_limit=observation_limit)
|
|
163
|
+
kwargs: dict[str, Any] = {
|
|
164
|
+
"model": model,
|
|
165
|
+
"messages": loop.build_context(task, system),
|
|
166
|
+
"stream": False,
|
|
167
|
+
}
|
|
168
|
+
if tools:
|
|
169
|
+
kwargs["tools"] = tools
|
|
170
|
+
|
|
171
|
+
for _ in range(max_steps):
|
|
172
|
+
try:
|
|
173
|
+
response = client.chat(**kwargs)
|
|
174
|
+
except Exception as exc:
|
|
175
|
+
loop.add_step(ReactStep(
|
|
176
|
+
thought="(LLM call failed)",
|
|
177
|
+
action="error",
|
|
178
|
+
action_input={},
|
|
179
|
+
observation=str(exc),
|
|
180
|
+
))
|
|
181
|
+
break
|
|
182
|
+
|
|
183
|
+
content = get_attr(get_attr(response, "message", {}), "content", "")
|
|
184
|
+
tool_calls = get_attr(get_attr(response, "message", {}), "tool_calls", None)
|
|
185
|
+
|
|
186
|
+
# If the model used structured tool calls, execute them
|
|
187
|
+
if tool_calls:
|
|
188
|
+
from ..tool_runtime import normalize_tool_call
|
|
189
|
+
for call in tool_calls:
|
|
190
|
+
name, args = normalize_tool_call(call)
|
|
191
|
+
thought = f"(model called tool: {name})"
|
|
192
|
+
obs = ""
|
|
193
|
+
if tool_map and name in tool_map:
|
|
194
|
+
try:
|
|
195
|
+
obs = str(tool_map[name](**args))
|
|
196
|
+
except Exception as exc:
|
|
197
|
+
obs = f"Tool error: {exc}"
|
|
198
|
+
step = ReactStep(thought=thought, action=name, action_input=args, observation=loop.truncate_observation(obs))
|
|
199
|
+
loop.add_step(step)
|
|
200
|
+
# Rebuild context and continue
|
|
201
|
+
kwargs["messages"] = loop.build_context(task, system)
|
|
202
|
+
continue
|
|
203
|
+
|
|
204
|
+
# Parse ReAct-formatted text
|
|
205
|
+
thought, action, action_input = parse_react_output(content)
|
|
206
|
+
|
|
207
|
+
if not action or action.lower() == "finish":
|
|
208
|
+
# Task complete or final answer
|
|
209
|
+
loop.add_step(ReactStep(
|
|
210
|
+
thought=thought,
|
|
211
|
+
action="finish",
|
|
212
|
+
action_input=action_input if isinstance(action_input, str) else json.dumps(action_input),
|
|
213
|
+
observation="Task complete.",
|
|
214
|
+
))
|
|
215
|
+
break
|
|
216
|
+
|
|
217
|
+
# Execute action if tool_map provided
|
|
218
|
+
obs = ""
|
|
219
|
+
if tool_map:
|
|
220
|
+
fn = tool_map.get(action)
|
|
221
|
+
if fn:
|
|
222
|
+
try:
|
|
223
|
+
args = action_input if isinstance(action_input, dict) else {"query": str(action_input)}
|
|
224
|
+
obs = str(fn(**args))
|
|
225
|
+
except Exception as exc:
|
|
226
|
+
obs = f"Tool error: {exc}"
|
|
227
|
+
else:
|
|
228
|
+
obs = f"Unknown action: {action}"
|
|
229
|
+
|
|
230
|
+
step = ReactStep(
|
|
231
|
+
thought=thought,
|
|
232
|
+
action=action,
|
|
233
|
+
action_input=action_input,
|
|
234
|
+
observation=loop.truncate_observation(obs),
|
|
235
|
+
)
|
|
236
|
+
loop.add_step(step)
|
|
237
|
+
|
|
238
|
+
# Loop detection
|
|
239
|
+
if loop.detect_loop():
|
|
240
|
+
step.observation += "\n[Loop detected: same action repeated. Try a different approach.]"
|
|
241
|
+
break
|
|
242
|
+
|
|
243
|
+
# Rebuild context
|
|
244
|
+
kwargs["messages"] = loop.build_context(task, system)
|
|
245
|
+
|
|
246
|
+
return loop.steps
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""Reflexion+ Verbal Self-Critique with Episodic Memory.
|
|
2
|
+
|
|
3
|
+
Extends the base reflex module with:
|
|
4
|
+
- Verbal self-evaluation after each attempt
|
|
5
|
+
- Episodic memory of past critiques for cross-attempt learning
|
|
6
|
+
- Retry with critique-guided modifications
|
|
7
|
+
- Convergence detection (when critiques stop surfacing new issues)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import time
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from ..chat_protocol import get_attr
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class ReflexionEpisode:
|
|
22
|
+
"""One attempt in a Reflexion loop."""
|
|
23
|
+
attempt: int
|
|
24
|
+
task: str
|
|
25
|
+
output: str
|
|
26
|
+
critique: str
|
|
27
|
+
score: float # 0.0-1.0 self-assessment
|
|
28
|
+
improved: bool # did this attempt improve over the previous?
|
|
29
|
+
timestamp: float = field(default_factory=time.time)
|
|
30
|
+
|
|
31
|
+
def critique_message(self) -> dict[str, Any]:
|
|
32
|
+
return {
|
|
33
|
+
"role": "user",
|
|
34
|
+
"content": (
|
|
35
|
+
f"## Self-Critique for Attempt {self.attempt}\n"
|
|
36
|
+
f"Score: {self.score:.2f}/1.0\n"
|
|
37
|
+
f"{self.critique}\n\n"
|
|
38
|
+
"Revise your approach to address these issues. Do not repeat the same mistakes."
|
|
39
|
+
),
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
CRITIQUE_PROMPT = """You are evaluating your own work. Be specific and constructive.
|
|
44
|
+
|
|
45
|
+
Rate your output on a scale of 0.0 to 1.0:
|
|
46
|
+
- 1.0: Fully correct, complete, no issues
|
|
47
|
+
- 0.7: Mostly correct with minor gaps
|
|
48
|
+
- 0.4: Partially correct with significant issues
|
|
49
|
+
- 0.0: Fundamentally wrong or missing
|
|
50
|
+
|
|
51
|
+
For each issue found:
|
|
52
|
+
1. State the specific problem
|
|
53
|
+
2. Explain why it is a problem
|
|
54
|
+
3. Suggest a concrete fix
|
|
55
|
+
|
|
56
|
+
Format your response as JSON:
|
|
57
|
+
{"score": <float>, "critique": "<specific issues and fixes>"}
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
CONVERGENCE_THRESHOLD = 0.05 # Score improvement below this = converged
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class ReflexionLoop:
|
|
65
|
+
"""Stateful Reflexion loop for agent harness integration."""
|
|
66
|
+
max_attempts: int = 3
|
|
67
|
+
convergence_threshold: float = CONVERGENCE_THRESHOLD
|
|
68
|
+
|
|
69
|
+
episodes: list[ReflexionEpisode] = field(default_factory=list)
|
|
70
|
+
best_output: str = ""
|
|
71
|
+
best_score: float = 0.0
|
|
72
|
+
|
|
73
|
+
def is_converged(self) -> bool:
|
|
74
|
+
"""Check if recent episodes show diminishing returns."""
|
|
75
|
+
if len(self.episodes) < 2:
|
|
76
|
+
return False
|
|
77
|
+
recent = self.episodes[-2:]
|
|
78
|
+
improvement = abs(recent[1].score - recent[0].score)
|
|
79
|
+
return improvement < self.convergence_threshold and recent[1].score >= recent[0].score
|
|
80
|
+
|
|
81
|
+
def add_episode(self, episode: ReflexionEpisode) -> None:
|
|
82
|
+
"""Record an episode and update the best-so-far result.
|
|
83
|
+
|
|
84
|
+
``improved`` describes progress over the immediately preceding attempt,
|
|
85
|
+
not whether the episode established a new all-time best. Keeping those
|
|
86
|
+
concepts separate matters after a regression followed by a partial
|
|
87
|
+
recovery (for example, scores of 0.8, 0.5, then 0.7).
|
|
88
|
+
"""
|
|
89
|
+
previous = self.episodes[-1] if self.episodes else None
|
|
90
|
+
episode.improved = previous is not None and episode.score > previous.score
|
|
91
|
+
self.episodes.append(episode)
|
|
92
|
+
if len(self.episodes) == 1 or episode.score > self.best_score:
|
|
93
|
+
self.best_score = episode.score
|
|
94
|
+
self.best_output = episode.output
|
|
95
|
+
|
|
96
|
+
def build_memory_context(self) -> str:
|
|
97
|
+
"""Build the episodic memory injection for the next attempt."""
|
|
98
|
+
if not self.episodes:
|
|
99
|
+
return ""
|
|
100
|
+
lines = ["## Previous Attempts and Self-Critiques"]
|
|
101
|
+
for ep in self.episodes:
|
|
102
|
+
lines.append(f"### Attempt {ep.attempt} (score: {ep.score:.2f})")
|
|
103
|
+
lines.append(f"Critique: {ep.critique[:500]}")
|
|
104
|
+
lines.append("\nLearn from these critiques. Avoid repeating the same errors.")
|
|
105
|
+
return "\n".join(lines)
|
|
106
|
+
|
|
107
|
+
def build_messages(self, task: str, system: str) -> list[dict[str, Any]]:
|
|
108
|
+
"""Build messages for the next attempt, including episodic memory."""
|
|
109
|
+
messages: list[dict[str, Any]] = [
|
|
110
|
+
{"role": "system", "content": system},
|
|
111
|
+
]
|
|
112
|
+
memory = self.build_memory_context()
|
|
113
|
+
if memory:
|
|
114
|
+
messages.append({"role": "user", "content": memory})
|
|
115
|
+
messages.append({"role": "user", "content": task})
|
|
116
|
+
return messages
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _parse_critique(response_text: str) -> tuple[float, str]:
|
|
120
|
+
"""Parse a critique response into (score, critique_text)."""
|
|
121
|
+
text = response_text.strip()
|
|
122
|
+
# Try JSON parse
|
|
123
|
+
try:
|
|
124
|
+
data = json.loads(text)
|
|
125
|
+
score = float(data.get("score", 0.5))
|
|
126
|
+
critique = str(data.get("critique", ""))
|
|
127
|
+
return max(0.0, min(1.0, score)), critique
|
|
128
|
+
except (json.JSONDecodeError, ValueError, TypeError):
|
|
129
|
+
pass
|
|
130
|
+
# Try extracting score from text
|
|
131
|
+
score_m = None
|
|
132
|
+
for pattern in [r'"score"\s*:\s*([0-9.]+)', r'score:\s*([0-9.]+)', r'([0-9.]+)\s*/\s*1\.0']:
|
|
133
|
+
import re
|
|
134
|
+
m = re.search(pattern, text)
|
|
135
|
+
if m:
|
|
136
|
+
score_m = m
|
|
137
|
+
break
|
|
138
|
+
score = float(score_m.group(1)) if score_m else 0.5
|
|
139
|
+
return max(0.0, min(1.0, score)), text[:1000]
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def run_reflexion_loop(
|
|
143
|
+
*,
|
|
144
|
+
task: str,
|
|
145
|
+
client: Any,
|
|
146
|
+
model: str,
|
|
147
|
+
critique_model: str | None = None,
|
|
148
|
+
system: str = "You are a capable reasoning agent. Produce your best work on the given task.",
|
|
149
|
+
max_attempts: int = 3,
|
|
150
|
+
tools: list[Any] | None = None,
|
|
151
|
+
score_threshold: float = 0.8,
|
|
152
|
+
) -> list[ReflexionEpisode]:
|
|
153
|
+
"""Run a Reflexion+ loop: attempt -> self-critique -> retry.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
task: The task to solve.
|
|
157
|
+
client: Ollama client instance.
|
|
158
|
+
model: Model for task attempts.
|
|
159
|
+
critique_model: Optional separate model for self-critique (defaults to model).
|
|
160
|
+
system: System prompt.
|
|
161
|
+
max_attempts: Maximum reflexion attempts.
|
|
162
|
+
tools: Optional tools for the model.
|
|
163
|
+
score_threshold: Stop early if score exceeds this.
|
|
164
|
+
|
|
165
|
+
Returns:
|
|
166
|
+
List of ReflexionEpisode records.
|
|
167
|
+
"""
|
|
168
|
+
critique_model = critique_model or model
|
|
169
|
+
loop = ReflexionLoop(max_attempts=max_attempts)
|
|
170
|
+
|
|
171
|
+
for attempt in range(1, max_attempts + 1):
|
|
172
|
+
# 1. Attempt the task
|
|
173
|
+
messages = loop.build_messages(task, system)
|
|
174
|
+
kwargs: dict[str, Any] = {
|
|
175
|
+
"model": model,
|
|
176
|
+
"messages": messages,
|
|
177
|
+
"stream": False,
|
|
178
|
+
}
|
|
179
|
+
if tools:
|
|
180
|
+
kwargs["tools"] = tools
|
|
181
|
+
|
|
182
|
+
try:
|
|
183
|
+
response = client.chat(**kwargs)
|
|
184
|
+
output = get_attr(get_attr(response, "message", {}), "content", "")
|
|
185
|
+
except Exception as exc:
|
|
186
|
+
output = f"Error during attempt {attempt}: {exc}"
|
|
187
|
+
|
|
188
|
+
# 2. Self-critique
|
|
189
|
+
critique_messages = [
|
|
190
|
+
{"role": "system", "content": CRITIQUE_PROMPT},
|
|
191
|
+
{"role": "user", "content": f"## Task\n{task}\n\n## Your Output\n{output}\n\nEvaluate your output above."},
|
|
192
|
+
]
|
|
193
|
+
try:
|
|
194
|
+
critique_response = client.chat(
|
|
195
|
+
model=critique_model,
|
|
196
|
+
messages=critique_messages,
|
|
197
|
+
stream=False,
|
|
198
|
+
format="json",
|
|
199
|
+
)
|
|
200
|
+
critique_text = get_attr(get_attr(critique_response, "message", {}), "content", "")
|
|
201
|
+
score, critique = _parse_critique(critique_text)
|
|
202
|
+
except Exception:
|
|
203
|
+
score = 0.5
|
|
204
|
+
critique = "(critique generation failed)"
|
|
205
|
+
|
|
206
|
+
episode = ReflexionEpisode(
|
|
207
|
+
attempt=attempt,
|
|
208
|
+
task=task,
|
|
209
|
+
output=output,
|
|
210
|
+
critique=critique,
|
|
211
|
+
score=score,
|
|
212
|
+
# add_episode derives the value from the previous episode. Supply
|
|
213
|
+
# the neutral value here so construction cannot fail before that
|
|
214
|
+
# comparison is made.
|
|
215
|
+
improved=False,
|
|
216
|
+
)
|
|
217
|
+
loop.add_episode(episode)
|
|
218
|
+
|
|
219
|
+
# 3. Check termination
|
|
220
|
+
if score >= score_threshold:
|
|
221
|
+
break
|
|
222
|
+
if loop.is_converged() and attempt >= 2:
|
|
223
|
+
break
|
|
224
|
+
|
|
225
|
+
return loop.episodes
|