algo-cli-runtime 0.14.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. algo_cli/__init__.py +3 -0
  2. algo_cli/__main__.py +7 -0
  3. algo_cli/_internal/__init__.py +12 -0
  4. algo_cli/_internal/policy_chain.py +259 -0
  5. algo_cli/action_registry.py +1047 -0
  6. algo_cli/agent_blocks.py +550 -0
  7. algo_cli/agent_pipeline.py +1457 -0
  8. algo_cli/agent_threads.py +308 -0
  9. algo_cli/animations.py +316 -0
  10. algo_cli/cache_admission.py +209 -0
  11. algo_cli/capability_mask.py +66 -0
  12. algo_cli/chat_protocol.py +116 -0
  13. algo_cli/chatgpt_auth.py +510 -0
  14. algo_cli/chatgpt_client.py +657 -0
  15. algo_cli/code_rag.py +479 -0
  16. algo_cli/config.py +651 -0
  17. algo_cli/context_budget.py +679 -0
  18. algo_cli/credential_helpers.py +315 -0
  19. algo_cli/deliberation.py +29 -0
  20. algo_cli/display.py +1470 -0
  21. algo_cli/evals/__init__.py +21 -0
  22. algo_cli/evals/algorithm_effectiveness.py +560 -0
  23. algo_cli/evals/competitive_harness_rating.py +702 -0
  24. algo_cli/evals/cot_quality.py +220 -0
  25. algo_cli/evals/harness_retrieval_benchmark.py +401 -0
  26. algo_cli/evals/performance_regression.py +136 -0
  27. algo_cli/evals/scorecard_grading.py +308 -0
  28. algo_cli/evals/session_distribution.py +84 -0
  29. algo_cli/execution_guardrails.py +806 -0
  30. algo_cli/extensions_manifest.py +84 -0
  31. algo_cli/git_evidence.py +227 -0
  32. algo_cli/google_workspace.py +407 -0
  33. algo_cli/google_workspace_auth.py +523 -0
  34. algo_cli/harness.py +2587 -0
  35. algo_cli/identity.py +557 -0
  36. algo_cli/index_compute_lab.py +228 -0
  37. algo_cli/inference_harness.py +70 -0
  38. algo_cli/intelligence/__init__.py +1103 -0
  39. algo_cli/intelligence/acrobat_config.py +307 -0
  40. algo_cli/intelligence/acrobat_manifests.py +338 -0
  41. algo_cli/intelligence/acrobat_models.py +195 -0
  42. algo_cli/intelligence/acrobat_pipeline.py +295 -0
  43. algo_cli/intelligence/acrobat_runtime.py +302 -0
  44. algo_cli/intelligence/acrobat_security.py +261 -0
  45. algo_cli/intelligence/acrobat_workflows.py +226 -0
  46. algo_cli/intelligence/actionability.py +165 -0
  47. algo_cli/intelligence/adversarial_audit.py +136 -0
  48. algo_cli/intelligence/agent_arena.py +92 -0
  49. algo_cli/intelligence/agent_benchmark.py +236 -0
  50. algo_cli/intelligence/agent_runtime.py +171 -0
  51. algo_cli/intelligence/agents_as_tools.py +70 -0
  52. algo_cli/intelligence/artifact_binding.py +80 -0
  53. algo_cli/intelligence/autonomous_engineer.py +1976 -0
  54. algo_cli/intelligence/backpressure.py +99 -0
  55. algo_cli/intelligence/bloom_filter.py +186 -0
  56. algo_cli/intelligence/bonferroni.py +66 -0
  57. algo_cli/intelligence/boundary_compaction.py +98 -0
  58. algo_cli/intelligence/catalog_verifier.py +172 -0
  59. algo_cli/intelligence/cavecrew.py +118 -0
  60. algo_cli/intelligence/changelog.py +176 -0
  61. algo_cli/intelligence/checkpoint_resume.py +92 -0
  62. algo_cli/intelligence/circuit_breaker.py +88 -0
  63. algo_cli/intelligence/clarification_gate.py +101 -0
  64. algo_cli/intelligence/code_graph.py +180 -0
  65. algo_cli/intelligence/coderank.py +97 -0
  66. algo_cli/intelligence/consistent_hash.py +150 -0
  67. algo_cli/intelligence/consortium_synthesis.py +139 -0
  68. algo_cli/intelligence/construction/__init__.py +241 -0
  69. algo_cli/intelligence/construction/common.py +273 -0
  70. algo_cli/intelligence/construction/documents.py +496 -0
  71. algo_cli/intelligence/construction/labor_units.py +1395 -0
  72. algo_cli/intelligence/construction/payments.py +470 -0
  73. algo_cli/intelligence/construction/risk.py +784 -0
  74. algo_cli/intelligence/content_extractor.py +132 -0
  75. algo_cli/intelligence/context_adaptive.py +102 -0
  76. algo_cli/intelligence/context_ops.py +95 -0
  77. algo_cli/intelligence/count_min.py +145 -0
  78. algo_cli/intelligence/cow_state.py +103 -0
  79. algo_cli/intelligence/critic_loop.py +119 -0
  80. algo_cli/intelligence/cross_source.py +113 -0
  81. algo_cli/intelligence/daemon_mode.py +99 -0
  82. algo_cli/intelligence/dag_orchestration.py +151 -0
  83. algo_cli/intelligence/deep_research.py +155 -0
  84. algo_cli/intelligence/degenerate_detector.py +78 -0
  85. algo_cli/intelligence/delta_report.py +92 -0
  86. algo_cli/intelligence/discovery_event_log.py +92 -0
  87. algo_cli/intelligence/document_ingest.py +298 -0
  88. algo_cli/intelligence/dual_layer_validate.py +151 -0
  89. algo_cli/intelligence/echo_fidelity.py +73 -0
  90. algo_cli/intelligence/ema_tuning.py +104 -0
  91. algo_cli/intelligence/event_log.py +92 -0
  92. algo_cli/intelligence/evidence_graph.py +114 -0
  93. algo_cli/intelligence/extension_host.py +162 -0
  94. algo_cli/intelligence/extension_manifest.py +115 -0
  95. algo_cli/intelligence/falsification_suite.py +178 -0
  96. algo_cli/intelligence/finance/__init__.py +169 -0
  97. algo_cli/intelligence/finance/anomalies.py +135 -0
  98. algo_cli/intelligence/finance/ap_ar.py +351 -0
  99. algo_cli/intelligence/finance/cash.py +162 -0
  100. algo_cli/intelligence/finance/close.py +332 -0
  101. algo_cli/intelligence/finance/common.py +244 -0
  102. algo_cli/intelligence/finance/construction.py +135 -0
  103. algo_cli/intelligence/finance/controls.py +172 -0
  104. algo_cli/intelligence/finance/evidence.py +119 -0
  105. algo_cli/intelligence/finance/exceptions.py +157 -0
  106. algo_cli/intelligence/finance/reconciliations.py +254 -0
  107. algo_cli/intelligence/finance/revenue.py +109 -0
  108. algo_cli/intelligence/finance/tax.py +74 -0
  109. algo_cli/intelligence/finance/workpapers.py +111 -0
  110. algo_cli/intelligence/finding_record.py +120 -0
  111. algo_cli/intelligence/flow_dag.py +267 -0
  112. algo_cli/intelligence/gatherer.py +223 -0
  113. algo_cli/intelligence/golden_master.py +98 -0
  114. algo_cli/intelligence/graph_rag.py +195 -0
  115. algo_cli/intelligence/group_chat.py +143 -0
  116. algo_cli/intelligence/hash_dedup.py +145 -0
  117. algo_cli/intelligence/hyperloglog.py +128 -0
  118. algo_cli/intelligence/incremental_index.py +316 -0
  119. algo_cli/intelligence/index_store.py +16 -0
  120. algo_cli/intelligence/iteration_plan.py +133 -0
  121. algo_cli/intelligence/kernel_plugins.py +167 -0
  122. algo_cli/intelligence/lesson_catalog.py +135 -0
  123. algo_cli/intelligence/llm_fallback.py +169 -0
  124. algo_cli/intelligence/log2_histogram.py +267 -0
  125. algo_cli/intelligence/lsp_integration.py +147 -0
  126. algo_cli/intelligence/memory_evolution.py +117 -0
  127. algo_cli/intelligence/minhash_lsh.py +182 -0
  128. algo_cli/intelligence/multi_model_score.py +174 -0
  129. algo_cli/intelligence/multi_tier_grade.py +211 -0
  130. algo_cli/intelligence/negative_controls.py +113 -0
  131. algo_cli/intelligence/numeric_clamp.py +63 -0
  132. algo_cli/intelligence/occ_editor.py +66 -0
  133. algo_cli/intelligence/output_normalize.py +112 -0
  134. algo_cli/intelligence/parallel_delegation.py +98 -0
  135. algo_cli/intelligence/parallel_fanout.py +104 -0
  136. algo_cli/intelligence/permission_modes.py +105 -0
  137. algo_cli/intelligence/pre_push_gate.py +68 -0
  138. algo_cli/intelligence/prefetch.py +171 -0
  139. algo_cli/intelligence/process_framework.py +217 -0
  140. algo_cli/intelligence/project_graph.py +387 -0
  141. algo_cli/intelligence/query_expansion.py +146 -0
  142. algo_cli/intelligence/ralph_loop.py +117 -0
  143. algo_cli/intelligence/rate_limiter.py +153 -0
  144. algo_cli/intelligence/refactor_transaction.py +94 -0
  145. algo_cli/intelligence/research_workspace.py +108 -0
  146. algo_cli/intelligence/retraction_ledger.py +72 -0
  147. algo_cli/intelligence/saga_pattern.py +88 -0
  148. algo_cli/intelligence/session_fork.py +100 -0
  149. algo_cli/intelligence/shadow_editor.py +67 -0
  150. algo_cli/intelligence/shell_session.py +213 -0
  151. algo_cli/intelligence/source_registry.py +143 -0
  152. algo_cli/intelligence/spawn_scales.py +99 -0
  153. algo_cli/intelligence/stat_stability.py +104 -0
  154. algo_cli/intelligence/structural_validator.py +148 -0
  155. algo_cli/intelligence/subagent_spawner.py +111 -0
  156. algo_cli/intelligence/symmetric_verify.py +70 -0
  157. algo_cli/intelligence/task_classifier.py +129 -0
  158. algo_cli/intelligence/team_execution.py +122 -0
  159. algo_cli/intelligence/tiered_access.py +121 -0
  160. algo_cli/intelligence/utility_registry.py +159 -0
  161. algo_cli/intuition_engine.py +560 -0
  162. algo_cli/intuition_injector.py +82 -0
  163. algo_cli/kernels/__init__.py +5 -0
  164. algo_cli/kernels/manifest.py +763 -0
  165. algo_cli/main.py +3903 -0
  166. algo_cli/memory_candidates.py +541 -0
  167. algo_cli/memory_echo_veil.py +394 -0
  168. algo_cli/memory_runtime.py +112 -0
  169. algo_cli/model_info.py +548 -0
  170. algo_cli/model_profile.py +160 -0
  171. algo_cli/model_routing.py +74 -0
  172. algo_cli/oneshot.py +331 -0
  173. algo_cli/perf_telemetry.py +389 -0
  174. algo_cli/plugins.py +245 -0
  175. algo_cli/private_event_store.py +654 -0
  176. algo_cli/quantization/__init__.py +24 -0
  177. algo_cli/quantization/lloyd_max.py +98 -0
  178. algo_cli/quantization/turbo_quant.py +308 -0
  179. algo_cli/reasoning/__init__.py +46 -0
  180. algo_cli/reasoning/combinatorial.py +356 -0
  181. algo_cli/reasoning/graph_of_thought.py +297 -0
  182. algo_cli/reasoning/mcts.py +220 -0
  183. algo_cli/reasoning/neuro_symbolic.py +250 -0
  184. algo_cli/reasoning/react.py +246 -0
  185. algo_cli/reasoning/reflexion.py +225 -0
  186. algo_cli/reasoning/tree_of_thought.py +241 -0
  187. algo_cli/reasoning_bridge.py +150 -0
  188. algo_cli/reconciliation.py +284 -0
  189. algo_cli/reflex.py +385 -0
  190. algo_cli/resources/docs/ALGO.md +13958 -0
  191. algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
  192. algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
  193. algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
  194. algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
  195. algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
  196. algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
  197. algo_cli/resources/docs/main-split-map.md +35 -0
  198. algo_cli/resources/docs/privacy-and-context.md +48 -0
  199. algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
  200. algo_cli/resources/skills/README.md +26 -0
  201. algo_cli/resources/skills/algo-cli.md +59 -0
  202. algo_cli/resources/skills/edit-file-precision.md +49 -0
  203. algo_cli/resources/skills/harness-search-first.md +47 -0
  204. algo_cli/resources/skills/memory-recall-ritual.md +51 -0
  205. algo_cli/resources/skills/qol-algorithms.md +224 -0
  206. algo_cli/resources/skills/smart-error-recovery.md +56 -0
  207. algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
  208. algo_cli/retrieval_algorithms.py +127 -0
  209. algo_cli/runtime_qos.py +236 -0
  210. algo_cli/runtime_services.py +320 -0
  211. algo_cli/session_commands.py +95 -0
  212. algo_cli/session_mode.py +113 -0
  213. algo_cli/skills.py +430 -0
  214. algo_cli/slash_dispatch.py +1265 -0
  215. algo_cli/small_context.py +206 -0
  216. algo_cli/spawn_budget.py +89 -0
  217. algo_cli/task_ledger.py +84 -0
  218. algo_cli/task_router.py +197 -0
  219. algo_cli/tool_context.py +94 -0
  220. algo_cli/tool_contract.py +99 -0
  221. algo_cli/tool_policy.py +357 -0
  222. algo_cli/tool_runtime.py +647 -0
  223. algo_cli/tools.py +3056 -0
  224. algo_cli/url_scheme.py +174 -0
  225. algo_cli/verify.py +154 -0
  226. algo_cli/version_manifest.py +178 -0
  227. algo_cli/vision_screenshot_verify.py +76 -0
  228. algo_cli/workspace_resolver.py +68 -0
  229. algo_cli/x_account.py +209 -0
  230. algo_cli/xai_auth.py +374 -0
  231. algo_cli/xai_client.py +600 -0
  232. algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
  233. algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
  234. algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
  235. algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
  236. algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
  237. ollama_cli/__init__.py +67 -0
@@ -0,0 +1,246 @@
1
+ """ReAct+ Enhanced Reasoning-Action Loop.
2
+
3
+ Interleaves structured Thought, Action, and Observation steps with:
4
+ - Typed action parsing (tool calls, sub-questions, assertions)
5
+ - Observation summarization to prevent context bloat
6
+ - Automatic thought-chain compaction for long episodes
7
+ - Loop detection with strategy broadening
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import re
14
+ import time
15
+ from dataclasses import dataclass, field
16
+ from typing import Any, Callable
17
+
18
+ from ..chat_protocol import get_attr
19
+
20
+
21
+ @dataclass
22
+ class ReactStep:
23
+ """One step in a ReAct episode."""
24
+ thought: str
25
+ action: str
26
+ action_input: dict[str, Any] | str
27
+ observation: str
28
+ timestamp: float = field(default_factory=time.time)
29
+
30
+ def to_message(self) -> dict[str, Any]:
31
+ return {
32
+ "role": "assistant",
33
+ "content": f"Thought: {self.thought}\nAction: {self.action}\nAction Input: {self._format_input()}",
34
+ }
35
+
36
+ def observation_message(self) -> dict[str, Any]:
37
+ return {
38
+ "role": "user",
39
+ "content": f"Observation: {self.observation}",
40
+ }
41
+
42
+ def _format_input(self) -> str:
43
+ if isinstance(self.action_input, dict):
44
+ return json.dumps(self.action_input)
45
+ return str(self.action_input)
46
+
47
+
48
+ THOUGHT_RE = re.compile(r"Thought:\s*(.+?)(?=\nAction:|$)", re.DOTALL)
49
+ ACTION_RE = re.compile(r"Action:\s*(.+?)(?=\nAction Input:|$)", re.DOTALL)
50
+ ACTION_INPUT_RE = re.compile(r"Action Input:\s*(.+?)$", re.DOTALL)
51
+
52
+
53
+ def parse_react_output(text: str) -> tuple[str, str, dict[str, Any] | str]:
54
+ """Parse a ReAct-formatted response into (thought, action, action_input)."""
55
+ thought_m = THOUGHT_RE.search(text)
56
+ action_m = ACTION_RE.search(text)
57
+ input_m = ACTION_INPUT_RE.search(text)
58
+
59
+ thought = thought_m.group(1).strip() if thought_m else ""
60
+ action = action_m.group(1).strip() if action_m else ""
61
+ raw_input = input_m.group(1).strip() if input_m else ""
62
+
63
+ # Try parsing action input as JSON; fall back to string
64
+ action_input: dict[str, Any] | str = raw_input
65
+ if raw_input:
66
+ try:
67
+ parsed = json.loads(raw_input)
68
+ if isinstance(parsed, dict):
69
+ action_input = parsed
70
+ except (json.JSONDecodeError, ValueError):
71
+ pass
72
+
73
+ return thought, action, action_input
74
+
75
+
76
+ def compact_observations(steps: list[ReactStep], max_chars: int = 4000) -> str:
77
+ """Compact a ReAct episode into a summary for context injection."""
78
+ if not steps:
79
+ return ""
80
+ lines: list[str] = []
81
+ total = 0
82
+ for i, step in enumerate(steps):
83
+ entry = f"[{i+1}] Thought: {step.thought[:200]}\n Action: {step.action} -> Obs: {step.observation[:300]}"
84
+ if total + len(entry) > max_chars:
85
+ remaining = len(steps) - i
86
+ lines.append(f"... ({remaining} earlier steps compacted)")
87
+ break
88
+ lines.append(entry)
89
+ total += len(entry)
90
+ return "\n".join(lines)
91
+
92
+
93
+ @dataclass
94
+ class ReactLoop:
95
+ """Stateful ReAct+ loop for agent harness integration."""
96
+ max_steps: int = 10
97
+ observation_limit: int = 2000
98
+ loop_detection_window: int = 3
99
+
100
+ steps: list[ReactStep] = field(default_factory=list)
101
+ _action_history: list[str] = field(default_factory=list)
102
+
103
+ def detect_loop(self) -> bool:
104
+ """Detect if the last N actions are identical (stuck loop)."""
105
+ if len(self._action_history) < self.loop_detection_window:
106
+ return False
107
+ window = self._action_history[-self.loop_detection_window:]
108
+ return len(set(window)) == 1
109
+
110
+ def add_step(self, step: ReactStep) -> None:
111
+ self.steps.append(step)
112
+ self._action_history.append(f"{step.action}:{str(step.action_input)[:80]}")
113
+
114
+ def truncate_observation(self, obs: str) -> str:
115
+ if len(obs) > self.observation_limit:
116
+ return obs[:self.observation_limit - 20] + "\n...[truncated]"
117
+ return obs
118
+
119
+ def build_context(self, task: str, system: str) -> list[dict[str, Any]]:
120
+ """Build the message list for the next LLM call."""
121
+ messages: list[dict[str, Any]] = [
122
+ {"role": "system", "content": system},
123
+ {"role": "user", "content": task},
124
+ ]
125
+ for step in self.steps:
126
+ messages.append(step.to_message())
127
+ messages.append(step.observation_message())
128
+ # Final prompt to continue reasoning
129
+ messages.append({
130
+ "role": "user",
131
+ "content": "Continue with your next Thought and Action. If you have enough information to answer, respond with just the final answer.",
132
+ })
133
+ return messages
134
+
135
+
136
+ def run_react_loop(
137
+ *,
138
+ task: str,
139
+ client: Any,
140
+ model: str,
141
+ tools: list[Any] | None = None,
142
+ system: str = "You are a reasoning agent. Use the ReAct format:\nThought: <your reasoning>\nAction: <tool name or 'finish'>\nAction Input: <JSON args or final answer>",
143
+ max_steps: int = 10,
144
+ tool_map: dict[str, Callable] | None = None,
145
+ observation_limit: int = 2000,
146
+ ) -> list[ReactStep]:
147
+ """Run a complete ReAct+ episode.
148
+
149
+ Args:
150
+ task: The task to solve.
151
+ client: Ollama client instance.
152
+ model: Model name.
153
+ tools: Optional list of tool functions for the model.
154
+ system: System prompt (ReAct format instructions).
155
+ max_steps: Maximum reasoning steps.
156
+ tool_map: Optional mapping of action names to callables for executing actions.
157
+ observation_limit: Max chars per observation.
158
+
159
+ Returns:
160
+ List of ReactStep records.
161
+ """
162
+ loop = ReactLoop(max_steps=max_steps, observation_limit=observation_limit)
163
+ kwargs: dict[str, Any] = {
164
+ "model": model,
165
+ "messages": loop.build_context(task, system),
166
+ "stream": False,
167
+ }
168
+ if tools:
169
+ kwargs["tools"] = tools
170
+
171
+ for _ in range(max_steps):
172
+ try:
173
+ response = client.chat(**kwargs)
174
+ except Exception as exc:
175
+ loop.add_step(ReactStep(
176
+ thought="(LLM call failed)",
177
+ action="error",
178
+ action_input={},
179
+ observation=str(exc),
180
+ ))
181
+ break
182
+
183
+ content = get_attr(get_attr(response, "message", {}), "content", "")
184
+ tool_calls = get_attr(get_attr(response, "message", {}), "tool_calls", None)
185
+
186
+ # If the model used structured tool calls, execute them
187
+ if tool_calls:
188
+ from ..tool_runtime import normalize_tool_call
189
+ for call in tool_calls:
190
+ name, args = normalize_tool_call(call)
191
+ thought = f"(model called tool: {name})"
192
+ obs = ""
193
+ if tool_map and name in tool_map:
194
+ try:
195
+ obs = str(tool_map[name](**args))
196
+ except Exception as exc:
197
+ obs = f"Tool error: {exc}"
198
+ step = ReactStep(thought=thought, action=name, action_input=args, observation=loop.truncate_observation(obs))
199
+ loop.add_step(step)
200
+ # Rebuild context and continue
201
+ kwargs["messages"] = loop.build_context(task, system)
202
+ continue
203
+
204
+ # Parse ReAct-formatted text
205
+ thought, action, action_input = parse_react_output(content)
206
+
207
+ if not action or action.lower() == "finish":
208
+ # Task complete or final answer
209
+ loop.add_step(ReactStep(
210
+ thought=thought,
211
+ action="finish",
212
+ action_input=action_input if isinstance(action_input, str) else json.dumps(action_input),
213
+ observation="Task complete.",
214
+ ))
215
+ break
216
+
217
+ # Execute action if tool_map provided
218
+ obs = ""
219
+ if tool_map:
220
+ fn = tool_map.get(action)
221
+ if fn:
222
+ try:
223
+ args = action_input if isinstance(action_input, dict) else {"query": str(action_input)}
224
+ obs = str(fn(**args))
225
+ except Exception as exc:
226
+ obs = f"Tool error: {exc}"
227
+ else:
228
+ obs = f"Unknown action: {action}"
229
+
230
+ step = ReactStep(
231
+ thought=thought,
232
+ action=action,
233
+ action_input=action_input,
234
+ observation=loop.truncate_observation(obs),
235
+ )
236
+ loop.add_step(step)
237
+
238
+ # Loop detection
239
+ if loop.detect_loop():
240
+ step.observation += "\n[Loop detected: same action repeated. Try a different approach.]"
241
+ break
242
+
243
+ # Rebuild context
244
+ kwargs["messages"] = loop.build_context(task, system)
245
+
246
+ return loop.steps
@@ -0,0 +1,225 @@
1
+ """Reflexion+ Verbal Self-Critique with Episodic Memory.
2
+
3
+ Extends the base reflex module with:
4
+ - Verbal self-evaluation after each attempt
5
+ - Episodic memory of past critiques for cross-attempt learning
6
+ - Retry with critique-guided modifications
7
+ - Convergence detection (when critiques stop surfacing new issues)
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import time
14
+ from dataclasses import dataclass, field
15
+ from typing import Any
16
+
17
+ from ..chat_protocol import get_attr
18
+
19
+
20
+ @dataclass
21
+ class ReflexionEpisode:
22
+ """One attempt in a Reflexion loop."""
23
+ attempt: int
24
+ task: str
25
+ output: str
26
+ critique: str
27
+ score: float # 0.0-1.0 self-assessment
28
+ improved: bool # did this attempt improve over the previous?
29
+ timestamp: float = field(default_factory=time.time)
30
+
31
+ def critique_message(self) -> dict[str, Any]:
32
+ return {
33
+ "role": "user",
34
+ "content": (
35
+ f"## Self-Critique for Attempt {self.attempt}\n"
36
+ f"Score: {self.score:.2f}/1.0\n"
37
+ f"{self.critique}\n\n"
38
+ "Revise your approach to address these issues. Do not repeat the same mistakes."
39
+ ),
40
+ }
41
+
42
+
43
+ CRITIQUE_PROMPT = """You are evaluating your own work. Be specific and constructive.
44
+
45
+ Rate your output on a scale of 0.0 to 1.0:
46
+ - 1.0: Fully correct, complete, no issues
47
+ - 0.7: Mostly correct with minor gaps
48
+ - 0.4: Partially correct with significant issues
49
+ - 0.0: Fundamentally wrong or missing
50
+
51
+ For each issue found:
52
+ 1. State the specific problem
53
+ 2. Explain why it is a problem
54
+ 3. Suggest a concrete fix
55
+
56
+ Format your response as JSON:
57
+ {"score": <float>, "critique": "<specific issues and fixes>"}
58
+ """
59
+
60
+ CONVERGENCE_THRESHOLD = 0.05 # Score improvement below this = converged
61
+
62
+
63
+ @dataclass
64
+ class ReflexionLoop:
65
+ """Stateful Reflexion loop for agent harness integration."""
66
+ max_attempts: int = 3
67
+ convergence_threshold: float = CONVERGENCE_THRESHOLD
68
+
69
+ episodes: list[ReflexionEpisode] = field(default_factory=list)
70
+ best_output: str = ""
71
+ best_score: float = 0.0
72
+
73
+ def is_converged(self) -> bool:
74
+ """Check if recent episodes show diminishing returns."""
75
+ if len(self.episodes) < 2:
76
+ return False
77
+ recent = self.episodes[-2:]
78
+ improvement = abs(recent[1].score - recent[0].score)
79
+ return improvement < self.convergence_threshold and recent[1].score >= recent[0].score
80
+
81
+ def add_episode(self, episode: ReflexionEpisode) -> None:
82
+ """Record an episode and update the best-so-far result.
83
+
84
+ ``improved`` describes progress over the immediately preceding attempt,
85
+ not whether the episode established a new all-time best. Keeping those
86
+ concepts separate matters after a regression followed by a partial
87
+ recovery (for example, scores of 0.8, 0.5, then 0.7).
88
+ """
89
+ previous = self.episodes[-1] if self.episodes else None
90
+ episode.improved = previous is not None and episode.score > previous.score
91
+ self.episodes.append(episode)
92
+ if len(self.episodes) == 1 or episode.score > self.best_score:
93
+ self.best_score = episode.score
94
+ self.best_output = episode.output
95
+
96
+ def build_memory_context(self) -> str:
97
+ """Build the episodic memory injection for the next attempt."""
98
+ if not self.episodes:
99
+ return ""
100
+ lines = ["## Previous Attempts and Self-Critiques"]
101
+ for ep in self.episodes:
102
+ lines.append(f"### Attempt {ep.attempt} (score: {ep.score:.2f})")
103
+ lines.append(f"Critique: {ep.critique[:500]}")
104
+ lines.append("\nLearn from these critiques. Avoid repeating the same errors.")
105
+ return "\n".join(lines)
106
+
107
+ def build_messages(self, task: str, system: str) -> list[dict[str, Any]]:
108
+ """Build messages for the next attempt, including episodic memory."""
109
+ messages: list[dict[str, Any]] = [
110
+ {"role": "system", "content": system},
111
+ ]
112
+ memory = self.build_memory_context()
113
+ if memory:
114
+ messages.append({"role": "user", "content": memory})
115
+ messages.append({"role": "user", "content": task})
116
+ return messages
117
+
118
+
119
+ def _parse_critique(response_text: str) -> tuple[float, str]:
120
+ """Parse a critique response into (score, critique_text)."""
121
+ text = response_text.strip()
122
+ # Try JSON parse
123
+ try:
124
+ data = json.loads(text)
125
+ score = float(data.get("score", 0.5))
126
+ critique = str(data.get("critique", ""))
127
+ return max(0.0, min(1.0, score)), critique
128
+ except (json.JSONDecodeError, ValueError, TypeError):
129
+ pass
130
+ # Try extracting score from text
131
+ score_m = None
132
+ for pattern in [r'"score"\s*:\s*([0-9.]+)', r'score:\s*([0-9.]+)', r'([0-9.]+)\s*/\s*1\.0']:
133
+ import re
134
+ m = re.search(pattern, text)
135
+ if m:
136
+ score_m = m
137
+ break
138
+ score = float(score_m.group(1)) if score_m else 0.5
139
+ return max(0.0, min(1.0, score)), text[:1000]
140
+
141
+
142
+ def run_reflexion_loop(
143
+ *,
144
+ task: str,
145
+ client: Any,
146
+ model: str,
147
+ critique_model: str | None = None,
148
+ system: str = "You are a capable reasoning agent. Produce your best work on the given task.",
149
+ max_attempts: int = 3,
150
+ tools: list[Any] | None = None,
151
+ score_threshold: float = 0.8,
152
+ ) -> list[ReflexionEpisode]:
153
+ """Run a Reflexion+ loop: attempt -> self-critique -> retry.
154
+
155
+ Args:
156
+ task: The task to solve.
157
+ client: Ollama client instance.
158
+ model: Model for task attempts.
159
+ critique_model: Optional separate model for self-critique (defaults to model).
160
+ system: System prompt.
161
+ max_attempts: Maximum reflexion attempts.
162
+ tools: Optional tools for the model.
163
+ score_threshold: Stop early if score exceeds this.
164
+
165
+ Returns:
166
+ List of ReflexionEpisode records.
167
+ """
168
+ critique_model = critique_model or model
169
+ loop = ReflexionLoop(max_attempts=max_attempts)
170
+
171
+ for attempt in range(1, max_attempts + 1):
172
+ # 1. Attempt the task
173
+ messages = loop.build_messages(task, system)
174
+ kwargs: dict[str, Any] = {
175
+ "model": model,
176
+ "messages": messages,
177
+ "stream": False,
178
+ }
179
+ if tools:
180
+ kwargs["tools"] = tools
181
+
182
+ try:
183
+ response = client.chat(**kwargs)
184
+ output = get_attr(get_attr(response, "message", {}), "content", "")
185
+ except Exception as exc:
186
+ output = f"Error during attempt {attempt}: {exc}"
187
+
188
+ # 2. Self-critique
189
+ critique_messages = [
190
+ {"role": "system", "content": CRITIQUE_PROMPT},
191
+ {"role": "user", "content": f"## Task\n{task}\n\n## Your Output\n{output}\n\nEvaluate your output above."},
192
+ ]
193
+ try:
194
+ critique_response = client.chat(
195
+ model=critique_model,
196
+ messages=critique_messages,
197
+ stream=False,
198
+ format="json",
199
+ )
200
+ critique_text = get_attr(get_attr(critique_response, "message", {}), "content", "")
201
+ score, critique = _parse_critique(critique_text)
202
+ except Exception:
203
+ score = 0.5
204
+ critique = "(critique generation failed)"
205
+
206
+ episode = ReflexionEpisode(
207
+ attempt=attempt,
208
+ task=task,
209
+ output=output,
210
+ critique=critique,
211
+ score=score,
212
+ # add_episode derives the value from the previous episode. Supply
213
+ # the neutral value here so construction cannot fail before that
214
+ # comparison is made.
215
+ improved=False,
216
+ )
217
+ loop.add_episode(episode)
218
+
219
+ # 3. Check termination
220
+ if score >= score_threshold:
221
+ break
222
+ if loop.is_converged() and attempt >= 2:
223
+ break
224
+
225
+ return loop.episodes