halfabyte-blackbox 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- halfabyte_blackbox-0.1.0/PKG-INFO +135 -0
- halfabyte_blackbox-0.1.0/README.md +86 -0
- halfabyte_blackbox-0.1.0/blackbox/__init__.py +36 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/base.py +223 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/builtin.py +7 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/checkers.py +83 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/math_agent.py +104 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/parsing.py +24 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/qa.py +118 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/shop.py +172 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/tools/__init__.py +6 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/tools/calculator.py +75 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/tools/python_exec.py +32 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/tools/registry.py +179 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/tools/search.py +74 -0
- halfabyte_blackbox-0.1.0/blackbox/agents/tools/shop.py +172 -0
- halfabyte_blackbox-0.1.0/blackbox/api/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/api/investigate.py +149 -0
- halfabyte_blackbox-0.1.0/blackbox/api/jobs.py +82 -0
- halfabyte_blackbox-0.1.0/blackbox/api/main.py +787 -0
- halfabyte_blackbox-0.1.0/blackbox/api/showcase.py +245 -0
- halfabyte_blackbox-0.1.0/blackbox/api/snapshot.py +69 -0
- halfabyte_blackbox-0.1.0/blackbox/api/static/inspector.html +441 -0
- halfabyte_blackbox-0.1.0/blackbox/api/ui.py +258 -0
- halfabyte_blackbox-0.1.0/blackbox/ci.py +183 -0
- halfabyte_blackbox-0.1.0/blackbox/cli.py +260 -0
- halfabyte_blackbox-0.1.0/blackbox/core/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/core/hashing.py +34 -0
- halfabyte_blackbox-0.1.0/blackbox/core/privacy.py +70 -0
- halfabyte_blackbox-0.1.0/blackbox/core/schema.py +111 -0
- halfabyte_blackbox-0.1.0/blackbox/core/store.py +239 -0
- halfabyte_blackbox-0.1.0/blackbox/data/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/data/dataset.py +218 -0
- halfabyte_blackbox-0.1.0/blackbox/data/download.py +69 -0
- halfabyte_blackbox-0.1.0/blackbox/data/farm.py +246 -0
- halfabyte_blackbox-0.1.0/blackbox/data/importers/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/data/importers/external.py +207 -0
- halfabyte_blackbox-0.1.0/blackbox/data/importers/who_and_when.py +82 -0
- halfabyte_blackbox-0.1.0/blackbox/data/shards.py +135 -0
- halfabyte_blackbox-0.1.0/blackbox/data/splits.py +45 -0
- halfabyte_blackbox-0.1.0/blackbox/data/tasks/__init__.py +21 -0
- halfabyte_blackbox-0.1.0/blackbox/data/tasks/gsm8k.py +39 -0
- halfabyte_blackbox-0.1.0/blackbox/data/tasks/hotpot.py +35 -0
- halfabyte_blackbox-0.1.0/blackbox/data/tasks/ops_tasks.py +121 -0
- halfabyte_blackbox-0.1.0/blackbox/demo.py +118 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/__init__.py +22 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/conformal.py +94 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/dataset.py +112 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/diagnose.py +176 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/embeddings.py +85 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/evidence.py +100 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/explain.py +72 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/features.py +216 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/fingerprint.py +139 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/gnn.py +582 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/graph_data.py +263 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/predict.py +23 -0
- halfabyte_blackbox-0.1.0/blackbox/diagnosis/ranker.py +358 -0
- halfabyte_blackbox-0.1.0/blackbox/eval/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/eval/baselines.py +32 -0
- halfabyte_blackbox-0.1.0/blackbox/eval/llm_judge.py +157 -0
- halfabyte_blackbox-0.1.0/blackbox/eval/metrics.py +74 -0
- halfabyte_blackbox-0.1.0/blackbox/eval/report.py +350 -0
- halfabyte_blackbox-0.1.0/blackbox/faults/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/faults/catalog.py +426 -0
- halfabyte_blackbox-0.1.0/blackbox/guardian/__init__.py +14 -0
- halfabyte_blackbox-0.1.0/blackbox/guardian/channels.py +230 -0
- halfabyte_blackbox-0.1.0/blackbox/guardian/incident.py +44 -0
- halfabyte_blackbox-0.1.0/blackbox/guardian/render.py +192 -0
- halfabyte_blackbox-0.1.0/blackbox/guardian/service.py +238 -0
- halfabyte_blackbox-0.1.0/blackbox/guardian/setup.py +135 -0
- halfabyte_blackbox-0.1.0/blackbox/instrument.py +73 -0
- halfabyte_blackbox-0.1.0/blackbox/integrations.py +445 -0
- halfabyte_blackbox-0.1.0/blackbox/lab.py +353 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/__init__.py +19 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/__main__.py +5 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/prompts.py +48 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/protocol.py +30 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/registry.py +109 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/resources.py +57 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/server.py +117 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/tools/__init__.py +3 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/tools/developer.py +59 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/tools/diagnosis.py +176 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/tools/experiments.py +148 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/tools/guardian.py +42 -0
- halfabyte_blackbox-0.1.0/blackbox/mcp/tools/runs.py +94 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/__init__.py +4 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/__init__.py +1 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/jev_adapter.py +92 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/laya_adapter.py +98 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/llm_adapter.py +65 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/cascade.py +146 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/check.py +4 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/finetune.py +194 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/questions.py +84 -0
- halfabyte_blackbox-0.1.0/blackbox/monitor/state_builder.py +47 -0
- halfabyte_blackbox-0.1.0/blackbox/recorder/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/recorder/context.py +517 -0
- halfabyte_blackbox-0.1.0/blackbox/recorder/providers.py +285 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/__init__.py +0 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/advisor.py +71 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/autopilot.py +85 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/causal.py +157 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/chaos.py +96 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/diff.py +102 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/engine.py +86 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/fleet.py +194 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/label_natural.py +78 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/repair.py +139 -0
- halfabyte_blackbox-0.1.0/blackbox/replay/verify.py +167 -0
- halfabyte_blackbox-0.1.0/blackbox/status.py +201 -0
- halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/PKG-INFO +135 -0
- halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/SOURCES.txt +155 -0
- halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/dependency_links.txt +1 -0
- halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/entry_points.txt +2 -0
- halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/requires.txt +36 -0
- halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/top_level.txt +1 -0
- halfabyte_blackbox-0.1.0/pyproject.toml +58 -0
- halfabyte_blackbox-0.1.0/setup.cfg +4 -0
- halfabyte_blackbox-0.1.0/tests/test_advisor_privacy.py +47 -0
- halfabyte_blackbox-0.1.0/tests/test_agents.py +179 -0
- halfabyte_blackbox-0.1.0/tests/test_api.py +169 -0
- halfabyte_blackbox-0.1.0/tests/test_async_integrations.py +70 -0
- halfabyte_blackbox-0.1.0/tests/test_causal.py +41 -0
- halfabyte_blackbox-0.1.0/tests/test_chaos.py +57 -0
- halfabyte_blackbox-0.1.0/tests/test_checker_head.py +32 -0
- halfabyte_blackbox-0.1.0/tests/test_checkers.py +78 -0
- halfabyte_blackbox-0.1.0/tests/test_ci.py +68 -0
- halfabyte_blackbox-0.1.0/tests/test_conformal.py +50 -0
- halfabyte_blackbox-0.1.0/tests/test_core_replay.py +197 -0
- halfabyte_blackbox-0.1.0/tests/test_dataset.py +98 -0
- halfabyte_blackbox-0.1.0/tests/test_eval.py +263 -0
- halfabyte_blackbox-0.1.0/tests/test_evidence.py +33 -0
- halfabyte_blackbox-0.1.0/tests/test_farm.py +179 -0
- halfabyte_blackbox-0.1.0/tests/test_fast_status.py +54 -0
- halfabyte_blackbox-0.1.0/tests/test_faults.py +153 -0
- halfabyte_blackbox-0.1.0/tests/test_faults_unseen.py +265 -0
- halfabyte_blackbox-0.1.0/tests/test_features.py +93 -0
- halfabyte_blackbox-0.1.0/tests/test_fingerprint.py +59 -0
- halfabyte_blackbox-0.1.0/tests/test_fleet.py +63 -0
- halfabyte_blackbox-0.1.0/tests/test_gnn.py +78 -0
- halfabyte_blackbox-0.1.0/tests/test_guardian.py +122 -0
- halfabyte_blackbox-0.1.0/tests/test_guardian_channels.py +146 -0
- halfabyte_blackbox-0.1.0/tests/test_guardian_setup.py +73 -0
- halfabyte_blackbox-0.1.0/tests/test_importers_external.py +68 -0
- halfabyte_blackbox-0.1.0/tests/test_integrations.py +117 -0
- halfabyte_blackbox-0.1.0/tests/test_live_ollama.py +71 -0
- halfabyte_blackbox-0.1.0/tests/test_mcp.py +161 -0
- halfabyte_blackbox-0.1.0/tests/test_monitor.py +291 -0
- halfabyte_blackbox-0.1.0/tests/test_ranker.py +103 -0
- halfabyte_blackbox-0.1.0/tests/test_shards.py +52 -0
- halfabyte_blackbox-0.1.0/tests/test_shop.py +258 -0
- halfabyte_blackbox-0.1.0/tests/test_tools.py +116 -0
- halfabyte_blackbox-0.1.0/tests/test_ui_api.py +54 -0
- halfabyte_blackbox-0.1.0/tests/test_verify_autopilot_repair.py +191 -0
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: halfabyte-blackbox
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A flight recorder for AI agents: record every step, find the step that caused a failure, prove it by replay, and fix it without regressions.
|
|
5
|
+
Author: Aryan Lomte, Radhesh, Aditya, Advay Chavan
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: ai agents,llm,observability,debugging,replay,root cause analysis,mcp,evaluation
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
17
|
+
Classifier: Topic :: Software Development :: Debuggers
|
|
18
|
+
Classifier: Topic :: Software Development :: Testing
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
Requires-Dist: pydantic>=2.5
|
|
22
|
+
Requires-Dist: pyyaml>=6
|
|
23
|
+
Requires-Dist: python-dotenv>=1
|
|
24
|
+
Requires-Dist: openai>=1.40
|
|
25
|
+
Provides-Extra: checker
|
|
26
|
+
Requires-Dist: laya; extra == "checker"
|
|
27
|
+
Requires-Dist: system-one-adapter[openai]; extra == "checker"
|
|
28
|
+
Provides-Extra: jev
|
|
29
|
+
Requires-Dist: typesafe-sdk; extra == "jev"
|
|
30
|
+
Provides-Extra: ml
|
|
31
|
+
Requires-Dist: lightgbm>=4; extra == "ml"
|
|
32
|
+
Requires-Dist: scikit-learn>=1.4; extra == "ml"
|
|
33
|
+
Requires-Dist: shap>=0.45; extra == "ml"
|
|
34
|
+
Requires-Dist: sentence-transformers>=3; extra == "ml"
|
|
35
|
+
Requires-Dist: numpy; extra == "ml"
|
|
36
|
+
Provides-Extra: ui
|
|
37
|
+
Requires-Dist: fastapi>=0.110; extra == "ui"
|
|
38
|
+
Requires-Dist: uvicorn[standard]>=0.29; extra == "ui"
|
|
39
|
+
Requires-Dist: sse-starlette>=2; extra == "ui"
|
|
40
|
+
Provides-Extra: agents
|
|
41
|
+
Requires-Dist: rank-bm25>=0.2; extra == "agents"
|
|
42
|
+
Provides-Extra: data
|
|
43
|
+
Requires-Dist: datasets>=2.18; extra == "data"
|
|
44
|
+
Provides-Extra: dev
|
|
45
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
46
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
47
|
+
Provides-Extra: all
|
|
48
|
+
Requires-Dist: halfabyte-blackbox[agents,checker,data,ml,ui]; extra == "all"
|
|
49
|
+
|
|
50
|
+
# Black Box: a flight recorder for AI agents
|
|
51
|
+
|
|
52
|
+
When an AI agent fails, the mistake usually happened several steps before the wrong answer. Black Box records
|
|
53
|
+
every step of an agent run, **finds the step that caused the failure, and proves it**: it re-runs the agent from
|
|
54
|
+
its recording with only that step repaired. If the run now passes, that step was the cause. Unchanged steps come
|
|
55
|
+
from the recording, so a replay costs zero model calls and a fix re-runs only what changed.
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pip install halfabyte-blackbox # recorder, replay, proof, CLI, CI, MCP server, trace import
|
|
59
|
+
pip install "halfabyte-blackbox[ui]" # + the API / dashboard backend
|
|
60
|
+
pip install "halfabyte-blackbox[all]" # + step checker, ranking model, test agents, dataset loaders
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Add it to an existing agent (3 lines)
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
import blackbox
|
|
67
|
+
from openai import OpenAI
|
|
68
|
+
|
|
69
|
+
client = blackbox.wrap(OpenAI()) # 1. every model call is recorded
|
|
70
|
+
|
|
71
|
+
@blackbox.tool # 2. every tool call is recorded and replayable
|
|
72
|
+
def get_weather(city: str) -> dict:
|
|
73
|
+
...
|
|
74
|
+
|
|
75
|
+
def my_agent(question: str) -> str: # your agent, unchanged
|
|
76
|
+
...
|
|
77
|
+
|
|
78
|
+
trace = blackbox.run(my_agent, "Will it rain in Pune?") # 3. run it under the recorder
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
blackbox.replay(trace) # re-run from the recording: 0 model calls
|
|
83
|
+
child = blackbox.fork(trace, 2, blackbox.Change(kind="output", value={"result": {"rain": True}}))
|
|
84
|
+
blackbox.savings(child) # steps re-run vs re-used, tokens saved
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Works with agents you did not write (shown on Hugging Face smolagents), and with traces you already export:
|
|
88
|
+
`blackbox import otel traces.json` / `blackbox import langfuse trace.json`.
|
|
89
|
+
|
|
90
|
+
## What it does
|
|
91
|
+
|
|
92
|
+
| Capability | How |
|
|
93
|
+
|---|---|
|
|
94
|
+
| **Record** every model call, tool call, search and memory change, with a save-point after each step | `blackbox.wrap`, `@blackbox.tool`, `blackbox.run` |
|
|
95
|
+
| **Diagnose**: rank the steps most likely to have caused a failure, with plain-language evidence | `blackbox show <run>`, API `/api/runs/{id}/diagnosis` |
|
|
96
|
+
| **Prove** the cause by re-running with one suspect repaired at a time | Verify (API, MCP `verify`) |
|
|
97
|
+
| **Causal report**: necessary vs sufficient, joint causes, every recovery path, blast radius | API `/causal`, MCP `causal_report` |
|
|
98
|
+
| **One fix for many**: test one rule on similar past failures and passing runs; APPROVE only if nothing breaks | `blackbox fleet <run>` |
|
|
99
|
+
| **Crash test**: plant every known kind of mistake into a working agent, grade it 1–5 stars | `blackbox crash-test <run>` |
|
|
100
|
+
| **Seen this before?**: failure fingerprints, look-alike failures, novel failures, known fixes | MCP `similar_failures` |
|
|
101
|
+
| **Guardian**: incidents in plain words on email, WhatsApp, Slack, Telegram; money-moving agents paused until approved | `blackbox.Guardian(...)`, `blackbox guardian test` |
|
|
102
|
+
| **Black Box CI**: replay pinned recorded runs on every pull request; fail on regressions | `blackbox ci pin` / `blackbox ci run`, GitHub Action |
|
|
103
|
+
| **MCP server**: let Claude Code, Cursor or VS Code investigate failures with 28 tools | `blackbox mcp` |
|
|
104
|
+
| **Live Lab**: ask any question, plant a mistake live, watch the 14-stage investigation | `blackbox lab "question"` |
|
|
105
|
+
|
|
106
|
+
Safety: tools that move money never execute during any re-run, and Guardian never retries them without a human.
|
|
107
|
+
|
|
108
|
+
## Command line
|
|
109
|
+
|
|
110
|
+
```bash
|
|
111
|
+
blackbox runs --fail # failed runs
|
|
112
|
+
blackbox show <run_id> # step by step, with who-used-whose-output links
|
|
113
|
+
blackbox replay <run_id> # replay from the recording
|
|
114
|
+
blackbox crash-test <run_id> # red-team a passing run
|
|
115
|
+
blackbox fleet <run_id> --rule "..." # one fix for many, with a regression firewall
|
|
116
|
+
blackbox ci pin && blackbox ci run # regression firewall for agent code
|
|
117
|
+
blackbox guardian channels # which alert channels are configured
|
|
118
|
+
blackbox mcp # MCP server on stdio
|
|
119
|
+
blackbox ui # API on http://127.0.0.1:8000
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Models are named in one file, `models.yaml` (Ollama locally, Groq or OpenRouter hosted); keys live in `.env`.
|
|
123
|
+
|
|
124
|
+
## Development (this repository)
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
pip install -e ".[all,dev]"
|
|
128
|
+
py -3 -m pytest -q -m "not live"
|
|
129
|
+
cd web && npm install && npm run dev # dashboard on http://localhost:5173
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Design: `SYSTEM.md` · rules: `CLAUDE.md` · UI contract: `docs/API_FOR_UI.md` · MCP: `docs/MCP.md` · CI: `docs/CI.md` ·
|
|
133
|
+
Guardian setup: `docs/GUARDIAN_SETUP.md`. Branches `<name>/<feature>` → PR into `dev` → `main` at milestones.
|
|
134
|
+
|
|
135
|
+
Built by team Half a Byte: Aryan Lomte, Radhesh, Aditya, Advay Chavan.
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# Black Box: a flight recorder for AI agents
|
|
2
|
+
|
|
3
|
+
When an AI agent fails, the mistake usually happened several steps before the wrong answer. Black Box records
|
|
4
|
+
every step of an agent run, **finds the step that caused the failure, and proves it**: it re-runs the agent from
|
|
5
|
+
its recording with only that step repaired. If the run now passes, that step was the cause. Unchanged steps come
|
|
6
|
+
from the recording, so a replay costs zero model calls and a fix re-runs only what changed.
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
pip install halfabyte-blackbox # recorder, replay, proof, CLI, CI, MCP server, trace import
|
|
10
|
+
pip install "halfabyte-blackbox[ui]" # + the API / dashboard backend
|
|
11
|
+
pip install "halfabyte-blackbox[all]" # + step checker, ranking model, test agents, dataset loaders
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## Add it to an existing agent (3 lines)
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
import blackbox
|
|
18
|
+
from openai import OpenAI
|
|
19
|
+
|
|
20
|
+
client = blackbox.wrap(OpenAI()) # 1. every model call is recorded
|
|
21
|
+
|
|
22
|
+
@blackbox.tool # 2. every tool call is recorded and replayable
|
|
23
|
+
def get_weather(city: str) -> dict:
|
|
24
|
+
...
|
|
25
|
+
|
|
26
|
+
def my_agent(question: str) -> str: # your agent, unchanged
|
|
27
|
+
...
|
|
28
|
+
|
|
29
|
+
trace = blackbox.run(my_agent, "Will it rain in Pune?") # 3. run it under the recorder
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
blackbox.replay(trace) # re-run from the recording: 0 model calls
|
|
34
|
+
child = blackbox.fork(trace, 2, blackbox.Change(kind="output", value={"result": {"rain": True}}))
|
|
35
|
+
blackbox.savings(child) # steps re-run vs re-used, tokens saved
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Works with agents you did not write (shown on Hugging Face smolagents), and with traces you already export:
|
|
39
|
+
`blackbox import otel traces.json` / `blackbox import langfuse trace.json`.
|
|
40
|
+
|
|
41
|
+
## What it does
|
|
42
|
+
|
|
43
|
+
| Capability | How |
|
|
44
|
+
|---|---|
|
|
45
|
+
| **Record** every model call, tool call, search and memory change, with a save-point after each step | `blackbox.wrap`, `@blackbox.tool`, `blackbox.run` |
|
|
46
|
+
| **Diagnose**: rank the steps most likely to have caused a failure, with plain-language evidence | `blackbox show <run>`, API `/api/runs/{id}/diagnosis` |
|
|
47
|
+
| **Prove** the cause by re-running with one suspect repaired at a time | Verify (API, MCP `verify`) |
|
|
48
|
+
| **Causal report**: necessary vs sufficient, joint causes, every recovery path, blast radius | API `/causal`, MCP `causal_report` |
|
|
49
|
+
| **One fix for many**: test one rule on similar past failures and passing runs; APPROVE only if nothing breaks | `blackbox fleet <run>` |
|
|
50
|
+
| **Crash test**: plant every known kind of mistake into a working agent, grade it 1–5 stars | `blackbox crash-test <run>` |
|
|
51
|
+
| **Seen this before?**: failure fingerprints, look-alike failures, novel failures, known fixes | MCP `similar_failures` |
|
|
52
|
+
| **Guardian**: incidents in plain words on email, WhatsApp, Slack, Telegram; money-moving agents paused until approved | `blackbox.Guardian(...)`, `blackbox guardian test` |
|
|
53
|
+
| **Black Box CI**: replay pinned recorded runs on every pull request; fail on regressions | `blackbox ci pin` / `blackbox ci run`, GitHub Action |
|
|
54
|
+
| **MCP server**: let Claude Code, Cursor or VS Code investigate failures with 28 tools | `blackbox mcp` |
|
|
55
|
+
| **Live Lab**: ask any question, plant a mistake live, watch the 14-stage investigation | `blackbox lab "question"` |
|
|
56
|
+
|
|
57
|
+
Safety: tools that move money never execute during any re-run, and Guardian never retries them without a human.
|
|
58
|
+
|
|
59
|
+
## Command line
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
blackbox runs --fail # failed runs
|
|
63
|
+
blackbox show <run_id> # step by step, with who-used-whose-output links
|
|
64
|
+
blackbox replay <run_id> # replay from the recording
|
|
65
|
+
blackbox crash-test <run_id> # red-team a passing run
|
|
66
|
+
blackbox fleet <run_id> --rule "..." # one fix for many, with a regression firewall
|
|
67
|
+
blackbox ci pin && blackbox ci run # regression firewall for agent code
|
|
68
|
+
blackbox guardian channels # which alert channels are configured
|
|
69
|
+
blackbox mcp # MCP server on stdio
|
|
70
|
+
blackbox ui # API on http://127.0.0.1:8000
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Models are named in one file, `models.yaml` (Ollama locally, Groq or OpenRouter hosted); keys live in `.env`.
|
|
74
|
+
|
|
75
|
+
## Development (this repository)
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install -e ".[all,dev]"
|
|
79
|
+
py -3 -m pytest -q -m "not live"
|
|
80
|
+
cd web && npm install && npm run dev # dashboard on http://localhost:5173
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Design: `SYSTEM.md` · rules: `CLAUDE.md` · UI contract: `docs/API_FOR_UI.md` · MCP: `docs/MCP.md` · CI: `docs/CI.md` ·
|
|
84
|
+
Guardian setup: `docs/GUARDIAN_SETUP.md`. Branches `<name>/<feature>` → PR into `dev` → `main` at milestones.
|
|
85
|
+
|
|
86
|
+
Built by team Half a Byte: Aryan Lomte, Radhesh, Aditya, Advay Chavan.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Black Box: a flight recorder for AI agents.
|
|
2
|
+
|
|
3
|
+
Quick start for an existing project (see README.md):
|
|
4
|
+
|
|
5
|
+
import blackbox
|
|
6
|
+
client = blackbox.wrap(OpenAI())
|
|
7
|
+
trace = blackbox.run(my_agent, "question")
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from blackbox.agents.base import Agent, Task, register_agent, register_checker, run_task
|
|
11
|
+
from blackbox.core.schema import Change, Outcome, Run, Step, Trace
|
|
12
|
+
from blackbox.guardian import Guardian, Incident
|
|
13
|
+
from blackbox.core.store import Store
|
|
14
|
+
from blackbox.integrations import (
|
|
15
|
+
Session,
|
|
16
|
+
afork,
|
|
17
|
+
areplay,
|
|
18
|
+
arun,
|
|
19
|
+
configure,
|
|
20
|
+
default_store,
|
|
21
|
+
retriever,
|
|
22
|
+
run,
|
|
23
|
+
session,
|
|
24
|
+
tool,
|
|
25
|
+
wrap,
|
|
26
|
+
)
|
|
27
|
+
from blackbox.recorder.context import RunContext, ToolError
|
|
28
|
+
from blackbox.replay.engine import fork, replay, savings
|
|
29
|
+
|
|
30
|
+
__version__ = "0.1.0"
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"Agent", "Change", "Guardian", "Incident", "afork", "areplay", "arun", "Outcome", "Run", "RunContext", "Session", "Step", "Store", "Task", "ToolError",
|
|
34
|
+
"Trace", "configure", "default_store", "fork", "register_agent", "register_checker", "replay",
|
|
35
|
+
"retriever", "run", "run_task", "savings", "session", "tool", "wrap",
|
|
36
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
"""Agents are resumable state machines (CLAUDE.md hard rule 3).
|
|
2
|
+
|
|
3
|
+
class QAAgent(Agent):
|
|
4
|
+
name, version, family, entry = "qa_agent", "v1", "qa", "plan"
|
|
5
|
+
def init_state(self, task): return {"question": task.input["question"]}
|
|
6
|
+
def plan(self, ctx): ...; return "search" # next node name
|
|
7
|
+
def search(self, ctx): ...; return "answer"
|
|
8
|
+
def answer(self, ctx): ctx.final(...); return None # None = finished
|
|
9
|
+
nodes = {"plan": plan, "search": search, "answer": answer}
|
|
10
|
+
|
|
11
|
+
Rules:
|
|
12
|
+
- Every model/tool/search call goes through ctx (ctx.llm / ctx.tool / ctx.retrieve).
|
|
13
|
+
- All memory lives in ctx.state; assign values (state[k] = v), never mutate in place.
|
|
14
|
+
- Randomness: ctx.rng. Time: ctx.now(). Nothing else.
|
|
15
|
+
- Finish by calling ctx.final(answer) and returning None.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import socket
|
|
22
|
+
import traceback
|
|
23
|
+
import uuid
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from typing import Any, Callable, ClassVar
|
|
26
|
+
|
|
27
|
+
from blackbox.core.schema import Change, Outcome, Run, Trace
|
|
28
|
+
from blackbox.core.store import Store
|
|
29
|
+
from blackbox.recorder.context import AutopilotRewind, Monitor, ReplayPlan, RunContext, StepListener, current
|
|
30
|
+
|
|
31
|
+
MAX_NODE_VISITS = 60
|
|
32
|
+
|
|
33
|
+
Checker = Callable[[Any, Any], Outcome] # (final_answer, expected) -> Outcome
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class Task:
|
|
38
|
+
family: str
|
|
39
|
+
task_id: str
|
|
40
|
+
input: dict # what the agent may see
|
|
41
|
+
expected: Any # what the checker compares against; the agent never sees it
|
|
42
|
+
meta: dict = field(default_factory=dict)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Agent:
|
|
46
|
+
name: ClassVar[str] = "base"
|
|
47
|
+
version: ClassVar[str] = "v1"
|
|
48
|
+
family: ClassVar[str] = ""
|
|
49
|
+
entry: ClassVar[str] = ""
|
|
50
|
+
nodes: ClassVar[dict[str, Callable[["Agent", RunContext], str | None]]] = {}
|
|
51
|
+
|
|
52
|
+
def init_state(self, task: Task) -> dict:
|
|
53
|
+
return {}
|
|
54
|
+
|
|
55
|
+
def tool_caller(self, task: Task) -> Callable[[str, dict], Any] | None:
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
def retriever(self, task: Task) -> Callable[..., list[dict]] | None:
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
@classmethod
|
|
62
|
+
def ref(cls) -> str:
|
|
63
|
+
return f"{cls.name}@{cls.version}"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# ---- registries (agents and checkers are looked up by name on replay) ----------
|
|
67
|
+
_AGENTS: dict[str, type[Agent]] = {}
|
|
68
|
+
_CHECKERS: dict[str, Checker] = {}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def register_agent(cls: type[Agent]) -> type[Agent]:
|
|
72
|
+
_AGENTS[cls.ref()] = cls
|
|
73
|
+
return cls
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def register_checker(family: str, fn: Checker) -> Checker:
|
|
77
|
+
_CHECKERS[family] = fn
|
|
78
|
+
return fn
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _load_builtin() -> None:
|
|
82
|
+
"""Register our test agents on first miss, so a process that only replays
|
|
83
|
+
stored runs (API, CLI) finds them. Skipped when the `agents` extra is absent."""
|
|
84
|
+
try:
|
|
85
|
+
import blackbox.agents.builtin # noqa: F401
|
|
86
|
+
except ModuleNotFoundError as exc:
|
|
87
|
+
if exc.name is None or exc.name.startswith("blackbox"):
|
|
88
|
+
raise
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def get_agent(ref: str) -> Agent:
|
|
92
|
+
if ref not in _AGENTS and ref.endswith("@fn"):
|
|
93
|
+
from blackbox.integrations import resolve_function_agent # user function agents
|
|
94
|
+
|
|
95
|
+
resolve_function_agent(ref)
|
|
96
|
+
if ref not in _AGENTS:
|
|
97
|
+
_load_builtin()
|
|
98
|
+
if ref not in _AGENTS:
|
|
99
|
+
raise KeyError(f"agent {ref!r} not registered (import its module). Known: {sorted(_AGENTS)}")
|
|
100
|
+
return _AGENTS[ref]()
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def get_checker(family: str) -> Checker | None:
|
|
104
|
+
"""None for a user's project that registered no checker: the run is recorded
|
|
105
|
+
without a pass/fail verdict rather than with an invented one."""
|
|
106
|
+
return _CHECKERS.get(family)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def machine_name() -> str:
|
|
110
|
+
return os.environ.get("BLACKBOX_MACHINE") or socket.gethostname()
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def run_task(
|
|
114
|
+
agent: Agent,
|
|
115
|
+
task: Task,
|
|
116
|
+
model_id: str,
|
|
117
|
+
*,
|
|
118
|
+
store: Store | None = None,
|
|
119
|
+
plan: ReplayPlan | None = None,
|
|
120
|
+
monitor: Monitor | None = None,
|
|
121
|
+
on_step: list[StepListener] | None = None,
|
|
122
|
+
run_id: str | None = None,
|
|
123
|
+
parent_run_id: str | None = None,
|
|
124
|
+
fork_step_idx: int | None = None,
|
|
125
|
+
patch: Change | None = None,
|
|
126
|
+
split: str | None = None,
|
|
127
|
+
autopilot=None,
|
|
128
|
+
) -> Trace:
|
|
129
|
+
"""Run (or replay/fork) one task and return its trace. Saved if a store is given.
|
|
130
|
+
|
|
131
|
+
A crash (exception escaping the agent) marks the run "crashed" with the
|
|
132
|
+
traceback. It is our bug, never a task failure, and never a label
|
|
133
|
+
(CLAUDE.md review rule 3).
|
|
134
|
+
"""
|
|
135
|
+
run = Run(
|
|
136
|
+
run_id=run_id or uuid.uuid4().hex[:16],
|
|
137
|
+
task_family=task.family,
|
|
138
|
+
task_id=task.task_id,
|
|
139
|
+
task_input=task.input,
|
|
140
|
+
agent=agent.ref(),
|
|
141
|
+
model_id=model_id,
|
|
142
|
+
parent_run_id=parent_run_id,
|
|
143
|
+
fork_step_idx=fork_step_idx,
|
|
144
|
+
patch=patch,
|
|
145
|
+
split=split,
|
|
146
|
+
machine=machine_name(),
|
|
147
|
+
task_expected=task.expected,
|
|
148
|
+
)
|
|
149
|
+
ctx = RunContext(
|
|
150
|
+
run, agent.init_state(task), store=store,
|
|
151
|
+
tool_caller=agent.tool_caller(task), retriever=agent.retriever(task),
|
|
152
|
+
plan=plan, monitor=monitor, on_step=on_step, autopilot=autopilot,
|
|
153
|
+
)
|
|
154
|
+
token = current.set(ctx)
|
|
155
|
+
try:
|
|
156
|
+
node: str | None = agent.entry
|
|
157
|
+
visits = 0
|
|
158
|
+
while node is not None:
|
|
159
|
+
if node not in agent.nodes:
|
|
160
|
+
raise KeyError(f"{agent.ref()}: unknown node {node!r}")
|
|
161
|
+
visits += 1
|
|
162
|
+
if visits > MAX_NODE_VISITS:
|
|
163
|
+
# Looping forever is a genuine agent failure, not a crash.
|
|
164
|
+
run.stats["stopped"] = f"exceeded {MAX_NODE_VISITS} node visits"
|
|
165
|
+
if not ctx.steps or ctx.steps[-1].kind != "final":
|
|
166
|
+
ctx.final(None)
|
|
167
|
+
break
|
|
168
|
+
ctx.node = node
|
|
169
|
+
node = agent.nodes[node](agent, ctx)
|
|
170
|
+
ctx.close()
|
|
171
|
+
final_steps = [s for s in ctx.steps if s.kind == "final"]
|
|
172
|
+
answer = final_steps[-1].output["answer"] if final_steps else None
|
|
173
|
+
checker = get_checker(task.family)
|
|
174
|
+
run.outcome = checker(answer, task.expected) if checker else None
|
|
175
|
+
run.status = "done"
|
|
176
|
+
except AutopilotRewind as rewind:
|
|
177
|
+
current.reset(token)
|
|
178
|
+
token = None
|
|
179
|
+
return _take_over(agent, task, model_id, ctx, run, rewind, store=store, monitor=monitor,
|
|
180
|
+
on_step=on_step, split=split, autopilot=autopilot)
|
|
181
|
+
except Exception: # noqa: BLE001 - recorded loudly as a crash, see docstring
|
|
182
|
+
ctx.close(monitor_timeout=5.0)
|
|
183
|
+
run.status = "crashed"
|
|
184
|
+
run.error = traceback.format_exc()
|
|
185
|
+
finally:
|
|
186
|
+
if token is not None:
|
|
187
|
+
current.reset(token)
|
|
188
|
+
run.stats = {**ctx.stats(), **run.stats}
|
|
189
|
+
trace = Trace(run=run, steps=ctx.steps)
|
|
190
|
+
if store is not None:
|
|
191
|
+
store.save_trace(trace)
|
|
192
|
+
return trace
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _take_over(agent, task, model_id, ctx, run, rewind: AutopilotRewind, *, store, monitor, on_step, split,
|
|
196
|
+
autopilot) -> Trace:
|
|
197
|
+
"""Autopilot (USP X1): the run is stopped before it can answer. Its steps so
|
|
198
|
+
far are kept as a 'rewound' run, and a fork from the flagged decision takes
|
|
199
|
+
over, re-using everything before it from the recording."""
|
|
200
|
+
ctx.close(monitor_timeout=5.0)
|
|
201
|
+
flagged = ctx.steps[rewind.step_idx]
|
|
202
|
+
flagged.monitor = rewind.answers if rewind.answers is not None else flagged.monitor
|
|
203
|
+
run.status = "rewound"
|
|
204
|
+
run.stats = {**ctx.stats(), "autopilot": {"flagged_step": rewind.step_idx, "rewound_to": rewind.target_idx}}
|
|
205
|
+
partial = Trace(run=run, steps=ctx.steps)
|
|
206
|
+
if store is not None:
|
|
207
|
+
store.save_trace(partial)
|
|
208
|
+
child = run_task(
|
|
209
|
+
agent, task, model_id, store=store, monitor=monitor, on_step=on_step, split=split,
|
|
210
|
+
plan=ReplayPlan(parent=partial, fork_idx=rewind.target_idx, change=rewind.change),
|
|
211
|
+
parent_run_id=run.run_id, fork_step_idx=rewind.target_idx, patch=rewind.change, autopilot=autopilot,
|
|
212
|
+
)
|
|
213
|
+
child.run.stats.setdefault("autopilot_rescue", {
|
|
214
|
+
"from_run": run.run_id, "flagged_step": rewind.step_idx, "rewound_to": rewind.target_idx,
|
|
215
|
+
"change": rewind.change.model_dump(mode="json") if rewind.change else None,
|
|
216
|
+
})
|
|
217
|
+
if store is not None:
|
|
218
|
+
store.save_trace(child)
|
|
219
|
+
return child
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def task_from_run(run: Run) -> Task:
|
|
223
|
+
return Task(family=run.task_family, task_id=run.task_id, input=run.task_input, expected=run.task_expected)
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Importing this module registers our test agents and their checkers.
|
|
2
|
+
|
|
3
|
+
Replay looks agents up by name (`qa_agent@v1`), so any process that replays
|
|
4
|
+
or forks a stored run (farm, API, CLI) needs them registered first.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from blackbox.agents import checkers, math_agent, qa, shop # noqa: F401 - registration side effect
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Pass/fail checkers. Plain code: no model decides what counts as a failure
|
|
2
|
+
(SYSTEM.md §3.1, CLAUDE.md hard rule 5).
|
|
3
|
+
|
|
4
|
+
QA : SQuAD-normalised word-overlap F1 >= 0.6 against the gold answer
|
|
5
|
+
Math : the final number equals the gold number exactly
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
import string
|
|
12
|
+
from collections import Counter
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from blackbox.agents.base import register_checker
|
|
16
|
+
from blackbox.core.schema import Outcome
|
|
17
|
+
|
|
18
|
+
QA_F1_THRESHOLD = 0.6
|
|
19
|
+
|
|
20
|
+
_ARTICLES = re.compile(r"\b(a|an|the)\b")
|
|
21
|
+
_PUNCT = str.maketrans("", "", string.punctuation)
|
|
22
|
+
_NUMBER = re.compile(r"-?\d[\d,]*\.?\d*|-?\.\d+")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def normalize_answer(text: Any) -> str:
|
|
26
|
+
"""SQuAD normalisation: lowercase, drop punctuation, drop articles, squeeze spaces."""
|
|
27
|
+
s = "" if text is None else str(text)
|
|
28
|
+
s = s.lower().translate(_PUNCT)
|
|
29
|
+
s = _ARTICLES.sub(" ", s)
|
|
30
|
+
return " ".join(s.split())
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def f1_score(prediction: Any, gold: Any) -> float:
|
|
34
|
+
pred, ref = normalize_answer(prediction).split(), normalize_answer(gold).split()
|
|
35
|
+
if not pred or not ref:
|
|
36
|
+
return float(pred == ref)
|
|
37
|
+
common = sum((Counter(pred) & Counter(ref)).values())
|
|
38
|
+
if common == 0:
|
|
39
|
+
return 0.0
|
|
40
|
+
precision, recall = common / len(pred), common / len(ref)
|
|
41
|
+
return 2 * precision * recall / (precision + recall)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _gold(expected: Any) -> Any:
|
|
45
|
+
return expected.get("answer") if isinstance(expected, dict) else expected
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def check_qa(answer: Any, expected: Any) -> Outcome:
|
|
49
|
+
gold = _gold(expected)
|
|
50
|
+
score = f1_score(answer, gold)
|
|
51
|
+
# yes/no questions: "yes, they are" must not pass on partial overlap, nor "no" match "yes"
|
|
52
|
+
if normalize_answer(gold) in ("yes", "no"):
|
|
53
|
+
words = normalize_answer(answer).split()
|
|
54
|
+
score = float(bool(words) and words[0] == normalize_answer(gold))
|
|
55
|
+
return Outcome(success=score >= QA_F1_THRESHOLD, score=round(score, 4),
|
|
56
|
+
final_answer=answer, expected=gold, checker="qa_f1")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def parse_number(value: Any) -> float | None:
|
|
60
|
+
"""The last number in the text ("So she has $1,250.00 left." -> 1250.0)."""
|
|
61
|
+
if isinstance(value, bool) or value is None:
|
|
62
|
+
return None
|
|
63
|
+
if isinstance(value, (int, float)):
|
|
64
|
+
return float(value)
|
|
65
|
+
found = _NUMBER.findall(str(value))
|
|
66
|
+
for raw in reversed(found):
|
|
67
|
+
cleaned = raw.replace(",", "").rstrip(".")
|
|
68
|
+
try:
|
|
69
|
+
return float(cleaned)
|
|
70
|
+
except ValueError:
|
|
71
|
+
continue
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def check_math(answer: Any, expected: Any) -> Outcome:
|
|
76
|
+
gold = parse_number(_gold(expected))
|
|
77
|
+
got = parse_number(answer)
|
|
78
|
+
ok = gold is not None and got is not None and abs(got - gold) < 1e-6
|
|
79
|
+
return Outcome(success=ok, score=float(ok), final_answer=answer, expected=_gold(expected), checker="math_exact")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
register_checker("qa", check_qa)
|
|
83
|
+
register_checker("math", check_math)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Math agent for GSM8K: plan -> (step -> calculate)* -> check -> answer.
|
|
2
|
+
|
|
3
|
+
The model never does arithmetic itself: it names one calculation at a time
|
|
4
|
+
and the calculator tool computes it. A run is 6 to 15 steps:
|
|
5
|
+
|
|
6
|
+
plan(llm) [step(llm) calculator(tool)] x 1..MAX_CALCS step(llm, DONE) check(llm) final
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from blackbox.agents import checkers # noqa: F401 - registers the "math" checker
|
|
12
|
+
from blackbox.agents.base import Agent, Task, register_agent
|
|
13
|
+
from blackbox.agents.parsing import clip, field
|
|
14
|
+
from blackbox.agents.tools import call_tool
|
|
15
|
+
from blackbox.recorder.context import RunContext, ToolError
|
|
16
|
+
|
|
17
|
+
MAX_CALCS = 5
|
|
18
|
+
MAX_TURNS = 6 # step-node visits, including the one that says DONE
|
|
19
|
+
|
|
20
|
+
PLAN_SYSTEM = (
|
|
21
|
+
"You plan how to solve a math word problem. List the calculations needed, in order, "
|
|
22
|
+
"one short line each: what is being found and from which numbers in the problem. "
|
|
23
|
+
"Do not compute the results. At most 5 lines. The last line must find what the question asks."
|
|
24
|
+
)
|
|
25
|
+
STEP_SYSTEM = (
|
|
26
|
+
"You solve a math word problem one calculation at a time. A calculator does the arithmetic. "
|
|
27
|
+
"Do the next line of the plan that has no result yet. Reply with exactly one line:\n"
|
|
28
|
+
"CALC: <one arithmetic expression using only numbers and + - * / ( )>\n"
|
|
29
|
+
"or, when every line of the plan has a result:\nDONE"
|
|
30
|
+
)
|
|
31
|
+
CHECK_SYSTEM = (
|
|
32
|
+
"You check the solution of a math word problem. Given the problem and the calculator results, "
|
|
33
|
+
"state the final answer to the question asked. Reply with exactly one line:\n"
|
|
34
|
+
"FINAL: <number only, no units>"
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _format_results(results: list[dict]) -> str:
|
|
39
|
+
if not results:
|
|
40
|
+
return "(none yet)"
|
|
41
|
+
return "\n".join(f"{i + 1}. {r['expr']} = {r['value']}" for i, r in enumerate(results))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@register_agent
|
|
45
|
+
class MathAgent(Agent):
|
|
46
|
+
name, version, family, entry = "math_agent", "v1", "math", "plan"
|
|
47
|
+
|
|
48
|
+
def init_state(self, task: Task) -> dict:
|
|
49
|
+
return {"question": task.input["question"], "results": [], "turns": 0}
|
|
50
|
+
|
|
51
|
+
def tool_caller(self, task: Task):
|
|
52
|
+
return call_tool
|
|
53
|
+
|
|
54
|
+
def plan(self, ctx: RunContext) -> str:
|
|
55
|
+
res = ctx.llm(ctx.model_id, [
|
|
56
|
+
{"role": "system", "content": PLAN_SYSTEM},
|
|
57
|
+
{"role": "user", "content": f"Problem: {ctx.state['question']}"},
|
|
58
|
+
], max_tokens=160)
|
|
59
|
+
ctx.state["plan"] = clip(res.text, 600)
|
|
60
|
+
return "step"
|
|
61
|
+
|
|
62
|
+
def step(self, ctx: RunContext) -> str:
|
|
63
|
+
results, turns = ctx.state["results"], ctx.state["turns"]
|
|
64
|
+
res = ctx.llm(ctx.model_id, [
|
|
65
|
+
{"role": "system", "content": STEP_SYSTEM},
|
|
66
|
+
{"role": "user", "content": (
|
|
67
|
+
f"Problem: {ctx.state['question']}\n\nPlan:\n{ctx.state['plan']}\n\n"
|
|
68
|
+
f"Results so far:\n{_format_results(results)}"
|
|
69
|
+
)},
|
|
70
|
+
], max_tokens=48)
|
|
71
|
+
expr = field(res.text, "CALC")
|
|
72
|
+
ctx.state["turns"] = turns + 1
|
|
73
|
+
if expr is None or len(results) >= MAX_CALCS or turns + 1 >= MAX_TURNS:
|
|
74
|
+
return "check"
|
|
75
|
+
ctx.state["expr"] = expr
|
|
76
|
+
return "calculate"
|
|
77
|
+
|
|
78
|
+
def calculate(self, ctx: RunContext) -> str:
|
|
79
|
+
expr = ctx.state["expr"]
|
|
80
|
+
try:
|
|
81
|
+
value = ctx.tool("calculator", expression=expr)
|
|
82
|
+
except ToolError as exc:
|
|
83
|
+
value = f"error: {clip(str(exc), 120)}" # recorded on the step; the model sees it next turn
|
|
84
|
+
# read after the call: the calculator depends on `expr` only, so its
|
|
85
|
+
# parents and its re-use fingerprint must not include earlier results
|
|
86
|
+
ctx.state["results"] = ctx.state["results"] + [{"expr": expr, "value": value}]
|
|
87
|
+
return "step"
|
|
88
|
+
|
|
89
|
+
def check(self, ctx: RunContext) -> str:
|
|
90
|
+
results = ctx.state["results"]
|
|
91
|
+
res = ctx.llm(ctx.model_id, [
|
|
92
|
+
{"role": "system", "content": CHECK_SYSTEM},
|
|
93
|
+
{"role": "user", "content": (
|
|
94
|
+
f"Problem: {ctx.state['question']}\n\nCalculator results:\n{_format_results(results)}"
|
|
95
|
+
)},
|
|
96
|
+
], max_tokens=32)
|
|
97
|
+
ctx.state["answer"] = field(res.text, "FINAL") or clip(res.text, 80)
|
|
98
|
+
return "answer"
|
|
99
|
+
|
|
100
|
+
def answer(self, ctx: RunContext) -> None:
|
|
101
|
+
ctx.final(ctx.state["answer"])
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
nodes = {"plan": plan, "step": step, "calculate": calculate, "check": check, "answer": answer}
|