halfabyte-blackbox 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. halfabyte_blackbox-0.1.0/PKG-INFO +135 -0
  2. halfabyte_blackbox-0.1.0/README.md +86 -0
  3. halfabyte_blackbox-0.1.0/blackbox/__init__.py +36 -0
  4. halfabyte_blackbox-0.1.0/blackbox/agents/__init__.py +0 -0
  5. halfabyte_blackbox-0.1.0/blackbox/agents/base.py +223 -0
  6. halfabyte_blackbox-0.1.0/blackbox/agents/builtin.py +7 -0
  7. halfabyte_blackbox-0.1.0/blackbox/agents/checkers.py +83 -0
  8. halfabyte_blackbox-0.1.0/blackbox/agents/math_agent.py +104 -0
  9. halfabyte_blackbox-0.1.0/blackbox/agents/parsing.py +24 -0
  10. halfabyte_blackbox-0.1.0/blackbox/agents/qa.py +118 -0
  11. halfabyte_blackbox-0.1.0/blackbox/agents/shop.py +172 -0
  12. halfabyte_blackbox-0.1.0/blackbox/agents/tools/__init__.py +6 -0
  13. halfabyte_blackbox-0.1.0/blackbox/agents/tools/calculator.py +75 -0
  14. halfabyte_blackbox-0.1.0/blackbox/agents/tools/python_exec.py +32 -0
  15. halfabyte_blackbox-0.1.0/blackbox/agents/tools/registry.py +179 -0
  16. halfabyte_blackbox-0.1.0/blackbox/agents/tools/search.py +74 -0
  17. halfabyte_blackbox-0.1.0/blackbox/agents/tools/shop.py +172 -0
  18. halfabyte_blackbox-0.1.0/blackbox/api/__init__.py +0 -0
  19. halfabyte_blackbox-0.1.0/blackbox/api/investigate.py +149 -0
  20. halfabyte_blackbox-0.1.0/blackbox/api/jobs.py +82 -0
  21. halfabyte_blackbox-0.1.0/blackbox/api/main.py +787 -0
  22. halfabyte_blackbox-0.1.0/blackbox/api/showcase.py +245 -0
  23. halfabyte_blackbox-0.1.0/blackbox/api/snapshot.py +69 -0
  24. halfabyte_blackbox-0.1.0/blackbox/api/static/inspector.html +441 -0
  25. halfabyte_blackbox-0.1.0/blackbox/api/ui.py +258 -0
  26. halfabyte_blackbox-0.1.0/blackbox/ci.py +183 -0
  27. halfabyte_blackbox-0.1.0/blackbox/cli.py +260 -0
  28. halfabyte_blackbox-0.1.0/blackbox/core/__init__.py +0 -0
  29. halfabyte_blackbox-0.1.0/blackbox/core/hashing.py +34 -0
  30. halfabyte_blackbox-0.1.0/blackbox/core/privacy.py +70 -0
  31. halfabyte_blackbox-0.1.0/blackbox/core/schema.py +111 -0
  32. halfabyte_blackbox-0.1.0/blackbox/core/store.py +239 -0
  33. halfabyte_blackbox-0.1.0/blackbox/data/__init__.py +0 -0
  34. halfabyte_blackbox-0.1.0/blackbox/data/dataset.py +218 -0
  35. halfabyte_blackbox-0.1.0/blackbox/data/download.py +69 -0
  36. halfabyte_blackbox-0.1.0/blackbox/data/farm.py +246 -0
  37. halfabyte_blackbox-0.1.0/blackbox/data/importers/__init__.py +0 -0
  38. halfabyte_blackbox-0.1.0/blackbox/data/importers/external.py +207 -0
  39. halfabyte_blackbox-0.1.0/blackbox/data/importers/who_and_when.py +82 -0
  40. halfabyte_blackbox-0.1.0/blackbox/data/shards.py +135 -0
  41. halfabyte_blackbox-0.1.0/blackbox/data/splits.py +45 -0
  42. halfabyte_blackbox-0.1.0/blackbox/data/tasks/__init__.py +21 -0
  43. halfabyte_blackbox-0.1.0/blackbox/data/tasks/gsm8k.py +39 -0
  44. halfabyte_blackbox-0.1.0/blackbox/data/tasks/hotpot.py +35 -0
  45. halfabyte_blackbox-0.1.0/blackbox/data/tasks/ops_tasks.py +121 -0
  46. halfabyte_blackbox-0.1.0/blackbox/demo.py +118 -0
  47. halfabyte_blackbox-0.1.0/blackbox/diagnosis/__init__.py +22 -0
  48. halfabyte_blackbox-0.1.0/blackbox/diagnosis/conformal.py +94 -0
  49. halfabyte_blackbox-0.1.0/blackbox/diagnosis/dataset.py +112 -0
  50. halfabyte_blackbox-0.1.0/blackbox/diagnosis/diagnose.py +176 -0
  51. halfabyte_blackbox-0.1.0/blackbox/diagnosis/embeddings.py +85 -0
  52. halfabyte_blackbox-0.1.0/blackbox/diagnosis/evidence.py +100 -0
  53. halfabyte_blackbox-0.1.0/blackbox/diagnosis/explain.py +72 -0
  54. halfabyte_blackbox-0.1.0/blackbox/diagnosis/features.py +216 -0
  55. halfabyte_blackbox-0.1.0/blackbox/diagnosis/fingerprint.py +139 -0
  56. halfabyte_blackbox-0.1.0/blackbox/diagnosis/gnn.py +582 -0
  57. halfabyte_blackbox-0.1.0/blackbox/diagnosis/graph_data.py +263 -0
  58. halfabyte_blackbox-0.1.0/blackbox/diagnosis/predict.py +23 -0
  59. halfabyte_blackbox-0.1.0/blackbox/diagnosis/ranker.py +358 -0
  60. halfabyte_blackbox-0.1.0/blackbox/eval/__init__.py +0 -0
  61. halfabyte_blackbox-0.1.0/blackbox/eval/baselines.py +32 -0
  62. halfabyte_blackbox-0.1.0/blackbox/eval/llm_judge.py +157 -0
  63. halfabyte_blackbox-0.1.0/blackbox/eval/metrics.py +74 -0
  64. halfabyte_blackbox-0.1.0/blackbox/eval/report.py +350 -0
  65. halfabyte_blackbox-0.1.0/blackbox/faults/__init__.py +0 -0
  66. halfabyte_blackbox-0.1.0/blackbox/faults/catalog.py +426 -0
  67. halfabyte_blackbox-0.1.0/blackbox/guardian/__init__.py +14 -0
  68. halfabyte_blackbox-0.1.0/blackbox/guardian/channels.py +230 -0
  69. halfabyte_blackbox-0.1.0/blackbox/guardian/incident.py +44 -0
  70. halfabyte_blackbox-0.1.0/blackbox/guardian/render.py +192 -0
  71. halfabyte_blackbox-0.1.0/blackbox/guardian/service.py +238 -0
  72. halfabyte_blackbox-0.1.0/blackbox/guardian/setup.py +135 -0
  73. halfabyte_blackbox-0.1.0/blackbox/instrument.py +73 -0
  74. halfabyte_blackbox-0.1.0/blackbox/integrations.py +445 -0
  75. halfabyte_blackbox-0.1.0/blackbox/lab.py +353 -0
  76. halfabyte_blackbox-0.1.0/blackbox/mcp/__init__.py +19 -0
  77. halfabyte_blackbox-0.1.0/blackbox/mcp/__main__.py +5 -0
  78. halfabyte_blackbox-0.1.0/blackbox/mcp/prompts.py +48 -0
  79. halfabyte_blackbox-0.1.0/blackbox/mcp/protocol.py +30 -0
  80. halfabyte_blackbox-0.1.0/blackbox/mcp/registry.py +109 -0
  81. halfabyte_blackbox-0.1.0/blackbox/mcp/resources.py +57 -0
  82. halfabyte_blackbox-0.1.0/blackbox/mcp/server.py +117 -0
  83. halfabyte_blackbox-0.1.0/blackbox/mcp/tools/__init__.py +3 -0
  84. halfabyte_blackbox-0.1.0/blackbox/mcp/tools/developer.py +59 -0
  85. halfabyte_blackbox-0.1.0/blackbox/mcp/tools/diagnosis.py +176 -0
  86. halfabyte_blackbox-0.1.0/blackbox/mcp/tools/experiments.py +148 -0
  87. halfabyte_blackbox-0.1.0/blackbox/mcp/tools/guardian.py +42 -0
  88. halfabyte_blackbox-0.1.0/blackbox/mcp/tools/runs.py +94 -0
  89. halfabyte_blackbox-0.1.0/blackbox/monitor/__init__.py +4 -0
  90. halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/__init__.py +1 -0
  91. halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/jev_adapter.py +92 -0
  92. halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/laya_adapter.py +98 -0
  93. halfabyte_blackbox-0.1.0/blackbox/monitor/adapters/llm_adapter.py +65 -0
  94. halfabyte_blackbox-0.1.0/blackbox/monitor/cascade.py +146 -0
  95. halfabyte_blackbox-0.1.0/blackbox/monitor/check.py +4 -0
  96. halfabyte_blackbox-0.1.0/blackbox/monitor/finetune.py +194 -0
  97. halfabyte_blackbox-0.1.0/blackbox/monitor/questions.py +84 -0
  98. halfabyte_blackbox-0.1.0/blackbox/monitor/state_builder.py +47 -0
  99. halfabyte_blackbox-0.1.0/blackbox/recorder/__init__.py +0 -0
  100. halfabyte_blackbox-0.1.0/blackbox/recorder/context.py +517 -0
  101. halfabyte_blackbox-0.1.0/blackbox/recorder/providers.py +285 -0
  102. halfabyte_blackbox-0.1.0/blackbox/replay/__init__.py +0 -0
  103. halfabyte_blackbox-0.1.0/blackbox/replay/advisor.py +71 -0
  104. halfabyte_blackbox-0.1.0/blackbox/replay/autopilot.py +85 -0
  105. halfabyte_blackbox-0.1.0/blackbox/replay/causal.py +157 -0
  106. halfabyte_blackbox-0.1.0/blackbox/replay/chaos.py +96 -0
  107. halfabyte_blackbox-0.1.0/blackbox/replay/diff.py +102 -0
  108. halfabyte_blackbox-0.1.0/blackbox/replay/engine.py +86 -0
  109. halfabyte_blackbox-0.1.0/blackbox/replay/fleet.py +194 -0
  110. halfabyte_blackbox-0.1.0/blackbox/replay/label_natural.py +78 -0
  111. halfabyte_blackbox-0.1.0/blackbox/replay/repair.py +139 -0
  112. halfabyte_blackbox-0.1.0/blackbox/replay/verify.py +167 -0
  113. halfabyte_blackbox-0.1.0/blackbox/status.py +201 -0
  114. halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/PKG-INFO +135 -0
  115. halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/SOURCES.txt +155 -0
  116. halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/dependency_links.txt +1 -0
  117. halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/entry_points.txt +2 -0
  118. halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/requires.txt +36 -0
  119. halfabyte_blackbox-0.1.0/halfabyte_blackbox.egg-info/top_level.txt +1 -0
  120. halfabyte_blackbox-0.1.0/pyproject.toml +58 -0
  121. halfabyte_blackbox-0.1.0/setup.cfg +4 -0
  122. halfabyte_blackbox-0.1.0/tests/test_advisor_privacy.py +47 -0
  123. halfabyte_blackbox-0.1.0/tests/test_agents.py +179 -0
  124. halfabyte_blackbox-0.1.0/tests/test_api.py +169 -0
  125. halfabyte_blackbox-0.1.0/tests/test_async_integrations.py +70 -0
  126. halfabyte_blackbox-0.1.0/tests/test_causal.py +41 -0
  127. halfabyte_blackbox-0.1.0/tests/test_chaos.py +57 -0
  128. halfabyte_blackbox-0.1.0/tests/test_checker_head.py +32 -0
  129. halfabyte_blackbox-0.1.0/tests/test_checkers.py +78 -0
  130. halfabyte_blackbox-0.1.0/tests/test_ci.py +68 -0
  131. halfabyte_blackbox-0.1.0/tests/test_conformal.py +50 -0
  132. halfabyte_blackbox-0.1.0/tests/test_core_replay.py +197 -0
  133. halfabyte_blackbox-0.1.0/tests/test_dataset.py +98 -0
  134. halfabyte_blackbox-0.1.0/tests/test_eval.py +263 -0
  135. halfabyte_blackbox-0.1.0/tests/test_evidence.py +33 -0
  136. halfabyte_blackbox-0.1.0/tests/test_farm.py +179 -0
  137. halfabyte_blackbox-0.1.0/tests/test_fast_status.py +54 -0
  138. halfabyte_blackbox-0.1.0/tests/test_faults.py +153 -0
  139. halfabyte_blackbox-0.1.0/tests/test_faults_unseen.py +265 -0
  140. halfabyte_blackbox-0.1.0/tests/test_features.py +93 -0
  141. halfabyte_blackbox-0.1.0/tests/test_fingerprint.py +59 -0
  142. halfabyte_blackbox-0.1.0/tests/test_fleet.py +63 -0
  143. halfabyte_blackbox-0.1.0/tests/test_gnn.py +78 -0
  144. halfabyte_blackbox-0.1.0/tests/test_guardian.py +122 -0
  145. halfabyte_blackbox-0.1.0/tests/test_guardian_channels.py +146 -0
  146. halfabyte_blackbox-0.1.0/tests/test_guardian_setup.py +73 -0
  147. halfabyte_blackbox-0.1.0/tests/test_importers_external.py +68 -0
  148. halfabyte_blackbox-0.1.0/tests/test_integrations.py +117 -0
  149. halfabyte_blackbox-0.1.0/tests/test_live_ollama.py +71 -0
  150. halfabyte_blackbox-0.1.0/tests/test_mcp.py +161 -0
  151. halfabyte_blackbox-0.1.0/tests/test_monitor.py +291 -0
  152. halfabyte_blackbox-0.1.0/tests/test_ranker.py +103 -0
  153. halfabyte_blackbox-0.1.0/tests/test_shards.py +52 -0
  154. halfabyte_blackbox-0.1.0/tests/test_shop.py +258 -0
  155. halfabyte_blackbox-0.1.0/tests/test_tools.py +116 -0
  156. halfabyte_blackbox-0.1.0/tests/test_ui_api.py +54 -0
  157. halfabyte_blackbox-0.1.0/tests/test_verify_autopilot_repair.py +191 -0
@@ -0,0 +1,135 @@
1
+ Metadata-Version: 2.4
2
+ Name: halfabyte-blackbox
3
+ Version: 0.1.0
4
+ Summary: A flight recorder for AI agents: record every step, find the step that caused a failure, prove it by replay, and fix it without regressions.
5
+ Author: Aryan Lomte, Radhesh, Aditya, Advay Chavan
6
+ License: MIT
7
+ Keywords: ai agents,llm,observability,debugging,replay,root cause analysis,mcp,evaluation
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
17
+ Classifier: Topic :: Software Development :: Debuggers
18
+ Classifier: Topic :: Software Development :: Testing
19
+ Requires-Python: >=3.10
20
+ Description-Content-Type: text/markdown
21
+ Requires-Dist: pydantic>=2.5
22
+ Requires-Dist: pyyaml>=6
23
+ Requires-Dist: python-dotenv>=1
24
+ Requires-Dist: openai>=1.40
25
+ Provides-Extra: checker
26
+ Requires-Dist: laya; extra == "checker"
27
+ Requires-Dist: system-one-adapter[openai]; extra == "checker"
28
+ Provides-Extra: jev
29
+ Requires-Dist: typesafe-sdk; extra == "jev"
30
+ Provides-Extra: ml
31
+ Requires-Dist: lightgbm>=4; extra == "ml"
32
+ Requires-Dist: scikit-learn>=1.4; extra == "ml"
33
+ Requires-Dist: shap>=0.45; extra == "ml"
34
+ Requires-Dist: sentence-transformers>=3; extra == "ml"
35
+ Requires-Dist: numpy; extra == "ml"
36
+ Provides-Extra: ui
37
+ Requires-Dist: fastapi>=0.110; extra == "ui"
38
+ Requires-Dist: uvicorn[standard]>=0.29; extra == "ui"
39
+ Requires-Dist: sse-starlette>=2; extra == "ui"
40
+ Provides-Extra: agents
41
+ Requires-Dist: rank-bm25>=0.2; extra == "agents"
42
+ Provides-Extra: data
43
+ Requires-Dist: datasets>=2.18; extra == "data"
44
+ Provides-Extra: dev
45
+ Requires-Dist: pytest>=8; extra == "dev"
46
+ Requires-Dist: ruff>=0.5; extra == "dev"
47
+ Provides-Extra: all
48
+ Requires-Dist: halfabyte-blackbox[agents,checker,data,ml,ui]; extra == "all"
49
+
50
+ # Black Box: a flight recorder for AI agents
51
+
52
+ When an AI agent fails, the mistake usually happened several steps before the wrong answer. Black Box records
53
+ every step of an agent run, **finds the step that caused the failure, and proves it**: it re-runs the agent from
54
+ its recording with only that step repaired. If the run now passes, that step was the cause. Unchanged steps come
55
+ from the recording, so a replay costs zero model calls and a fix re-runs only what changed.
56
+
57
+ ```bash
58
+ pip install halfabyte-blackbox # recorder, replay, proof, CLI, CI, MCP server, trace import
59
+ pip install "halfabyte-blackbox[ui]" # + the API / dashboard backend
60
+ pip install "halfabyte-blackbox[all]" # + step checker, ranking model, test agents, dataset loaders
61
+ ```
62
+
63
+ ## Add it to an existing agent (3 lines)
64
+
65
+ ```python
66
+ import blackbox
67
+ from openai import OpenAI
68
+
69
+ client = blackbox.wrap(OpenAI()) # 1. every model call is recorded
70
+
71
+ @blackbox.tool # 2. every tool call is recorded and replayable
72
+ def get_weather(city: str) -> dict:
73
+ ...
74
+
75
+ def my_agent(question: str) -> str: # your agent, unchanged
76
+ ...
77
+
78
+ trace = blackbox.run(my_agent, "Will it rain in Pune?") # 3. run it under the recorder
79
+ ```
80
+
81
+ ```python
82
+ blackbox.replay(trace) # re-run from the recording: 0 model calls
83
+ child = blackbox.fork(trace, 2, blackbox.Change(kind="output", value={"result": {"rain": True}}))
84
+ blackbox.savings(child) # steps re-run vs re-used, tokens saved
85
+ ```
86
+
87
+ Works with agents you did not write (shown on Hugging Face smolagents), and with traces you already export:
88
+ `blackbox import otel traces.json` / `blackbox import langfuse trace.json`.
89
+
90
+ ## What it does
91
+
92
+ | Capability | How |
93
+ |---|---|
94
+ | **Record** every model call, tool call, search and memory change, with a save-point after each step | `blackbox.wrap`, `@blackbox.tool`, `blackbox.run` |
95
+ | **Diagnose**: rank the steps most likely to have caused a failure, with plain-language evidence | `blackbox show <run>`, API `/api/runs/{id}/diagnosis` |
96
+ | **Prove** the cause by re-running with one suspect repaired at a time | Verify (API, MCP `verify`) |
97
+ | **Causal report**: necessary vs sufficient, joint causes, every recovery path, blast radius | API `/causal`, MCP `causal_report` |
98
+ | **One fix for many**: test one rule on similar past failures and passing runs; APPROVE only if nothing breaks | `blackbox fleet <run>` |
99
+ | **Crash test**: plant every known kind of mistake into a working agent, grade it 1–5 stars | `blackbox crash-test <run>` |
100
+ | **Seen this before?**: failure fingerprints, look-alike failures, novel failures, known fixes | MCP `similar_failures` |
101
+ | **Guardian**: incidents in plain words on email, WhatsApp, Slack, Telegram; money-moving agents paused until approved | `blackbox.Guardian(...)`, `blackbox guardian test` |
102
+ | **Black Box CI**: replay pinned recorded runs on every pull request; fail on regressions | `blackbox ci pin` / `blackbox ci run`, GitHub Action |
103
+ | **MCP server**: let Claude Code, Cursor or VS Code investigate failures with 28 tools | `blackbox mcp` |
104
+ | **Live Lab**: ask any question, plant a mistake live, watch the 14-stage investigation | `blackbox lab "question"` |
105
+
106
+ Safety: tools that move money never execute during any re-run, and Guardian never retries them without a human.
107
+
108
+ ## Command line
109
+
110
+ ```bash
111
+ blackbox runs --fail # failed runs
112
+ blackbox show <run_id> # step by step, with who-used-whose-output links
113
+ blackbox replay <run_id> # replay from the recording
114
+ blackbox crash-test <run_id> # red-team a passing run
115
+ blackbox fleet <run_id> --rule "..." # one fix for many, with a regression firewall
116
+ blackbox ci pin && blackbox ci run # regression firewall for agent code
117
+ blackbox guardian channels # which alert channels are configured
118
+ blackbox mcp # MCP server on stdio
119
+ blackbox ui # API on http://127.0.0.1:8000
120
+ ```
121
+
122
+ Models are named in one file, `models.yaml` (Ollama locally, Groq or OpenRouter hosted); keys live in `.env`.
123
+
124
+ ## Development (this repository)
125
+
126
+ ```bash
127
+ pip install -e ".[all,dev]"
128
+ py -3 -m pytest -q -m "not live"
129
+ cd web && npm install && npm run dev # dashboard on http://localhost:5173
130
+ ```
131
+
132
+ Design: `SYSTEM.md` · rules: `CLAUDE.md` · UI contract: `docs/API_FOR_UI.md` · MCP: `docs/MCP.md` · CI: `docs/CI.md` ·
133
+ Guardian setup: `docs/GUARDIAN_SETUP.md`. Branches `<name>/<feature>` → PR into `dev` → `main` at milestones.
134
+
135
+ Built by team Half a Byte: Aryan Lomte, Radhesh, Aditya, Advay Chavan.
@@ -0,0 +1,86 @@
1
+ # Black Box: a flight recorder for AI agents
2
+
3
+ When an AI agent fails, the mistake usually happened several steps before the wrong answer. Black Box records
4
+ every step of an agent run, **finds the step that caused the failure, and proves it**: it re-runs the agent from
5
+ its recording with only that step repaired. If the run now passes, that step was the cause. Unchanged steps come
6
+ from the recording, so a replay costs zero model calls and a fix re-runs only what changed.
7
+
8
+ ```bash
9
+ pip install halfabyte-blackbox # recorder, replay, proof, CLI, CI, MCP server, trace import
10
+ pip install "halfabyte-blackbox[ui]" # + the API / dashboard backend
11
+ pip install "halfabyte-blackbox[all]" # + step checker, ranking model, test agents, dataset loaders
12
+ ```
13
+
14
+ ## Add it to an existing agent (3 lines)
15
+
16
+ ```python
17
+ import blackbox
18
+ from openai import OpenAI
19
+
20
+ client = blackbox.wrap(OpenAI()) # 1. every model call is recorded
21
+
22
+ @blackbox.tool # 2. every tool call is recorded and replayable
23
+ def get_weather(city: str) -> dict:
24
+ ...
25
+
26
+ def my_agent(question: str) -> str: # your agent, unchanged
27
+ ...
28
+
29
+ trace = blackbox.run(my_agent, "Will it rain in Pune?") # 3. run it under the recorder
30
+ ```
31
+
32
+ ```python
33
+ blackbox.replay(trace) # re-run from the recording: 0 model calls
34
+ child = blackbox.fork(trace, 2, blackbox.Change(kind="output", value={"result": {"rain": True}}))
35
+ blackbox.savings(child) # steps re-run vs re-used, tokens saved
36
+ ```
37
+
38
+ Works with agents you did not write (shown on Hugging Face smolagents), and with traces you already export:
39
+ `blackbox import otel traces.json` / `blackbox import langfuse trace.json`.
40
+
41
+ ## What it does
42
+
43
+ | Capability | How |
44
+ |---|---|
45
+ | **Record** every model call, tool call, search and memory change, with a save-point after each step | `blackbox.wrap`, `@blackbox.tool`, `blackbox.run` |
46
+ | **Diagnose**: rank the steps most likely to have caused a failure, with plain-language evidence | `blackbox show <run>`, API `/api/runs/{id}/diagnosis` |
47
+ | **Prove** the cause by re-running with one suspect repaired at a time | Verify (API, MCP `verify`) |
48
+ | **Causal report**: necessary vs sufficient, joint causes, every recovery path, blast radius | API `/causal`, MCP `causal_report` |
49
+ | **One fix for many**: test one rule on similar past failures and passing runs; APPROVE only if nothing breaks | `blackbox fleet <run>` |
50
+ | **Crash test**: plant every known kind of mistake into a working agent, grade it 1–5 stars | `blackbox crash-test <run>` |
51
+ | **Seen this before?**: failure fingerprints, look-alike failures, novel failures, known fixes | MCP `similar_failures` |
52
+ | **Guardian**: incidents in plain words on email, WhatsApp, Slack, Telegram; money-moving agents paused until approved | `blackbox.Guardian(...)`, `blackbox guardian test` |
53
+ | **Black Box CI**: replay pinned recorded runs on every pull request; fail on regressions | `blackbox ci pin` / `blackbox ci run`, GitHub Action |
54
+ | **MCP server**: let Claude Code, Cursor or VS Code investigate failures with 28 tools | `blackbox mcp` |
55
+ | **Live Lab**: ask any question, plant a mistake live, watch the 14-stage investigation | `blackbox lab "question"` |
56
+
57
+ Safety: tools that move money never execute during any re-run, and Guardian never retries them without a human.
58
+
59
+ ## Command line
60
+
61
+ ```bash
62
+ blackbox runs --fail # failed runs
63
+ blackbox show <run_id> # step by step, with who-used-whose-output links
64
+ blackbox replay <run_id> # replay from the recording
65
+ blackbox crash-test <run_id> # red-team a passing run
66
+ blackbox fleet <run_id> --rule "..." # one fix for many, with a regression firewall
67
+ blackbox ci pin && blackbox ci run # regression firewall for agent code
68
+ blackbox guardian channels # which alert channels are configured
69
+ blackbox mcp # MCP server on stdio
70
+ blackbox ui # API on http://127.0.0.1:8000
71
+ ```
72
+
73
+ Models are named in one file, `models.yaml` (Ollama locally, Groq or OpenRouter hosted); keys live in `.env`.
74
+
75
+ ## Development (this repository)
76
+
77
+ ```bash
78
+ pip install -e ".[all,dev]"
79
+ py -3 -m pytest -q -m "not live"
80
+ cd web && npm install && npm run dev # dashboard on http://localhost:5173
81
+ ```
82
+
83
+ Design: `SYSTEM.md` · rules: `CLAUDE.md` · UI contract: `docs/API_FOR_UI.md` · MCP: `docs/MCP.md` · CI: `docs/CI.md` ·
84
+ Guardian setup: `docs/GUARDIAN_SETUP.md`. Branches `<name>/<feature>` → PR into `dev` → `main` at milestones.
85
+
86
+ Built by team Half a Byte: Aryan Lomte, Radhesh, Aditya, Advay Chavan.
@@ -0,0 +1,36 @@
1
+ """Black Box: a flight recorder for AI agents.
2
+
3
+ Quick start for an existing project (see README.md):
4
+
5
+ import blackbox
6
+ client = blackbox.wrap(OpenAI())
7
+ trace = blackbox.run(my_agent, "question")
8
+ """
9
+
10
+ from blackbox.agents.base import Agent, Task, register_agent, register_checker, run_task
11
+ from blackbox.core.schema import Change, Outcome, Run, Step, Trace
12
+ from blackbox.guardian import Guardian, Incident
13
+ from blackbox.core.store import Store
14
+ from blackbox.integrations import (
15
+ Session,
16
+ afork,
17
+ areplay,
18
+ arun,
19
+ configure,
20
+ default_store,
21
+ retriever,
22
+ run,
23
+ session,
24
+ tool,
25
+ wrap,
26
+ )
27
+ from blackbox.recorder.context import RunContext, ToolError
28
+ from blackbox.replay.engine import fork, replay, savings
29
+
30
+ __version__ = "0.1.0"
31
+
32
+ __all__ = [
33
+ "Agent", "Change", "Guardian", "Incident", "afork", "areplay", "arun", "Outcome", "Run", "RunContext", "Session", "Step", "Store", "Task", "ToolError",
34
+ "Trace", "configure", "default_store", "fork", "register_agent", "register_checker", "replay",
35
+ "retriever", "run", "run_task", "savings", "session", "tool", "wrap",
36
+ ]
File without changes
@@ -0,0 +1,223 @@
1
+ """Agents are resumable state machines (CLAUDE.md hard rule 3).
2
+
3
+ class QAAgent(Agent):
4
+ name, version, family, entry = "qa_agent", "v1", "qa", "plan"
5
+ def init_state(self, task): return {"question": task.input["question"]}
6
+ def plan(self, ctx): ...; return "search" # next node name
7
+ def search(self, ctx): ...; return "answer"
8
+ def answer(self, ctx): ctx.final(...); return None # None = finished
9
+ nodes = {"plan": plan, "search": search, "answer": answer}
10
+
11
+ Rules:
12
+ - Every model/tool/search call goes through ctx (ctx.llm / ctx.tool / ctx.retrieve).
13
+ - All memory lives in ctx.state; assign values (state[k] = v), never mutate in place.
14
+ - Randomness: ctx.rng. Time: ctx.now(). Nothing else.
15
+ - Finish by calling ctx.final(answer) and returning None.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ import socket
22
+ import traceback
23
+ import uuid
24
+ from dataclasses import dataclass, field
25
+ from typing import Any, Callable, ClassVar
26
+
27
+ from blackbox.core.schema import Change, Outcome, Run, Trace
28
+ from blackbox.core.store import Store
29
+ from blackbox.recorder.context import AutopilotRewind, Monitor, ReplayPlan, RunContext, StepListener, current
30
+
31
+ MAX_NODE_VISITS = 60
32
+
33
+ Checker = Callable[[Any, Any], Outcome] # (final_answer, expected) -> Outcome
34
+
35
+
36
+ @dataclass
37
+ class Task:
38
+ family: str
39
+ task_id: str
40
+ input: dict # what the agent may see
41
+ expected: Any # what the checker compares against; the agent never sees it
42
+ meta: dict = field(default_factory=dict)
43
+
44
+
45
+ class Agent:
46
+ name: ClassVar[str] = "base"
47
+ version: ClassVar[str] = "v1"
48
+ family: ClassVar[str] = ""
49
+ entry: ClassVar[str] = ""
50
+ nodes: ClassVar[dict[str, Callable[["Agent", RunContext], str | None]]] = {}
51
+
52
+ def init_state(self, task: Task) -> dict:
53
+ return {}
54
+
55
+ def tool_caller(self, task: Task) -> Callable[[str, dict], Any] | None:
56
+ return None
57
+
58
+ def retriever(self, task: Task) -> Callable[..., list[dict]] | None:
59
+ return None
60
+
61
+ @classmethod
62
+ def ref(cls) -> str:
63
+ return f"{cls.name}@{cls.version}"
64
+
65
+
66
+ # ---- registries (agents and checkers are looked up by name on replay) ----------
67
+ _AGENTS: dict[str, type[Agent]] = {}
68
+ _CHECKERS: dict[str, Checker] = {}
69
+
70
+
71
+ def register_agent(cls: type[Agent]) -> type[Agent]:
72
+ _AGENTS[cls.ref()] = cls
73
+ return cls
74
+
75
+
76
+ def register_checker(family: str, fn: Checker) -> Checker:
77
+ _CHECKERS[family] = fn
78
+ return fn
79
+
80
+
81
+ def _load_builtin() -> None:
82
+ """Register our test agents on first miss, so a process that only replays
83
+ stored runs (API, CLI) finds them. Skipped when the `agents` extra is absent."""
84
+ try:
85
+ import blackbox.agents.builtin # noqa: F401
86
+ except ModuleNotFoundError as exc:
87
+ if exc.name is None or exc.name.startswith("blackbox"):
88
+ raise
89
+
90
+
91
+ def get_agent(ref: str) -> Agent:
92
+ if ref not in _AGENTS and ref.endswith("@fn"):
93
+ from blackbox.integrations import resolve_function_agent # user function agents
94
+
95
+ resolve_function_agent(ref)
96
+ if ref not in _AGENTS:
97
+ _load_builtin()
98
+ if ref not in _AGENTS:
99
+ raise KeyError(f"agent {ref!r} not registered (import its module). Known: {sorted(_AGENTS)}")
100
+ return _AGENTS[ref]()
101
+
102
+
103
+ def get_checker(family: str) -> Checker | None:
104
+ """None for a user's project that registered no checker: the run is recorded
105
+ without a pass/fail verdict rather than with an invented one."""
106
+ return _CHECKERS.get(family)
107
+
108
+
109
+ def machine_name() -> str:
110
+ return os.environ.get("BLACKBOX_MACHINE") or socket.gethostname()
111
+
112
+
113
+ def run_task(
114
+ agent: Agent,
115
+ task: Task,
116
+ model_id: str,
117
+ *,
118
+ store: Store | None = None,
119
+ plan: ReplayPlan | None = None,
120
+ monitor: Monitor | None = None,
121
+ on_step: list[StepListener] | None = None,
122
+ run_id: str | None = None,
123
+ parent_run_id: str | None = None,
124
+ fork_step_idx: int | None = None,
125
+ patch: Change | None = None,
126
+ split: str | None = None,
127
+ autopilot=None,
128
+ ) -> Trace:
129
+ """Run (or replay/fork) one task and return its trace. Saved if a store is given.
130
+
131
+ A crash (exception escaping the agent) marks the run "crashed" with the
132
+ traceback. It is our bug, never a task failure, and never a label
133
+ (CLAUDE.md review rule 3).
134
+ """
135
+ run = Run(
136
+ run_id=run_id or uuid.uuid4().hex[:16],
137
+ task_family=task.family,
138
+ task_id=task.task_id,
139
+ task_input=task.input,
140
+ agent=agent.ref(),
141
+ model_id=model_id,
142
+ parent_run_id=parent_run_id,
143
+ fork_step_idx=fork_step_idx,
144
+ patch=patch,
145
+ split=split,
146
+ machine=machine_name(),
147
+ task_expected=task.expected,
148
+ )
149
+ ctx = RunContext(
150
+ run, agent.init_state(task), store=store,
151
+ tool_caller=agent.tool_caller(task), retriever=agent.retriever(task),
152
+ plan=plan, monitor=monitor, on_step=on_step, autopilot=autopilot,
153
+ )
154
+ token = current.set(ctx)
155
+ try:
156
+ node: str | None = agent.entry
157
+ visits = 0
158
+ while node is not None:
159
+ if node not in agent.nodes:
160
+ raise KeyError(f"{agent.ref()}: unknown node {node!r}")
161
+ visits += 1
162
+ if visits > MAX_NODE_VISITS:
163
+ # Looping forever is a genuine agent failure, not a crash.
164
+ run.stats["stopped"] = f"exceeded {MAX_NODE_VISITS} node visits"
165
+ if not ctx.steps or ctx.steps[-1].kind != "final":
166
+ ctx.final(None)
167
+ break
168
+ ctx.node = node
169
+ node = agent.nodes[node](agent, ctx)
170
+ ctx.close()
171
+ final_steps = [s for s in ctx.steps if s.kind == "final"]
172
+ answer = final_steps[-1].output["answer"] if final_steps else None
173
+ checker = get_checker(task.family)
174
+ run.outcome = checker(answer, task.expected) if checker else None
175
+ run.status = "done"
176
+ except AutopilotRewind as rewind:
177
+ current.reset(token)
178
+ token = None
179
+ return _take_over(agent, task, model_id, ctx, run, rewind, store=store, monitor=monitor,
180
+ on_step=on_step, split=split, autopilot=autopilot)
181
+ except Exception: # noqa: BLE001 - recorded loudly as a crash, see docstring
182
+ ctx.close(monitor_timeout=5.0)
183
+ run.status = "crashed"
184
+ run.error = traceback.format_exc()
185
+ finally:
186
+ if token is not None:
187
+ current.reset(token)
188
+ run.stats = {**ctx.stats(), **run.stats}
189
+ trace = Trace(run=run, steps=ctx.steps)
190
+ if store is not None:
191
+ store.save_trace(trace)
192
+ return trace
193
+
194
+
195
+ def _take_over(agent, task, model_id, ctx, run, rewind: AutopilotRewind, *, store, monitor, on_step, split,
196
+ autopilot) -> Trace:
197
+ """Autopilot (USP X1): the run is stopped before it can answer. Its steps so
198
+ far are kept as a 'rewound' run, and a fork from the flagged decision takes
199
+ over, re-using everything before it from the recording."""
200
+ ctx.close(monitor_timeout=5.0)
201
+ flagged = ctx.steps[rewind.step_idx]
202
+ flagged.monitor = rewind.answers if rewind.answers is not None else flagged.monitor
203
+ run.status = "rewound"
204
+ run.stats = {**ctx.stats(), "autopilot": {"flagged_step": rewind.step_idx, "rewound_to": rewind.target_idx}}
205
+ partial = Trace(run=run, steps=ctx.steps)
206
+ if store is not None:
207
+ store.save_trace(partial)
208
+ child = run_task(
209
+ agent, task, model_id, store=store, monitor=monitor, on_step=on_step, split=split,
210
+ plan=ReplayPlan(parent=partial, fork_idx=rewind.target_idx, change=rewind.change),
211
+ parent_run_id=run.run_id, fork_step_idx=rewind.target_idx, patch=rewind.change, autopilot=autopilot,
212
+ )
213
+ child.run.stats.setdefault("autopilot_rescue", {
214
+ "from_run": run.run_id, "flagged_step": rewind.step_idx, "rewound_to": rewind.target_idx,
215
+ "change": rewind.change.model_dump(mode="json") if rewind.change else None,
216
+ })
217
+ if store is not None:
218
+ store.save_trace(child)
219
+ return child
220
+
221
+
222
+ def task_from_run(run: Run) -> Task:
223
+ return Task(family=run.task_family, task_id=run.task_id, input=run.task_input, expected=run.task_expected)
@@ -0,0 +1,7 @@
1
+ """Importing this module registers our test agents and their checkers.
2
+
3
+ Replay looks agents up by name (`qa_agent@v1`), so any process that replays
4
+ or forks a stored run (farm, API, CLI) needs them registered first.
5
+ """
6
+
7
+ from blackbox.agents import checkers, math_agent, qa, shop # noqa: F401 - registration side effect
@@ -0,0 +1,83 @@
1
+ """Pass/fail checkers. Plain code: no model decides what counts as a failure
2
+ (SYSTEM.md §3.1, CLAUDE.md hard rule 5).
3
+
4
+ QA : SQuAD-normalised word-overlap F1 >= 0.6 against the gold answer
5
+ Math : the final number equals the gold number exactly
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ import string
12
+ from collections import Counter
13
+ from typing import Any
14
+
15
+ from blackbox.agents.base import register_checker
16
+ from blackbox.core.schema import Outcome
17
+
18
+ QA_F1_THRESHOLD = 0.6
19
+
20
+ _ARTICLES = re.compile(r"\b(a|an|the)\b")
21
+ _PUNCT = str.maketrans("", "", string.punctuation)
22
+ _NUMBER = re.compile(r"-?\d[\d,]*\.?\d*|-?\.\d+")
23
+
24
+
25
+ def normalize_answer(text: Any) -> str:
26
+ """SQuAD normalisation: lowercase, drop punctuation, drop articles, squeeze spaces."""
27
+ s = "" if text is None else str(text)
28
+ s = s.lower().translate(_PUNCT)
29
+ s = _ARTICLES.sub(" ", s)
30
+ return " ".join(s.split())
31
+
32
+
33
+ def f1_score(prediction: Any, gold: Any) -> float:
34
+ pred, ref = normalize_answer(prediction).split(), normalize_answer(gold).split()
35
+ if not pred or not ref:
36
+ return float(pred == ref)
37
+ common = sum((Counter(pred) & Counter(ref)).values())
38
+ if common == 0:
39
+ return 0.0
40
+ precision, recall = common / len(pred), common / len(ref)
41
+ return 2 * precision * recall / (precision + recall)
42
+
43
+
44
+ def _gold(expected: Any) -> Any:
45
+ return expected.get("answer") if isinstance(expected, dict) else expected
46
+
47
+
48
+ def check_qa(answer: Any, expected: Any) -> Outcome:
49
+ gold = _gold(expected)
50
+ score = f1_score(answer, gold)
51
+ # yes/no questions: "yes, they are" must not pass on partial overlap, nor "no" match "yes"
52
+ if normalize_answer(gold) in ("yes", "no"):
53
+ words = normalize_answer(answer).split()
54
+ score = float(bool(words) and words[0] == normalize_answer(gold))
55
+ return Outcome(success=score >= QA_F1_THRESHOLD, score=round(score, 4),
56
+ final_answer=answer, expected=gold, checker="qa_f1")
57
+
58
+
59
+ def parse_number(value: Any) -> float | None:
60
+ """The last number in the text ("So she has $1,250.00 left." -> 1250.0)."""
61
+ if isinstance(value, bool) or value is None:
62
+ return None
63
+ if isinstance(value, (int, float)):
64
+ return float(value)
65
+ found = _NUMBER.findall(str(value))
66
+ for raw in reversed(found):
67
+ cleaned = raw.replace(",", "").rstrip(".")
68
+ try:
69
+ return float(cleaned)
70
+ except ValueError:
71
+ continue
72
+ return None
73
+
74
+
75
+ def check_math(answer: Any, expected: Any) -> Outcome:
76
+ gold = parse_number(_gold(expected))
77
+ got = parse_number(answer)
78
+ ok = gold is not None and got is not None and abs(got - gold) < 1e-6
79
+ return Outcome(success=ok, score=float(ok), final_answer=answer, expected=_gold(expected), checker="math_exact")
80
+
81
+
82
+ register_checker("qa", check_qa)
83
+ register_checker("math", check_math)
@@ -0,0 +1,104 @@
1
+ """Math agent for GSM8K: plan -> (step -> calculate)* -> check -> answer.
2
+
3
+ The model never does arithmetic itself: it names one calculation at a time
4
+ and the calculator tool computes it. A run is 6 to 15 steps:
5
+
6
+ plan(llm) [step(llm) calculator(tool)] x 1..MAX_CALCS step(llm, DONE) check(llm) final
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from blackbox.agents import checkers # noqa: F401 - registers the "math" checker
12
+ from blackbox.agents.base import Agent, Task, register_agent
13
+ from blackbox.agents.parsing import clip, field
14
+ from blackbox.agents.tools import call_tool
15
+ from blackbox.recorder.context import RunContext, ToolError
16
+
17
+ MAX_CALCS = 5
18
+ MAX_TURNS = 6 # step-node visits, including the one that says DONE
19
+
20
+ PLAN_SYSTEM = (
21
+ "You plan how to solve a math word problem. List the calculations needed, in order, "
22
+ "one short line each: what is being found and from which numbers in the problem. "
23
+ "Do not compute the results. At most 5 lines. The last line must find what the question asks."
24
+ )
25
+ STEP_SYSTEM = (
26
+ "You solve a math word problem one calculation at a time. A calculator does the arithmetic. "
27
+ "Do the next line of the plan that has no result yet. Reply with exactly one line:\n"
28
+ "CALC: <one arithmetic expression using only numbers and + - * / ( )>\n"
29
+ "or, when every line of the plan has a result:\nDONE"
30
+ )
31
+ CHECK_SYSTEM = (
32
+ "You check the solution of a math word problem. Given the problem and the calculator results, "
33
+ "state the final answer to the question asked. Reply with exactly one line:\n"
34
+ "FINAL: <number only, no units>"
35
+ )
36
+
37
+
38
+ def _format_results(results: list[dict]) -> str:
39
+ if not results:
40
+ return "(none yet)"
41
+ return "\n".join(f"{i + 1}. {r['expr']} = {r['value']}" for i, r in enumerate(results))
42
+
43
+
44
+ @register_agent
45
+ class MathAgent(Agent):
46
+ name, version, family, entry = "math_agent", "v1", "math", "plan"
47
+
48
+ def init_state(self, task: Task) -> dict:
49
+ return {"question": task.input["question"], "results": [], "turns": 0}
50
+
51
+ def tool_caller(self, task: Task):
52
+ return call_tool
53
+
54
+ def plan(self, ctx: RunContext) -> str:
55
+ res = ctx.llm(ctx.model_id, [
56
+ {"role": "system", "content": PLAN_SYSTEM},
57
+ {"role": "user", "content": f"Problem: {ctx.state['question']}"},
58
+ ], max_tokens=160)
59
+ ctx.state["plan"] = clip(res.text, 600)
60
+ return "step"
61
+
62
+ def step(self, ctx: RunContext) -> str:
63
+ results, turns = ctx.state["results"], ctx.state["turns"]
64
+ res = ctx.llm(ctx.model_id, [
65
+ {"role": "system", "content": STEP_SYSTEM},
66
+ {"role": "user", "content": (
67
+ f"Problem: {ctx.state['question']}\n\nPlan:\n{ctx.state['plan']}\n\n"
68
+ f"Results so far:\n{_format_results(results)}"
69
+ )},
70
+ ], max_tokens=48)
71
+ expr = field(res.text, "CALC")
72
+ ctx.state["turns"] = turns + 1
73
+ if expr is None or len(results) >= MAX_CALCS or turns + 1 >= MAX_TURNS:
74
+ return "check"
75
+ ctx.state["expr"] = expr
76
+ return "calculate"
77
+
78
+ def calculate(self, ctx: RunContext) -> str:
79
+ expr = ctx.state["expr"]
80
+ try:
81
+ value = ctx.tool("calculator", expression=expr)
82
+ except ToolError as exc:
83
+ value = f"error: {clip(str(exc), 120)}" # recorded on the step; the model sees it next turn
84
+ # read after the call: the calculator depends on `expr` only, so its
85
+ # parents and its re-use fingerprint must not include earlier results
86
+ ctx.state["results"] = ctx.state["results"] + [{"expr": expr, "value": value}]
87
+ return "step"
88
+
89
+ def check(self, ctx: RunContext) -> str:
90
+ results = ctx.state["results"]
91
+ res = ctx.llm(ctx.model_id, [
92
+ {"role": "system", "content": CHECK_SYSTEM},
93
+ {"role": "user", "content": (
94
+ f"Problem: {ctx.state['question']}\n\nCalculator results:\n{_format_results(results)}"
95
+ )},
96
+ ], max_tokens=32)
97
+ ctx.state["answer"] = field(res.text, "FINAL") or clip(res.text, 80)
98
+ return "answer"
99
+
100
+ def answer(self, ctx: RunContext) -> None:
101
+ ctx.final(ctx.state["answer"])
102
+ return None
103
+
104
+ nodes = {"plan": plan, "step": step, "calculate": calculate, "check": check, "answer": answer}