vex-harness 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. vex_harness-0.1.0/PKG-INFO +284 -0
  2. vex_harness-0.1.0/README.md +253 -0
  3. vex_harness-0.1.0/cli/__init__.py +8 -0
  4. vex_harness-0.1.0/cli/__main__.py +24 -0
  5. vex_harness-0.1.0/cli/_stubs/__init__.py +5 -0
  6. vex_harness-0.1.0/cli/_stubs/scheduler.py +87 -0
  7. vex_harness-0.1.0/cli/deps.py +67 -0
  8. vex_harness-0.1.0/cli/errors.py +217 -0
  9. vex_harness-0.1.0/cli/interactive.py +1066 -0
  10. vex_harness-0.1.0/cli/main.py +1188 -0
  11. vex_harness-0.1.0/cli/ui.py +172 -0
  12. vex_harness-0.1.0/cli/vexconfig.py +162 -0
  13. vex_harness-0.1.0/dashboard/__init__.py +5 -0
  14. vex_harness-0.1.0/dashboard/__main__.py +5 -0
  15. vex_harness-0.1.0/dashboard/collect.py +299 -0
  16. vex_harness-0.1.0/dashboard/server.py +291 -0
  17. vex_harness-0.1.0/execution/__init__.py +17 -0
  18. vex_harness-0.1.0/execution/env_snapshot.py +249 -0
  19. vex_harness-0.1.0/execution/feedback.py +565 -0
  20. vex_harness-0.1.0/execution/git_output.py +288 -0
  21. vex_harness-0.1.0/execution/rationale.py +179 -0
  22. vex_harness-0.1.0/execution/sandbox.py +979 -0
  23. vex_harness-0.1.0/execution/sandbox_adversarial.py +784 -0
  24. vex_harness-0.1.0/execution/sandbox_perf.py +298 -0
  25. vex_harness-0.1.0/execution/sandbox_stress.py +319 -0
  26. vex_harness-0.1.0/execution/sandbox_sustained.py +461 -0
  27. vex_harness-0.1.0/execution/verify.py +177 -0
  28. vex_harness-0.1.0/harness/__init__.py +4 -0
  29. vex_harness-0.1.0/harness/_stubs/__init__.py +6 -0
  30. vex_harness-0.1.0/harness/_stubs/model_router.py +98 -0
  31. vex_harness-0.1.0/harness/_stubs/sandbox.py +103 -0
  32. vex_harness-0.1.0/harness/_stubs/scripted_model.py +107 -0
  33. vex_harness-0.1.0/harness/_stubs/verify.py +128 -0
  34. vex_harness-0.1.0/harness/ablation_agenttests.py +371 -0
  35. vex_harness-0.1.0/harness/agent_tests.py +251 -0
  36. vex_harness-0.1.0/harness/config.py +162 -0
  37. vex_harness-0.1.0/harness/context.py +332 -0
  38. vex_harness-0.1.0/harness/coordination.py +395 -0
  39. vex_harness-0.1.0/harness/core.py +1793 -0
  40. vex_harness-0.1.0/harness/decision_memory.py +133 -0
  41. vex_harness-0.1.0/harness/deps.py +144 -0
  42. vex_harness-0.1.0/harness/docs_lookup.py +295 -0
  43. vex_harness-0.1.0/harness/editor.py +357 -0
  44. vex_harness-0.1.0/harness/lint.py +273 -0
  45. vex_harness-0.1.0/harness/model_client.py +98 -0
  46. vex_harness-0.1.0/harness/prompts.py +461 -0
  47. vex_harness-0.1.0/harness/retrieval.py +420 -0
  48. vex_harness-0.1.0/harness/state_machine.py +326 -0
  49. vex_harness-0.1.0/harness/tool_errors.py +329 -0
  50. vex_harness-0.1.0/harness/tools.py +378 -0
  51. vex_harness-0.1.0/harness/trace.py +136 -0
  52. vex_harness-0.1.0/harness/webfetch.py +499 -0
  53. vex_harness-0.1.0/mcp_server/__init__.py +5 -0
  54. vex_harness-0.1.0/mcp_server/__main__.py +5 -0
  55. vex_harness-0.1.0/mcp_server/server.py +246 -0
  56. vex_harness-0.1.0/memory/__init__.py +7 -0
  57. vex_harness-0.1.0/memory/code_graph.py +852 -0
  58. vex_harness-0.1.0/memory/decision_store.py +367 -0
  59. vex_harness-0.1.0/memory/mcp_client.py +184 -0
  60. vex_harness-0.1.0/memory/paths.py +88 -0
  61. vex_harness-0.1.0/pyproject.toml +160 -0
  62. vex_harness-0.1.0/runtime/__init__.py +29 -0
  63. vex_harness-0.1.0/runtime/ablation.py +615 -0
  64. vex_harness-0.1.0/runtime/ablation_memplan.py +419 -0
  65. vex_harness-0.1.0/runtime/ablation_tasks.py +515 -0
  66. vex_harness-0.1.0/runtime/abuse.py +632 -0
  67. vex_harness-0.1.0/runtime/approval.py +110 -0
  68. vex_harness-0.1.0/runtime/checkpoint.py +120 -0
  69. vex_harness-0.1.0/runtime/config.py +89 -0
  70. vex_harness-0.1.0/runtime/difficulty.py +275 -0
  71. vex_harness-0.1.0/runtime/ensemble.py +389 -0
  72. vex_harness-0.1.0/runtime/fake_harness.py +163 -0
  73. vex_harness-0.1.0/runtime/fsutil.py +148 -0
  74. vex_harness-0.1.0/runtime/mock_provider.py +129 -0
  75. vex_harness-0.1.0/runtime/model_router.py +422 -0
  76. vex_harness-0.1.0/runtime/multirepo_tasks.py +511 -0
  77. vex_harness-0.1.0/runtime/paths.py +70 -0
  78. vex_harness-0.1.0/runtime/scheduler.py +333 -0
  79. vex_harness-0.1.0/runtime/serialize.py +72 -0
  80. vex_harness-0.1.0/runtime/soak.py +593 -0
  81. vex_harness-0.1.0/runtime/stress.py +529 -0
  82. vex_harness-0.1.0/runtime/worker.py +290 -0
  83. vex_harness-0.1.0/setup.cfg +4 -0
  84. vex_harness-0.1.0/shared/__init__.py +1 -0
  85. vex_harness-0.1.0/shared/traceview.py +480 -0
  86. vex_harness-0.1.0/shared/tracing.py +303 -0
  87. vex_harness-0.1.0/shared/types.py +87 -0
  88. vex_harness-0.1.0/tests/test_adversarial.py +365 -0
  89. vex_harness-0.1.0/tests/test_agent_tests.py +436 -0
  90. vex_harness-0.1.0/tests/test_batch_docs_lint.py +675 -0
  91. vex_harness-0.1.0/tests/test_cli.py +468 -0
  92. vex_harness-0.1.0/tests/test_cli_adversarial.py +466 -0
  93. vex_harness-0.1.0/tests/test_cli_errors.py +240 -0
  94. vex_harness-0.1.0/tests/test_cli_vex.py +333 -0
  95. vex_harness-0.1.0/tests/test_cli_vex2.py +498 -0
  96. vex_harness-0.1.0/tests/test_code_graph.py +226 -0
  97. vex_harness-0.1.0/tests/test_config_trace_state.py +209 -0
  98. vex_harness-0.1.0/tests/test_context_budget_rerank.py +217 -0
  99. vex_harness-0.1.0/tests/test_coordination.py +464 -0
  100. vex_harness-0.1.0/tests/test_coordination_e2e.py +423 -0
  101. vex_harness-0.1.0/tests/test_dashboard.py +235 -0
  102. vex_harness-0.1.0/tests/test_decision_memory_planning.py +396 -0
  103. vex_harness-0.1.0/tests/test_decision_store.py +227 -0
  104. vex_harness-0.1.0/tests/test_difficulty_approval.py +113 -0
  105. vex_harness-0.1.0/tests/test_e2e_run_task.py +1034 -0
  106. vex_harness-0.1.0/tests/test_editor_prompts.py +284 -0
  107. vex_harness-0.1.0/tests/test_ensemble.py +368 -0
  108. vex_harness-0.1.0/tests/test_env_snapshot.py +181 -0
  109. vex_harness-0.1.0/tests/test_evals_tasks.py +197 -0
  110. vex_harness-0.1.0/tests/test_feedback.py +439 -0
  111. vex_harness-0.1.0/tests/test_git_output_rationale.py +226 -0
  112. vex_harness-0.1.0/tests/test_mcp_adversarial.py +323 -0
  113. vex_harness-0.1.0/tests/test_mcp_client.py +115 -0
  114. vex_harness-0.1.0/tests/test_mcp_server.py +204 -0
  115. vex_harness-0.1.0/tests/test_mcp_stdio_fileno.py +67 -0
  116. vex_harness-0.1.0/tests/test_model_router.py +321 -0
  117. vex_harness-0.1.0/tests/test_provider_smoke.py +120 -0
  118. vex_harness-0.1.0/tests/test_recall_unit.py +274 -0
  119. vex_harness-0.1.0/tests/test_retrieval_tools.py +252 -0
  120. vex_harness-0.1.0/tests/test_sandbox.py +765 -0
  121. vex_harness-0.1.0/tests/test_scheduler_integration.py +414 -0
  122. vex_harness-0.1.0/tests/test_self_critique.py +349 -0
  123. vex_harness-0.1.0/tests/test_state_machine.py +246 -0
  124. vex_harness-0.1.0/tests/test_stubs_and_deps.py +88 -0
  125. vex_harness-0.1.0/tests/test_tool_errors.py +251 -0
  126. vex_harness-0.1.0/tests/test_tracing.py +472 -0
  127. vex_harness-0.1.0/tests/test_verify.py +256 -0
  128. vex_harness-0.1.0/tests/test_webfetch.py +761 -0
  129. vex_harness-0.1.0/vex_harness.egg-info/PKG-INFO +284 -0
  130. vex_harness-0.1.0/vex_harness.egg-info/SOURCES.txt +132 -0
  131. vex_harness-0.1.0/vex_harness.egg-info/dependency_links.txt +1 -0
  132. vex_harness-0.1.0/vex_harness.egg-info/entry_points.txt +3 -0
  133. vex_harness-0.1.0/vex_harness.egg-info/requires.txt +9 -0
  134. vex_harness-0.1.0/vex_harness.egg-info/top_level.txt +8 -0
@@ -0,0 +1,284 @@
1
+ Metadata-Version: 2.4
2
+ Name: vex-harness
3
+ Version: 0.1.0
4
+ Summary: Vex — AI coding agent harness with adaptive runtime and persistent memory
5
+ Author: Pavanteja2007
6
+ Project-URL: Homepage, https://github.com/Pavanteja2007/coding-harness
7
+ Project-URL: Repository, https://github.com/Pavanteja2007/coding-harness
8
+ Project-URL: Issues, https://github.com/Pavanteja2007/coding-harness/issues
9
+ Project-URL: Changelog, https://github.com/Pavanteja2007/coding-harness/blob/main/CHANGELOG.md
10
+ Keywords: ai,agent,coding-agent,bug-fixing,harness,llm,mcp,tree-sitter,sandbox,docker
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Software Development :: Bug Tracking
20
+ Classifier: Topic :: Software Development :: Quality Assurance
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ Requires-Dist: tree-sitter>=0.23
24
+ Requires-Dist: tree-sitter-python>=0.23
25
+ Requires-Dist: mcp>=1.2
26
+ Requires-Dist: litellm==1.74.9
27
+ Requires-Dist: rich>=13.0
28
+ Provides-Extra: dev
29
+ Requires-Dist: pytest>=8.0; extra == "dev"
30
+ Requires-Dist: ruff>=0.16; extra == "dev"
31
+
32
+ # coding-harness
33
+
34
+ [![CI (harness + runtime)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/ci.yml/badge.svg)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/ci.yml)
35
+ [![CI (memory + MCP + CLI)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/memory-cli-ci.yml/badge.svg)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/memory-cli-ci.yml)
36
+
37
+ An AI coding agent harness that fixes real software bugs end-to-end:
38
+ one system where a Docker-sandboxed agent loop, a concurrent
39
+ checkpointing runtime, a persistent cross-agent memory layer (exposed
40
+ via MCP), and **adaptive model routing by predicted difficulty** are
41
+ integrated deliberately — the integration is the point, not any one
42
+ piece.
43
+
44
+ ## What it does
45
+
46
+ ```
47
+ pip install vex-harness # PyPI distribution name; installs the `vex` command
48
+ vex # interactive mode, or the subcommands below
49
+ ```
50
+
51
+ ```
52
+ vex fix --repo <path> --issue "<bug report>" # one bug, one agent
53
+ vex run-benchmark --subset tasks.json # N bugs, N supervised agents
54
+ vex status --task-id <id> # structured progress view
55
+ vex dashboard # read-only web view of a run
56
+ vex memory query-decisions "<topic>" # the persistent memory layer
57
+ vex mcp call "<server cmd>" <tool> [--args '{..}'] # consume any external MCP server
58
+ ```
59
+
60
+ > The distribution name is `vex-harness` (`vex`, `vex-cli`, `vexx`, and
61
+ > `pyvex` were already taken on PyPI by unrelated packages); the installed
62
+ > command is `vex` — like `beautifulsoup4` installing as `bs4`. From a
63
+ > clone, the legacy alias `harness ...` and `python -m cli ...` still work
64
+ > too.
65
+
66
+ Under the hood, per task: the repo is snapshotted (the original is
67
+ never touched), a planner decomposes the fix into small verifiable
68
+ steps, an agent executes them with bash inside a locked-down Docker
69
+ sandbox (read-only rootfs, no network, resource limits, capability
70
+ drop), and **completion is verifier-gated** — a task only reports
71
+ `success` when the target test passes AND the full suite shows no
72
+ regressions. On verified fixes the harness additionally writes
73
+ git-native output (branch + commit + PR description), a human-readable
74
+ rationale.md, and a structured trace of every prompt/response/tool
75
+ call.
76
+
77
+ ## The four layers
78
+
79
+ | Layer | What it is | Where |
80
+ |---|---|---|
81
+ | **Harness** | planner / step agent / verifier gate, repo snapshot + diff, git-native output, rationale log, resume contract | `harness/` ([architecture doc](docs/architecture-harness.md)) |
82
+ | **Execution** | Docker sandbox (fresh container per command, orphan reaping, serialized image builds), stateless verify + flake detection | `execution/` |
83
+ | **Runtime** | process-per-task scheduler (proven at 10–50 concurrent), checkpoint/resume across hard kills, approval gate, adaptive model router + per-call cost ledger | `runtime/` |
84
+ | **Memory + MCP** | tree-sitter code graph, SQLite decision memory, MCP server exposing 5 tools to any MCP client (Claude Code, Cursor, …), MCP client for consuming external servers | `memory/`, `mcp_server/` |
85
+
86
+ ## The novel mechanism: adaptive model routing
87
+
88
+ Per call, the runtime predicts difficulty (intrinsic signal from the
89
+ issue text + struggle signal from the conversation tail — failing test
90
+ output, burned turns) and routes easy/medium calls to a cheap model,
91
+ hard calls to an expensive one. Every call lands in a per-task JSONL
92
+ ledger (model, tokens, cost, hint) — the mechanism is measurable, not
93
+ asserted.
94
+
95
+ **Ablation results (real bugs, real models, full real stack —
96
+ scheduler subprocesses → real harness → Docker verify):**
97
+
98
+ | run | arm | success | calls | tokens | cost* | wall |
99
+ |---|---|---|---|---|---|---|
100
+ | 5 fixture bugs | always-expensive | 5/5 | 17 | 38,680 | $0.0528 | 575s |
101
+ | 5 fixture bugs | adaptive | 5/5 | 31 | 69,615 | $0.0237 | 300s |
102
+ | 16-task expanded set | always-expensive | 16/16 | 81 | 138,526 | $0.1505 | 2717s |
103
+ | 16-task expanded set | adaptive | 16/16 | 71 | 136,436 | $0.0581 | 812s |
104
+ | 5 real OSS repos (Round 6) | always-expensive | 2/5 | 71 | 329,438 | $0.3059 | 2992s |
105
+ | 5 real OSS repos (Round 6) | adaptive | 3/5 | 75 | 302,801 | $0.0730 | 581s |
106
+
107
+ (Aritfacts: `logs/ablations/v2-heuristic-*` (n=5) and
108
+ `logs/ablations/v4` (n=16; an earlier `v3-expanded` run is invalid —
109
+ fixture-path bug — superseded by `v3-expanded-fixed` and `v4`.)
110
+
111
+ - Same 100% success rate in every arm/run, at **45% (n=5) / 39%
112
+ (n=16) of baseline cost** — same direction, growing margin with a
113
+ more varied task set (16 tasks: 5 real fixture bugs + 11 synthesized
114
+ repos with varied bug classes AND varied issue-text styles).
115
+ - Escalations did what they should: one genuine struggle escalation
116
+ (cheap attempts failed → hard-tier call finished the task), zero
117
+ cost-wasting escalations across the 8 easy-styled texts, and 1 of 2
118
+ deliberately-SCARY texts (stack trace + "race" wording over a
119
+ one-token bug) tricked the intrinsic scorer into one expensive call —
120
+ the honest false-escalation data point (1/16 tasks).
121
+
122
+ \* Honesty notes: endpoints are free-tier BYO routers; token counts and
123
+ model-choice data are raw measurements from ledgers, costs use proxy
124
+ price rates for comparable model classes (both endpoints report no
125
+ cost) — the cost **delta** is a price-model delta, not a bill. n=5 and
126
+ n=16 × 1 rep are directional, not benchmark-grade. The Round-6
127
+ multi-repo arm additionally suffered endpoint degradation (2 timeouts in
128
+ the OFF arm were 0-call wall-clock kills, not model failures). Phase 6
129
+ should re-run on SWE-bench subsets with paid tiers.
130
+ ## Multi-repo validation: the system beyond its home turf
131
+
132
+ Beyond the fixture set, the full stack has been validated against real,
133
+ unfamiliar OSS code — not just the original repo:
134
+
135
+ - **5 real OSS repos through the routing ablation (Round 6)** —
136
+ more-itertools, arrow, inflect, boltons, python-semver (pinned SHAs,
137
+ one genuine introduced bug each, real suites, per-repo suite pins for
138
+ dev-only deps): the adaptive arm went 3/5 (vs 2/5 always-expensive)
139
+ at **24% of the cost and 5x faster wall** — first multi-repo evidence
140
+ that the routing margin holds (and the failure modes are endpoint
141
+ timeouts, not harness bugs; honest data point: multi-hundred-K-token
142
+ real repos are simply harder than the fixture set, in BOTH arms).
143
+ (`logs/ablations/v6-multirepo/`)
144
+ - **jaraco/path (full DoD)** — unfamiliar real OSS repo end-to-end:
145
+ real cloud model, Docker sandbox, verifier-gated success in 1 attempt
146
+ ($0.053, 6 calls), git-native branch/commit/PR, rationale.md,
147
+ approval gate, memory ingestion — plus one honest first failure that
148
+ exposed and fixed a real harness bug (binary-artifact diff crash).
149
+ (`logs/oss-round4/`)
150
+ - **python-semver (module DoD)** — pristine baseline → broken-state
151
+ detection → in-sandbox fix → verified, flake-flagging, git output,
152
+ grounded rationale: 15/15 checks. (`logs/dod/`)
153
+ - **3 more real repos staged for a final sweep (bottle, click, parse)**
154
+ — full-stack runs in flight; first attempts showed honest failures
155
+ (unparseable plans from the degraded free-tier endpoint → the
156
+ harness correctly refused to claim success). Numbers land when the
157
+ endpoint stabilizes; artifacts will live under `logs/oss-round6/`.
158
+
159
+ Net: 7 real OSS repos have been driven by the actual harness/verifier
160
+ stack (plus 3 in flight), spanning plugin, date/time, inflection, and
161
+ versioning domains — with the same verifier-gated honesty rules as the
162
+ fixture set: nothing above is a claimed success without the gate.
163
+
164
+ ## Runtime reliability (proven, not claimed)
165
+
166
+ - **45 tasks @ concurrency 45, 8 simultaneous mid-run hard kills**:
167
+ 45/45 success, 8/8 genuine resumes — verified from trace events
168
+ (`plan_reused`, `step_skipped_resume`, pre-kill trace survival), zero
169
+ leaked containers (`logs/stress/real-45/`).
170
+ - Concurrency cap proven from the event journal (max overlap ≤ cap);
171
+ every kill recorded + requeued; parallel beats the serial floor.
172
+ - Scheduler + worker share one log tree (spawn pins `resume_dir` /
173
+ `log_root`); a mid-run kill resumes from completed steps with all
174
+ artifacts under the caller's `--log-root`.
175
+
176
+ ## Memory layer (MCP)
177
+
178
+ - `query_structure` — tree-sitter code graph (functions, classes,
179
+ calls, imports; persistent index)
180
+ - `query_decisions` / `record_decision` — decision/pattern memory
181
+ (auto-ingests every task's structured state)
182
+ - `task_status`, `list_repos`
183
+
184
+ Run it: `python -m mcp_server` (stdio) and connect any MCP client. The
185
+ harness's retrieval consumes the same graph programmatically, so
186
+ structural context rides into prompts without re-reading files.
187
+
188
+ ## Demo script (5 minutes)
189
+
190
+ Two variants: **zero-setup offline** (deterministic, no key/Docker) and
191
+ the real-model walkthrough.
192
+
193
+ ```bash
194
+ # 0) The whole story, offline in one command (scripted model; the loop,
195
+ # verifier gate, git output, rationale, and memory are all REAL):
196
+ python demo/run_demo.py # fix -> git/PR -> routing numbers -> memory -> dashboard hint
197
+ # per-step talking points: demo/README.md
198
+ ```
199
+
200
+ With a real model (BYO endpoint/key):
201
+
202
+ ```bash
203
+ # one-time: pip install vex-harness
204
+ # (from a clone instead: pip install -e . — or use python -m cli everywhere)
205
+
206
+ # 1) Fix a real bug with a real model (needs a BYO endpoint/key):
207
+ export MY_KEY=... # your openai-compatible router key
208
+ python -m cli fix \
209
+ --repo cli/fixtures/smoke_repo \
210
+ --issue "The mean() function in mathutil.py returns the sum instead of the arithmetic mean. Fix it so tests/test_mathutil.py::test_mean passes." \
211
+ --provider openai --model <model> --api-key $MY_KEY --api-base <base-url>
212
+ # → status, cost, diff; then inspect the artifacts:
213
+ python -m cli status --task-id <task_id> # plan checklist + decisions
214
+ type logs\<task_id>\rationale.md # what was wrong / what changed / why
215
+ type logs\<task_id>\git.json # branch + commit + PR description
216
+
217
+ # 2) Adaptive routing vs always-expensive, measured (the ablation):
218
+ # endpoints/keys are configured in runtime/ablation.py (BYO, env keys)
219
+ python -m runtime.ablation --tasks all --concurrency 2 # both arms, one summary
220
+ type logs\ablations\<ts>\summary.json # per-arm cost/token table
221
+
222
+ # 3) Concurrency + crash-resume at target scale (offline, scripted model):
223
+ python -m runtime.stress --mode real --tasks 45 --concurrency 45 --kill 8
224
+
225
+ # 4) The memory layer, queried over MCP (our own server, external client style):
226
+ python -m cli mcp call "python -m mcp_server" query_decisions --args "{\"query\": \"pytest\"}"
227
+ python -m cli mcp list-tools "python -m mcp_server"
228
+
229
+ # 5) Read-only dashboard over any run's logs:
230
+ python -m cli dashboard --logs-dir logs/ablations/<ts>/tasklogs
231
+ ```
232
+
233
+ ## Repo layout
234
+
235
+ ```
236
+ harness/ agent loop: planner, steps, verifier gate, resume, git output
237
+ execution/ Docker sandbox + verify + git-native output + rationale
238
+ runtime/ scheduler, worker, checkpoint, router (novel mechanism), ablation
239
+ memory/ code graph (tree-sitter), decision store (SQLite), MCP client
240
+ mcp_server/ MCP exposure of memory/status (stdio)
241
+ cli/ the `harness` command (fix / run-benchmark / status / mcp / dashboard)
242
+ dashboard/ read-only web view of existing logs
243
+ demo/ one-command offline demo + walkthrough (run_demo.py)
244
+ tests/ ~300 tests incl. real e2e bug-fix runs and real process-kill resumes
245
+ logs/ (gitignored) per-task state, traces, ledgers, run journals
246
+ ```
247
+
248
+ ## Status & verification
249
+
250
+ - CI on every push: two workflow files (kept separate — the four
251
+ modules were built in parallel terminals): `ci.yml` (harness +
252
+ runtime suites, OS matrix, nightly full stress + adversarial
253
+ abuse) and `memory-cli-ci.yml` (memory/MCP incl. real stdio
254
+ round-trip, CLI offline e2e through the real Docker sandbox,
255
+ dashboard — across Linux/Windows/macOS). Badges above.
256
+ - Full test suite green (scheduler integration with real process
257
+ kills, router, memory, MCP incl. real stdio round-trip, dashboard).
258
+ - CI (`.github/workflows/ci.yml`): the harness suite runs on every
259
+ push across Linux/macOS/Windows × Python 3.10/3.12 — Docker-gated
260
+ e2e tests self-skip with an explicit reason on runners without
261
+ Docker; full-scale stress + abuse suites run nightly. Harness
262
+ internals: [docs/architecture-harness.md](docs/architecture-harness.md).
263
+ - **Adversarially tested (Round 6)**: the MCP server and CLI were
264
+ probed with crafted/hostile inputs — path traversal, shell-injection
265
+ payloads, SQL injection, malformed subsets, null bytes. One real
266
+ data leak (task-id path traversal in `task_status`/`harness status`)
267
+ was found live, fixed, and pinned by 101 adversarial tests; all other
268
+ surfaces held (per-probe outcomes in each module's AGENTS.md; the
269
+ Docker sandbox was adversarially confirmed separately — 24/24
270
+ sequential + concurrent attack suites).
271
+ - Contract between modules: `INTERFACES.md`. Module-by-module state
272
+ (what's built, what's stubbed, decisions): each module's
273
+ `AGENTS.md`. High-level build history: `CHANGELOG.md` (current
274
+ release: **v0.1.0**).
275
+ - Deferred per spec: SWE-bench Lite numbers (Phase 6), multi-language,
276
+ plugin marketplace.
277
+
278
+ ## Tech
279
+
280
+ Python 3.10 · litellm (multi-provider, BYO-key) · Docker · tree-sitter
281
+ · MCP (official Python SDK) · argparse CLI · stdlib HTTP dashboard.
282
+ `litellm` is a real-model dependency (in `pyproject.toml`) pinned to
283
+ `1.74.9` on Python 3.10 (newer breaks the `typing` import on 3.10);
284
+ the offline/demo paths work without it (lazy import).
@@ -0,0 +1,253 @@
1
+ # coding-harness
2
+
3
+ [![CI (harness + runtime)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/ci.yml/badge.svg)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/ci.yml)
4
+ [![CI (memory + MCP + CLI)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/memory-cli-ci.yml/badge.svg)](https://github.com/Pavanteja2007/coding-harness/actions/workflows/memory-cli-ci.yml)
5
+
6
+ An AI coding agent harness that fixes real software bugs end-to-end:
7
+ one system where a Docker-sandboxed agent loop, a concurrent
8
+ checkpointing runtime, a persistent cross-agent memory layer (exposed
9
+ via MCP), and **adaptive model routing by predicted difficulty** are
10
+ integrated deliberately — the integration is the point, not any one
11
+ piece.
12
+
13
+ ## What it does
14
+
15
+ ```
16
+ pip install vex-harness # PyPI distribution name; installs the `vex` command
17
+ vex # interactive mode, or the subcommands below
18
+ ```
19
+
20
+ ```
21
+ vex fix --repo <path> --issue "<bug report>" # one bug, one agent
22
+ vex run-benchmark --subset tasks.json # N bugs, N supervised agents
23
+ vex status --task-id <id> # structured progress view
24
+ vex dashboard # read-only web view of a run
25
+ vex memory query-decisions "<topic>" # the persistent memory layer
26
+ vex mcp call "<server cmd>" <tool> [--args '{..}'] # consume any external MCP server
27
+ ```
28
+
29
+ > The distribution name is `vex-harness` (`vex`, `vex-cli`, `vexx`, and
30
+ > `pyvex` were already taken on PyPI by unrelated packages); the installed
31
+ > command is `vex` — like `beautifulsoup4` installing as `bs4`. From a
32
+ > clone, the legacy alias `harness ...` and `python -m cli ...` still work
33
+ > too.
34
+
35
+ Under the hood, per task: the repo is snapshotted (the original is
36
+ never touched), a planner decomposes the fix into small verifiable
37
+ steps, an agent executes them with bash inside a locked-down Docker
38
+ sandbox (read-only rootfs, no network, resource limits, capability
39
+ drop), and **completion is verifier-gated** — a task only reports
40
+ `success` when the target test passes AND the full suite shows no
41
+ regressions. On verified fixes the harness additionally writes
42
+ git-native output (branch + commit + PR description), a human-readable
43
+ rationale.md, and a structured trace of every prompt/response/tool
44
+ call.
45
+
46
+ ## The four layers
47
+
48
+ | Layer | What it is | Where |
49
+ |---|---|---|
50
+ | **Harness** | planner / step agent / verifier gate, repo snapshot + diff, git-native output, rationale log, resume contract | `harness/` ([architecture doc](docs/architecture-harness.md)) |
51
+ | **Execution** | Docker sandbox (fresh container per command, orphan reaping, serialized image builds), stateless verify + flake detection | `execution/` |
52
+ | **Runtime** | process-per-task scheduler (proven at 10–50 concurrent), checkpoint/resume across hard kills, approval gate, adaptive model router + per-call cost ledger | `runtime/` |
53
+ | **Memory + MCP** | tree-sitter code graph, SQLite decision memory, MCP server exposing 5 tools to any MCP client (Claude Code, Cursor, …), MCP client for consuming external servers | `memory/`, `mcp_server/` |
54
+
55
+ ## The novel mechanism: adaptive model routing
56
+
57
+ Per call, the runtime predicts difficulty (intrinsic signal from the
58
+ issue text + struggle signal from the conversation tail — failing test
59
+ output, burned turns) and routes easy/medium calls to a cheap model,
60
+ hard calls to an expensive one. Every call lands in a per-task JSONL
61
+ ledger (model, tokens, cost, hint) — the mechanism is measurable, not
62
+ asserted.
63
+
64
+ **Ablation results (real bugs, real models, full real stack —
65
+ scheduler subprocesses → real harness → Docker verify):**
66
+
67
+ | run | arm | success | calls | tokens | cost* | wall |
68
+ |---|---|---|---|---|---|---|
69
+ | 5 fixture bugs | always-expensive | 5/5 | 17 | 38,680 | $0.0528 | 575s |
70
+ | 5 fixture bugs | adaptive | 5/5 | 31 | 69,615 | $0.0237 | 300s |
71
+ | 16-task expanded set | always-expensive | 16/16 | 81 | 138,526 | $0.1505 | 2717s |
72
+ | 16-task expanded set | adaptive | 16/16 | 71 | 136,436 | $0.0581 | 812s |
73
+ | 5 real OSS repos (Round 6) | always-expensive | 2/5 | 71 | 329,438 | $0.3059 | 2992s |
74
+ | 5 real OSS repos (Round 6) | adaptive | 3/5 | 75 | 302,801 | $0.0730 | 581s |
75
+
76
+ (Aritfacts: `logs/ablations/v2-heuristic-*` (n=5) and
77
+ `logs/ablations/v4` (n=16; an earlier `v3-expanded` run is invalid —
78
+ fixture-path bug — superseded by `v3-expanded-fixed` and `v4`.)
79
+
80
+ - Same 100% success rate in every arm/run, at **45% (n=5) / 39%
81
+ (n=16) of baseline cost** — same direction, growing margin with a
82
+ more varied task set (16 tasks: 5 real fixture bugs + 11 synthesized
83
+ repos with varied bug classes AND varied issue-text styles).
84
+ - Escalations did what they should: one genuine struggle escalation
85
+ (cheap attempts failed → hard-tier call finished the task), zero
86
+ cost-wasting escalations across the 8 easy-styled texts, and 1 of 2
87
+ deliberately-SCARY texts (stack trace + "race" wording over a
88
+ one-token bug) tricked the intrinsic scorer into one expensive call —
89
+ the honest false-escalation data point (1/16 tasks).
90
+
91
+ \* Honesty notes: endpoints are free-tier BYO routers; token counts and
92
+ model-choice data are raw measurements from ledgers, costs use proxy
93
+ price rates for comparable model classes (both endpoints report no
94
+ cost) — the cost **delta** is a price-model delta, not a bill. n=5 and
95
+ n=16 × 1 rep are directional, not benchmark-grade. The Round-6
96
+ multi-repo arm additionally suffered endpoint degradation (2 timeouts in
97
+ the OFF arm were 0-call wall-clock kills, not model failures). Phase 6
98
+ should re-run on SWE-bench subsets with paid tiers.
99
+ ## Multi-repo validation: the system beyond its home turf
100
+
101
+ Beyond the fixture set, the full stack has been validated against real,
102
+ unfamiliar OSS code — not just the original repo:
103
+
104
+ - **5 real OSS repos through the routing ablation (Round 6)** —
105
+ more-itertools, arrow, inflect, boltons, python-semver (pinned SHAs,
106
+ one genuine introduced bug each, real suites, per-repo suite pins for
107
+ dev-only deps): the adaptive arm went 3/5 (vs 2/5 always-expensive)
108
+ at **24% of the cost and 5x faster wall** — first multi-repo evidence
109
+ that the routing margin holds (and the failure modes are endpoint
110
+ timeouts, not harness bugs; honest data point: multi-hundred-K-token
111
+ real repos are simply harder than the fixture set, in BOTH arms).
112
+ (`logs/ablations/v6-multirepo/`)
113
+ - **jaraco/path (full DoD)** — unfamiliar real OSS repo end-to-end:
114
+ real cloud model, Docker sandbox, verifier-gated success in 1 attempt
115
+ ($0.053, 6 calls), git-native branch/commit/PR, rationale.md,
116
+ approval gate, memory ingestion — plus one honest first failure that
117
+ exposed and fixed a real harness bug (binary-artifact diff crash).
118
+ (`logs/oss-round4/`)
119
+ - **python-semver (module DoD)** — pristine baseline → broken-state
120
+ detection → in-sandbox fix → verified, flake-flagging, git output,
121
+ grounded rationale: 15/15 checks. (`logs/dod/`)
122
+ - **3 more real repos staged for a final sweep (bottle, click, parse)**
123
+ — full-stack runs in flight; first attempts showed honest failures
124
+ (unparseable plans from the degraded free-tier endpoint → the
125
+ harness correctly refused to claim success). Numbers land when the
126
+ endpoint stabilizes; artifacts will live under `logs/oss-round6/`.
127
+
128
+ Net: 7 real OSS repos have been driven by the actual harness/verifier
129
+ stack (plus 3 in flight), spanning plugin, date/time, inflection, and
130
+ versioning domains — with the same verifier-gated honesty rules as the
131
+ fixture set: nothing above is a claimed success without the gate.
132
+
133
+ ## Runtime reliability (proven, not claimed)
134
+
135
+ - **45 tasks @ concurrency 45, 8 simultaneous mid-run hard kills**:
136
+ 45/45 success, 8/8 genuine resumes — verified from trace events
137
+ (`plan_reused`, `step_skipped_resume`, pre-kill trace survival), zero
138
+ leaked containers (`logs/stress/real-45/`).
139
+ - Concurrency cap proven from the event journal (max overlap ≤ cap);
140
+ every kill recorded + requeued; parallel beats the serial floor.
141
+ - Scheduler + worker share one log tree (spawn pins `resume_dir` /
142
+ `log_root`); a mid-run kill resumes from completed steps with all
143
+ artifacts under the caller's `--log-root`.
144
+
145
+ ## Memory layer (MCP)
146
+
147
+ - `query_structure` — tree-sitter code graph (functions, classes,
148
+ calls, imports; persistent index)
149
+ - `query_decisions` / `record_decision` — decision/pattern memory
150
+ (auto-ingests every task's structured state)
151
+ - `task_status`, `list_repos`
152
+
153
+ Run it: `python -m mcp_server` (stdio) and connect any MCP client. The
154
+ harness's retrieval consumes the same graph programmatically, so
155
+ structural context rides into prompts without re-reading files.
156
+
157
+ ## Demo script (5 minutes)
158
+
159
+ Two variants: **zero-setup offline** (deterministic, no key/Docker) and
160
+ the real-model walkthrough.
161
+
162
+ ```bash
163
+ # 0) The whole story, offline in one command (scripted model; the loop,
164
+ # verifier gate, git output, rationale, and memory are all REAL):
165
+ python demo/run_demo.py # fix -> git/PR -> routing numbers -> memory -> dashboard hint
166
+ # per-step talking points: demo/README.md
167
+ ```
168
+
169
+ With a real model (BYO endpoint/key):
170
+
171
+ ```bash
172
+ # one-time: pip install vex-harness
173
+ # (from a clone instead: pip install -e . — or use python -m cli everywhere)
174
+
175
+ # 1) Fix a real bug with a real model (needs a BYO endpoint/key):
176
+ export MY_KEY=... # your openai-compatible router key
177
+ python -m cli fix \
178
+ --repo cli/fixtures/smoke_repo \
179
+ --issue "The mean() function in mathutil.py returns the sum instead of the arithmetic mean. Fix it so tests/test_mathutil.py::test_mean passes." \
180
+ --provider openai --model <model> --api-key $MY_KEY --api-base <base-url>
181
+ # → status, cost, diff; then inspect the artifacts:
182
+ python -m cli status --task-id <task_id> # plan checklist + decisions
183
+ type logs\<task_id>\rationale.md # what was wrong / what changed / why
184
+ type logs\<task_id>\git.json # branch + commit + PR description
185
+
186
+ # 2) Adaptive routing vs always-expensive, measured (the ablation):
187
+ # endpoints/keys are configured in runtime/ablation.py (BYO, env keys)
188
+ python -m runtime.ablation --tasks all --concurrency 2 # both arms, one summary
189
+ type logs\ablations\<ts>\summary.json # per-arm cost/token table
190
+
191
+ # 3) Concurrency + crash-resume at target scale (offline, scripted model):
192
+ python -m runtime.stress --mode real --tasks 45 --concurrency 45 --kill 8
193
+
194
+ # 4) The memory layer, queried over MCP (our own server, external client style):
195
+ python -m cli mcp call "python -m mcp_server" query_decisions --args "{\"query\": \"pytest\"}"
196
+ python -m cli mcp list-tools "python -m mcp_server"
197
+
198
+ # 5) Read-only dashboard over any run's logs:
199
+ python -m cli dashboard --logs-dir logs/ablations/<ts>/tasklogs
200
+ ```
201
+
202
+ ## Repo layout
203
+
204
+ ```
205
+ harness/ agent loop: planner, steps, verifier gate, resume, git output
206
+ execution/ Docker sandbox + verify + git-native output + rationale
207
+ runtime/ scheduler, worker, checkpoint, router (novel mechanism), ablation
208
+ memory/ code graph (tree-sitter), decision store (SQLite), MCP client
209
+ mcp_server/ MCP exposure of memory/status (stdio)
210
+ cli/ the `harness` command (fix / run-benchmark / status / mcp / dashboard)
211
+ dashboard/ read-only web view of existing logs
212
+ demo/ one-command offline demo + walkthrough (run_demo.py)
213
+ tests/ ~300 tests incl. real e2e bug-fix runs and real process-kill resumes
214
+ logs/ (gitignored) per-task state, traces, ledgers, run journals
215
+ ```
216
+
217
+ ## Status & verification
218
+
219
+ - CI on every push: two workflow files (kept separate — the four
220
+ modules were built in parallel terminals): `ci.yml` (harness +
221
+ runtime suites, OS matrix, nightly full stress + adversarial
222
+ abuse) and `memory-cli-ci.yml` (memory/MCP incl. real stdio
223
+ round-trip, CLI offline e2e through the real Docker sandbox,
224
+ dashboard — across Linux/Windows/macOS). Badges above.
225
+ - Full test suite green (scheduler integration with real process
226
+ kills, router, memory, MCP incl. real stdio round-trip, dashboard).
227
+ - CI (`.github/workflows/ci.yml`): the harness suite runs on every
228
+ push across Linux/macOS/Windows × Python 3.10/3.12 — Docker-gated
229
+ e2e tests self-skip with an explicit reason on runners without
230
+ Docker; full-scale stress + abuse suites run nightly. Harness
231
+ internals: [docs/architecture-harness.md](docs/architecture-harness.md).
232
+ - **Adversarially tested (Round 6)**: the MCP server and CLI were
233
+ probed with crafted/hostile inputs — path traversal, shell-injection
234
+ payloads, SQL injection, malformed subsets, null bytes. One real
235
+ data leak (task-id path traversal in `task_status`/`harness status`)
236
+ was found live, fixed, and pinned by 101 adversarial tests; all other
237
+ surfaces held (per-probe outcomes in each module's AGENTS.md; the
238
+ Docker sandbox was adversarially confirmed separately — 24/24
239
+ sequential + concurrent attack suites).
240
+ - Contract between modules: `INTERFACES.md`. Module-by-module state
241
+ (what's built, what's stubbed, decisions): each module's
242
+ `AGENTS.md`. High-level build history: `CHANGELOG.md` (current
243
+ release: **v0.1.0**).
244
+ - Deferred per spec: SWE-bench Lite numbers (Phase 6), multi-language,
245
+ plugin marketplace.
246
+
247
+ ## Tech
248
+
249
+ Python 3.10 · litellm (multi-provider, BYO-key) · Docker · tree-sitter
250
+ · MCP (official Python SDK) · argparse CLI · stdlib HTTP dashboard.
251
+ `litellm` is a real-model dependency (in `pyproject.toml`) pinned to
252
+ `1.74.9` on Python 3.10 (newer breaks the `typing` import on 3.10);
253
+ the offline/demo paths work without it (lazy import).
@@ -0,0 +1,8 @@
1
+ """Terminal 4 — CLI: the human-facing interface tying the system together.
2
+
3
+ Entry point: cli.main (console script ``vex`` — legacy alias ``harness``
4
+ kept during migration — or ``python -m cli``).
5
+ Commands (INTERFACES.md Boundary 6): fix, run-benchmark, status, memory,
6
+ dashboard, mcp; no-args interactive natural-language mode (see
7
+ cli/interactive.py — the primary UX).
8
+ """
@@ -0,0 +1,24 @@
1
+ """Allow ``python -m cli`` alongside the ``vex`` console script."""
2
+
3
+ import sys
4
+
5
+
6
+ def _run() -> int:
7
+ try:
8
+ from cli.main import main
9
+ except ImportError as exc:
10
+ # Broken install / missing dependency (Task D): plain language, no
11
+ # traceback. This is the FIRST import of the package a user can
12
+ # hit, so an install problem lands exactly here.
13
+ print(
14
+ f"error: the Vex CLI could not be imported: {exc}\n"
15
+ 'check: was the package installed? Run: pip install -e ".[dev]"\n'
16
+ "check: are you in the right environment (venv active)?",
17
+ file=sys.stderr,
18
+ )
19
+ return 2
20
+ return main()
21
+
22
+
23
+ if __name__ == "__main__":
24
+ sys.exit(_run())
@@ -0,0 +1,5 @@
1
+ """Local stubs for boundaries whose real modules don't exist yet.
2
+
3
+ Mirrors the harness/_stubs convention: exact contract signatures so the
4
+ swap to the real module needs no caller changes.
5
+ """
@@ -0,0 +1,87 @@
1
+ """STUB for the scheduler boundary (INTERFACES.md Boundary 6) — Terminal 3
2
+ owns the real runtime.scheduler.run.
3
+
4
+ Concurrent fan-out via ThreadPoolExecutor calling run_task per task, with
5
+ per-task status lines printed to stdout — enough to exercise the CLI's
6
+ run-benchmark path end-to-end today. When runtime/scheduler.py lands,
7
+ cli.deps.get_scheduler_run() picks it up automatically (import-probe
8
+ first, stub second) — no CLI code changes.
9
+
10
+ Signature contract (Boundary 6):
11
+ run(tasks: list[Task], concurrency: int = 10, **kwargs) -> list[TaskResult]
12
+
13
+ Differences from the future real scheduler (documented for Terminal 3):
14
+ - No checkpoint/resume, no adaptive routing, no per-task worker processes
15
+ (threads, not processes — fine for a stub, since run_task is thread-safe
16
+ by its own docstring).
17
+ - Benchmarks: --subset is handled in the CLI; this stub receives the
18
+ already-materialized Task list.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import threading
23
+ from concurrent.futures import ThreadPoolExecutor, as_completed
24
+ from typing import Any, Callable, Dict, List
25
+
26
+ from shared.types import Task, TaskResult
27
+
28
+
29
+ def run(
30
+ tasks: List[Task],
31
+ concurrency: int = 10,
32
+ run_task: Callable[[Task], TaskResult] | None = None,
33
+ **kwargs: Any,
34
+ ) -> List[TaskResult]:
35
+ """STUB: run tasks concurrently via a thread pool; returns results in
36
+ task-list order (not completion order) for deterministic output.
37
+
38
+ Assumes `tasks` is a list of Task with distinct task_ids (run_task's
39
+ thread-safety requirement) and concurrency >= 1. Optionally accepts
40
+ an injected run_task (the CLI passes the resolved Boundary 3 callable
41
+ so tests can stub one level down).
42
+ """
43
+ if not tasks:
44
+ return []
45
+ concurrency = max(1, min(int(concurrency), len(tasks)))
46
+ fn = run_task or _resolve_run_task()
47
+ results: Dict[str, TaskResult] = {}
48
+ lock = threading.Lock()
49
+ done_count = [0]
50
+
51
+ with ThreadPoolExecutor(max_workers=concurrency) as pool:
52
+ futures = {pool.submit(fn, t): t for t in tasks}
53
+ for fut in as_completed(futures):
54
+ task = futures[fut]
55
+ try:
56
+ result = fut.result()
57
+ except Exception as exc: # a crashed worker must not lose the batch
58
+ result = TaskResult(
59
+ task_id=task.task_id,
60
+ status="error",
61
+ attempts=0,
62
+ diff=None,
63
+ verification=None,
64
+ cost_usd=0.0,
65
+ model_calls=[],
66
+ log_path="",
67
+ )
68
+ with lock:
69
+ print(f"[scheduler-stub] task {task.task_id} crashed: {exc}")
70
+ with lock:
71
+ results[task.task_id] = result
72
+ done_count[0] += 1
73
+ status = getattr(result, "status", "?")
74
+ print(
75
+ f"[scheduler-stub] [{done_count[0]}/{len(tasks)}] "
76
+ f"{task.task_id}: {status}"
77
+ )
78
+
79
+ return [results[t.task_id] for t in tasks]
80
+
81
+
82
+ def _resolve_run_task() -> Callable[[Task], TaskResult]:
83
+ """Same resolution as cli.deps (kept local to avoid an import cycle
84
+ when cli.deps itself probes this stub)."""
85
+ from harness.core import run_task
86
+
87
+ return run_task