vex-harness 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vex_harness-0.1.0/PKG-INFO +284 -0
- vex_harness-0.1.0/README.md +253 -0
- vex_harness-0.1.0/cli/__init__.py +8 -0
- vex_harness-0.1.0/cli/__main__.py +24 -0
- vex_harness-0.1.0/cli/_stubs/__init__.py +5 -0
- vex_harness-0.1.0/cli/_stubs/scheduler.py +87 -0
- vex_harness-0.1.0/cli/deps.py +67 -0
- vex_harness-0.1.0/cli/errors.py +217 -0
- vex_harness-0.1.0/cli/interactive.py +1066 -0
- vex_harness-0.1.0/cli/main.py +1188 -0
- vex_harness-0.1.0/cli/ui.py +172 -0
- vex_harness-0.1.0/cli/vexconfig.py +162 -0
- vex_harness-0.1.0/dashboard/__init__.py +5 -0
- vex_harness-0.1.0/dashboard/__main__.py +5 -0
- vex_harness-0.1.0/dashboard/collect.py +299 -0
- vex_harness-0.1.0/dashboard/server.py +291 -0
- vex_harness-0.1.0/execution/__init__.py +17 -0
- vex_harness-0.1.0/execution/env_snapshot.py +249 -0
- vex_harness-0.1.0/execution/feedback.py +565 -0
- vex_harness-0.1.0/execution/git_output.py +288 -0
- vex_harness-0.1.0/execution/rationale.py +179 -0
- vex_harness-0.1.0/execution/sandbox.py +979 -0
- vex_harness-0.1.0/execution/sandbox_adversarial.py +784 -0
- vex_harness-0.1.0/execution/sandbox_perf.py +298 -0
- vex_harness-0.1.0/execution/sandbox_stress.py +319 -0
- vex_harness-0.1.0/execution/sandbox_sustained.py +461 -0
- vex_harness-0.1.0/execution/verify.py +177 -0
- vex_harness-0.1.0/harness/__init__.py +4 -0
- vex_harness-0.1.0/harness/_stubs/__init__.py +6 -0
- vex_harness-0.1.0/harness/_stubs/model_router.py +98 -0
- vex_harness-0.1.0/harness/_stubs/sandbox.py +103 -0
- vex_harness-0.1.0/harness/_stubs/scripted_model.py +107 -0
- vex_harness-0.1.0/harness/_stubs/verify.py +128 -0
- vex_harness-0.1.0/harness/ablation_agenttests.py +371 -0
- vex_harness-0.1.0/harness/agent_tests.py +251 -0
- vex_harness-0.1.0/harness/config.py +162 -0
- vex_harness-0.1.0/harness/context.py +332 -0
- vex_harness-0.1.0/harness/coordination.py +395 -0
- vex_harness-0.1.0/harness/core.py +1793 -0
- vex_harness-0.1.0/harness/decision_memory.py +133 -0
- vex_harness-0.1.0/harness/deps.py +144 -0
- vex_harness-0.1.0/harness/docs_lookup.py +295 -0
- vex_harness-0.1.0/harness/editor.py +357 -0
- vex_harness-0.1.0/harness/lint.py +273 -0
- vex_harness-0.1.0/harness/model_client.py +98 -0
- vex_harness-0.1.0/harness/prompts.py +461 -0
- vex_harness-0.1.0/harness/retrieval.py +420 -0
- vex_harness-0.1.0/harness/state_machine.py +326 -0
- vex_harness-0.1.0/harness/tool_errors.py +329 -0
- vex_harness-0.1.0/harness/tools.py +378 -0
- vex_harness-0.1.0/harness/trace.py +136 -0
- vex_harness-0.1.0/harness/webfetch.py +499 -0
- vex_harness-0.1.0/mcp_server/__init__.py +5 -0
- vex_harness-0.1.0/mcp_server/__main__.py +5 -0
- vex_harness-0.1.0/mcp_server/server.py +246 -0
- vex_harness-0.1.0/memory/__init__.py +7 -0
- vex_harness-0.1.0/memory/code_graph.py +852 -0
- vex_harness-0.1.0/memory/decision_store.py +367 -0
- vex_harness-0.1.0/memory/mcp_client.py +184 -0
- vex_harness-0.1.0/memory/paths.py +88 -0
- vex_harness-0.1.0/pyproject.toml +160 -0
- vex_harness-0.1.0/runtime/__init__.py +29 -0
- vex_harness-0.1.0/runtime/ablation.py +615 -0
- vex_harness-0.1.0/runtime/ablation_memplan.py +419 -0
- vex_harness-0.1.0/runtime/ablation_tasks.py +515 -0
- vex_harness-0.1.0/runtime/abuse.py +632 -0
- vex_harness-0.1.0/runtime/approval.py +110 -0
- vex_harness-0.1.0/runtime/checkpoint.py +120 -0
- vex_harness-0.1.0/runtime/config.py +89 -0
- vex_harness-0.1.0/runtime/difficulty.py +275 -0
- vex_harness-0.1.0/runtime/ensemble.py +389 -0
- vex_harness-0.1.0/runtime/fake_harness.py +163 -0
- vex_harness-0.1.0/runtime/fsutil.py +148 -0
- vex_harness-0.1.0/runtime/mock_provider.py +129 -0
- vex_harness-0.1.0/runtime/model_router.py +422 -0
- vex_harness-0.1.0/runtime/multirepo_tasks.py +511 -0
- vex_harness-0.1.0/runtime/paths.py +70 -0
- vex_harness-0.1.0/runtime/scheduler.py +333 -0
- vex_harness-0.1.0/runtime/serialize.py +72 -0
- vex_harness-0.1.0/runtime/soak.py +593 -0
- vex_harness-0.1.0/runtime/stress.py +529 -0
- vex_harness-0.1.0/runtime/worker.py +290 -0
- vex_harness-0.1.0/setup.cfg +4 -0
- vex_harness-0.1.0/shared/__init__.py +1 -0
- vex_harness-0.1.0/shared/traceview.py +480 -0
- vex_harness-0.1.0/shared/tracing.py +303 -0
- vex_harness-0.1.0/shared/types.py +87 -0
- vex_harness-0.1.0/tests/test_adversarial.py +365 -0
- vex_harness-0.1.0/tests/test_agent_tests.py +436 -0
- vex_harness-0.1.0/tests/test_batch_docs_lint.py +675 -0
- vex_harness-0.1.0/tests/test_cli.py +468 -0
- vex_harness-0.1.0/tests/test_cli_adversarial.py +466 -0
- vex_harness-0.1.0/tests/test_cli_errors.py +240 -0
- vex_harness-0.1.0/tests/test_cli_vex.py +333 -0
- vex_harness-0.1.0/tests/test_cli_vex2.py +498 -0
- vex_harness-0.1.0/tests/test_code_graph.py +226 -0
- vex_harness-0.1.0/tests/test_config_trace_state.py +209 -0
- vex_harness-0.1.0/tests/test_context_budget_rerank.py +217 -0
- vex_harness-0.1.0/tests/test_coordination.py +464 -0
- vex_harness-0.1.0/tests/test_coordination_e2e.py +423 -0
- vex_harness-0.1.0/tests/test_dashboard.py +235 -0
- vex_harness-0.1.0/tests/test_decision_memory_planning.py +396 -0
- vex_harness-0.1.0/tests/test_decision_store.py +227 -0
- vex_harness-0.1.0/tests/test_difficulty_approval.py +113 -0
- vex_harness-0.1.0/tests/test_e2e_run_task.py +1034 -0
- vex_harness-0.1.0/tests/test_editor_prompts.py +284 -0
- vex_harness-0.1.0/tests/test_ensemble.py +368 -0
- vex_harness-0.1.0/tests/test_env_snapshot.py +181 -0
- vex_harness-0.1.0/tests/test_evals_tasks.py +197 -0
- vex_harness-0.1.0/tests/test_feedback.py +439 -0
- vex_harness-0.1.0/tests/test_git_output_rationale.py +226 -0
- vex_harness-0.1.0/tests/test_mcp_adversarial.py +323 -0
- vex_harness-0.1.0/tests/test_mcp_client.py +115 -0
- vex_harness-0.1.0/tests/test_mcp_server.py +204 -0
- vex_harness-0.1.0/tests/test_mcp_stdio_fileno.py +67 -0
- vex_harness-0.1.0/tests/test_model_router.py +321 -0
- vex_harness-0.1.0/tests/test_provider_smoke.py +120 -0
- vex_harness-0.1.0/tests/test_recall_unit.py +274 -0
- vex_harness-0.1.0/tests/test_retrieval_tools.py +252 -0
- vex_harness-0.1.0/tests/test_sandbox.py +765 -0
- vex_harness-0.1.0/tests/test_scheduler_integration.py +414 -0
- vex_harness-0.1.0/tests/test_self_critique.py +349 -0
- vex_harness-0.1.0/tests/test_state_machine.py +246 -0
- vex_harness-0.1.0/tests/test_stubs_and_deps.py +88 -0
- vex_harness-0.1.0/tests/test_tool_errors.py +251 -0
- vex_harness-0.1.0/tests/test_tracing.py +472 -0
- vex_harness-0.1.0/tests/test_verify.py +256 -0
- vex_harness-0.1.0/tests/test_webfetch.py +761 -0
- vex_harness-0.1.0/vex_harness.egg-info/PKG-INFO +284 -0
- vex_harness-0.1.0/vex_harness.egg-info/SOURCES.txt +132 -0
- vex_harness-0.1.0/vex_harness.egg-info/dependency_links.txt +1 -0
- vex_harness-0.1.0/vex_harness.egg-info/entry_points.txt +3 -0
- vex_harness-0.1.0/vex_harness.egg-info/requires.txt +9 -0
- vex_harness-0.1.0/vex_harness.egg-info/top_level.txt +8 -0
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: vex-harness
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Vex — AI coding agent harness with adaptive runtime and persistent memory
|
|
5
|
+
Author: Pavanteja2007
|
|
6
|
+
Project-URL: Homepage, https://github.com/Pavanteja2007/coding-harness
|
|
7
|
+
Project-URL: Repository, https://github.com/Pavanteja2007/coding-harness
|
|
8
|
+
Project-URL: Issues, https://github.com/Pavanteja2007/coding-harness/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/Pavanteja2007/coding-harness/blob/main/CHANGELOG.md
|
|
10
|
+
Keywords: ai,agent,coding-agent,bug-fixing,harness,llm,mcp,tree-sitter,sandbox,docker
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Software Development :: Bug Tracking
|
|
20
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
Requires-Dist: tree-sitter>=0.23
|
|
24
|
+
Requires-Dist: tree-sitter-python>=0.23
|
|
25
|
+
Requires-Dist: mcp>=1.2
|
|
26
|
+
Requires-Dist: litellm==1.74.9
|
|
27
|
+
Requires-Dist: rich>=13.0
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
30
|
+
Requires-Dist: ruff>=0.16; extra == "dev"
|
|
31
|
+
|
|
32
|
+
# coding-harness
|
|
33
|
+
|
|
34
|
+
[](https://github.com/Pavanteja2007/coding-harness/actions/workflows/ci.yml)
|
|
35
|
+
[](https://github.com/Pavanteja2007/coding-harness/actions/workflows/memory-cli-ci.yml)
|
|
36
|
+
|
|
37
|
+
An AI coding agent harness that fixes real software bugs end-to-end:
|
|
38
|
+
one system where a Docker-sandboxed agent loop, a concurrent
|
|
39
|
+
checkpointing runtime, a persistent cross-agent memory layer (exposed
|
|
40
|
+
via MCP), and **adaptive model routing by predicted difficulty** are
|
|
41
|
+
integrated deliberately — the integration is the point, not any one
|
|
42
|
+
piece.
|
|
43
|
+
|
|
44
|
+
## What it does
|
|
45
|
+
|
|
46
|
+
```
|
|
47
|
+
pip install vex-harness # PyPI distribution name; installs the `vex` command
|
|
48
|
+
vex # interactive mode, or the subcommands below
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```
|
|
52
|
+
vex fix --repo <path> --issue "<bug report>" # one bug, one agent
|
|
53
|
+
vex run-benchmark --subset tasks.json # N bugs, N supervised agents
|
|
54
|
+
vex status --task-id <id> # structured progress view
|
|
55
|
+
vex dashboard # read-only web view of a run
|
|
56
|
+
vex memory query-decisions "<topic>" # the persistent memory layer
|
|
57
|
+
vex mcp call "<server cmd>" <tool> [--args '{..}'] # consume any external MCP server
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
> The distribution name is `vex-harness` (`vex`, `vex-cli`, `vexx`, and
|
|
61
|
+
> `pyvex` were already taken on PyPI by unrelated packages); the installed
|
|
62
|
+
> command is `vex` — like `beautifulsoup4` installing as `bs4`. From a
|
|
63
|
+
> clone, the legacy alias `harness ...` and `python -m cli ...` still work
|
|
64
|
+
> too.
|
|
65
|
+
|
|
66
|
+
Under the hood, per task: the repo is snapshotted (the original is
|
|
67
|
+
never touched), a planner decomposes the fix into small verifiable
|
|
68
|
+
steps, an agent executes them with bash inside a locked-down Docker
|
|
69
|
+
sandbox (read-only rootfs, no network, resource limits, capability
|
|
70
|
+
drop), and **completion is verifier-gated** — a task only reports
|
|
71
|
+
`success` when the target test passes AND the full suite shows no
|
|
72
|
+
regressions. On verified fixes the harness additionally writes
|
|
73
|
+
git-native output (branch + commit + PR description), a human-readable
|
|
74
|
+
rationale.md, and a structured trace of every prompt/response/tool
|
|
75
|
+
call.
|
|
76
|
+
|
|
77
|
+
## The four layers
|
|
78
|
+
|
|
79
|
+
| Layer | What it is | Where |
|
|
80
|
+
|---|---|---|
|
|
81
|
+
| **Harness** | planner / step agent / verifier gate, repo snapshot + diff, git-native output, rationale log, resume contract | `harness/` ([architecture doc](docs/architecture-harness.md)) |
|
|
82
|
+
| **Execution** | Docker sandbox (fresh container per command, orphan reaping, serialized image builds), stateless verify + flake detection | `execution/` |
|
|
83
|
+
| **Runtime** | process-per-task scheduler (proven at 10–50 concurrent), checkpoint/resume across hard kills, approval gate, adaptive model router + per-call cost ledger | `runtime/` |
|
|
84
|
+
| **Memory + MCP** | tree-sitter code graph, SQLite decision memory, MCP server exposing 5 tools to any MCP client (Claude Code, Cursor, …), MCP client for consuming external servers | `memory/`, `mcp_server/` |
|
|
85
|
+
|
|
86
|
+
## The novel mechanism: adaptive model routing
|
|
87
|
+
|
|
88
|
+
Per call, the runtime predicts difficulty (intrinsic signal from the
|
|
89
|
+
issue text + struggle signal from the conversation tail — failing test
|
|
90
|
+
output, burned turns) and routes easy/medium calls to a cheap model,
|
|
91
|
+
hard calls to an expensive one. Every call lands in a per-task JSONL
|
|
92
|
+
ledger (model, tokens, cost, hint) — the mechanism is measurable, not
|
|
93
|
+
asserted.
|
|
94
|
+
|
|
95
|
+
**Ablation results (real bugs, real models, full real stack —
|
|
96
|
+
scheduler subprocesses → real harness → Docker verify):**
|
|
97
|
+
|
|
98
|
+
| run | arm | success | calls | tokens | cost* | wall |
|
|
99
|
+
|---|---|---|---|---|---|---|
|
|
100
|
+
| 5 fixture bugs | always-expensive | 5/5 | 17 | 38,680 | $0.0528 | 575s |
|
|
101
|
+
| 5 fixture bugs | adaptive | 5/5 | 31 | 69,615 | $0.0237 | 300s |
|
|
102
|
+
| 16-task expanded set | always-expensive | 16/16 | 81 | 138,526 | $0.1505 | 2717s |
|
|
103
|
+
| 16-task expanded set | adaptive | 16/16 | 71 | 136,436 | $0.0581 | 812s |
|
|
104
|
+
| 5 real OSS repos (Round 6) | always-expensive | 2/5 | 71 | 329,438 | $0.3059 | 2992s |
|
|
105
|
+
| 5 real OSS repos (Round 6) | adaptive | 3/5 | 75 | 302,801 | $0.0730 | 581s |
|
|
106
|
+
|
|
107
|
+
(Aritfacts: `logs/ablations/v2-heuristic-*` (n=5) and
|
|
108
|
+
`logs/ablations/v4` (n=16; an earlier `v3-expanded` run is invalid —
|
|
109
|
+
fixture-path bug — superseded by `v3-expanded-fixed` and `v4`.)
|
|
110
|
+
|
|
111
|
+
- Same 100% success rate in every arm/run, at **45% (n=5) / 39%
|
|
112
|
+
(n=16) of baseline cost** — same direction, growing margin with a
|
|
113
|
+
more varied task set (16 tasks: 5 real fixture bugs + 11 synthesized
|
|
114
|
+
repos with varied bug classes AND varied issue-text styles).
|
|
115
|
+
- Escalations did what they should: one genuine struggle escalation
|
|
116
|
+
(cheap attempts failed → hard-tier call finished the task), zero
|
|
117
|
+
cost-wasting escalations across the 8 easy-styled texts, and 1 of 2
|
|
118
|
+
deliberately-SCARY texts (stack trace + "race" wording over a
|
|
119
|
+
one-token bug) tricked the intrinsic scorer into one expensive call —
|
|
120
|
+
the honest false-escalation data point (1/16 tasks).
|
|
121
|
+
|
|
122
|
+
\* Honesty notes: endpoints are free-tier BYO routers; token counts and
|
|
123
|
+
model-choice data are raw measurements from ledgers, costs use proxy
|
|
124
|
+
price rates for comparable model classes (both endpoints report no
|
|
125
|
+
cost) — the cost **delta** is a price-model delta, not a bill. n=5 and
|
|
126
|
+
n=16 × 1 rep are directional, not benchmark-grade. The Round-6
|
|
127
|
+
multi-repo arm additionally suffered endpoint degradation (2 timeouts in
|
|
128
|
+
the OFF arm were 0-call wall-clock kills, not model failures). Phase 6
|
|
129
|
+
should re-run on SWE-bench subsets with paid tiers.
|
|
130
|
+
## Multi-repo validation: the system beyond its home turf
|
|
131
|
+
|
|
132
|
+
Beyond the fixture set, the full stack has been validated against real,
|
|
133
|
+
unfamiliar OSS code — not just the original repo:
|
|
134
|
+
|
|
135
|
+
- **5 real OSS repos through the routing ablation (Round 6)** —
|
|
136
|
+
more-itertools, arrow, inflect, boltons, python-semver (pinned SHAs,
|
|
137
|
+
one genuine introduced bug each, real suites, per-repo suite pins for
|
|
138
|
+
dev-only deps): the adaptive arm went 3/5 (vs 2/5 always-expensive)
|
|
139
|
+
at **24% of the cost and 5x faster wall** — first multi-repo evidence
|
|
140
|
+
that the routing margin holds (and the failure modes are endpoint
|
|
141
|
+
timeouts, not harness bugs; honest data point: multi-hundred-K-token
|
|
142
|
+
real repos are simply harder than the fixture set, in BOTH arms).
|
|
143
|
+
(`logs/ablations/v6-multirepo/`)
|
|
144
|
+
- **jaraco/path (full DoD)** — unfamiliar real OSS repo end-to-end:
|
|
145
|
+
real cloud model, Docker sandbox, verifier-gated success in 1 attempt
|
|
146
|
+
($0.053, 6 calls), git-native branch/commit/PR, rationale.md,
|
|
147
|
+
approval gate, memory ingestion — plus one honest first failure that
|
|
148
|
+
exposed and fixed a real harness bug (binary-artifact diff crash).
|
|
149
|
+
(`logs/oss-round4/`)
|
|
150
|
+
- **python-semver (module DoD)** — pristine baseline → broken-state
|
|
151
|
+
detection → in-sandbox fix → verified, flake-flagging, git output,
|
|
152
|
+
grounded rationale: 15/15 checks. (`logs/dod/`)
|
|
153
|
+
- **3 more real repos staged for a final sweep (bottle, click, parse)**
|
|
154
|
+
— full-stack runs in flight; first attempts showed honest failures
|
|
155
|
+
(unparseable plans from the degraded free-tier endpoint → the
|
|
156
|
+
harness correctly refused to claim success). Numbers land when the
|
|
157
|
+
endpoint stabilizes; artifacts will live under `logs/oss-round6/`.
|
|
158
|
+
|
|
159
|
+
Net: 7 real OSS repos have been driven by the actual harness/verifier
|
|
160
|
+
stack (plus 3 in flight), spanning plugin, date/time, inflection, and
|
|
161
|
+
versioning domains — with the same verifier-gated honesty rules as the
|
|
162
|
+
fixture set: nothing above is a claimed success without the gate.
|
|
163
|
+
|
|
164
|
+
## Runtime reliability (proven, not claimed)
|
|
165
|
+
|
|
166
|
+
- **45 tasks @ concurrency 45, 8 simultaneous mid-run hard kills**:
|
|
167
|
+
45/45 success, 8/8 genuine resumes — verified from trace events
|
|
168
|
+
(`plan_reused`, `step_skipped_resume`, pre-kill trace survival), zero
|
|
169
|
+
leaked containers (`logs/stress/real-45/`).
|
|
170
|
+
- Concurrency cap proven from the event journal (max overlap ≤ cap);
|
|
171
|
+
every kill recorded + requeued; parallel beats the serial floor.
|
|
172
|
+
- Scheduler + worker share one log tree (spawn pins `resume_dir` /
|
|
173
|
+
`log_root`); a mid-run kill resumes from completed steps with all
|
|
174
|
+
artifacts under the caller's `--log-root`.
|
|
175
|
+
|
|
176
|
+
## Memory layer (MCP)
|
|
177
|
+
|
|
178
|
+
- `query_structure` — tree-sitter code graph (functions, classes,
|
|
179
|
+
calls, imports; persistent index)
|
|
180
|
+
- `query_decisions` / `record_decision` — decision/pattern memory
|
|
181
|
+
(auto-ingests every task's structured state)
|
|
182
|
+
- `task_status`, `list_repos`
|
|
183
|
+
|
|
184
|
+
Run it: `python -m mcp_server` (stdio) and connect any MCP client. The
|
|
185
|
+
harness's retrieval consumes the same graph programmatically, so
|
|
186
|
+
structural context rides into prompts without re-reading files.
|
|
187
|
+
|
|
188
|
+
## Demo script (5 minutes)
|
|
189
|
+
|
|
190
|
+
Two variants: **zero-setup offline** (deterministic, no key/Docker) and
|
|
191
|
+
the real-model walkthrough.
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
# 0) The whole story, offline in one command (scripted model; the loop,
|
|
195
|
+
# verifier gate, git output, rationale, and memory are all REAL):
|
|
196
|
+
python demo/run_demo.py # fix -> git/PR -> routing numbers -> memory -> dashboard hint
|
|
197
|
+
# per-step talking points: demo/README.md
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
With a real model (BYO endpoint/key):
|
|
201
|
+
|
|
202
|
+
```bash
|
|
203
|
+
# one-time: pip install vex-harness
|
|
204
|
+
# (from a clone instead: pip install -e . — or use python -m cli everywhere)
|
|
205
|
+
|
|
206
|
+
# 1) Fix a real bug with a real model (needs a BYO endpoint/key):
|
|
207
|
+
export MY_KEY=... # your openai-compatible router key
|
|
208
|
+
python -m cli fix \
|
|
209
|
+
--repo cli/fixtures/smoke_repo \
|
|
210
|
+
--issue "The mean() function in mathutil.py returns the sum instead of the arithmetic mean. Fix it so tests/test_mathutil.py::test_mean passes." \
|
|
211
|
+
--provider openai --model <model> --api-key $MY_KEY --api-base <base-url>
|
|
212
|
+
# → status, cost, diff; then inspect the artifacts:
|
|
213
|
+
python -m cli status --task-id <task_id> # plan checklist + decisions
|
|
214
|
+
type logs\<task_id>\rationale.md # what was wrong / what changed / why
|
|
215
|
+
type logs\<task_id>\git.json # branch + commit + PR description
|
|
216
|
+
|
|
217
|
+
# 2) Adaptive routing vs always-expensive, measured (the ablation):
|
|
218
|
+
# endpoints/keys are configured in runtime/ablation.py (BYO, env keys)
|
|
219
|
+
python -m runtime.ablation --tasks all --concurrency 2 # both arms, one summary
|
|
220
|
+
type logs\ablations\<ts>\summary.json # per-arm cost/token table
|
|
221
|
+
|
|
222
|
+
# 3) Concurrency + crash-resume at target scale (offline, scripted model):
|
|
223
|
+
python -m runtime.stress --mode real --tasks 45 --concurrency 45 --kill 8
|
|
224
|
+
|
|
225
|
+
# 4) The memory layer, queried over MCP (our own server, external client style):
|
|
226
|
+
python -m cli mcp call "python -m mcp_server" query_decisions --args "{\"query\": \"pytest\"}"
|
|
227
|
+
python -m cli mcp list-tools "python -m mcp_server"
|
|
228
|
+
|
|
229
|
+
# 5) Read-only dashboard over any run's logs:
|
|
230
|
+
python -m cli dashboard --logs-dir logs/ablations/<ts>/tasklogs
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
## Repo layout
|
|
234
|
+
|
|
235
|
+
```
|
|
236
|
+
harness/ agent loop: planner, steps, verifier gate, resume, git output
|
|
237
|
+
execution/ Docker sandbox + verify + git-native output + rationale
|
|
238
|
+
runtime/ scheduler, worker, checkpoint, router (novel mechanism), ablation
|
|
239
|
+
memory/ code graph (tree-sitter), decision store (SQLite), MCP client
|
|
240
|
+
mcp_server/ MCP exposure of memory/status (stdio)
|
|
241
|
+
cli/ the `harness` command (fix / run-benchmark / status / mcp / dashboard)
|
|
242
|
+
dashboard/ read-only web view of existing logs
|
|
243
|
+
demo/ one-command offline demo + walkthrough (run_demo.py)
|
|
244
|
+
tests/ ~300 tests incl. real e2e bug-fix runs and real process-kill resumes
|
|
245
|
+
logs/ (gitignored) per-task state, traces, ledgers, run journals
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
## Status & verification
|
|
249
|
+
|
|
250
|
+
- CI on every push: two workflow files (kept separate — the four
|
|
251
|
+
modules were built in parallel terminals): `ci.yml` (harness +
|
|
252
|
+
runtime suites, OS matrix, nightly full stress + adversarial
|
|
253
|
+
abuse) and `memory-cli-ci.yml` (memory/MCP incl. real stdio
|
|
254
|
+
round-trip, CLI offline e2e through the real Docker sandbox,
|
|
255
|
+
dashboard — across Linux/Windows/macOS). Badges above.
|
|
256
|
+
- Full test suite green (scheduler integration with real process
|
|
257
|
+
kills, router, memory, MCP incl. real stdio round-trip, dashboard).
|
|
258
|
+
- CI (`.github/workflows/ci.yml`): the harness suite runs on every
|
|
259
|
+
push across Linux/macOS/Windows × Python 3.10/3.12 — Docker-gated
|
|
260
|
+
e2e tests self-skip with an explicit reason on runners without
|
|
261
|
+
Docker; full-scale stress + abuse suites run nightly. Harness
|
|
262
|
+
internals: [docs/architecture-harness.md](docs/architecture-harness.md).
|
|
263
|
+
- **Adversarially tested (Round 6)**: the MCP server and CLI were
|
|
264
|
+
probed with crafted/hostile inputs — path traversal, shell-injection
|
|
265
|
+
payloads, SQL injection, malformed subsets, null bytes. One real
|
|
266
|
+
data leak (task-id path traversal in `task_status`/`harness status`)
|
|
267
|
+
was found live, fixed, and pinned by 101 adversarial tests; all other
|
|
268
|
+
surfaces held (per-probe outcomes in each module's AGENTS.md; the
|
|
269
|
+
Docker sandbox was adversarially confirmed separately — 24/24
|
|
270
|
+
sequential + concurrent attack suites).
|
|
271
|
+
- Contract between modules: `INTERFACES.md`. Module-by-module state
|
|
272
|
+
(what's built, what's stubbed, decisions): each module's
|
|
273
|
+
`AGENTS.md`. High-level build history: `CHANGELOG.md` (current
|
|
274
|
+
release: **v0.1.0**).
|
|
275
|
+
- Deferred per spec: SWE-bench Lite numbers (Phase 6), multi-language,
|
|
276
|
+
plugin marketplace.
|
|
277
|
+
|
|
278
|
+
## Tech
|
|
279
|
+
|
|
280
|
+
Python 3.10 · litellm (multi-provider, BYO-key) · Docker · tree-sitter
|
|
281
|
+
· MCP (official Python SDK) · argparse CLI · stdlib HTTP dashboard.
|
|
282
|
+
`litellm` is a real-model dependency (in `pyproject.toml`) pinned to
|
|
283
|
+
`1.74.9` on Python 3.10 (newer breaks the `typing` import on 3.10);
|
|
284
|
+
the offline/demo paths work without it (lazy import).
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
# coding-harness
|
|
2
|
+
|
|
3
|
+
[](https://github.com/Pavanteja2007/coding-harness/actions/workflows/ci.yml)
|
|
4
|
+
[](https://github.com/Pavanteja2007/coding-harness/actions/workflows/memory-cli-ci.yml)
|
|
5
|
+
|
|
6
|
+
An AI coding agent harness that fixes real software bugs end-to-end:
|
|
7
|
+
one system where a Docker-sandboxed agent loop, a concurrent
|
|
8
|
+
checkpointing runtime, a persistent cross-agent memory layer (exposed
|
|
9
|
+
via MCP), and **adaptive model routing by predicted difficulty** are
|
|
10
|
+
integrated deliberately — the integration is the point, not any one
|
|
11
|
+
piece.
|
|
12
|
+
|
|
13
|
+
## What it does
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
pip install vex-harness # PyPI distribution name; installs the `vex` command
|
|
17
|
+
vex # interactive mode, or the subcommands below
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
vex fix --repo <path> --issue "<bug report>" # one bug, one agent
|
|
22
|
+
vex run-benchmark --subset tasks.json # N bugs, N supervised agents
|
|
23
|
+
vex status --task-id <id> # structured progress view
|
|
24
|
+
vex dashboard # read-only web view of a run
|
|
25
|
+
vex memory query-decisions "<topic>" # the persistent memory layer
|
|
26
|
+
vex mcp call "<server cmd>" <tool> [--args '{..}'] # consume any external MCP server
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
> The distribution name is `vex-harness` (`vex`, `vex-cli`, `vexx`, and
|
|
30
|
+
> `pyvex` were already taken on PyPI by unrelated packages); the installed
|
|
31
|
+
> command is `vex` — like `beautifulsoup4` installing as `bs4`. From a
|
|
32
|
+
> clone, the legacy alias `harness ...` and `python -m cli ...` still work
|
|
33
|
+
> too.
|
|
34
|
+
|
|
35
|
+
Under the hood, per task: the repo is snapshotted (the original is
|
|
36
|
+
never touched), a planner decomposes the fix into small verifiable
|
|
37
|
+
steps, an agent executes them with bash inside a locked-down Docker
|
|
38
|
+
sandbox (read-only rootfs, no network, resource limits, capability
|
|
39
|
+
drop), and **completion is verifier-gated** — a task only reports
|
|
40
|
+
`success` when the target test passes AND the full suite shows no
|
|
41
|
+
regressions. On verified fixes the harness additionally writes
|
|
42
|
+
git-native output (branch + commit + PR description), a human-readable
|
|
43
|
+
rationale.md, and a structured trace of every prompt/response/tool
|
|
44
|
+
call.
|
|
45
|
+
|
|
46
|
+
## The four layers
|
|
47
|
+
|
|
48
|
+
| Layer | What it is | Where |
|
|
49
|
+
|---|---|---|
|
|
50
|
+
| **Harness** | planner / step agent / verifier gate, repo snapshot + diff, git-native output, rationale log, resume contract | `harness/` ([architecture doc](docs/architecture-harness.md)) |
|
|
51
|
+
| **Execution** | Docker sandbox (fresh container per command, orphan reaping, serialized image builds), stateless verify + flake detection | `execution/` |
|
|
52
|
+
| **Runtime** | process-per-task scheduler (proven at 10–50 concurrent), checkpoint/resume across hard kills, approval gate, adaptive model router + per-call cost ledger | `runtime/` |
|
|
53
|
+
| **Memory + MCP** | tree-sitter code graph, SQLite decision memory, MCP server exposing 5 tools to any MCP client (Claude Code, Cursor, …), MCP client for consuming external servers | `memory/`, `mcp_server/` |
|
|
54
|
+
|
|
55
|
+
## The novel mechanism: adaptive model routing
|
|
56
|
+
|
|
57
|
+
Per call, the runtime predicts difficulty (intrinsic signal from the
|
|
58
|
+
issue text + struggle signal from the conversation tail — failing test
|
|
59
|
+
output, burned turns) and routes easy/medium calls to a cheap model,
|
|
60
|
+
hard calls to an expensive one. Every call lands in a per-task JSONL
|
|
61
|
+
ledger (model, tokens, cost, hint) — the mechanism is measurable, not
|
|
62
|
+
asserted.
|
|
63
|
+
|
|
64
|
+
**Ablation results (real bugs, real models, full real stack —
|
|
65
|
+
scheduler subprocesses → real harness → Docker verify):**
|
|
66
|
+
|
|
67
|
+
| run | arm | success | calls | tokens | cost* | wall |
|
|
68
|
+
|---|---|---|---|---|---|---|
|
|
69
|
+
| 5 fixture bugs | always-expensive | 5/5 | 17 | 38,680 | $0.0528 | 575s |
|
|
70
|
+
| 5 fixture bugs | adaptive | 5/5 | 31 | 69,615 | $0.0237 | 300s |
|
|
71
|
+
| 16-task expanded set | always-expensive | 16/16 | 81 | 138,526 | $0.1505 | 2717s |
|
|
72
|
+
| 16-task expanded set | adaptive | 16/16 | 71 | 136,436 | $0.0581 | 812s |
|
|
73
|
+
| 5 real OSS repos (Round 6) | always-expensive | 2/5 | 71 | 329,438 | $0.3059 | 2992s |
|
|
74
|
+
| 5 real OSS repos (Round 6) | adaptive | 3/5 | 75 | 302,801 | $0.0730 | 581s |
|
|
75
|
+
|
|
76
|
+
(Aritfacts: `logs/ablations/v2-heuristic-*` (n=5) and
|
|
77
|
+
`logs/ablations/v4` (n=16; an earlier `v3-expanded` run is invalid —
|
|
78
|
+
fixture-path bug — superseded by `v3-expanded-fixed` and `v4`.)
|
|
79
|
+
|
|
80
|
+
- Same 100% success rate in every arm/run, at **45% (n=5) / 39%
|
|
81
|
+
(n=16) of baseline cost** — same direction, growing margin with a
|
|
82
|
+
more varied task set (16 tasks: 5 real fixture bugs + 11 synthesized
|
|
83
|
+
repos with varied bug classes AND varied issue-text styles).
|
|
84
|
+
- Escalations did what they should: one genuine struggle escalation
|
|
85
|
+
(cheap attempts failed → hard-tier call finished the task), zero
|
|
86
|
+
cost-wasting escalations across the 8 easy-styled texts, and 1 of 2
|
|
87
|
+
deliberately-SCARY texts (stack trace + "race" wording over a
|
|
88
|
+
one-token bug) tricked the intrinsic scorer into one expensive call —
|
|
89
|
+
the honest false-escalation data point (1/16 tasks).
|
|
90
|
+
|
|
91
|
+
\* Honesty notes: endpoints are free-tier BYO routers; token counts and
|
|
92
|
+
model-choice data are raw measurements from ledgers, costs use proxy
|
|
93
|
+
price rates for comparable model classes (both endpoints report no
|
|
94
|
+
cost) — the cost **delta** is a price-model delta, not a bill. n=5 and
|
|
95
|
+
n=16 × 1 rep are directional, not benchmark-grade. The Round-6
|
|
96
|
+
multi-repo arm additionally suffered endpoint degradation (2 timeouts in
|
|
97
|
+
the OFF arm were 0-call wall-clock kills, not model failures). Phase 6
|
|
98
|
+
should re-run on SWE-bench subsets with paid tiers.
|
|
99
|
+
## Multi-repo validation: the system beyond its home turf
|
|
100
|
+
|
|
101
|
+
Beyond the fixture set, the full stack has been validated against real,
|
|
102
|
+
unfamiliar OSS code — not just the original repo:
|
|
103
|
+
|
|
104
|
+
- **5 real OSS repos through the routing ablation (Round 6)** —
|
|
105
|
+
more-itertools, arrow, inflect, boltons, python-semver (pinned SHAs,
|
|
106
|
+
one genuine introduced bug each, real suites, per-repo suite pins for
|
|
107
|
+
dev-only deps): the adaptive arm went 3/5 (vs 2/5 always-expensive)
|
|
108
|
+
at **24% of the cost and 5x faster wall** — first multi-repo evidence
|
|
109
|
+
that the routing margin holds (and the failure modes are endpoint
|
|
110
|
+
timeouts, not harness bugs; honest data point: multi-hundred-K-token
|
|
111
|
+
real repos are simply harder than the fixture set, in BOTH arms).
|
|
112
|
+
(`logs/ablations/v6-multirepo/`)
|
|
113
|
+
- **jaraco/path (full DoD)** — unfamiliar real OSS repo end-to-end:
|
|
114
|
+
real cloud model, Docker sandbox, verifier-gated success in 1 attempt
|
|
115
|
+
($0.053, 6 calls), git-native branch/commit/PR, rationale.md,
|
|
116
|
+
approval gate, memory ingestion — plus one honest first failure that
|
|
117
|
+
exposed and fixed a real harness bug (binary-artifact diff crash).
|
|
118
|
+
(`logs/oss-round4/`)
|
|
119
|
+
- **python-semver (module DoD)** — pristine baseline → broken-state
|
|
120
|
+
detection → in-sandbox fix → verified, flake-flagging, git output,
|
|
121
|
+
grounded rationale: 15/15 checks. (`logs/dod/`)
|
|
122
|
+
- **3 more real repos staged for a final sweep (bottle, click, parse)**
|
|
123
|
+
— full-stack runs in flight; first attempts showed honest failures
|
|
124
|
+
(unparseable plans from the degraded free-tier endpoint → the
|
|
125
|
+
harness correctly refused to claim success). Numbers land when the
|
|
126
|
+
endpoint stabilizes; artifacts will live under `logs/oss-round6/`.
|
|
127
|
+
|
|
128
|
+
Net: 7 real OSS repos have been driven by the actual harness/verifier
|
|
129
|
+
stack (plus 3 in flight), spanning plugin, date/time, inflection, and
|
|
130
|
+
versioning domains — with the same verifier-gated honesty rules as the
|
|
131
|
+
fixture set: nothing above is a claimed success without the gate.
|
|
132
|
+
|
|
133
|
+
## Runtime reliability (proven, not claimed)
|
|
134
|
+
|
|
135
|
+
- **45 tasks @ concurrency 45, 8 simultaneous mid-run hard kills**:
|
|
136
|
+
45/45 success, 8/8 genuine resumes — verified from trace events
|
|
137
|
+
(`plan_reused`, `step_skipped_resume`, pre-kill trace survival), zero
|
|
138
|
+
leaked containers (`logs/stress/real-45/`).
|
|
139
|
+
- Concurrency cap proven from the event journal (max overlap ≤ cap);
|
|
140
|
+
every kill recorded + requeued; parallel beats the serial floor.
|
|
141
|
+
- Scheduler + worker share one log tree (spawn pins `resume_dir` /
|
|
142
|
+
`log_root`); a mid-run kill resumes from completed steps with all
|
|
143
|
+
artifacts under the caller's `--log-root`.
|
|
144
|
+
|
|
145
|
+
## Memory layer (MCP)
|
|
146
|
+
|
|
147
|
+
- `query_structure` — tree-sitter code graph (functions, classes,
|
|
148
|
+
calls, imports; persistent index)
|
|
149
|
+
- `query_decisions` / `record_decision` — decision/pattern memory
|
|
150
|
+
(auto-ingests every task's structured state)
|
|
151
|
+
- `task_status`, `list_repos`
|
|
152
|
+
|
|
153
|
+
Run it: `python -m mcp_server` (stdio) and connect any MCP client. The
|
|
154
|
+
harness's retrieval consumes the same graph programmatically, so
|
|
155
|
+
structural context rides into prompts without re-reading files.
|
|
156
|
+
|
|
157
|
+
## Demo script (5 minutes)
|
|
158
|
+
|
|
159
|
+
Two variants: **zero-setup offline** (deterministic, no key/Docker) and
|
|
160
|
+
the real-model walkthrough.
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
# 0) The whole story, offline in one command (scripted model; the loop,
|
|
164
|
+
# verifier gate, git output, rationale, and memory are all REAL):
|
|
165
|
+
python demo/run_demo.py # fix -> git/PR -> routing numbers -> memory -> dashboard hint
|
|
166
|
+
# per-step talking points: demo/README.md
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
With a real model (BYO endpoint/key):
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
# one-time: pip install vex-harness
|
|
173
|
+
# (from a clone instead: pip install -e . — or use python -m cli everywhere)
|
|
174
|
+
|
|
175
|
+
# 1) Fix a real bug with a real model (needs a BYO endpoint/key):
|
|
176
|
+
export MY_KEY=... # your openai-compatible router key
|
|
177
|
+
python -m cli fix \
|
|
178
|
+
--repo cli/fixtures/smoke_repo \
|
|
179
|
+
--issue "The mean() function in mathutil.py returns the sum instead of the arithmetic mean. Fix it so tests/test_mathutil.py::test_mean passes." \
|
|
180
|
+
--provider openai --model <model> --api-key $MY_KEY --api-base <base-url>
|
|
181
|
+
# → status, cost, diff; then inspect the artifacts:
|
|
182
|
+
python -m cli status --task-id <task_id> # plan checklist + decisions
|
|
183
|
+
type logs\<task_id>\rationale.md # what was wrong / what changed / why
|
|
184
|
+
type logs\<task_id>\git.json # branch + commit + PR description
|
|
185
|
+
|
|
186
|
+
# 2) Adaptive routing vs always-expensive, measured (the ablation):
|
|
187
|
+
# endpoints/keys are configured in runtime/ablation.py (BYO, env keys)
|
|
188
|
+
python -m runtime.ablation --tasks all --concurrency 2 # both arms, one summary
|
|
189
|
+
type logs\ablations\<ts>\summary.json # per-arm cost/token table
|
|
190
|
+
|
|
191
|
+
# 3) Concurrency + crash-resume at target scale (offline, scripted model):
|
|
192
|
+
python -m runtime.stress --mode real --tasks 45 --concurrency 45 --kill 8
|
|
193
|
+
|
|
194
|
+
# 4) The memory layer, queried over MCP (our own server, external client style):
|
|
195
|
+
python -m cli mcp call "python -m mcp_server" query_decisions --args "{\"query\": \"pytest\"}"
|
|
196
|
+
python -m cli mcp list-tools "python -m mcp_server"
|
|
197
|
+
|
|
198
|
+
# 5) Read-only dashboard over any run's logs:
|
|
199
|
+
python -m cli dashboard --logs-dir logs/ablations/<ts>/tasklogs
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
## Repo layout
|
|
203
|
+
|
|
204
|
+
```
|
|
205
|
+
harness/ agent loop: planner, steps, verifier gate, resume, git output
|
|
206
|
+
execution/ Docker sandbox + verify + git-native output + rationale
|
|
207
|
+
runtime/ scheduler, worker, checkpoint, router (novel mechanism), ablation
|
|
208
|
+
memory/ code graph (tree-sitter), decision store (SQLite), MCP client
|
|
209
|
+
mcp_server/ MCP exposure of memory/status (stdio)
|
|
210
|
+
cli/ the `harness` command (fix / run-benchmark / status / mcp / dashboard)
|
|
211
|
+
dashboard/ read-only web view of existing logs
|
|
212
|
+
demo/ one-command offline demo + walkthrough (run_demo.py)
|
|
213
|
+
tests/ ~300 tests incl. real e2e bug-fix runs and real process-kill resumes
|
|
214
|
+
logs/ (gitignored) per-task state, traces, ledgers, run journals
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
## Status & verification
|
|
218
|
+
|
|
219
|
+
- CI on every push: two workflow files (kept separate — the four
|
|
220
|
+
modules were built in parallel terminals): `ci.yml` (harness +
|
|
221
|
+
runtime suites, OS matrix, nightly full stress + adversarial
|
|
222
|
+
abuse) and `memory-cli-ci.yml` (memory/MCP incl. real stdio
|
|
223
|
+
round-trip, CLI offline e2e through the real Docker sandbox,
|
|
224
|
+
dashboard — across Linux/Windows/macOS). Badges above.
|
|
225
|
+
- Full test suite green (scheduler integration with real process
|
|
226
|
+
kills, router, memory, MCP incl. real stdio round-trip, dashboard).
|
|
227
|
+
- CI (`.github/workflows/ci.yml`): the harness suite runs on every
|
|
228
|
+
push across Linux/macOS/Windows × Python 3.10/3.12 — Docker-gated
|
|
229
|
+
e2e tests self-skip with an explicit reason on runners without
|
|
230
|
+
Docker; full-scale stress + abuse suites run nightly. Harness
|
|
231
|
+
internals: [docs/architecture-harness.md](docs/architecture-harness.md).
|
|
232
|
+
- **Adversarially tested (Round 6)**: the MCP server and CLI were
|
|
233
|
+
probed with crafted/hostile inputs — path traversal, shell-injection
|
|
234
|
+
payloads, SQL injection, malformed subsets, null bytes. One real
|
|
235
|
+
data leak (task-id path traversal in `task_status`/`harness status`)
|
|
236
|
+
was found live, fixed, and pinned by 101 adversarial tests; all other
|
|
237
|
+
surfaces held (per-probe outcomes in each module's AGENTS.md; the
|
|
238
|
+
Docker sandbox was adversarially confirmed separately — 24/24
|
|
239
|
+
sequential + concurrent attack suites).
|
|
240
|
+
- Contract between modules: `INTERFACES.md`. Module-by-module state
|
|
241
|
+
(what's built, what's stubbed, decisions): each module's
|
|
242
|
+
`AGENTS.md`. High-level build history: `CHANGELOG.md` (current
|
|
243
|
+
release: **v0.1.0**).
|
|
244
|
+
- Deferred per spec: SWE-bench Lite numbers (Phase 6), multi-language,
|
|
245
|
+
plugin marketplace.
|
|
246
|
+
|
|
247
|
+
## Tech
|
|
248
|
+
|
|
249
|
+
Python 3.10 · litellm (multi-provider, BYO-key) · Docker · tree-sitter
|
|
250
|
+
· MCP (official Python SDK) · argparse CLI · stdlib HTTP dashboard.
|
|
251
|
+
`litellm` is a real-model dependency (in `pyproject.toml`) pinned to
|
|
252
|
+
`1.74.9` on Python 3.10 (newer breaks the `typing` import on 3.10);
|
|
253
|
+
the offline/demo paths work without it (lazy import).
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Terminal 4 — CLI: the human-facing interface tying the system together.
|
|
2
|
+
|
|
3
|
+
Entry point: cli.main (console script ``vex`` — legacy alias ``harness``
|
|
4
|
+
kept during migration — or ``python -m cli``).
|
|
5
|
+
Commands (INTERFACES.md Boundary 6): fix, run-benchmark, status, memory,
|
|
6
|
+
dashboard, mcp; no-args interactive natural-language mode (see
|
|
7
|
+
cli/interactive.py — the primary UX).
|
|
8
|
+
"""
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Allow ``python -m cli`` alongside the ``vex`` console script."""
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _run() -> int:
|
|
7
|
+
try:
|
|
8
|
+
from cli.main import main
|
|
9
|
+
except ImportError as exc:
|
|
10
|
+
# Broken install / missing dependency (Task D): plain language, no
|
|
11
|
+
# traceback. This is the FIRST import of the package a user can
|
|
12
|
+
# hit, so an install problem lands exactly here.
|
|
13
|
+
print(
|
|
14
|
+
f"error: the Vex CLI could not be imported: {exc}\n"
|
|
15
|
+
'check: was the package installed? Run: pip install -e ".[dev]"\n'
|
|
16
|
+
"check: are you in the right environment (venv active)?",
|
|
17
|
+
file=sys.stderr,
|
|
18
|
+
)
|
|
19
|
+
return 2
|
|
20
|
+
return main()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
if __name__ == "__main__":
|
|
24
|
+
sys.exit(_run())
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""STUB for the scheduler boundary (INTERFACES.md Boundary 6) — Terminal 3
|
|
2
|
+
owns the real runtime.scheduler.run.
|
|
3
|
+
|
|
4
|
+
Concurrent fan-out via ThreadPoolExecutor calling run_task per task, with
|
|
5
|
+
per-task status lines printed to stdout — enough to exercise the CLI's
|
|
6
|
+
run-benchmark path end-to-end today. When runtime/scheduler.py lands,
|
|
7
|
+
cli.deps.get_scheduler_run() picks it up automatically (import-probe
|
|
8
|
+
first, stub second) — no CLI code changes.
|
|
9
|
+
|
|
10
|
+
Signature contract (Boundary 6):
|
|
11
|
+
run(tasks: list[Task], concurrency: int = 10, **kwargs) -> list[TaskResult]
|
|
12
|
+
|
|
13
|
+
Differences from the future real scheduler (documented for Terminal 3):
|
|
14
|
+
- No checkpoint/resume, no adaptive routing, no per-task worker processes
|
|
15
|
+
(threads, not processes — fine for a stub, since run_task is thread-safe
|
|
16
|
+
by its own docstring).
|
|
17
|
+
- Benchmarks: --subset is handled in the CLI; this stub receives the
|
|
18
|
+
already-materialized Task list.
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import threading
|
|
23
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
24
|
+
from typing import Any, Callable, Dict, List
|
|
25
|
+
|
|
26
|
+
from shared.types import Task, TaskResult
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def run(
|
|
30
|
+
tasks: List[Task],
|
|
31
|
+
concurrency: int = 10,
|
|
32
|
+
run_task: Callable[[Task], TaskResult] | None = None,
|
|
33
|
+
**kwargs: Any,
|
|
34
|
+
) -> List[TaskResult]:
|
|
35
|
+
"""STUB: run tasks concurrently via a thread pool; returns results in
|
|
36
|
+
task-list order (not completion order) for deterministic output.
|
|
37
|
+
|
|
38
|
+
Assumes `tasks` is a list of Task with distinct task_ids (run_task's
|
|
39
|
+
thread-safety requirement) and concurrency >= 1. Optionally accepts
|
|
40
|
+
an injected run_task (the CLI passes the resolved Boundary 3 callable
|
|
41
|
+
so tests can stub one level down).
|
|
42
|
+
"""
|
|
43
|
+
if not tasks:
|
|
44
|
+
return []
|
|
45
|
+
concurrency = max(1, min(int(concurrency), len(tasks)))
|
|
46
|
+
fn = run_task or _resolve_run_task()
|
|
47
|
+
results: Dict[str, TaskResult] = {}
|
|
48
|
+
lock = threading.Lock()
|
|
49
|
+
done_count = [0]
|
|
50
|
+
|
|
51
|
+
with ThreadPoolExecutor(max_workers=concurrency) as pool:
|
|
52
|
+
futures = {pool.submit(fn, t): t for t in tasks}
|
|
53
|
+
for fut in as_completed(futures):
|
|
54
|
+
task = futures[fut]
|
|
55
|
+
try:
|
|
56
|
+
result = fut.result()
|
|
57
|
+
except Exception as exc: # a crashed worker must not lose the batch
|
|
58
|
+
result = TaskResult(
|
|
59
|
+
task_id=task.task_id,
|
|
60
|
+
status="error",
|
|
61
|
+
attempts=0,
|
|
62
|
+
diff=None,
|
|
63
|
+
verification=None,
|
|
64
|
+
cost_usd=0.0,
|
|
65
|
+
model_calls=[],
|
|
66
|
+
log_path="",
|
|
67
|
+
)
|
|
68
|
+
with lock:
|
|
69
|
+
print(f"[scheduler-stub] task {task.task_id} crashed: {exc}")
|
|
70
|
+
with lock:
|
|
71
|
+
results[task.task_id] = result
|
|
72
|
+
done_count[0] += 1
|
|
73
|
+
status = getattr(result, "status", "?")
|
|
74
|
+
print(
|
|
75
|
+
f"[scheduler-stub] [{done_count[0]}/{len(tasks)}] "
|
|
76
|
+
f"{task.task_id}: {status}"
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
return [results[t.task_id] for t in tasks]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _resolve_run_task() -> Callable[[Task], TaskResult]:
|
|
83
|
+
"""Same resolution as cli.deps (kept local to avoid an import cycle
|
|
84
|
+
when cli.deps itself probes this stub)."""
|
|
85
|
+
from harness.core import run_task
|
|
86
|
+
|
|
87
|
+
return run_task
|