neosloc 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- neosloc-0.3.0/LICENSE +21 -0
- neosloc-0.3.0/PKG-INFO +141 -0
- neosloc-0.3.0/README.md +114 -0
- neosloc-0.3.0/neosloc/__init__.py +2 -0
- neosloc-0.3.0/neosloc/__main__.py +5 -0
- neosloc-0.3.0/neosloc/agentic/__init__.py +10 -0
- neosloc-0.3.0/neosloc/agentic/grader.py +186 -0
- neosloc-0.3.0/neosloc/agentic/judge.py +59 -0
- neosloc-0.3.0/neosloc/agentic/llm.py +392 -0
- neosloc-0.3.0/neosloc/agentic/loop.py +79 -0
- neosloc-0.3.0/neosloc/agentic/review.py +122 -0
- neosloc-0.3.0/neosloc/agentic/runner.py +176 -0
- neosloc-0.3.0/neosloc/agentic/tasks.py +60 -0
- neosloc-0.3.0/neosloc/agentic/workspace.py +261 -0
- neosloc-0.3.0/neosloc/cli.py +171 -0
- neosloc-0.3.0/neosloc/detectors/__init__.py +18 -0
- neosloc-0.3.0/neosloc/detectors/base.py +25 -0
- neosloc-0.3.0/neosloc/detectors/embeddability.py +114 -0
- neosloc-0.3.0/neosloc/detectors/ergonomics.py +59 -0
- neosloc-0.3.0/neosloc/detectors/events.py +61 -0
- neosloc-0.3.0/neosloc/detectors/extensibility.py +48 -0
- neosloc-0.3.0/neosloc/detectors/identity.py +55 -0
- neosloc-0.3.0/neosloc/detectors/interface.py +223 -0
- neosloc-0.3.0/neosloc/detectors/legibility.py +301 -0
- neosloc-0.3.0/neosloc/detectors/observability.py +44 -0
- neosloc-0.3.0/neosloc/detectors/portability.py +49 -0
- neosloc-0.3.0/neosloc/detectors/signals.py +90 -0
- neosloc-0.3.0/neosloc/detectors/stability.py +139 -0
- neosloc-0.3.0/neosloc/estimate.py +80 -0
- neosloc-0.3.0/neosloc/model.py +72 -0
- neosloc-0.3.0/neosloc/repo.py +228 -0
- neosloc-0.3.0/neosloc/report.py +159 -0
- neosloc-0.3.0/neosloc/specs.py +112 -0
- neosloc-0.3.0/neosloc/value.py +139 -0
- neosloc-0.3.0/neosloc.egg-info/PKG-INFO +141 -0
- neosloc-0.3.0/neosloc.egg-info/SOURCES.txt +44 -0
- neosloc-0.3.0/neosloc.egg-info/dependency_links.txt +1 -0
- neosloc-0.3.0/neosloc.egg-info/entry_points.txt +2 -0
- neosloc-0.3.0/neosloc.egg-info/requires.txt +5 -0
- neosloc-0.3.0/neosloc.egg-info/top_level.txt +1 -0
- neosloc-0.3.0/pyproject.toml +45 -0
- neosloc-0.3.0/setup.cfg +4 -0
- neosloc-0.3.0/tests/test_agentic.py +309 -0
- neosloc-0.3.0/tests/test_neosloc.py +327 -0
- neosloc-0.3.0/tests/test_openrouter.py +271 -0
- neosloc-0.3.0/tests/test_sdk_contract.py +94 -0
neosloc-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Marco Montanari
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
neosloc-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: neosloc
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Integrability evaluation for software repositories: what sloccount measured, re-asked for the age of LLMs.
|
|
5
|
+
Author: Marco Montanari
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/sirmmo/neosloc
|
|
8
|
+
Project-URL: Documentation, https://ingmmo.com/neosloc/
|
|
9
|
+
Project-URL: Repository, https://github.com/sirmmo/neosloc
|
|
10
|
+
Project-URL: Issues, https://github.com/sirmmo/neosloc/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/sirmmo/neosloc/blob/main/CHANGELOG.md
|
|
12
|
+
Keywords: sloccount,cocomo,integrability,software metrics,llm,agents,api
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Environment :: Console
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
21
|
+
Requires-Python: >=3.8
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Provides-Extra: agentic
|
|
25
|
+
Requires-Dist: anthropic>=1.11; python_version >= "3.10" and extra == "agentic"
|
|
26
|
+
Dynamic: license-file
|
|
27
|
+
|
|
28
|
+
# neosloc
|
|
29
|
+
|
|
30
|
+
[](https://github.com/sirmmo/neosloc/actions/workflows/ci.yml)
|
|
31
|
+
[](https://pypi.org/project/neosloc/)
|
|
32
|
+
[](https://ingmmo.com/neosloc/)
|
|
33
|
+
|
|
34
|
+
`sloccount` counted lines of code and fed them to COCOMO to estimate the effort to *build*
|
|
35
|
+
software. With LLMs writing code, lines are cheap and that estimate has lost its meaning.
|
|
36
|
+
What is still expensive is everything *around* the code: how easily other systems and
|
|
37
|
+
agents can call it, run it, rely on it and change it safely, and the knowledge that exists
|
|
38
|
+
only in the code and its history.
|
|
39
|
+
|
|
40
|
+
neosloc measures that:
|
|
41
|
+
|
|
42
|
+
- **ten integrability dimensions**, each scored 0–4 with the evidence behind the level and
|
|
43
|
+
the gaps phrased as actions;
|
|
44
|
+
- the **retrofit effort** to bring every dimension up to "solid", and whether the system
|
|
45
|
+
is cheap to wrap;
|
|
46
|
+
- **neoCOCOMO**: what the codebase is worth when an agent can re-type it, next to classic
|
|
47
|
+
COCOMO;
|
|
48
|
+
- optional **LLM evaluators** on Anthropic or [OpenRouter](https://openrouter.ai): a
|
|
49
|
+
*probe* attempts real integration tasks using only the docs, a *judge* checks the
|
|
50
|
+
answers against the source, and a *review* audits every static level.
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
pip install neosloc
|
|
54
|
+
neosloc path/to/repo
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
```text
|
|
58
|
+
Interface surface ■■■□ 3 solid The framework generates a contract from code, so it tracks the implementation.
|
|
59
|
+
+ routes:fastapi/flask: 15 route declarations in 3 files (app/routes/admin.py)
|
|
60
|
+
+ generated-spec: FastAPI (auto OpenAPI) (app/main.py)
|
|
61
|
+
- Check the generated spec into the repo so changes are reviewable and diffable.
|
|
62
|
+
...
|
|
63
|
+
Integrability index (0-4, assessed dimensions): 1.80
|
|
64
|
+
Retrofit effort to level 3 (agent-assisted): 7.5 person-days
|
|
65
|
+
|
|
66
|
+
Value (neoCOCOMO, cost approach; see neosloc/value.py)
|
|
67
|
+
Classic COCOMO (sloccount): 4.6 KSLOC -> 11.8 person-months, 6.4 months, $133k
|
|
68
|
+
Behaviour captured: 38% (tests 0.14, contract 0.25, docs 1.00)
|
|
69
|
+
Reproduce with agents: x0.38 of classic -> 4.5 PM
|
|
70
|
+
Rediscover uncaptured history: 102 fix / 474 commits, 19 authors -> 4.4 PM knowledge, 2.7 PM uncaptured
|
|
71
|
+
Value: 5.2 PM ~ $59k (at $11k per PM)
|
|
72
|
+
Knowledge at risk: 38% of replacement cost lives only in code and history
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
**Documentation: <https://ingmmo.com/neosloc/>**
|
|
76
|
+
|
|
77
|
+
## Install
|
|
78
|
+
|
|
79
|
+
| | |
|
|
80
|
+
|---|---|
|
|
81
|
+
| `pip install neosloc` (or `pipx install neosloc`, `uv tool install neosloc`) | the `neosloc` command; the static analysis has **no dependencies** and runs on Python ≥ 3.8 |
|
|
82
|
+
| `pip install 'neosloc[agentic]'` | adds the anthropic SDK, needed for `anthropic:` models (Python ≥ 3.10) |
|
|
83
|
+
| `pip install git+https://github.com/sirmmo/neosloc` | the development version |
|
|
84
|
+
|
|
85
|
+
OpenRouter models need no extra package, only `OPENROUTER_API_KEY`.
|
|
86
|
+
|
|
87
|
+
## Dimensions
|
|
88
|
+
|
|
89
|
+
| Key | Asks |
|
|
90
|
+
|---|---|
|
|
91
|
+
| `interface` | Is there a machine-readable contract covering the implemented surface, and more than one way in (MCP, CLI `--json`, SDK, typed library)? |
|
|
92
|
+
| `stability` | Semver, changelog, API versioning, deprecations, and the real git history of the OpenAPI files. |
|
|
93
|
+
| `events` | Outbound webhooks (signed, self-service), streams, brokers, CDC, AsyncAPI/CloudEvents. |
|
|
94
|
+
| `identity` | Token auth, OAuth2/OIDC, scopes, managed API keys or service accounts, SCIM. |
|
|
95
|
+
| `portability` | Bulk export/import, open formats, schema in the repo. |
|
|
96
|
+
| `ergonomics` | RFC 9457 errors, validation, idempotency keys, pagination, rate-limit headers, dry-run, `llms.txt`. |
|
|
97
|
+
| `embeddability` | Container, compose/Helm/IaC, documented env config, headless entry point, library packaging. |
|
|
98
|
+
| `extensibility` | Plugin discovery, hooks, scripting, extension docs. |
|
|
99
|
+
| `observability` | Health/readiness, metrics, tracing, structured logs, error tracking. |
|
|
100
|
+
| `legibility` | The heir to SLOC: tokens per module, import cycles, tests, types, CI, lockfiles, agent docs. |
|
|
101
|
+
|
|
102
|
+
Details: [dimensions](https://ingmmo.com/neosloc/dimensions/),
|
|
103
|
+
[retrofit effort](https://ingmmo.com/neosloc/estimate/),
|
|
104
|
+
[neoCOCOMO](https://ingmmo.com/neosloc/value/).
|
|
105
|
+
|
|
106
|
+
## LLM evaluators
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
# Probe: a model attempts 8 integration tasks from the docs; answers are graded against the code
|
|
110
|
+
neosloc --agentic --scope both --transcripts runs/ .
|
|
111
|
+
|
|
112
|
+
# Probe with Claude, judge with another vendor through OpenRouter
|
|
113
|
+
neosloc --judge --judge-model openrouter:openai/gpt-5.6-luna .
|
|
114
|
+
|
|
115
|
+
# Review: a panel audits every static level against the code
|
|
116
|
+
neosloc --review --review-model openrouter:google/gemini-3.8-flash,openrouter:deepseek/deepseek-v4-pro-0813 .
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Models are written `provider:model`. The default is `anthropic:claude-opus-5-5` at effort
|
|
120
|
+
`medium`. A comma-separated list runs a panel and reports agreement. Every model call
|
|
121
|
+
shares one `--max-cost` budget (default $5), and the report states the total. See
|
|
122
|
+
[LLM evaluators](https://ingmmo.com/neosloc/evaluators/) and
|
|
123
|
+
[Using OpenRouter](https://ingmmo.com/neosloc/openrouter/).
|
|
124
|
+
|
|
125
|
+
## Status
|
|
126
|
+
|
|
127
|
+
All thresholds and coefficients are uncalibrated, so treat the numbers as rankings.
|
|
128
|
+
[Calibration and limits](https://ingmmo.com/neosloc/calibration/) describes how to fix
|
|
129
|
+
that; `--review` and `--agentic` exist partly to do it.
|
|
130
|
+
|
|
131
|
+
## Development
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
python -m unittest discover -s tests -t . # no network, no API spend
|
|
135
|
+
python -m neosloc .
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
See [Writing detectors](https://ingmmo.com/neosloc/extending/) and `AGENTS.md`.
|
|
139
|
+
Releases: bump `neosloc/__init__.py`, add a `CHANGELOG.md` section, push a `vX.Y.Z` tag.
|
|
140
|
+
|
|
141
|
+
MIT licensed.
|
neosloc-0.3.0/README.md
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# neosloc
|
|
2
|
+
|
|
3
|
+
[](https://github.com/sirmmo/neosloc/actions/workflows/ci.yml)
|
|
4
|
+
[](https://pypi.org/project/neosloc/)
|
|
5
|
+
[](https://ingmmo.com/neosloc/)
|
|
6
|
+
|
|
7
|
+
`sloccount` counted lines of code and fed them to COCOMO to estimate the effort to *build*
|
|
8
|
+
software. With LLMs writing code, lines are cheap and that estimate has lost its meaning.
|
|
9
|
+
What is still expensive is everything *around* the code: how easily other systems and
|
|
10
|
+
agents can call it, run it, rely on it and change it safely, and the knowledge that exists
|
|
11
|
+
only in the code and its history.
|
|
12
|
+
|
|
13
|
+
neosloc measures that:
|
|
14
|
+
|
|
15
|
+
- **ten integrability dimensions**, each scored 0–4 with the evidence behind the level and
|
|
16
|
+
the gaps phrased as actions;
|
|
17
|
+
- the **retrofit effort** to bring every dimension up to "solid", and whether the system
|
|
18
|
+
is cheap to wrap;
|
|
19
|
+
- **neoCOCOMO**: what the codebase is worth when an agent can re-type it, next to classic
|
|
20
|
+
COCOMO;
|
|
21
|
+
- optional **LLM evaluators** on Anthropic or [OpenRouter](https://openrouter.ai): a
|
|
22
|
+
*probe* attempts real integration tasks using only the docs, a *judge* checks the
|
|
23
|
+
answers against the source, and a *review* audits every static level.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install neosloc
|
|
27
|
+
neosloc path/to/repo
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
```text
|
|
31
|
+
Interface surface ■■■□ 3 solid The framework generates a contract from code, so it tracks the implementation.
|
|
32
|
+
+ routes:fastapi/flask: 15 route declarations in 3 files (app/routes/admin.py)
|
|
33
|
+
+ generated-spec: FastAPI (auto OpenAPI) (app/main.py)
|
|
34
|
+
- Check the generated spec into the repo so changes are reviewable and diffable.
|
|
35
|
+
...
|
|
36
|
+
Integrability index (0-4, assessed dimensions): 1.80
|
|
37
|
+
Retrofit effort to level 3 (agent-assisted): 7.5 person-days
|
|
38
|
+
|
|
39
|
+
Value (neoCOCOMO, cost approach; see neosloc/value.py)
|
|
40
|
+
Classic COCOMO (sloccount): 4.6 KSLOC -> 11.8 person-months, 6.4 months, $133k
|
|
41
|
+
Behaviour captured: 38% (tests 0.14, contract 0.25, docs 1.00)
|
|
42
|
+
Reproduce with agents: x0.38 of classic -> 4.5 PM
|
|
43
|
+
Rediscover uncaptured history: 102 fix / 474 commits, 19 authors -> 4.4 PM knowledge, 2.7 PM uncaptured
|
|
44
|
+
Value: 5.2 PM ~ $59k (at $11k per PM)
|
|
45
|
+
Knowledge at risk: 38% of replacement cost lives only in code and history
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
**Documentation: <https://ingmmo.com/neosloc/>**
|
|
49
|
+
|
|
50
|
+
## Install
|
|
51
|
+
|
|
52
|
+
| | |
|
|
53
|
+
|---|---|
|
|
54
|
+
| `pip install neosloc` (or `pipx install neosloc`, `uv tool install neosloc`) | the `neosloc` command; the static analysis has **no dependencies** and runs on Python ≥ 3.8 |
|
|
55
|
+
| `pip install 'neosloc[agentic]'` | adds the anthropic SDK, needed for `anthropic:` models (Python ≥ 3.10) |
|
|
56
|
+
| `pip install git+https://github.com/sirmmo/neosloc` | the development version |
|
|
57
|
+
|
|
58
|
+
OpenRouter models need no extra package, only `OPENROUTER_API_KEY`.
|
|
59
|
+
|
|
60
|
+
## Dimensions
|
|
61
|
+
|
|
62
|
+
| Key | Asks |
|
|
63
|
+
|---|---|
|
|
64
|
+
| `interface` | Is there a machine-readable contract covering the implemented surface, and more than one way in (MCP, CLI `--json`, SDK, typed library)? |
|
|
65
|
+
| `stability` | Semver, changelog, API versioning, deprecations, and the real git history of the OpenAPI files. |
|
|
66
|
+
| `events` | Outbound webhooks (signed, self-service), streams, brokers, CDC, AsyncAPI/CloudEvents. |
|
|
67
|
+
| `identity` | Token auth, OAuth2/OIDC, scopes, managed API keys or service accounts, SCIM. |
|
|
68
|
+
| `portability` | Bulk export/import, open formats, schema in the repo. |
|
|
69
|
+
| `ergonomics` | RFC 9457 errors, validation, idempotency keys, pagination, rate-limit headers, dry-run, `llms.txt`. |
|
|
70
|
+
| `embeddability` | Container, compose/Helm/IaC, documented env config, headless entry point, library packaging. |
|
|
71
|
+
| `extensibility` | Plugin discovery, hooks, scripting, extension docs. |
|
|
72
|
+
| `observability` | Health/readiness, metrics, tracing, structured logs, error tracking. |
|
|
73
|
+
| `legibility` | The heir to SLOC: tokens per module, import cycles, tests, types, CI, lockfiles, agent docs. |
|
|
74
|
+
|
|
75
|
+
Details: [dimensions](https://ingmmo.com/neosloc/dimensions/),
|
|
76
|
+
[retrofit effort](https://ingmmo.com/neosloc/estimate/),
|
|
77
|
+
[neoCOCOMO](https://ingmmo.com/neosloc/value/).
|
|
78
|
+
|
|
79
|
+
## LLM evaluators
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
# Probe: a model attempts 8 integration tasks from the docs; answers are graded against the code
|
|
83
|
+
neosloc --agentic --scope both --transcripts runs/ .
|
|
84
|
+
|
|
85
|
+
# Probe with Claude, judge with another vendor through OpenRouter
|
|
86
|
+
neosloc --judge --judge-model openrouter:openai/gpt-5.6-luna .
|
|
87
|
+
|
|
88
|
+
# Review: a panel audits every static level against the code
|
|
89
|
+
neosloc --review --review-model openrouter:google/gemini-3.8-flash,openrouter:deepseek/deepseek-v4-pro-0813 .
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Models are written `provider:model`. The default is `anthropic:claude-opus-5-5` at effort
|
|
93
|
+
`medium`. A comma-separated list runs a panel and reports agreement. Every model call
|
|
94
|
+
shares one `--max-cost` budget (default $5), and the report states the total. See
|
|
95
|
+
[LLM evaluators](https://ingmmo.com/neosloc/evaluators/) and
|
|
96
|
+
[Using OpenRouter](https://ingmmo.com/neosloc/openrouter/).
|
|
97
|
+
|
|
98
|
+
## Status
|
|
99
|
+
|
|
100
|
+
All thresholds and coefficients are uncalibrated, so treat the numbers as rankings.
|
|
101
|
+
[Calibration and limits](https://ingmmo.com/neosloc/calibration/) describes how to fix
|
|
102
|
+
that; `--review` and `--agentic` exist partly to do it.
|
|
103
|
+
|
|
104
|
+
## Development
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
python -m unittest discover -s tests -t . # no network, no API spend
|
|
108
|
+
python -m neosloc .
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
See [Writing detectors](https://ingmmo.com/neosloc/extending/) and `AGENTS.md`.
|
|
112
|
+
Releases: bump `neosloc/__init__.py`, add a `CHANGELOG.md` section, push a `vX.Y.Z` tag.
|
|
113
|
+
|
|
114
|
+
MIT licensed.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""LLM evaluators: probe (attempt integration tasks), judge (check answers
|
|
2
|
+
against the source) and review (audit static levels), on Anthropic or OpenRouter."""
|
|
3
|
+
from .judge import Judge
|
|
4
|
+
from .llm import Budget, make_backend, parse_spec
|
|
5
|
+
from .review import Reviewer, review_all
|
|
6
|
+
from .runner import DEFAULT_EFFORT, DEFAULT_MODEL, Probe, panel
|
|
7
|
+
from .tasks import TASKS, select
|
|
8
|
+
|
|
9
|
+
__all__ = ["Budget", "DEFAULT_EFFORT", "DEFAULT_MODEL", "Judge", "Probe", "Reviewer", "TASKS",
|
|
10
|
+
"make_backend", "panel", "parse_spec", "review_all", "select"]
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Grounding: check an agent's answer against the implementation.
|
|
2
|
+
|
|
3
|
+
The agent may only have seen the docs; the grader always sees everything. A
|
|
4
|
+
step is grounded when the thing it names exists: the HTTP route is declared
|
|
5
|
+
(or in the contract), the CLI program and its flags exist, the env vars are
|
|
6
|
+
read somewhere, the symbol is defined. With a live instance, HTTP steps must
|
|
7
|
+
also have been exercised successfully.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import posixpath
|
|
12
|
+
import re
|
|
13
|
+
import shlex
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from typing import Dict, List, Optional, Set
|
|
16
|
+
|
|
17
|
+
from ..repo import Repo
|
|
18
|
+
from ..specs import find_specs
|
|
19
|
+
from .workspace import HttpCall
|
|
20
|
+
|
|
21
|
+
PARAM_RE = re.compile(r"\{[^}/]+\}|<[^>/]+>|:[A-Za-z_]\w*|\[[^\]/]+\]|\$\{[^}]+\}")
|
|
22
|
+
DYNAMIC_SEG = re.compile(r"^(\d+|[0-9a-f]{8}-[0-9a-f-]{27,}|[0-9a-f]{24,})$", re.I)
|
|
23
|
+
ROUTE_STRING = re.compile(r"""['"`](\^?/?[A-Za-z0-9_{}<>:\[\]$./-]*)['"`]""")
|
|
24
|
+
FLAG_RE = re.compile(r"(?<![\w-])(--[A-Za-z][\w-]*)")
|
|
25
|
+
ENV_NAME = re.compile(r"^[A-Z][A-Z0-9_]{2,}$")
|
|
26
|
+
LAUNCHERS = {"python", "python3", "-m", "npx", "pnpm", "yarn", "npm", "bunx", "uv", "poetry", "pipx",
|
|
27
|
+
"run", "exec", "sudo", "env", "bundle", "go", "cargo", "dotnet", "java", "-jar", "php"}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def normalize_path(path: str) -> str:
|
|
31
|
+
path = re.sub(r"^\w+://[^/]+", "", path.strip()).split("?", 1)[0].split("#", 1)[0]
|
|
32
|
+
path = path.lstrip("^").rstrip("$")
|
|
33
|
+
path = PARAM_RE.sub("{}", path)
|
|
34
|
+
segs = [("{}" if DYNAMIC_SEG.match(s) else s) for s in path.strip("/").split("/") if s]
|
|
35
|
+
return "/" + "/".join(segs)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class Grade:
|
|
40
|
+
grounded: int = 0
|
|
41
|
+
checked: int = 0
|
|
42
|
+
problems: List[str] = field(default_factory=list)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Grader:
|
|
46
|
+
def __init__(self, repo: Repo, route_files: Optional[List[str]] = None):
|
|
47
|
+
self.repo = repo
|
|
48
|
+
self.src = repo.source_files(include_tests=False)
|
|
49
|
+
self.spec_ops: Set[tuple] = set()
|
|
50
|
+
for s in find_specs(repo):
|
|
51
|
+
if s.kind == "openapi":
|
|
52
|
+
for op in s.operations:
|
|
53
|
+
m, p = op.split(" ", 1)
|
|
54
|
+
self.spec_ops.add((m, normalize_path(p)))
|
|
55
|
+
self.route_templates: Set[str] = set()
|
|
56
|
+
for f in route_files or []:
|
|
57
|
+
for line in repo.read(f).splitlines():
|
|
58
|
+
for lit in ROUTE_STRING.findall(line):
|
|
59
|
+
if lit.strip("/^$") and not lit.startswith("."):
|
|
60
|
+
self.route_templates.add(normalize_path(lit))
|
|
61
|
+
self._all_text = None
|
|
62
|
+
|
|
63
|
+
# ---- helpers -----------------------------------------------------------
|
|
64
|
+
|
|
65
|
+
def _text(self) -> str:
|
|
66
|
+
if self._all_text is None:
|
|
67
|
+
files = self.src + self.repo.manifests() + self.repo.config_files() + self.repo.doc_files()
|
|
68
|
+
self._all_text = "\n".join(self.repo.read(f) for f in dict.fromkeys(files))
|
|
69
|
+
return self._all_text
|
|
70
|
+
|
|
71
|
+
def _code_text(self) -> str:
|
|
72
|
+
return "\n".join(self.repo.read(f) for f in self.src + self.repo.config_files())
|
|
73
|
+
|
|
74
|
+
@staticmethod
|
|
75
|
+
def _suffix_match(candidate: str, template: str) -> bool:
|
|
76
|
+
c = candidate.strip("/").split("/")
|
|
77
|
+
t = template.strip("/").split("/")
|
|
78
|
+
if not t or t == [""] or len(t) > len(c):
|
|
79
|
+
return False
|
|
80
|
+
return all(a == b or b == "{}" or a == "{}" for a, b in zip(c[-len(t):], t))
|
|
81
|
+
|
|
82
|
+
# ---- step checks -------------------------------------------------------
|
|
83
|
+
|
|
84
|
+
def http(self, method: Optional[str], path: Optional[str], live: Optional[List[HttpCall]]) -> Optional[str]:
|
|
85
|
+
if not path:
|
|
86
|
+
return "http step without a path"
|
|
87
|
+
cand = normalize_path(path)
|
|
88
|
+
method = (method or "").upper()
|
|
89
|
+
in_spec = any((not method or m == method) and self._suffix_match(cand, p) for m, p in self.spec_ops)
|
|
90
|
+
in_code = any(self._suffix_match(cand, t) for t in self.route_templates)
|
|
91
|
+
if not (in_spec or in_code):
|
|
92
|
+
return "%s %s is not a declared route" % (method or "?", path)
|
|
93
|
+
if live is not None:
|
|
94
|
+
ok = [c for c in live if self._suffix_match(normalize_path(c.path), cand)
|
|
95
|
+
and (not method or c.method == method)]
|
|
96
|
+
if not ok:
|
|
97
|
+
return "%s %s was never exercised against the live instance" % (method or "?", path)
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
def cli(self, command: Optional[str]) -> Optional[str]:
|
|
101
|
+
if not command:
|
|
102
|
+
return "cli step without a command"
|
|
103
|
+
try:
|
|
104
|
+
words = shlex.split(command.splitlines()[0])
|
|
105
|
+
except ValueError:
|
|
106
|
+
words = command.split()
|
|
107
|
+
if not words:
|
|
108
|
+
return "empty command"
|
|
109
|
+
if words[0] == "docker" or words[:2] == ["docker-compose"]:
|
|
110
|
+
ok = self.repo.glob(r"(^|/)(Dockerfile|Containerfile|(docker-)?compose[\w.-]*\.ya?ml)$")
|
|
111
|
+
return None if ok else "docker command but no Dockerfile/compose file"
|
|
112
|
+
if words[0] in ("make", "just"):
|
|
113
|
+
target = words[1] if len(words) > 1 else None
|
|
114
|
+
mk = self.repo.glob(r"(^|/)(Makefile|justfile)$")
|
|
115
|
+
if not mk:
|
|
116
|
+
return "%s command but no %s" % (words[0], "Makefile" if words[0] == "make" else "justfile")
|
|
117
|
+
if target and not re.search(r"^%s\s*:" % re.escape(target), self.repo.read(mk[0]), re.M):
|
|
118
|
+
return "no %s target %r" % (words[0], target)
|
|
119
|
+
return None
|
|
120
|
+
program = next((w for w in words if w not in LAUNCHERS and not w.startswith("-")
|
|
121
|
+
and "=" not in w), None)
|
|
122
|
+
if program is None:
|
|
123
|
+
return "could not identify the program in %r" % command
|
|
124
|
+
text = self._text()
|
|
125
|
+
base = posixpath.basename(program.replace("\\", "/"))
|
|
126
|
+
stem = posixpath.splitext(base)[0]
|
|
127
|
+
known = (self.repo.exists(program.lstrip("./")) or any(posixpath.basename(f) == base for f in self.repo.files)
|
|
128
|
+
or re.search(r"""(^|[\s"'\[])%s\s*=|['"]%s['"]\s*:""" % (re.escape(stem), re.escape(stem)),
|
|
129
|
+
"\n".join(self.repo.read(m) for m in self.repo.manifests()), re.M)
|
|
130
|
+
or any(f.split("/")[0] == stem or ("/%s/" % stem) in ("/" + f) for f in self.src))
|
|
131
|
+
if not known:
|
|
132
|
+
return "program %r is not provided by this repository" % program
|
|
133
|
+
missing = [fl for fl in FLAG_RE.findall(command) if fl not in text]
|
|
134
|
+
if missing:
|
|
135
|
+
return "flags not found in the code: %s" % ", ".join(missing)
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
def library(self, symbol: Optional[str]) -> Optional[str]:
|
|
139
|
+
if not symbol:
|
|
140
|
+
return "library step without a symbol"
|
|
141
|
+
name = re.split(r"[.:/#]", symbol.strip().rstrip("()"))[-1]
|
|
142
|
+
if not name:
|
|
143
|
+
return "unparseable symbol %r" % symbol
|
|
144
|
+
rx = re.compile(r"\b(def|class|function|func|fn|const|let|var|interface|type|struct|trait|export\s+\w+)\s+"
|
|
145
|
+
r"(\([^)]*\)\s*)?%s\b|\b%s\s*[:=]\s*(function|\(|async)" % (re.escape(name), re.escape(name)))
|
|
146
|
+
if not any(rx.search(self.repo.read(f)) for f in self.src):
|
|
147
|
+
return "symbol %r is not defined in the code" % symbol
|
|
148
|
+
return None
|
|
149
|
+
|
|
150
|
+
def env(self, names: List[str]) -> Optional[str]:
|
|
151
|
+
names = [n for n in names if ENV_NAME.match(n)]
|
|
152
|
+
if not names:
|
|
153
|
+
return None
|
|
154
|
+
code = self._code_text()
|
|
155
|
+
missing = [n for n in names if n not in code]
|
|
156
|
+
return "not read by the code: %s" % ", ".join(missing) if missing else None
|
|
157
|
+
|
|
158
|
+
# ---- whole answer ------------------------------------------------------
|
|
159
|
+
|
|
160
|
+
def grade(self, answer: Dict, live: Optional[List[HttpCall]] = None) -> Grade:
|
|
161
|
+
g = Grade()
|
|
162
|
+
for step in answer.get("steps", []):
|
|
163
|
+
kind = step.get("kind")
|
|
164
|
+
checks = []
|
|
165
|
+
if kind == "http":
|
|
166
|
+
checks.append(self.http(step.get("method"), step.get("path"), live))
|
|
167
|
+
elif kind == "cli":
|
|
168
|
+
checks.append(self.cli(step.get("command")))
|
|
169
|
+
elif kind == "library":
|
|
170
|
+
checks.append(self.library(step.get("symbol")))
|
|
171
|
+
elif kind == "config" and not step.get("env_vars"):
|
|
172
|
+
checks.append("config step names no settings")
|
|
173
|
+
if step.get("env_vars"):
|
|
174
|
+
checks.append(self.env(step["env_vars"]))
|
|
175
|
+
if not checks:
|
|
176
|
+
continue # "other" steps can't be verified; they neither help nor hurt
|
|
177
|
+
g.checked += 1
|
|
178
|
+
errors = [c for c in checks if c]
|
|
179
|
+
if errors:
|
|
180
|
+
g.problems.extend(errors)
|
|
181
|
+
else:
|
|
182
|
+
g.grounded += 1
|
|
183
|
+
missing = [p for p in answer.get("evidence", []) if not self.repo.exists(p.lstrip("./").split(":")[0])]
|
|
184
|
+
if missing:
|
|
185
|
+
g.problems.append("cited files do not exist: %s" % ", ".join(missing[:3]))
|
|
186
|
+
return g
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""The judge role: would a grounded answer actually accomplish the task?
|
|
2
|
+
|
|
3
|
+
Grounding proves that what the probe named exists. The judge, which may be a
|
|
4
|
+
different model and provider, reads the full source and decides whether the
|
|
5
|
+
steps would work as described (right method, required fields, correct order,
|
|
6
|
+
auth). A probe answer only counts as a success when both agree.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from typing import Any, Dict
|
|
12
|
+
|
|
13
|
+
from ..repo import Repo
|
|
14
|
+
from .llm import Budget
|
|
15
|
+
from .loop import run_loop
|
|
16
|
+
from .tasks import Task
|
|
17
|
+
from .workspace import Workspace
|
|
18
|
+
|
|
19
|
+
VERDICT_TOOL = {
|
|
20
|
+
"name": "submit_verdict",
|
|
21
|
+
"description": "Submit your verdict on the proposed integration once you have checked it against the code.",
|
|
22
|
+
"strict": True,
|
|
23
|
+
"input_schema": {
|
|
24
|
+
"type": "object", "additionalProperties": False,
|
|
25
|
+
"required": ["works", "reason", "issues"],
|
|
26
|
+
"properties": {
|
|
27
|
+
"works": {"type": "boolean",
|
|
28
|
+
"description": "True if following the steps as written would accomplish the task."},
|
|
29
|
+
"reason": {"type": "string"},
|
|
30
|
+
"issues": {"type": "array", "items": {"type": "string"},
|
|
31
|
+
"description": "Concrete problems: wrong paths, missing fields, missing auth, wrong order."},
|
|
32
|
+
},
|
|
33
|
+
},
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
SYSTEM = """You review integration instructions written by another engineer for the software \
|
|
37
|
+
in this repository. You can read the whole repository, implementation included.
|
|
38
|
+
|
|
39
|
+
Check the proposed steps against the code: do the routes, commands, settings and symbols \
|
|
40
|
+
behave the way the steps assume, are required inputs and authentication covered, and would \
|
|
41
|
+
following the steps in order accomplish the task? Minor omissions a competent engineer \
|
|
42
|
+
would fill in without reading the code don't make an answer wrong. Then call submit_verdict."""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Judge:
|
|
46
|
+
def __init__(self, repo: Repo, backend: Any, budget: Budget, max_turns: int = 20):
|
|
47
|
+
self.repo, self.backend, self.budget, self.max_turns = repo, backend, budget, max_turns
|
|
48
|
+
|
|
49
|
+
def judge(self, task: Task, answer: Dict) -> Dict:
|
|
50
|
+
ws = Workspace(self.repo, "source")
|
|
51
|
+
tools = [t for t in ws.tools() if t["name"] != "submit_result"] + [VERDICT_TOOL]
|
|
52
|
+
conv = self.backend.conversation(SYSTEM, tools)
|
|
53
|
+
prompt = "Task given to the engineer:\n%s\n\nTheir answer:\n%s" % (
|
|
54
|
+
task.prompt, json.dumps(answer, indent=2, sort_keys=True))
|
|
55
|
+
res = run_loop(conv, prompt, ws.run, VERDICT_TOOL, self.budget, self.max_turns)
|
|
56
|
+
verdict = dict(res.answer) if res.answer else {"works": None, "reason": "judge did not decide (%s)"
|
|
57
|
+
% res.outcome, "issues": []}
|
|
58
|
+
verdict.update({"model": self.backend.label, "turns": res.turns, "cost_usd": res.cost_usd})
|
|
59
|
+
return verdict
|