neosloc 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. neosloc-0.3.0/LICENSE +21 -0
  2. neosloc-0.3.0/PKG-INFO +141 -0
  3. neosloc-0.3.0/README.md +114 -0
  4. neosloc-0.3.0/neosloc/__init__.py +2 -0
  5. neosloc-0.3.0/neosloc/__main__.py +5 -0
  6. neosloc-0.3.0/neosloc/agentic/__init__.py +10 -0
  7. neosloc-0.3.0/neosloc/agentic/grader.py +186 -0
  8. neosloc-0.3.0/neosloc/agentic/judge.py +59 -0
  9. neosloc-0.3.0/neosloc/agentic/llm.py +392 -0
  10. neosloc-0.3.0/neosloc/agentic/loop.py +79 -0
  11. neosloc-0.3.0/neosloc/agentic/review.py +122 -0
  12. neosloc-0.3.0/neosloc/agentic/runner.py +176 -0
  13. neosloc-0.3.0/neosloc/agentic/tasks.py +60 -0
  14. neosloc-0.3.0/neosloc/agentic/workspace.py +261 -0
  15. neosloc-0.3.0/neosloc/cli.py +171 -0
  16. neosloc-0.3.0/neosloc/detectors/__init__.py +18 -0
  17. neosloc-0.3.0/neosloc/detectors/base.py +25 -0
  18. neosloc-0.3.0/neosloc/detectors/embeddability.py +114 -0
  19. neosloc-0.3.0/neosloc/detectors/ergonomics.py +59 -0
  20. neosloc-0.3.0/neosloc/detectors/events.py +61 -0
  21. neosloc-0.3.0/neosloc/detectors/extensibility.py +48 -0
  22. neosloc-0.3.0/neosloc/detectors/identity.py +55 -0
  23. neosloc-0.3.0/neosloc/detectors/interface.py +223 -0
  24. neosloc-0.3.0/neosloc/detectors/legibility.py +301 -0
  25. neosloc-0.3.0/neosloc/detectors/observability.py +44 -0
  26. neosloc-0.3.0/neosloc/detectors/portability.py +49 -0
  27. neosloc-0.3.0/neosloc/detectors/signals.py +90 -0
  28. neosloc-0.3.0/neosloc/detectors/stability.py +139 -0
  29. neosloc-0.3.0/neosloc/estimate.py +80 -0
  30. neosloc-0.3.0/neosloc/model.py +72 -0
  31. neosloc-0.3.0/neosloc/repo.py +228 -0
  32. neosloc-0.3.0/neosloc/report.py +159 -0
  33. neosloc-0.3.0/neosloc/specs.py +112 -0
  34. neosloc-0.3.0/neosloc/value.py +139 -0
  35. neosloc-0.3.0/neosloc.egg-info/PKG-INFO +141 -0
  36. neosloc-0.3.0/neosloc.egg-info/SOURCES.txt +44 -0
  37. neosloc-0.3.0/neosloc.egg-info/dependency_links.txt +1 -0
  38. neosloc-0.3.0/neosloc.egg-info/entry_points.txt +2 -0
  39. neosloc-0.3.0/neosloc.egg-info/requires.txt +5 -0
  40. neosloc-0.3.0/neosloc.egg-info/top_level.txt +1 -0
  41. neosloc-0.3.0/pyproject.toml +45 -0
  42. neosloc-0.3.0/setup.cfg +4 -0
  43. neosloc-0.3.0/tests/test_agentic.py +309 -0
  44. neosloc-0.3.0/tests/test_neosloc.py +327 -0
  45. neosloc-0.3.0/tests/test_openrouter.py +271 -0
  46. neosloc-0.3.0/tests/test_sdk_contract.py +94 -0
neosloc-0.3.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Marco Montanari
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
neosloc-0.3.0/PKG-INFO ADDED
@@ -0,0 +1,141 @@
1
+ Metadata-Version: 2.4
2
+ Name: neosloc
3
+ Version: 0.3.0
4
+ Summary: Integrability evaluation for software repositories: what sloccount measured, re-asked for the age of LLMs.
5
+ Author: Marco Montanari
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/sirmmo/neosloc
8
+ Project-URL: Documentation, https://ingmmo.com/neosloc/
9
+ Project-URL: Repository, https://github.com/sirmmo/neosloc
10
+ Project-URL: Issues, https://github.com/sirmmo/neosloc/issues
11
+ Project-URL: Changelog, https://github.com/sirmmo/neosloc/blob/main/CHANGELOG.md
12
+ Keywords: sloccount,cocomo,integrability,software metrics,llm,agents,api
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Environment :: Console
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3 :: Only
20
+ Classifier: Topic :: Software Development :: Quality Assurance
21
+ Requires-Python: >=3.8
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Provides-Extra: agentic
25
+ Requires-Dist: anthropic>=1.11; python_version >= "3.10" and extra == "agentic"
26
+ Dynamic: license-file
27
+
28
+ # neosloc
29
+
30
+ [![ci](https://github.com/sirmmo/neosloc/actions/workflows/ci.yml/badge.svg)](https://github.com/sirmmo/neosloc/actions/workflows/ci.yml)
31
+ [![PyPI](https://img.shields.io/pypi/v/neosloc)](https://pypi.org/project/neosloc/)
32
+ [![docs](https://img.shields.io/badge/docs-ingmmo.com%2Fneosloc-blue)](https://ingmmo.com/neosloc/)
33
+
34
+ `sloccount` counted lines of code and fed them to COCOMO to estimate the effort to *build*
35
+ software. With LLMs writing code, lines are cheap and that estimate has lost its meaning.
36
+ What is still expensive is everything *around* the code: how easily other systems and
37
+ agents can call it, run it, rely on it and change it safely, and the knowledge that exists
38
+ only in the code and its history.
39
+
40
+ neosloc measures that:
41
+
42
+ - **ten integrability dimensions**, each scored 0–4 with the evidence behind the level and
43
+ the gaps phrased as actions;
44
+ - the **retrofit effort** to bring every dimension up to "solid", and whether the system
45
+ is cheap to wrap;
46
+ - **neoCOCOMO**: what the codebase is worth when an agent can re-type it, next to classic
47
+ COCOMO;
48
+ - optional **LLM evaluators** on Anthropic or [OpenRouter](https://openrouter.ai): a
49
+ *probe* attempts real integration tasks using only the docs, a *judge* checks the
50
+ answers against the source, and a *review* audits every static level.
51
+
52
+ ```bash
53
+ pip install neosloc
54
+ neosloc path/to/repo
55
+ ```
56
+
57
+ ```text
58
+ Interface surface ■■■□ 3 solid The framework generates a contract from code, so it tracks the implementation.
59
+ + routes:fastapi/flask: 15 route declarations in 3 files (app/routes/admin.py)
60
+ + generated-spec: FastAPI (auto OpenAPI) (app/main.py)
61
+ - Check the generated spec into the repo so changes are reviewable and diffable.
62
+ ...
63
+ Integrability index (0-4, assessed dimensions): 1.80
64
+ Retrofit effort to level 3 (agent-assisted): 7.5 person-days
65
+
66
+ Value (neoCOCOMO, cost approach; see neosloc/value.py)
67
+ Classic COCOMO (sloccount): 4.6 KSLOC -> 11.8 person-months, 6.4 months, $133k
68
+ Behaviour captured: 38% (tests 0.14, contract 0.25, docs 1.00)
69
+ Reproduce with agents: x0.38 of classic -> 4.5 PM
70
+ Rediscover uncaptured history: 102 fix / 474 commits, 19 authors -> 4.4 PM knowledge, 2.7 PM uncaptured
71
+ Value: 5.2 PM ~ $59k (at $11k per PM)
72
+ Knowledge at risk: 38% of replacement cost lives only in code and history
73
+ ```
74
+
75
+ **Documentation: <https://ingmmo.com/neosloc/>**
76
+
77
+ ## Install
78
+
79
+ | | |
80
+ |---|---|
81
+ | `pip install neosloc` (or `pipx install neosloc`, `uv tool install neosloc`) | the `neosloc` command; the static analysis has **no dependencies** and runs on Python ≥ 3.8 |
82
+ | `pip install 'neosloc[agentic]'` | adds the anthropic SDK, needed for `anthropic:` models (Python ≥ 3.10) |
83
+ | `pip install git+https://github.com/sirmmo/neosloc` | the development version |
84
+
85
+ OpenRouter models need no extra package, only `OPENROUTER_API_KEY`.
86
+
87
+ ## Dimensions
88
+
89
+ | Key | Asks |
90
+ |---|---|
91
+ | `interface` | Is there a machine-readable contract covering the implemented surface, and more than one way in (MCP, CLI `--json`, SDK, typed library)? |
92
+ | `stability` | Semver, changelog, API versioning, deprecations, and the real git history of the OpenAPI files. |
93
+ | `events` | Outbound webhooks (signed, self-service), streams, brokers, CDC, AsyncAPI/CloudEvents. |
94
+ | `identity` | Token auth, OAuth2/OIDC, scopes, managed API keys or service accounts, SCIM. |
95
+ | `portability` | Bulk export/import, open formats, schema in the repo. |
96
+ | `ergonomics` | RFC 9457 errors, validation, idempotency keys, pagination, rate-limit headers, dry-run, `llms.txt`. |
97
+ | `embeddability` | Container, compose/Helm/IaC, documented env config, headless entry point, library packaging. |
98
+ | `extensibility` | Plugin discovery, hooks, scripting, extension docs. |
99
+ | `observability` | Health/readiness, metrics, tracing, structured logs, error tracking. |
100
+ | `legibility` | The heir to SLOC: tokens per module, import cycles, tests, types, CI, lockfiles, agent docs. |
101
+
102
+ Details: [dimensions](https://ingmmo.com/neosloc/dimensions/),
103
+ [retrofit effort](https://ingmmo.com/neosloc/estimate/),
104
+ [neoCOCOMO](https://ingmmo.com/neosloc/value/).
105
+
106
+ ## LLM evaluators
107
+
108
+ ```bash
109
+ # Probe: a model attempts 8 integration tasks from the docs; answers are graded against the code
110
+ neosloc --agentic --scope both --transcripts runs/ .
111
+
112
+ # Probe with Claude, judge with another vendor through OpenRouter
113
+ neosloc --judge --judge-model openrouter:openai/gpt-5.6-luna .
114
+
115
+ # Review: a panel audits every static level against the code
116
+ neosloc --review --review-model openrouter:google/gemini-3.8-flash,openrouter:deepseek/deepseek-v4-pro-0813 .
117
+ ```
118
+
119
+ Models are written `provider:model`. The default is `anthropic:claude-opus-5-5` at effort
120
+ `medium`. A comma-separated list runs a panel and reports agreement. Every model call
121
+ shares one `--max-cost` budget (default $5), and the report states the total. See
122
+ [LLM evaluators](https://ingmmo.com/neosloc/evaluators/) and
123
+ [Using OpenRouter](https://ingmmo.com/neosloc/openrouter/).
124
+
125
+ ## Status
126
+
127
+ All thresholds and coefficients are uncalibrated, so treat the numbers as rankings.
128
+ [Calibration and limits](https://ingmmo.com/neosloc/calibration/) describes how to fix
129
+ that; `--review` and `--agentic` exist partly to do it.
130
+
131
+ ## Development
132
+
133
+ ```bash
134
+ python -m unittest discover -s tests -t . # no network, no API spend
135
+ python -m neosloc .
136
+ ```
137
+
138
+ See [Writing detectors](https://ingmmo.com/neosloc/extending/) and `AGENTS.md`.
139
+ Releases: bump `neosloc/__init__.py`, add a `CHANGELOG.md` section, push a `vX.Y.Z` tag.
140
+
141
+ MIT licensed.
@@ -0,0 +1,114 @@
1
+ # neosloc
2
+
3
+ [![ci](https://github.com/sirmmo/neosloc/actions/workflows/ci.yml/badge.svg)](https://github.com/sirmmo/neosloc/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/neosloc)](https://pypi.org/project/neosloc/)
5
+ [![docs](https://img.shields.io/badge/docs-ingmmo.com%2Fneosloc-blue)](https://ingmmo.com/neosloc/)
6
+
7
+ `sloccount` counted lines of code and fed them to COCOMO to estimate the effort to *build*
8
+ software. With LLMs writing code, lines are cheap and that estimate has lost its meaning.
9
+ What is still expensive is everything *around* the code: how easily other systems and
10
+ agents can call it, run it, rely on it and change it safely, and the knowledge that exists
11
+ only in the code and its history.
12
+
13
+ neosloc measures that:
14
+
15
+ - **ten integrability dimensions**, each scored 0–4 with the evidence behind the level and
16
+ the gaps phrased as actions;
17
+ - the **retrofit effort** to bring every dimension up to "solid", and whether the system
18
+ is cheap to wrap;
19
+ - **neoCOCOMO**: what the codebase is worth when an agent can re-type it, next to classic
20
+ COCOMO;
21
+ - optional **LLM evaluators** on Anthropic or [OpenRouter](https://openrouter.ai): a
22
+ *probe* attempts real integration tasks using only the docs, a *judge* checks the
23
+ answers against the source, and a *review* audits every static level.
24
+
25
+ ```bash
26
+ pip install neosloc
27
+ neosloc path/to/repo
28
+ ```
29
+
30
+ ```text
31
+ Interface surface ■■■□ 3 solid The framework generates a contract from code, so it tracks the implementation.
32
+ + routes:fastapi/flask: 15 route declarations in 3 files (app/routes/admin.py)
33
+ + generated-spec: FastAPI (auto OpenAPI) (app/main.py)
34
+ - Check the generated spec into the repo so changes are reviewable and diffable.
35
+ ...
36
+ Integrability index (0-4, assessed dimensions): 1.80
37
+ Retrofit effort to level 3 (agent-assisted): 7.5 person-days
38
+
39
+ Value (neoCOCOMO, cost approach; see neosloc/value.py)
40
+ Classic COCOMO (sloccount): 4.6 KSLOC -> 11.8 person-months, 6.4 months, $133k
41
+ Behaviour captured: 38% (tests 0.14, contract 0.25, docs 1.00)
42
+ Reproduce with agents: x0.38 of classic -> 4.5 PM
43
+ Rediscover uncaptured history: 102 fix / 474 commits, 19 authors -> 4.4 PM knowledge, 2.7 PM uncaptured
44
+ Value: 5.2 PM ~ $59k (at $11k per PM)
45
+ Knowledge at risk: 38% of replacement cost lives only in code and history
46
+ ```
47
+
48
+ **Documentation: <https://ingmmo.com/neosloc/>**
49
+
50
+ ## Install
51
+
52
+ | | |
53
+ |---|---|
54
+ | `pip install neosloc` (or `pipx install neosloc`, `uv tool install neosloc`) | the `neosloc` command; the static analysis has **no dependencies** and runs on Python ≥ 3.8 |
55
+ | `pip install 'neosloc[agentic]'` | adds the anthropic SDK, needed for `anthropic:` models (Python ≥ 3.10) |
56
+ | `pip install git+https://github.com/sirmmo/neosloc` | the development version |
57
+
58
+ OpenRouter models need no extra package, only `OPENROUTER_API_KEY`.
59
+
60
+ ## Dimensions
61
+
62
+ | Key | Asks |
63
+ |---|---|
64
+ | `interface` | Is there a machine-readable contract covering the implemented surface, and more than one way in (MCP, CLI `--json`, SDK, typed library)? |
65
+ | `stability` | Semver, changelog, API versioning, deprecations, and the real git history of the OpenAPI files. |
66
+ | `events` | Outbound webhooks (signed, self-service), streams, brokers, CDC, AsyncAPI/CloudEvents. |
67
+ | `identity` | Token auth, OAuth2/OIDC, scopes, managed API keys or service accounts, SCIM. |
68
+ | `portability` | Bulk export/import, open formats, schema in the repo. |
69
+ | `ergonomics` | RFC 9457 errors, validation, idempotency keys, pagination, rate-limit headers, dry-run, `llms.txt`. |
70
+ | `embeddability` | Container, compose/Helm/IaC, documented env config, headless entry point, library packaging. |
71
+ | `extensibility` | Plugin discovery, hooks, scripting, extension docs. |
72
+ | `observability` | Health/readiness, metrics, tracing, structured logs, error tracking. |
73
+ | `legibility` | The heir to SLOC: tokens per module, import cycles, tests, types, CI, lockfiles, agent docs. |
74
+
75
+ Details: [dimensions](https://ingmmo.com/neosloc/dimensions/),
76
+ [retrofit effort](https://ingmmo.com/neosloc/estimate/),
77
+ [neoCOCOMO](https://ingmmo.com/neosloc/value/).
78
+
79
+ ## LLM evaluators
80
+
81
+ ```bash
82
+ # Probe: a model attempts 8 integration tasks from the docs; answers are graded against the code
83
+ neosloc --agentic --scope both --transcripts runs/ .
84
+
85
+ # Probe with Claude, judge with another vendor through OpenRouter
86
+ neosloc --judge --judge-model openrouter:openai/gpt-5.6-luna .
87
+
88
+ # Review: a panel audits every static level against the code
89
+ neosloc --review --review-model openrouter:google/gemini-3.8-flash,openrouter:deepseek/deepseek-v4-pro-0813 .
90
+ ```
91
+
92
+ Models are written `provider:model`. The default is `anthropic:claude-opus-5-5` at effort
93
+ `medium`. A comma-separated list runs a panel and reports agreement. Every model call
94
+ shares one `--max-cost` budget (default $5), and the report states the total. See
95
+ [LLM evaluators](https://ingmmo.com/neosloc/evaluators/) and
96
+ [Using OpenRouter](https://ingmmo.com/neosloc/openrouter/).
97
+
98
+ ## Status
99
+
100
+ All thresholds and coefficients are uncalibrated, so treat the numbers as rankings.
101
+ [Calibration and limits](https://ingmmo.com/neosloc/calibration/) describes how to fix
102
+ that; `--review` and `--agentic` exist partly to do it.
103
+
104
+ ## Development
105
+
106
+ ```bash
107
+ python -m unittest discover -s tests -t . # no network, no API spend
108
+ python -m neosloc .
109
+ ```
110
+
111
+ See [Writing detectors](https://ingmmo.com/neosloc/extending/) and `AGENTS.md`.
112
+ Releases: bump `neosloc/__init__.py`, add a `CHANGELOG.md` section, push a `vX.Y.Z` tag.
113
+
114
+ MIT licensed.
@@ -0,0 +1,2 @@
1
+ """neosloc - integrability evaluation for software, the successor question to SLOC."""
2
+ __version__ = "0.3.0"
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
@@ -0,0 +1,10 @@
1
+ """LLM evaluators: probe (attempt integration tasks), judge (check answers
2
+ against the source) and review (audit static levels), on Anthropic or OpenRouter."""
3
+ from .judge import Judge
4
+ from .llm import Budget, make_backend, parse_spec
5
+ from .review import Reviewer, review_all
6
+ from .runner import DEFAULT_EFFORT, DEFAULT_MODEL, Probe, panel
7
+ from .tasks import TASKS, select
8
+
9
+ __all__ = ["Budget", "DEFAULT_EFFORT", "DEFAULT_MODEL", "Judge", "Probe", "Reviewer", "TASKS",
10
+ "make_backend", "panel", "parse_spec", "review_all", "select"]
@@ -0,0 +1,186 @@
1
+ """Grounding: check an agent's answer against the implementation.
2
+
3
+ The agent may only have seen the docs; the grader always sees everything. A
4
+ step is grounded when the thing it names exists: the HTTP route is declared
5
+ (or in the contract), the CLI program and its flags exist, the env vars are
6
+ read somewhere, the symbol is defined. With a live instance, HTTP steps must
7
+ also have been exercised successfully.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import posixpath
12
+ import re
13
+ import shlex
14
+ from dataclasses import dataclass, field
15
+ from typing import Dict, List, Optional, Set
16
+
17
+ from ..repo import Repo
18
+ from ..specs import find_specs
19
+ from .workspace import HttpCall
20
+
21
+ PARAM_RE = re.compile(r"\{[^}/]+\}|<[^>/]+>|:[A-Za-z_]\w*|\[[^\]/]+\]|\$\{[^}]+\}")
22
+ DYNAMIC_SEG = re.compile(r"^(\d+|[0-9a-f]{8}-[0-9a-f-]{27,}|[0-9a-f]{24,})$", re.I)
23
+ ROUTE_STRING = re.compile(r"""['"`](\^?/?[A-Za-z0-9_{}<>:\[\]$./-]*)['"`]""")
24
+ FLAG_RE = re.compile(r"(?<![\w-])(--[A-Za-z][\w-]*)")
25
+ ENV_NAME = re.compile(r"^[A-Z][A-Z0-9_]{2,}$")
26
+ LAUNCHERS = {"python", "python3", "-m", "npx", "pnpm", "yarn", "npm", "bunx", "uv", "poetry", "pipx",
27
+ "run", "exec", "sudo", "env", "bundle", "go", "cargo", "dotnet", "java", "-jar", "php"}
28
+
29
+
30
+ def normalize_path(path: str) -> str:
31
+ path = re.sub(r"^\w+://[^/]+", "", path.strip()).split("?", 1)[0].split("#", 1)[0]
32
+ path = path.lstrip("^").rstrip("$")
33
+ path = PARAM_RE.sub("{}", path)
34
+ segs = [("{}" if DYNAMIC_SEG.match(s) else s) for s in path.strip("/").split("/") if s]
35
+ return "/" + "/".join(segs)
36
+
37
+
38
+ @dataclass
39
+ class Grade:
40
+ grounded: int = 0
41
+ checked: int = 0
42
+ problems: List[str] = field(default_factory=list)
43
+
44
+
45
+ class Grader:
46
+ def __init__(self, repo: Repo, route_files: Optional[List[str]] = None):
47
+ self.repo = repo
48
+ self.src = repo.source_files(include_tests=False)
49
+ self.spec_ops: Set[tuple] = set()
50
+ for s in find_specs(repo):
51
+ if s.kind == "openapi":
52
+ for op in s.operations:
53
+ m, p = op.split(" ", 1)
54
+ self.spec_ops.add((m, normalize_path(p)))
55
+ self.route_templates: Set[str] = set()
56
+ for f in route_files or []:
57
+ for line in repo.read(f).splitlines():
58
+ for lit in ROUTE_STRING.findall(line):
59
+ if lit.strip("/^$") and not lit.startswith("."):
60
+ self.route_templates.add(normalize_path(lit))
61
+ self._all_text = None
62
+
63
+ # ---- helpers -----------------------------------------------------------
64
+
65
+ def _text(self) -> str:
66
+ if self._all_text is None:
67
+ files = self.src + self.repo.manifests() + self.repo.config_files() + self.repo.doc_files()
68
+ self._all_text = "\n".join(self.repo.read(f) for f in dict.fromkeys(files))
69
+ return self._all_text
70
+
71
+ def _code_text(self) -> str:
72
+ return "\n".join(self.repo.read(f) for f in self.src + self.repo.config_files())
73
+
74
+ @staticmethod
75
+ def _suffix_match(candidate: str, template: str) -> bool:
76
+ c = candidate.strip("/").split("/")
77
+ t = template.strip("/").split("/")
78
+ if not t or t == [""] or len(t) > len(c):
79
+ return False
80
+ return all(a == b or b == "{}" or a == "{}" for a, b in zip(c[-len(t):], t))
81
+
82
+ # ---- step checks -------------------------------------------------------
83
+
84
+ def http(self, method: Optional[str], path: Optional[str], live: Optional[List[HttpCall]]) -> Optional[str]:
85
+ if not path:
86
+ return "http step without a path"
87
+ cand = normalize_path(path)
88
+ method = (method or "").upper()
89
+ in_spec = any((not method or m == method) and self._suffix_match(cand, p) for m, p in self.spec_ops)
90
+ in_code = any(self._suffix_match(cand, t) for t in self.route_templates)
91
+ if not (in_spec or in_code):
92
+ return "%s %s is not a declared route" % (method or "?", path)
93
+ if live is not None:
94
+ ok = [c for c in live if self._suffix_match(normalize_path(c.path), cand)
95
+ and (not method or c.method == method)]
96
+ if not ok:
97
+ return "%s %s was never exercised against the live instance" % (method or "?", path)
98
+ return None
99
+
100
+ def cli(self, command: Optional[str]) -> Optional[str]:
101
+ if not command:
102
+ return "cli step without a command"
103
+ try:
104
+ words = shlex.split(command.splitlines()[0])
105
+ except ValueError:
106
+ words = command.split()
107
+ if not words:
108
+ return "empty command"
109
+ if words[0] == "docker" or words[:2] == ["docker-compose"]:
110
+ ok = self.repo.glob(r"(^|/)(Dockerfile|Containerfile|(docker-)?compose[\w.-]*\.ya?ml)$")
111
+ return None if ok else "docker command but no Dockerfile/compose file"
112
+ if words[0] in ("make", "just"):
113
+ target = words[1] if len(words) > 1 else None
114
+ mk = self.repo.glob(r"(^|/)(Makefile|justfile)$")
115
+ if not mk:
116
+ return "%s command but no %s" % (words[0], "Makefile" if words[0] == "make" else "justfile")
117
+ if target and not re.search(r"^%s\s*:" % re.escape(target), self.repo.read(mk[0]), re.M):
118
+ return "no %s target %r" % (words[0], target)
119
+ return None
120
+ program = next((w for w in words if w not in LAUNCHERS and not w.startswith("-")
121
+ and "=" not in w), None)
122
+ if program is None:
123
+ return "could not identify the program in %r" % command
124
+ text = self._text()
125
+ base = posixpath.basename(program.replace("\\", "/"))
126
+ stem = posixpath.splitext(base)[0]
127
+ known = (self.repo.exists(program.lstrip("./")) or any(posixpath.basename(f) == base for f in self.repo.files)
128
+ or re.search(r"""(^|[\s"'\[])%s\s*=|['"]%s['"]\s*:""" % (re.escape(stem), re.escape(stem)),
129
+ "\n".join(self.repo.read(m) for m in self.repo.manifests()), re.M)
130
+ or any(f.split("/")[0] == stem or ("/%s/" % stem) in ("/" + f) for f in self.src))
131
+ if not known:
132
+ return "program %r is not provided by this repository" % program
133
+ missing = [fl for fl in FLAG_RE.findall(command) if fl not in text]
134
+ if missing:
135
+ return "flags not found in the code: %s" % ", ".join(missing)
136
+ return None
137
+
138
+ def library(self, symbol: Optional[str]) -> Optional[str]:
139
+ if not symbol:
140
+ return "library step without a symbol"
141
+ name = re.split(r"[.:/#]", symbol.strip().rstrip("()"))[-1]
142
+ if not name:
143
+ return "unparseable symbol %r" % symbol
144
+ rx = re.compile(r"\b(def|class|function|func|fn|const|let|var|interface|type|struct|trait|export\s+\w+)\s+"
145
+ r"(\([^)]*\)\s*)?%s\b|\b%s\s*[:=]\s*(function|\(|async)" % (re.escape(name), re.escape(name)))
146
+ if not any(rx.search(self.repo.read(f)) for f in self.src):
147
+ return "symbol %r is not defined in the code" % symbol
148
+ return None
149
+
150
+ def env(self, names: List[str]) -> Optional[str]:
151
+ names = [n for n in names if ENV_NAME.match(n)]
152
+ if not names:
153
+ return None
154
+ code = self._code_text()
155
+ missing = [n for n in names if n not in code]
156
+ return "not read by the code: %s" % ", ".join(missing) if missing else None
157
+
158
+ # ---- whole answer ------------------------------------------------------
159
+
160
+ def grade(self, answer: Dict, live: Optional[List[HttpCall]] = None) -> Grade:
161
+ g = Grade()
162
+ for step in answer.get("steps", []):
163
+ kind = step.get("kind")
164
+ checks = []
165
+ if kind == "http":
166
+ checks.append(self.http(step.get("method"), step.get("path"), live))
167
+ elif kind == "cli":
168
+ checks.append(self.cli(step.get("command")))
169
+ elif kind == "library":
170
+ checks.append(self.library(step.get("symbol")))
171
+ elif kind == "config" and not step.get("env_vars"):
172
+ checks.append("config step names no settings")
173
+ if step.get("env_vars"):
174
+ checks.append(self.env(step["env_vars"]))
175
+ if not checks:
176
+ continue # "other" steps can't be verified; they neither help nor hurt
177
+ g.checked += 1
178
+ errors = [c for c in checks if c]
179
+ if errors:
180
+ g.problems.extend(errors)
181
+ else:
182
+ g.grounded += 1
183
+ missing = [p for p in answer.get("evidence", []) if not self.repo.exists(p.lstrip("./").split(":")[0])]
184
+ if missing:
185
+ g.problems.append("cited files do not exist: %s" % ", ".join(missing[:3]))
186
+ return g
@@ -0,0 +1,59 @@
1
+ """The judge role: would a grounded answer actually accomplish the task?
2
+
3
+ Grounding proves that what the probe named exists. The judge, which may be a
4
+ different model and provider, reads the full source and decides whether the
5
+ steps would work as described (right method, required fields, correct order,
6
+ auth). A probe answer only counts as a success when both agree.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from typing import Any, Dict
12
+
13
+ from ..repo import Repo
14
+ from .llm import Budget
15
+ from .loop import run_loop
16
+ from .tasks import Task
17
+ from .workspace import Workspace
18
+
19
+ VERDICT_TOOL = {
20
+ "name": "submit_verdict",
21
+ "description": "Submit your verdict on the proposed integration once you have checked it against the code.",
22
+ "strict": True,
23
+ "input_schema": {
24
+ "type": "object", "additionalProperties": False,
25
+ "required": ["works", "reason", "issues"],
26
+ "properties": {
27
+ "works": {"type": "boolean",
28
+ "description": "True if following the steps as written would accomplish the task."},
29
+ "reason": {"type": "string"},
30
+ "issues": {"type": "array", "items": {"type": "string"},
31
+ "description": "Concrete problems: wrong paths, missing fields, missing auth, wrong order."},
32
+ },
33
+ },
34
+ }
35
+
36
+ SYSTEM = """You review integration instructions written by another engineer for the software \
37
+ in this repository. You can read the whole repository, implementation included.
38
+
39
+ Check the proposed steps against the code: do the routes, commands, settings and symbols \
40
+ behave the way the steps assume, are required inputs and authentication covered, and would \
41
+ following the steps in order accomplish the task? Minor omissions a competent engineer \
42
+ would fill in without reading the code don't make an answer wrong. Then call submit_verdict."""
43
+
44
+
45
+ class Judge:
46
+ def __init__(self, repo: Repo, backend: Any, budget: Budget, max_turns: int = 20):
47
+ self.repo, self.backend, self.budget, self.max_turns = repo, backend, budget, max_turns
48
+
49
+ def judge(self, task: Task, answer: Dict) -> Dict:
50
+ ws = Workspace(self.repo, "source")
51
+ tools = [t for t in ws.tools() if t["name"] != "submit_result"] + [VERDICT_TOOL]
52
+ conv = self.backend.conversation(SYSTEM, tools)
53
+ prompt = "Task given to the engineer:\n%s\n\nTheir answer:\n%s" % (
54
+ task.prompt, json.dumps(answer, indent=2, sort_keys=True))
55
+ res = run_loop(conv, prompt, ws.run, VERDICT_TOOL, self.budget, self.max_turns)
56
+ verdict = dict(res.answer) if res.answer else {"works": None, "reason": "judge did not decide (%s)"
57
+ % res.outcome, "issues": []}
58
+ verdict.update({"model": self.backend.label, "turns": res.turns, "cost_usd": res.cost_usd})
59
+ return verdict