py-harness-cli 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- finetune/__init__.py +1 -0
- finetune/agent_system.py +41 -0
- finetune/agent_traces.py +157 -0
- finetune/everyday.py +30 -0
- finetune/hf_ollama.py +158 -0
- finetune/huggingface_store.py +144 -0
- finetune/models.py +74 -0
- finetune/paths.py +11 -0
- finetune/python_vibe.py +788 -0
- finetune/splits.py +54 -0
- finetune/systems.py +9 -0
- harness/__init__.py +42 -0
- harness/__main__.py +8 -0
- harness/act/__init__.py +6 -0
- harness/act/autofix/__init__.py +110 -0
- harness/act/autofix/additions.py +217 -0
- harness/act/autofix/conflicts.py +124 -0
- harness/act/autofix/cover.py +419 -0
- harness/act/autofix/mechanical.py +151 -0
- harness/act/autofix/missing_imports.py +50 -0
- harness/act/autofix/moves.py +439 -0
- harness/act/autofix/names.py +339 -0
- harness/act/autofix/scaffold.py +224 -0
- harness/act/code.py +157 -0
- harness/act/gate.py +229 -0
- harness/act/parse.py +247 -0
- harness/act/patch_fix.py +138 -0
- harness/act/tools.py +244 -0
- harness/agent/__init__.py +11 -0
- harness/agent/dispatch.py +235 -0
- harness/agent/loop.py +699 -0
- harness/agent/options.py +144 -0
- harness/agent/policy.py +856 -0
- harness/agent/prompt.py +170 -0
- harness/cli.py +393 -0
- harness/editor_kit.py +265 -0
- harness/guard/__init__.py +6 -0
- harness/guard/fallbacks.py +6 -0
- harness/guard/loop_guard.py +57 -0
- harness/guard/python_vibe.py +68 -0
- harness/guard/run.py +41 -0
- harness/guard/types.py +19 -0
- harness/locate.py +767 -0
- harness/mcp_stdio.py +306 -0
- harness/memory/__init__.py +5 -0
- harness/memory/conversation.py +104 -0
- harness/model/__init__.py +6 -0
- harness/model/chat_backend.py +100 -0
- harness/model/engine.py +165 -0
- harness/model/ollama_generate.py +60 -0
- harness/model/openai_generate.py +156 -0
- harness/model/outbound.py +83 -0
- harness/model/route.py +90 -0
- harness/observe/__init__.py +6 -0
- harness/observe/eval_gate.py +80 -0
- harness/observe/eval_loop.py +185 -0
- harness/observe/eval_tasks.py +399 -0
- harness/observe/report_md.py +102 -0
- harness/observe/trace_record.py +79 -0
- harness/openai_api.py +81 -0
- harness/paths.py +88 -0
- harness/py.typed +0 -0
- harness/scan/__init__.py +6 -0
- harness/scan/app_spec.py +338 -0
- harness/scan/design.py +112 -0
- harness/scan/existing.py +131 -0
- harness/scan/layout.py +254 -0
- harness/scan/names.py +308 -0
- harness/scan/project_brief.py +287 -0
- harness/scan/project_docs.py +42 -0
- harness/scan/project_scan.py +49 -0
- harness/scan/repo_map.py +101 -0
- harness/secrets.py +39 -0
- harness/server.py +199 -0
- harness/ship/__init__.py +1 -0
- harness/ship/bot_pr.py +221 -0
- harness/ship/git_ship.py +262 -0
- harness/ship/identity.py +62 -0
- harness/ship/ticket.py +251 -0
- harness/skillkit/__init__.py +6 -0
- harness/skillkit/catalog.py +241 -0
- harness/skillkit/refuse_change.py +640 -0
- harness/skillkit/refuse_finish.py +295 -0
- harness/skillkit/target.py +238 -0
- harness/task.py +717 -0
- py_harness_cli-0.3.0.dist-info/METADATA +177 -0
- py_harness_cli-0.3.0.dist-info/RECORD +92 -0
- py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
- py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
- py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
- py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
- py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Call a local/cloud Ollama model. Weights stay in Ollama; this process is tiny."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from harness.model.chat_backend import ChatBackend
|
|
6
|
+
|
|
7
|
+
# How much the model is told it may read. Ollama's own default is 4096,
|
|
8
|
+
# small enough that a twenty-step run silently lost its opening.
|
|
9
|
+
CONTEXT_TOKENS = 8192
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import urllib.error
|
|
14
|
+
import urllib.request
|
|
15
|
+
from collections.abc import Sequence
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class OllamaGenerate(ChatBackend):
|
|
19
|
+
def __init__(
|
|
20
|
+
self,
|
|
21
|
+
model: str,
|
|
22
|
+
system: str,
|
|
23
|
+
host: str | None = None,
|
|
24
|
+
timeout: float = 180,
|
|
25
|
+
) -> None:
|
|
26
|
+
super().__init__(model, system, timeout=timeout)
|
|
27
|
+
self.host = (host or os.environ.get("OLLAMA_HOST") or "http://127.0.0.1:11434").rstrip(
|
|
28
|
+
"/"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
num_ctx = CONTEXT_TOKENS
|
|
32
|
+
|
|
33
|
+
def url(self) -> str:
|
|
34
|
+
return f"{self.host}/api/chat"
|
|
35
|
+
|
|
36
|
+
def body(self, messages: list[dict[str, str]]) -> dict[str, object]:
|
|
37
|
+
return {
|
|
38
|
+
"model": self.model,
|
|
39
|
+
"stream": False,
|
|
40
|
+
"messages": messages,
|
|
41
|
+
# Say the size rather than take the server's default. That
|
|
42
|
+
# default is 4096 for weights that accept 131072, and a run
|
|
43
|
+
# crossing it had its oldest messages dropped by the server
|
|
44
|
+
# without saying so. The harness decides what to forget.
|
|
45
|
+
"options": {"num_ctx": self.num_ctx},
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
def reply_from(self, payload: dict[str, object]) -> str:
|
|
49
|
+
message = payload.get("message") or {}
|
|
50
|
+
return str(message.get("content") or "") # type: ignore[union-attr]
|
|
51
|
+
|
|
52
|
+
def unreachable(self, exc: Exception) -> str:
|
|
53
|
+
return f"ollama {self.host} unreachable: {exc}"
|
|
54
|
+
|
|
55
|
+
def healthy(self) -> bool:
|
|
56
|
+
try:
|
|
57
|
+
with urllib.request.urlopen(f"{self.host}/api/tags", timeout=2) as resp:
|
|
58
|
+
return resp.status == 200
|
|
59
|
+
except (urllib.error.URLError, TimeoutError, OSError):
|
|
60
|
+
return False
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Call an OpenAI-compatible chat endpoint. Weights stay on that host.
|
|
2
|
+
|
|
3
|
+
The laptop harness does not change. This is how a 14B–70B that timed out
|
|
4
|
+
locally is reached: Hugging Face Inference, a rented vLLM box, or any
|
|
5
|
+
other /v1/chat/completions server. Tokens come from the environment.
|
|
6
|
+
They are never written into a trace, a test, or an error string.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from harness.model.chat_backend import ChatBackend
|
|
12
|
+
from harness.model.outbound import refuse_to_send, what_was_sent
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import sys
|
|
17
|
+
import urllib.error
|
|
18
|
+
import urllib.request
|
|
19
|
+
from collections.abc import Sequence
|
|
20
|
+
|
|
21
|
+
HF_ROUTER = "https://router.huggingface.co/v1"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def chat_url(base: str) -> str:
|
|
25
|
+
"""POST target. Accepts a host, a /v1 root, or the full chat path."""
|
|
26
|
+
base = base.rstrip("/")
|
|
27
|
+
if base.endswith("/chat/completions"):
|
|
28
|
+
return base
|
|
29
|
+
return base + "/chat/completions"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def resolve_openai_endpoint(
|
|
33
|
+
*,
|
|
34
|
+
base_url: str | None = None,
|
|
35
|
+
api_key: str | None = None,
|
|
36
|
+
) -> tuple[str, str]:
|
|
37
|
+
"""(base_url, api_key). Raises ValueError with no secret in the text."""
|
|
38
|
+
base = (
|
|
39
|
+
base_url
|
|
40
|
+
or os.environ.get("PYTHON_VIBE_BASE_URL")
|
|
41
|
+
or os.environ.get("OPENAI_BASE_URL")
|
|
42
|
+
or ""
|
|
43
|
+
).strip()
|
|
44
|
+
key = (
|
|
45
|
+
api_key
|
|
46
|
+
or os.environ.get("PYTHON_VIBE_API_KEY")
|
|
47
|
+
or os.environ.get("HF_TOKEN")
|
|
48
|
+
or os.environ.get("HUGGING_FACE_HUB_TOKEN")
|
|
49
|
+
or os.environ.get("OPENAI_API_KEY")
|
|
50
|
+
or ""
|
|
51
|
+
).strip()
|
|
52
|
+
if not base and key:
|
|
53
|
+
base = HF_ROUTER
|
|
54
|
+
if not base:
|
|
55
|
+
raise ValueError(
|
|
56
|
+
"PYTHON_VIBE_BASE_URL is required for --engine openai "
|
|
57
|
+
"(or set HF_TOKEN to use the Hugging Face router)"
|
|
58
|
+
)
|
|
59
|
+
return base.rstrip("/"), key
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def looks_like_an_ollama_tag(model: str) -> bool:
|
|
63
|
+
"""`llama3.1:8b` names a local pull, not a model on someone's server.
|
|
64
|
+
|
|
65
|
+
The engine falls back to the everyday Ollama name when --model is not
|
|
66
|
+
given, so the first remote run sends a tag no remote host knows and
|
|
67
|
+
comes back with a bare 400. Saying so is cheaper than guessing.
|
|
68
|
+
"""
|
|
69
|
+
return ":" in model and "/" not in model
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
class OpenAIGenerate(ChatBackend):
|
|
73
|
+
def __init__(
|
|
74
|
+
self,
|
|
75
|
+
model: str,
|
|
76
|
+
system: str,
|
|
77
|
+
*,
|
|
78
|
+
max_tokens: int = 700,
|
|
79
|
+
base_url: str | None = None,
|
|
80
|
+
api_key: str | None = None,
|
|
81
|
+
timeout: float | None = None,
|
|
82
|
+
) -> None:
|
|
83
|
+
self.max_tokens = max_tokens
|
|
84
|
+
self.base_url, self.api_key = resolve_openai_endpoint(
|
|
85
|
+
base_url=base_url, api_key=api_key
|
|
86
|
+
)
|
|
87
|
+
if timeout is None:
|
|
88
|
+
raw = os.environ.get("PYTHON_VIBE_TIMEOUT", "180")
|
|
89
|
+
try:
|
|
90
|
+
timeout = float(raw)
|
|
91
|
+
except ValueError:
|
|
92
|
+
timeout = 180.0
|
|
93
|
+
self.last_sent = ""
|
|
94
|
+
self._said = False
|
|
95
|
+
super().__init__(model, system, timeout=timeout)
|
|
96
|
+
|
|
97
|
+
def _hint(self, code: int) -> str:
|
|
98
|
+
"""Say the likely cause. Never quote the key or the headers."""
|
|
99
|
+
if code in {401, 403}:
|
|
100
|
+
return (
|
|
101
|
+
". Check the token in PYTHON_VIBE_API_KEY or HF_TOKEN, "
|
|
102
|
+
"and that it may use this host"
|
|
103
|
+
)
|
|
104
|
+
if code in {400, 404} and looks_like_an_ollama_tag(self.model):
|
|
105
|
+
return (
|
|
106
|
+
f". `{self.model}` is an Ollama tag; a remote host wants its "
|
|
107
|
+
"own id, such as meta-llama/Llama-3.1-8B-Instruct. "
|
|
108
|
+
"Pass --model"
|
|
109
|
+
)
|
|
110
|
+
return ""
|
|
111
|
+
|
|
112
|
+
def url(self) -> str:
|
|
113
|
+
return chat_url(self.base_url)
|
|
114
|
+
|
|
115
|
+
def body(self, messages: list[dict[str, str]]) -> dict[str, object]:
|
|
116
|
+
return {
|
|
117
|
+
"model": self.model,
|
|
118
|
+
"stream": False,
|
|
119
|
+
"max_tokens": self.max_tokens,
|
|
120
|
+
"messages": messages,
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
def headers(self) -> dict[str, str]:
|
|
124
|
+
headers = {"Content-Type": "application/json"}
|
|
125
|
+
if self.api_key:
|
|
126
|
+
headers["Authorization"] = f"Bearer {self.api_key}"
|
|
127
|
+
return headers
|
|
128
|
+
|
|
129
|
+
def before_send(self, messages: list[dict[str, str]]) -> None:
|
|
130
|
+
"""Nothing goes to somebody else's host unread.
|
|
131
|
+
|
|
132
|
+
The summary is said once, on the first send. A run takes up to
|
|
133
|
+
twenty turns and a line per turn would be noise, but a person
|
|
134
|
+
who wants to know whether their file left the laptop should not
|
|
135
|
+
have to go looking.
|
|
136
|
+
"""
|
|
137
|
+
blocked = refuse_to_send(messages, self.base_url)
|
|
138
|
+
if blocked:
|
|
139
|
+
raise RuntimeError(blocked)
|
|
140
|
+
self.last_sent = what_was_sent(messages, self.base_url)
|
|
141
|
+
if self.last_sent and not self._said:
|
|
142
|
+
self._said = True
|
|
143
|
+
print(self.last_sent, file=sys.stderr)
|
|
144
|
+
|
|
145
|
+
def reply_from(self, payload: dict[str, object]) -> str:
|
|
146
|
+
choices = payload.get("choices") or []
|
|
147
|
+
if not choices:
|
|
148
|
+
return ""
|
|
149
|
+
message = choices[0].get("message") or {} # type: ignore[index]
|
|
150
|
+
return str(message.get("content") or "")
|
|
151
|
+
|
|
152
|
+
def unreachable(self, exc: Exception) -> str:
|
|
153
|
+
return "remote model unreachable"
|
|
154
|
+
|
|
155
|
+
def refused(self, code: int) -> str:
|
|
156
|
+
return f"remote model HTTP {code}{self._hint(code)}"
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""What may leave this machine, and what was sent when it did.
|
|
2
|
+
|
|
3
|
+
The default engine talks to Ollama on this laptop, so nothing leaves and
|
|
4
|
+
none of this runs. `--engine openai` posts to a host somebody else runs,
|
|
5
|
+
and until now nothing looked at the bytes on the way out. The guard read
|
|
6
|
+
drafts coming back and never the prompt going out.
|
|
7
|
+
|
|
8
|
+
Two questions are answered here. May this go: not if it carries a secret
|
|
9
|
+
shape, and not if it is far larger than a task should need. And what
|
|
10
|
+
went: a sentence a person can check against what they expected, without
|
|
11
|
+
the prompt itself being printed back at them.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
from urllib.parse import urlparse
|
|
18
|
+
|
|
19
|
+
from harness.secrets import secret_in
|
|
20
|
+
|
|
21
|
+
# Hosts that are this machine. Sending to one of these is not sending.
|
|
22
|
+
LOCAL_HOSTS = frozenset({"localhost", "127.0.0.1", "::1", "0.0.0.0", ""})
|
|
23
|
+
|
|
24
|
+
# A single large file is ordinary. A whole repository is a mistake, and
|
|
25
|
+
# the point of a cap is to catch the mistake, not to ration the work.
|
|
26
|
+
DEFAULT_MAX_CHARS = 200_000
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def leaves_this_machine(base_url: str) -> bool:
|
|
30
|
+
"""True when this host is somebody else's."""
|
|
31
|
+
host = urlparse(base_url if "//" in base_url else f"//{base_url}").hostname
|
|
32
|
+
return (host or "") not in LOCAL_HOSTS
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def max_chars() -> int:
|
|
36
|
+
"""The cap, which a caller who means it can raise."""
|
|
37
|
+
try:
|
|
38
|
+
return int(os.environ.get("PYTHON_VIBE_MAX_SEND", DEFAULT_MAX_CHARS))
|
|
39
|
+
except ValueError:
|
|
40
|
+
return DEFAULT_MAX_CHARS
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _joined(messages: list[dict[str, str]]) -> str:
|
|
44
|
+
return "\n".join(str(m.get("content") or "") for m in messages)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def refuse_to_send(messages: list[dict[str, str]], base_url: str) -> str:
|
|
48
|
+
"""Why these messages must not go to that host. "" when they may."""
|
|
49
|
+
if not leaves_this_machine(base_url):
|
|
50
|
+
return ""
|
|
51
|
+
text = _joined(messages)
|
|
52
|
+
found = secret_in(text)
|
|
53
|
+
if found:
|
|
54
|
+
return (
|
|
55
|
+
f"Refusing to send: the prompt contains {found}. It would go "
|
|
56
|
+
"to a host this machine does not control. Take it out of the "
|
|
57
|
+
"file, or run without --engine openai."
|
|
58
|
+
)
|
|
59
|
+
size = len(text)
|
|
60
|
+
cap = max_chars()
|
|
61
|
+
if size > cap:
|
|
62
|
+
return (
|
|
63
|
+
f"Refusing to send {size} characters to a remote host; the "
|
|
64
|
+
f"limit is {cap}. Narrow the task with --scope, or raise "
|
|
65
|
+
"PYTHON_VIBE_MAX_SEND if that is really the intent."
|
|
66
|
+
)
|
|
67
|
+
return ""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def what_was_sent(messages: list[dict[str, str]], base_url: str) -> str:
|
|
71
|
+
"""One line naming where it went and how much of it. "" when local.
|
|
72
|
+
|
|
73
|
+
Deliberately a size and a destination rather than the prompt. A
|
|
74
|
+
person who wants to know whether their file went can read this; a
|
|
75
|
+
person who wants the prompt back has it already.
|
|
76
|
+
"""
|
|
77
|
+
if not leaves_this_machine(base_url):
|
|
78
|
+
return ""
|
|
79
|
+
host = urlparse(base_url if "//" in base_url else f"//{base_url}").hostname
|
|
80
|
+
text = _joined(messages)
|
|
81
|
+
return (
|
|
82
|
+
f"sent {len(text)} characters in {len(messages)} message(s) to {host}"
|
|
83
|
+
)
|
harness/model/route.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Which local weight a task should use. Deterministic. No LLM router.
|
|
2
|
+
|
|
3
|
+
Papers call this routing (pick one model first) versus cascading (cheap
|
|
4
|
+
model first, escalate if a judge fails). This harness already cascades on
|
|
5
|
+
compiler oracles, same model. The router only names a *lane*. It never
|
|
6
|
+
selects the 0.5B sidecar or the 30B that timed out on this laptop.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from harness.task import (
|
|
12
|
+
looks_like_add_feature,
|
|
13
|
+
looks_like_bugfix,
|
|
14
|
+
looks_like_design_loop,
|
|
15
|
+
looks_like_everyday_code,
|
|
16
|
+
looks_like_fix_smell,
|
|
17
|
+
looks_like_question,
|
|
18
|
+
looks_like_review_code,
|
|
19
|
+
looks_like_ship,
|
|
20
|
+
looks_like_write_tests,
|
|
21
|
+
task_paths,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
EVERYDAY = "llama3.1:8b"
|
|
25
|
+
TINY = "qwen2.5-coder:0.5b"
|
|
26
|
+
FAST = "qwen2.5-coder:1.5b"
|
|
27
|
+
CODER = "qwen2.5-coder:7b"
|
|
28
|
+
HEAVY = "qwen3coder:latest"
|
|
29
|
+
|
|
30
|
+
LANES = ("none", "read", "write", "structure")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def model_lane(task: str) -> str:
|
|
34
|
+
"""One of LANES. Ship and one-file reviews are not write jobs."""
|
|
35
|
+
if looks_like_ship(task):
|
|
36
|
+
return "none"
|
|
37
|
+
if looks_like_question(task):
|
|
38
|
+
return "read"
|
|
39
|
+
if looks_like_design_loop(task) and not task_paths(task):
|
|
40
|
+
return "structure"
|
|
41
|
+
if looks_like_review_code(task) and task_paths(task):
|
|
42
|
+
return "read"
|
|
43
|
+
if (
|
|
44
|
+
looks_like_add_feature(task)
|
|
45
|
+
or looks_like_everyday_code(task)
|
|
46
|
+
or looks_like_write_tests(task)
|
|
47
|
+
or looks_like_fix_smell(task)
|
|
48
|
+
or looks_like_bugfix(task)
|
|
49
|
+
):
|
|
50
|
+
return "write"
|
|
51
|
+
if looks_like_design_loop(task):
|
|
52
|
+
return "structure"
|
|
53
|
+
return "write"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def suggest_ollama(task: str) -> str:
|
|
57
|
+
"""Ollama name to run, or empty when the harness needs no model."""
|
|
58
|
+
if model_lane(task) == "none":
|
|
59
|
+
return ""
|
|
60
|
+
return EVERYDAY
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def route_advice(task: str) -> str:
|
|
64
|
+
"""Human-readable pick. Safe to print from `brief` / `route`."""
|
|
65
|
+
lane = model_lane(task)
|
|
66
|
+
model = suggest_ollama(task)
|
|
67
|
+
if lane == "none":
|
|
68
|
+
return (
|
|
69
|
+
"Lane: none. No model. Ship actions are limited git/gh. "
|
|
70
|
+
"Do not pull a larger weight for this."
|
|
71
|
+
)
|
|
72
|
+
extra = {
|
|
73
|
+
"read": (
|
|
74
|
+
f"Lane: read. Use {model}. "
|
|
75
|
+
"A chat 8B is enough for a typed question. "
|
|
76
|
+
"Do not use the 0.5B sidecar (misses Action:). "
|
|
77
|
+
f"{FAST} is on disk but unproven on this protocol."
|
|
78
|
+
),
|
|
79
|
+
"write": (
|
|
80
|
+
f"Lane: write. Use {model}. "
|
|
81
|
+
f"Optional specialist when pulled: {CODER} via --model. "
|
|
82
|
+
f"Do not auto-switch to {HEAVY} (180s timeout on this laptop). "
|
|
83
|
+
"Oracles stay on the same model — do not load a second weight mid-run."
|
|
84
|
+
),
|
|
85
|
+
"structure": (
|
|
86
|
+
f"Lane: structure. Use {model}. "
|
|
87
|
+
"The design scan is deterministic. A 30B does not replace it."
|
|
88
|
+
),
|
|
89
|
+
}
|
|
90
|
+
return extra[lane]
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Offline everyday gate. Live Ollama comparison is scripts/measure/eval_everyday.py --live."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from harness.act.parse import parse_turn
|
|
9
|
+
from harness.paths import EVAL_DIR, REPO_ROOT
|
|
10
|
+
from harness.act.code import apply_source, write_and_run
|
|
11
|
+
|
|
12
|
+
ROOT = REPO_ROOT
|
|
13
|
+
EVAL = EVAL_DIR
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def action_parse_rate(path: Path | None = None) -> tuple[int, int]:
|
|
17
|
+
drafts = path or EVAL / "action_drafts.jsonl"
|
|
18
|
+
ok = 0
|
|
19
|
+
total = 0
|
|
20
|
+
for line in drafts.read_text(encoding="utf-8").splitlines():
|
|
21
|
+
if not line.strip():
|
|
22
|
+
continue
|
|
23
|
+
row = json.loads(line)
|
|
24
|
+
total += 1
|
|
25
|
+
turn = parse_turn(row["draft"])
|
|
26
|
+
if turn and turn.action == row["action"]:
|
|
27
|
+
ok += 1
|
|
28
|
+
return ok, total
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def held_out_run_pass(gold_dir: Path | None = None) -> tuple[int, int]:
|
|
32
|
+
import tempfile
|
|
33
|
+
|
|
34
|
+
gold = gold_dir or EVAL / "gold"
|
|
35
|
+
ok = 0
|
|
36
|
+
tasks = list((EVAL / "held_out").glob("*.json")) if (EVAL / "held_out").is_dir() else []
|
|
37
|
+
if not tasks:
|
|
38
|
+
return 0, 0
|
|
39
|
+
for spec_path in tasks:
|
|
40
|
+
spec = json.loads(spec_path.read_text(encoding="utf-8"))
|
|
41
|
+
script = gold / spec["script"]
|
|
42
|
+
argv: list[str] = []
|
|
43
|
+
for item in spec.get("argv") or []:
|
|
44
|
+
cand = EVAL / str(item)
|
|
45
|
+
argv.append(str(cand) if cand.exists() else str(item))
|
|
46
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
47
|
+
dest = Path(tmp) / script.name
|
|
48
|
+
result = write_and_run(
|
|
49
|
+
script.read_text(encoding="utf-8"),
|
|
50
|
+
dest,
|
|
51
|
+
argv,
|
|
52
|
+
)
|
|
53
|
+
expected = spec["stdout"].strip()
|
|
54
|
+
if result.code == 0 and result.stdout.strip() == expected:
|
|
55
|
+
ok += 1
|
|
56
|
+
return ok, len(tasks)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def bugfix_fixture_ready(project: Path | None = None) -> tuple[bool, str]:
|
|
60
|
+
root = project or EVAL / "fixtures" / "nameerror_pkg"
|
|
61
|
+
broken = root / "pkg" / "util_stats.py"
|
|
62
|
+
gold = EVAL / "gold" / "util_stats.py"
|
|
63
|
+
if not broken.is_file() or not gold.is_file():
|
|
64
|
+
return False, "missing fixture or gold"
|
|
65
|
+
if broken.stat().st_size < 1000:
|
|
66
|
+
return False, f"{broken} is under 1 KB"
|
|
67
|
+
original = broken.read_text(encoding="utf-8")
|
|
68
|
+
if "tota" not in original or "return tota" not in original:
|
|
69
|
+
return False, "fixture no longer has the NameError"
|
|
70
|
+
apply_source(broken, gold.read_text(encoding="utf-8"), original=original)
|
|
71
|
+
try:
|
|
72
|
+
text = broken.read_text(encoding="utf-8")
|
|
73
|
+
if "return total" not in text and "return sum(" not in text:
|
|
74
|
+
return False, "gold apply did not fix the name"
|
|
75
|
+
finally:
|
|
76
|
+
broken.write_text(original, encoding="utf-8")
|
|
77
|
+
bak = broken.with_suffix(broken.suffix + ".bak")
|
|
78
|
+
if bak.is_file():
|
|
79
|
+
bak.unlink()
|
|
80
|
+
return True, "ok"
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Score one draft against a held-out task. No model. No network."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import subprocess
|
|
6
|
+
import tempfile
|
|
7
|
+
from collections.abc import Callable, Iterator
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from harness.act.code import RunResult, extract_python, write_and_run
|
|
12
|
+
from harness.guard.fallbacks import PYTHON_VIBE_FALLBACK
|
|
13
|
+
from harness.guard.python_vibe import PythonVibeGuard
|
|
14
|
+
from harness.guard.run import complete
|
|
15
|
+
from harness.guard.types import Outcome
|
|
16
|
+
from harness.observe.eval_tasks import REPAIR_PREFIX, RUN_PREFIX, Task
|
|
17
|
+
|
|
18
|
+
GenerateFn = Callable[[str], str]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class Score:
|
|
23
|
+
task_id: str
|
|
24
|
+
passed: bool
|
|
25
|
+
repaired: bool
|
|
26
|
+
reason: str
|
|
27
|
+
stdout: str
|
|
28
|
+
stderr: str
|
|
29
|
+
exit_code: int
|
|
30
|
+
fallback: bool
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def stdout_matches(expect: str, got: str) -> bool:
|
|
34
|
+
return got.rstrip("\n") == expect.rstrip("\n")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _remember(generate: GenerateFn, prompt: str, draft: str) -> None:
|
|
38
|
+
memory = getattr(generate, "memory", None)
|
|
39
|
+
if memory is None:
|
|
40
|
+
return
|
|
41
|
+
memory.remember(prompt, draft)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _reset(generate: GenerateFn) -> None:
|
|
45
|
+
memory = getattr(generate, "memory", None)
|
|
46
|
+
if memory is not None:
|
|
47
|
+
memory.clear()
|
|
48
|
+
return
|
|
49
|
+
history = getattr(generate, "history", None)
|
|
50
|
+
if history is not None:
|
|
51
|
+
history.clear()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def run_source(task: Task, source: str, dest: Path) -> tuple[bool, RunResult]:
|
|
55
|
+
work = dest.parent
|
|
56
|
+
for rel, content in task.files:
|
|
57
|
+
path = work / rel
|
|
58
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
path.write_text(content, encoding="utf-8")
|
|
60
|
+
try:
|
|
61
|
+
result = write_and_run(
|
|
62
|
+
source,
|
|
63
|
+
dest,
|
|
64
|
+
list(task.argv),
|
|
65
|
+
cwd=work,
|
|
66
|
+
timeout=task.timeout,
|
|
67
|
+
stdin=task.stdin or None,
|
|
68
|
+
)
|
|
69
|
+
except subprocess.TimeoutExpired as exc:
|
|
70
|
+
result = RunResult(
|
|
71
|
+
124,
|
|
72
|
+
(exc.stdout or "") if isinstance(exc.stdout, str) else "",
|
|
73
|
+
(exc.stderr or "") if isinstance(exc.stderr, str) else "timed out",
|
|
74
|
+
)
|
|
75
|
+
ok = result.code == 0 and stdout_matches(task.expect_stdout, result.stdout)
|
|
76
|
+
return ok, result
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def score_source(task: Task, source: str) -> Score:
|
|
80
|
+
if not source.strip():
|
|
81
|
+
return Score(task.id, False, False, "empty source", "", "", 1, False)
|
|
82
|
+
with tempfile.TemporaryDirectory(prefix=f"pv-eval-{task.id}-") as tmp:
|
|
83
|
+
dest = Path(tmp) / "task.py"
|
|
84
|
+
ok, result = run_source(task, source, dest)
|
|
85
|
+
return Score(
|
|
86
|
+
task.id,
|
|
87
|
+
ok,
|
|
88
|
+
False,
|
|
89
|
+
(
|
|
90
|
+
"pass"
|
|
91
|
+
if ok
|
|
92
|
+
else "timeout"
|
|
93
|
+
if result.code == 124
|
|
94
|
+
else "wrong output"
|
|
95
|
+
if result.code == 0
|
|
96
|
+
else "nonzero exit"
|
|
97
|
+
),
|
|
98
|
+
result.stdout,
|
|
99
|
+
result.stderr,
|
|
100
|
+
result.code,
|
|
101
|
+
False,
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _draft_source(outcome: Outcome) -> tuple[str | None, Score | None]:
|
|
106
|
+
if outcome.fallback or not outcome.output:
|
|
107
|
+
return None, Score(
|
|
108
|
+
"",
|
|
109
|
+
False,
|
|
110
|
+
False,
|
|
111
|
+
"guard fallback" if outcome.fallback else "empty draft",
|
|
112
|
+
"",
|
|
113
|
+
"",
|
|
114
|
+
1,
|
|
115
|
+
outcome.fallback,
|
|
116
|
+
)
|
|
117
|
+
source = extract_python(outcome.output)
|
|
118
|
+
if not source:
|
|
119
|
+
return None, Score("", False, False, "no python block", "", "", 1, False)
|
|
120
|
+
return source, None
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def score_generate(
|
|
124
|
+
task: Task,
|
|
125
|
+
generate: GenerateFn,
|
|
126
|
+
*,
|
|
127
|
+
repair: bool = False,
|
|
128
|
+
guard: PythonVibeGuard | None = None,
|
|
129
|
+
) -> Score:
|
|
130
|
+
checker = guard or PythonVibeGuard()
|
|
131
|
+
prompt = RUN_PREFIX + task.prompt
|
|
132
|
+
try:
|
|
133
|
+
outcome = complete(generate, checker, PYTHON_VIBE_FALLBACK, prompt)
|
|
134
|
+
except (TimeoutError, RuntimeError, OSError) as exc:
|
|
135
|
+
return Score(task.id, False, False, f"generate error: {exc}", "", str(exc), 1, False)
|
|
136
|
+
source, early = _draft_source(outcome)
|
|
137
|
+
if early is not None:
|
|
138
|
+
return Score(task.id, early.passed, False, early.reason, "", "", 1, early.fallback)
|
|
139
|
+
assert source is not None
|
|
140
|
+
_remember(generate, prompt, outcome.output or source)
|
|
141
|
+
first = score_source(task, source)
|
|
142
|
+
if first.passed or not repair:
|
|
143
|
+
return first
|
|
144
|
+
err = (first.stderr or first.stdout).strip() or first.reason
|
|
145
|
+
repair_prompt = f"{REPAIR_PREFIX}```\n{err}\n```"
|
|
146
|
+
try:
|
|
147
|
+
repaired = complete(generate, checker, PYTHON_VIBE_FALLBACK, repair_prompt)
|
|
148
|
+
except (TimeoutError, RuntimeError, OSError) as exc:
|
|
149
|
+
return Score(
|
|
150
|
+
task.id, False, True, f"generate error: {exc}", first.stdout, first.stderr, first.exit_code, False
|
|
151
|
+
)
|
|
152
|
+
source2, early2 = _draft_source(repaired)
|
|
153
|
+
if early2 is not None:
|
|
154
|
+
return Score(
|
|
155
|
+
task.id, False, True, early2.reason, first.stdout, first.stderr, first.exit_code, early2.fallback
|
|
156
|
+
)
|
|
157
|
+
assert source2 is not None
|
|
158
|
+
second = score_source(task, source2)
|
|
159
|
+
return Score(
|
|
160
|
+
task.id,
|
|
161
|
+
second.passed,
|
|
162
|
+
True,
|
|
163
|
+
f"repair {second.reason}",
|
|
164
|
+
second.stdout,
|
|
165
|
+
second.stderr,
|
|
166
|
+
second.exit_code,
|
|
167
|
+
False,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def run_repeats(
|
|
172
|
+
tasks: list[Task],
|
|
173
|
+
generate: GenerateFn,
|
|
174
|
+
*,
|
|
175
|
+
repair: bool,
|
|
176
|
+
repeats: int,
|
|
177
|
+
reset: Callable[[], None] | None = None,
|
|
178
|
+
) -> Iterator[Score]:
|
|
179
|
+
for _ in range(repeats):
|
|
180
|
+
for task in tasks:
|
|
181
|
+
if reset is not None:
|
|
182
|
+
reset()
|
|
183
|
+
else:
|
|
184
|
+
_reset(generate)
|
|
185
|
+
yield score_generate(task, generate, repair=repair)
|