py-harness-cli 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. finetune/__init__.py +1 -0
  2. finetune/agent_system.py +41 -0
  3. finetune/agent_traces.py +157 -0
  4. finetune/everyday.py +30 -0
  5. finetune/hf_ollama.py +158 -0
  6. finetune/huggingface_store.py +144 -0
  7. finetune/models.py +74 -0
  8. finetune/paths.py +11 -0
  9. finetune/python_vibe.py +788 -0
  10. finetune/splits.py +54 -0
  11. finetune/systems.py +9 -0
  12. harness/__init__.py +42 -0
  13. harness/__main__.py +8 -0
  14. harness/act/__init__.py +6 -0
  15. harness/act/autofix/__init__.py +110 -0
  16. harness/act/autofix/additions.py +217 -0
  17. harness/act/autofix/conflicts.py +124 -0
  18. harness/act/autofix/cover.py +419 -0
  19. harness/act/autofix/mechanical.py +151 -0
  20. harness/act/autofix/missing_imports.py +50 -0
  21. harness/act/autofix/moves.py +439 -0
  22. harness/act/autofix/names.py +339 -0
  23. harness/act/autofix/scaffold.py +224 -0
  24. harness/act/code.py +157 -0
  25. harness/act/gate.py +229 -0
  26. harness/act/parse.py +247 -0
  27. harness/act/patch_fix.py +138 -0
  28. harness/act/tools.py +244 -0
  29. harness/agent/__init__.py +11 -0
  30. harness/agent/dispatch.py +235 -0
  31. harness/agent/loop.py +699 -0
  32. harness/agent/options.py +144 -0
  33. harness/agent/policy.py +856 -0
  34. harness/agent/prompt.py +170 -0
  35. harness/cli.py +393 -0
  36. harness/editor_kit.py +265 -0
  37. harness/guard/__init__.py +6 -0
  38. harness/guard/fallbacks.py +6 -0
  39. harness/guard/loop_guard.py +57 -0
  40. harness/guard/python_vibe.py +68 -0
  41. harness/guard/run.py +41 -0
  42. harness/guard/types.py +19 -0
  43. harness/locate.py +767 -0
  44. harness/mcp_stdio.py +306 -0
  45. harness/memory/__init__.py +5 -0
  46. harness/memory/conversation.py +104 -0
  47. harness/model/__init__.py +6 -0
  48. harness/model/chat_backend.py +100 -0
  49. harness/model/engine.py +165 -0
  50. harness/model/ollama_generate.py +60 -0
  51. harness/model/openai_generate.py +156 -0
  52. harness/model/outbound.py +83 -0
  53. harness/model/route.py +90 -0
  54. harness/observe/__init__.py +6 -0
  55. harness/observe/eval_gate.py +80 -0
  56. harness/observe/eval_loop.py +185 -0
  57. harness/observe/eval_tasks.py +399 -0
  58. harness/observe/report_md.py +102 -0
  59. harness/observe/trace_record.py +79 -0
  60. harness/openai_api.py +81 -0
  61. harness/paths.py +88 -0
  62. harness/py.typed +0 -0
  63. harness/scan/__init__.py +6 -0
  64. harness/scan/app_spec.py +338 -0
  65. harness/scan/design.py +112 -0
  66. harness/scan/existing.py +131 -0
  67. harness/scan/layout.py +254 -0
  68. harness/scan/names.py +308 -0
  69. harness/scan/project_brief.py +287 -0
  70. harness/scan/project_docs.py +42 -0
  71. harness/scan/project_scan.py +49 -0
  72. harness/scan/repo_map.py +101 -0
  73. harness/secrets.py +39 -0
  74. harness/server.py +199 -0
  75. harness/ship/__init__.py +1 -0
  76. harness/ship/bot_pr.py +221 -0
  77. harness/ship/git_ship.py +262 -0
  78. harness/ship/identity.py +62 -0
  79. harness/ship/ticket.py +251 -0
  80. harness/skillkit/__init__.py +6 -0
  81. harness/skillkit/catalog.py +241 -0
  82. harness/skillkit/refuse_change.py +640 -0
  83. harness/skillkit/refuse_finish.py +295 -0
  84. harness/skillkit/target.py +238 -0
  85. harness/task.py +717 -0
  86. py_harness_cli-0.3.0.dist-info/METADATA +177 -0
  87. py_harness_cli-0.3.0.dist-info/RECORD +92 -0
  88. py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
  89. py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
  90. py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
  91. py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
  92. py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,60 @@
1
+ """Call a local/cloud Ollama model. Weights stay in Ollama; this process is tiny."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from harness.model.chat_backend import ChatBackend
6
+
7
+ # How much the model is told it may read. Ollama's own default is 4096,
8
+ # small enough that a twenty-step run silently lost its opening.
9
+ CONTEXT_TOKENS = 8192
10
+
11
+ import json
12
+ import os
13
+ import urllib.error
14
+ import urllib.request
15
+ from collections.abc import Sequence
16
+
17
+
18
+ class OllamaGenerate(ChatBackend):
19
+ def __init__(
20
+ self,
21
+ model: str,
22
+ system: str,
23
+ host: str | None = None,
24
+ timeout: float = 180,
25
+ ) -> None:
26
+ super().__init__(model, system, timeout=timeout)
27
+ self.host = (host or os.environ.get("OLLAMA_HOST") or "http://127.0.0.1:11434").rstrip(
28
+ "/"
29
+ )
30
+
31
+ num_ctx = CONTEXT_TOKENS
32
+
33
+ def url(self) -> str:
34
+ return f"{self.host}/api/chat"
35
+
36
+ def body(self, messages: list[dict[str, str]]) -> dict[str, object]:
37
+ return {
38
+ "model": self.model,
39
+ "stream": False,
40
+ "messages": messages,
41
+ # Say the size rather than take the server's default. That
42
+ # default is 4096 for weights that accept 131072, and a run
43
+ # crossing it had its oldest messages dropped by the server
44
+ # without saying so. The harness decides what to forget.
45
+ "options": {"num_ctx": self.num_ctx},
46
+ }
47
+
48
+ def reply_from(self, payload: dict[str, object]) -> str:
49
+ message = payload.get("message") or {}
50
+ return str(message.get("content") or "") # type: ignore[union-attr]
51
+
52
+ def unreachable(self, exc: Exception) -> str:
53
+ return f"ollama {self.host} unreachable: {exc}"
54
+
55
+ def healthy(self) -> bool:
56
+ try:
57
+ with urllib.request.urlopen(f"{self.host}/api/tags", timeout=2) as resp:
58
+ return resp.status == 200
59
+ except (urllib.error.URLError, TimeoutError, OSError):
60
+ return False
@@ -0,0 +1,156 @@
1
+ """Call an OpenAI-compatible chat endpoint. Weights stay on that host.
2
+
3
+ The laptop harness does not change. This is how a 14B–70B that timed out
4
+ locally is reached: Hugging Face Inference, a rented vLLM box, or any
5
+ other /v1/chat/completions server. Tokens come from the environment.
6
+ They are never written into a trace, a test, or an error string.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from harness.model.chat_backend import ChatBackend
12
+ from harness.model.outbound import refuse_to_send, what_was_sent
13
+
14
+ import json
15
+ import os
16
+ import sys
17
+ import urllib.error
18
+ import urllib.request
19
+ from collections.abc import Sequence
20
+
21
+ HF_ROUTER = "https://router.huggingface.co/v1"
22
+
23
+
24
+ def chat_url(base: str) -> str:
25
+ """POST target. Accepts a host, a /v1 root, or the full chat path."""
26
+ base = base.rstrip("/")
27
+ if base.endswith("/chat/completions"):
28
+ return base
29
+ return base + "/chat/completions"
30
+
31
+
32
+ def resolve_openai_endpoint(
33
+ *,
34
+ base_url: str | None = None,
35
+ api_key: str | None = None,
36
+ ) -> tuple[str, str]:
37
+ """(base_url, api_key). Raises ValueError with no secret in the text."""
38
+ base = (
39
+ base_url
40
+ or os.environ.get("PYTHON_VIBE_BASE_URL")
41
+ or os.environ.get("OPENAI_BASE_URL")
42
+ or ""
43
+ ).strip()
44
+ key = (
45
+ api_key
46
+ or os.environ.get("PYTHON_VIBE_API_KEY")
47
+ or os.environ.get("HF_TOKEN")
48
+ or os.environ.get("HUGGING_FACE_HUB_TOKEN")
49
+ or os.environ.get("OPENAI_API_KEY")
50
+ or ""
51
+ ).strip()
52
+ if not base and key:
53
+ base = HF_ROUTER
54
+ if not base:
55
+ raise ValueError(
56
+ "PYTHON_VIBE_BASE_URL is required for --engine openai "
57
+ "(or set HF_TOKEN to use the Hugging Face router)"
58
+ )
59
+ return base.rstrip("/"), key
60
+
61
+
62
+ def looks_like_an_ollama_tag(model: str) -> bool:
63
+ """`llama3.1:8b` names a local pull, not a model on someone's server.
64
+
65
+ The engine falls back to the everyday Ollama name when --model is not
66
+ given, so the first remote run sends a tag no remote host knows and
67
+ comes back with a bare 400. Saying so is cheaper than guessing.
68
+ """
69
+ return ":" in model and "/" not in model
70
+
71
+
72
+ class OpenAIGenerate(ChatBackend):
73
+ def __init__(
74
+ self,
75
+ model: str,
76
+ system: str,
77
+ *,
78
+ max_tokens: int = 700,
79
+ base_url: str | None = None,
80
+ api_key: str | None = None,
81
+ timeout: float | None = None,
82
+ ) -> None:
83
+ self.max_tokens = max_tokens
84
+ self.base_url, self.api_key = resolve_openai_endpoint(
85
+ base_url=base_url, api_key=api_key
86
+ )
87
+ if timeout is None:
88
+ raw = os.environ.get("PYTHON_VIBE_TIMEOUT", "180")
89
+ try:
90
+ timeout = float(raw)
91
+ except ValueError:
92
+ timeout = 180.0
93
+ self.last_sent = ""
94
+ self._said = False
95
+ super().__init__(model, system, timeout=timeout)
96
+
97
+ def _hint(self, code: int) -> str:
98
+ """Say the likely cause. Never quote the key or the headers."""
99
+ if code in {401, 403}:
100
+ return (
101
+ ". Check the token in PYTHON_VIBE_API_KEY or HF_TOKEN, "
102
+ "and that it may use this host"
103
+ )
104
+ if code in {400, 404} and looks_like_an_ollama_tag(self.model):
105
+ return (
106
+ f". `{self.model}` is an Ollama tag; a remote host wants its "
107
+ "own id, such as meta-llama/Llama-3.1-8B-Instruct. "
108
+ "Pass --model"
109
+ )
110
+ return ""
111
+
112
+ def url(self) -> str:
113
+ return chat_url(self.base_url)
114
+
115
+ def body(self, messages: list[dict[str, str]]) -> dict[str, object]:
116
+ return {
117
+ "model": self.model,
118
+ "stream": False,
119
+ "max_tokens": self.max_tokens,
120
+ "messages": messages,
121
+ }
122
+
123
+ def headers(self) -> dict[str, str]:
124
+ headers = {"Content-Type": "application/json"}
125
+ if self.api_key:
126
+ headers["Authorization"] = f"Bearer {self.api_key}"
127
+ return headers
128
+
129
+ def before_send(self, messages: list[dict[str, str]]) -> None:
130
+ """Nothing goes to somebody else's host unread.
131
+
132
+ The summary is said once, on the first send. A run takes up to
133
+ twenty turns and a line per turn would be noise, but a person
134
+ who wants to know whether their file left the laptop should not
135
+ have to go looking.
136
+ """
137
+ blocked = refuse_to_send(messages, self.base_url)
138
+ if blocked:
139
+ raise RuntimeError(blocked)
140
+ self.last_sent = what_was_sent(messages, self.base_url)
141
+ if self.last_sent and not self._said:
142
+ self._said = True
143
+ print(self.last_sent, file=sys.stderr)
144
+
145
+ def reply_from(self, payload: dict[str, object]) -> str:
146
+ choices = payload.get("choices") or []
147
+ if not choices:
148
+ return ""
149
+ message = choices[0].get("message") or {} # type: ignore[index]
150
+ return str(message.get("content") or "")
151
+
152
+ def unreachable(self, exc: Exception) -> str:
153
+ return "remote model unreachable"
154
+
155
+ def refused(self, code: int) -> str:
156
+ return f"remote model HTTP {code}{self._hint(code)}"
@@ -0,0 +1,83 @@
1
+ """What may leave this machine, and what was sent when it did.
2
+
3
+ The default engine talks to Ollama on this laptop, so nothing leaves and
4
+ none of this runs. `--engine openai` posts to a host somebody else runs,
5
+ and until now nothing looked at the bytes on the way out. The guard read
6
+ drafts coming back and never the prompt going out.
7
+
8
+ Two questions are answered here. May this go: not if it carries a secret
9
+ shape, and not if it is far larger than a task should need. And what
10
+ went: a sentence a person can check against what they expected, without
11
+ the prompt itself being printed back at them.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ from urllib.parse import urlparse
18
+
19
+ from harness.secrets import secret_in
20
+
21
+ # Hosts that are this machine. Sending to one of these is not sending.
22
+ LOCAL_HOSTS = frozenset({"localhost", "127.0.0.1", "::1", "0.0.0.0", ""})
23
+
24
+ # A single large file is ordinary. A whole repository is a mistake, and
25
+ # the point of a cap is to catch the mistake, not to ration the work.
26
+ DEFAULT_MAX_CHARS = 200_000
27
+
28
+
29
+ def leaves_this_machine(base_url: str) -> bool:
30
+ """True when this host is somebody else's."""
31
+ host = urlparse(base_url if "//" in base_url else f"//{base_url}").hostname
32
+ return (host or "") not in LOCAL_HOSTS
33
+
34
+
35
+ def max_chars() -> int:
36
+ """The cap, which a caller who means it can raise."""
37
+ try:
38
+ return int(os.environ.get("PYTHON_VIBE_MAX_SEND", DEFAULT_MAX_CHARS))
39
+ except ValueError:
40
+ return DEFAULT_MAX_CHARS
41
+
42
+
43
+ def _joined(messages: list[dict[str, str]]) -> str:
44
+ return "\n".join(str(m.get("content") or "") for m in messages)
45
+
46
+
47
+ def refuse_to_send(messages: list[dict[str, str]], base_url: str) -> str:
48
+ """Why these messages must not go to that host. "" when they may."""
49
+ if not leaves_this_machine(base_url):
50
+ return ""
51
+ text = _joined(messages)
52
+ found = secret_in(text)
53
+ if found:
54
+ return (
55
+ f"Refusing to send: the prompt contains {found}. It would go "
56
+ "to a host this machine does not control. Take it out of the "
57
+ "file, or run without --engine openai."
58
+ )
59
+ size = len(text)
60
+ cap = max_chars()
61
+ if size > cap:
62
+ return (
63
+ f"Refusing to send {size} characters to a remote host; the "
64
+ f"limit is {cap}. Narrow the task with --scope, or raise "
65
+ "PYTHON_VIBE_MAX_SEND if that is really the intent."
66
+ )
67
+ return ""
68
+
69
+
70
+ def what_was_sent(messages: list[dict[str, str]], base_url: str) -> str:
71
+ """One line naming where it went and how much of it. "" when local.
72
+
73
+ Deliberately a size and a destination rather than the prompt. A
74
+ person who wants to know whether their file went can read this; a
75
+ person who wants the prompt back has it already.
76
+ """
77
+ if not leaves_this_machine(base_url):
78
+ return ""
79
+ host = urlparse(base_url if "//" in base_url else f"//{base_url}").hostname
80
+ text = _joined(messages)
81
+ return (
82
+ f"sent {len(text)} characters in {len(messages)} message(s) to {host}"
83
+ )
harness/model/route.py ADDED
@@ -0,0 +1,90 @@
1
+ """Which local weight a task should use. Deterministic. No LLM router.
2
+
3
+ Papers call this routing (pick one model first) versus cascading (cheap
4
+ model first, escalate if a judge fails). This harness already cascades on
5
+ compiler oracles, same model. The router only names a *lane*. It never
6
+ selects the 0.5B sidecar or the 30B that timed out on this laptop.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from harness.task import (
12
+ looks_like_add_feature,
13
+ looks_like_bugfix,
14
+ looks_like_design_loop,
15
+ looks_like_everyday_code,
16
+ looks_like_fix_smell,
17
+ looks_like_question,
18
+ looks_like_review_code,
19
+ looks_like_ship,
20
+ looks_like_write_tests,
21
+ task_paths,
22
+ )
23
+
24
+ EVERYDAY = "llama3.1:8b"
25
+ TINY = "qwen2.5-coder:0.5b"
26
+ FAST = "qwen2.5-coder:1.5b"
27
+ CODER = "qwen2.5-coder:7b"
28
+ HEAVY = "qwen3coder:latest"
29
+
30
+ LANES = ("none", "read", "write", "structure")
31
+
32
+
33
+ def model_lane(task: str) -> str:
34
+ """One of LANES. Ship and one-file reviews are not write jobs."""
35
+ if looks_like_ship(task):
36
+ return "none"
37
+ if looks_like_question(task):
38
+ return "read"
39
+ if looks_like_design_loop(task) and not task_paths(task):
40
+ return "structure"
41
+ if looks_like_review_code(task) and task_paths(task):
42
+ return "read"
43
+ if (
44
+ looks_like_add_feature(task)
45
+ or looks_like_everyday_code(task)
46
+ or looks_like_write_tests(task)
47
+ or looks_like_fix_smell(task)
48
+ or looks_like_bugfix(task)
49
+ ):
50
+ return "write"
51
+ if looks_like_design_loop(task):
52
+ return "structure"
53
+ return "write"
54
+
55
+
56
+ def suggest_ollama(task: str) -> str:
57
+ """Ollama name to run, or empty when the harness needs no model."""
58
+ if model_lane(task) == "none":
59
+ return ""
60
+ return EVERYDAY
61
+
62
+
63
+ def route_advice(task: str) -> str:
64
+ """Human-readable pick. Safe to print from `brief` / `route`."""
65
+ lane = model_lane(task)
66
+ model = suggest_ollama(task)
67
+ if lane == "none":
68
+ return (
69
+ "Lane: none. No model. Ship actions are limited git/gh. "
70
+ "Do not pull a larger weight for this."
71
+ )
72
+ extra = {
73
+ "read": (
74
+ f"Lane: read. Use {model}. "
75
+ "A chat 8B is enough for a typed question. "
76
+ "Do not use the 0.5B sidecar (misses Action:). "
77
+ f"{FAST} is on disk but unproven on this protocol."
78
+ ),
79
+ "write": (
80
+ f"Lane: write. Use {model}. "
81
+ f"Optional specialist when pulled: {CODER} via --model. "
82
+ f"Do not auto-switch to {HEAVY} (180s timeout on this laptop). "
83
+ "Oracles stay on the same model — do not load a second weight mid-run."
84
+ ),
85
+ "structure": (
86
+ f"Lane: structure. Use {model}. "
87
+ "The design scan is deterministic. A 30B does not replace it."
88
+ ),
89
+ }
90
+ return extra[lane]
@@ -0,0 +1,6 @@
1
+ """Record what a run did.
2
+
3
+ Holds the redacted trace writer, the Markdown report writer, the
4
+ offline checks used as the merge gate, and the held-out exact-stdout
5
+ scorer.
6
+ """
@@ -0,0 +1,80 @@
1
+ """Offline everyday gate. Live Ollama comparison is scripts/measure/eval_everyday.py --live."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from pathlib import Path
7
+
8
+ from harness.act.parse import parse_turn
9
+ from harness.paths import EVAL_DIR, REPO_ROOT
10
+ from harness.act.code import apply_source, write_and_run
11
+
12
+ ROOT = REPO_ROOT
13
+ EVAL = EVAL_DIR
14
+
15
+
16
+ def action_parse_rate(path: Path | None = None) -> tuple[int, int]:
17
+ drafts = path or EVAL / "action_drafts.jsonl"
18
+ ok = 0
19
+ total = 0
20
+ for line in drafts.read_text(encoding="utf-8").splitlines():
21
+ if not line.strip():
22
+ continue
23
+ row = json.loads(line)
24
+ total += 1
25
+ turn = parse_turn(row["draft"])
26
+ if turn and turn.action == row["action"]:
27
+ ok += 1
28
+ return ok, total
29
+
30
+
31
+ def held_out_run_pass(gold_dir: Path | None = None) -> tuple[int, int]:
32
+ import tempfile
33
+
34
+ gold = gold_dir or EVAL / "gold"
35
+ ok = 0
36
+ tasks = list((EVAL / "held_out").glob("*.json")) if (EVAL / "held_out").is_dir() else []
37
+ if not tasks:
38
+ return 0, 0
39
+ for spec_path in tasks:
40
+ spec = json.loads(spec_path.read_text(encoding="utf-8"))
41
+ script = gold / spec["script"]
42
+ argv: list[str] = []
43
+ for item in spec.get("argv") or []:
44
+ cand = EVAL / str(item)
45
+ argv.append(str(cand) if cand.exists() else str(item))
46
+ with tempfile.TemporaryDirectory() as tmp:
47
+ dest = Path(tmp) / script.name
48
+ result = write_and_run(
49
+ script.read_text(encoding="utf-8"),
50
+ dest,
51
+ argv,
52
+ )
53
+ expected = spec["stdout"].strip()
54
+ if result.code == 0 and result.stdout.strip() == expected:
55
+ ok += 1
56
+ return ok, len(tasks)
57
+
58
+
59
+ def bugfix_fixture_ready(project: Path | None = None) -> tuple[bool, str]:
60
+ root = project or EVAL / "fixtures" / "nameerror_pkg"
61
+ broken = root / "pkg" / "util_stats.py"
62
+ gold = EVAL / "gold" / "util_stats.py"
63
+ if not broken.is_file() or not gold.is_file():
64
+ return False, "missing fixture or gold"
65
+ if broken.stat().st_size < 1000:
66
+ return False, f"{broken} is under 1 KB"
67
+ original = broken.read_text(encoding="utf-8")
68
+ if "tota" not in original or "return tota" not in original:
69
+ return False, "fixture no longer has the NameError"
70
+ apply_source(broken, gold.read_text(encoding="utf-8"), original=original)
71
+ try:
72
+ text = broken.read_text(encoding="utf-8")
73
+ if "return total" not in text and "return sum(" not in text:
74
+ return False, "gold apply did not fix the name"
75
+ finally:
76
+ broken.write_text(original, encoding="utf-8")
77
+ bak = broken.with_suffix(broken.suffix + ".bak")
78
+ if bak.is_file():
79
+ bak.unlink()
80
+ return True, "ok"
@@ -0,0 +1,185 @@
1
+ """Score one draft against a held-out task. No model. No network."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import subprocess
6
+ import tempfile
7
+ from collections.abc import Callable, Iterator
8
+ from dataclasses import dataclass
9
+ from pathlib import Path
10
+
11
+ from harness.act.code import RunResult, extract_python, write_and_run
12
+ from harness.guard.fallbacks import PYTHON_VIBE_FALLBACK
13
+ from harness.guard.python_vibe import PythonVibeGuard
14
+ from harness.guard.run import complete
15
+ from harness.guard.types import Outcome
16
+ from harness.observe.eval_tasks import REPAIR_PREFIX, RUN_PREFIX, Task
17
+
18
+ GenerateFn = Callable[[str], str]
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class Score:
23
+ task_id: str
24
+ passed: bool
25
+ repaired: bool
26
+ reason: str
27
+ stdout: str
28
+ stderr: str
29
+ exit_code: int
30
+ fallback: bool
31
+
32
+
33
+ def stdout_matches(expect: str, got: str) -> bool:
34
+ return got.rstrip("\n") == expect.rstrip("\n")
35
+
36
+
37
+ def _remember(generate: GenerateFn, prompt: str, draft: str) -> None:
38
+ memory = getattr(generate, "memory", None)
39
+ if memory is None:
40
+ return
41
+ memory.remember(prompt, draft)
42
+
43
+
44
+ def _reset(generate: GenerateFn) -> None:
45
+ memory = getattr(generate, "memory", None)
46
+ if memory is not None:
47
+ memory.clear()
48
+ return
49
+ history = getattr(generate, "history", None)
50
+ if history is not None:
51
+ history.clear()
52
+
53
+
54
+ def run_source(task: Task, source: str, dest: Path) -> tuple[bool, RunResult]:
55
+ work = dest.parent
56
+ for rel, content in task.files:
57
+ path = work / rel
58
+ path.parent.mkdir(parents=True, exist_ok=True)
59
+ path.write_text(content, encoding="utf-8")
60
+ try:
61
+ result = write_and_run(
62
+ source,
63
+ dest,
64
+ list(task.argv),
65
+ cwd=work,
66
+ timeout=task.timeout,
67
+ stdin=task.stdin or None,
68
+ )
69
+ except subprocess.TimeoutExpired as exc:
70
+ result = RunResult(
71
+ 124,
72
+ (exc.stdout or "") if isinstance(exc.stdout, str) else "",
73
+ (exc.stderr or "") if isinstance(exc.stderr, str) else "timed out",
74
+ )
75
+ ok = result.code == 0 and stdout_matches(task.expect_stdout, result.stdout)
76
+ return ok, result
77
+
78
+
79
+ def score_source(task: Task, source: str) -> Score:
80
+ if not source.strip():
81
+ return Score(task.id, False, False, "empty source", "", "", 1, False)
82
+ with tempfile.TemporaryDirectory(prefix=f"pv-eval-{task.id}-") as tmp:
83
+ dest = Path(tmp) / "task.py"
84
+ ok, result = run_source(task, source, dest)
85
+ return Score(
86
+ task.id,
87
+ ok,
88
+ False,
89
+ (
90
+ "pass"
91
+ if ok
92
+ else "timeout"
93
+ if result.code == 124
94
+ else "wrong output"
95
+ if result.code == 0
96
+ else "nonzero exit"
97
+ ),
98
+ result.stdout,
99
+ result.stderr,
100
+ result.code,
101
+ False,
102
+ )
103
+
104
+
105
+ def _draft_source(outcome: Outcome) -> tuple[str | None, Score | None]:
106
+ if outcome.fallback or not outcome.output:
107
+ return None, Score(
108
+ "",
109
+ False,
110
+ False,
111
+ "guard fallback" if outcome.fallback else "empty draft",
112
+ "",
113
+ "",
114
+ 1,
115
+ outcome.fallback,
116
+ )
117
+ source = extract_python(outcome.output)
118
+ if not source:
119
+ return None, Score("", False, False, "no python block", "", "", 1, False)
120
+ return source, None
121
+
122
+
123
+ def score_generate(
124
+ task: Task,
125
+ generate: GenerateFn,
126
+ *,
127
+ repair: bool = False,
128
+ guard: PythonVibeGuard | None = None,
129
+ ) -> Score:
130
+ checker = guard or PythonVibeGuard()
131
+ prompt = RUN_PREFIX + task.prompt
132
+ try:
133
+ outcome = complete(generate, checker, PYTHON_VIBE_FALLBACK, prompt)
134
+ except (TimeoutError, RuntimeError, OSError) as exc:
135
+ return Score(task.id, False, False, f"generate error: {exc}", "", str(exc), 1, False)
136
+ source, early = _draft_source(outcome)
137
+ if early is not None:
138
+ return Score(task.id, early.passed, False, early.reason, "", "", 1, early.fallback)
139
+ assert source is not None
140
+ _remember(generate, prompt, outcome.output or source)
141
+ first = score_source(task, source)
142
+ if first.passed or not repair:
143
+ return first
144
+ err = (first.stderr or first.stdout).strip() or first.reason
145
+ repair_prompt = f"{REPAIR_PREFIX}```\n{err}\n```"
146
+ try:
147
+ repaired = complete(generate, checker, PYTHON_VIBE_FALLBACK, repair_prompt)
148
+ except (TimeoutError, RuntimeError, OSError) as exc:
149
+ return Score(
150
+ task.id, False, True, f"generate error: {exc}", first.stdout, first.stderr, first.exit_code, False
151
+ )
152
+ source2, early2 = _draft_source(repaired)
153
+ if early2 is not None:
154
+ return Score(
155
+ task.id, False, True, early2.reason, first.stdout, first.stderr, first.exit_code, early2.fallback
156
+ )
157
+ assert source2 is not None
158
+ second = score_source(task, source2)
159
+ return Score(
160
+ task.id,
161
+ second.passed,
162
+ True,
163
+ f"repair {second.reason}",
164
+ second.stdout,
165
+ second.stderr,
166
+ second.exit_code,
167
+ False,
168
+ )
169
+
170
+
171
+ def run_repeats(
172
+ tasks: list[Task],
173
+ generate: GenerateFn,
174
+ *,
175
+ repair: bool,
176
+ repeats: int,
177
+ reset: Callable[[], None] | None = None,
178
+ ) -> Iterator[Score]:
179
+ for _ in range(repeats):
180
+ for task in tasks:
181
+ if reset is not None:
182
+ reset()
183
+ else:
184
+ _reset(generate)
185
+ yield score_generate(task, generate, repair=repair)