py-harness-cli 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- finetune/__init__.py +1 -0
- finetune/agent_system.py +41 -0
- finetune/agent_traces.py +157 -0
- finetune/everyday.py +30 -0
- finetune/hf_ollama.py +158 -0
- finetune/huggingface_store.py +144 -0
- finetune/models.py +74 -0
- finetune/paths.py +11 -0
- finetune/python_vibe.py +788 -0
- finetune/splits.py +54 -0
- finetune/systems.py +9 -0
- harness/__init__.py +42 -0
- harness/__main__.py +8 -0
- harness/act/__init__.py +6 -0
- harness/act/autofix/__init__.py +110 -0
- harness/act/autofix/additions.py +217 -0
- harness/act/autofix/conflicts.py +124 -0
- harness/act/autofix/cover.py +419 -0
- harness/act/autofix/mechanical.py +151 -0
- harness/act/autofix/missing_imports.py +50 -0
- harness/act/autofix/moves.py +439 -0
- harness/act/autofix/names.py +339 -0
- harness/act/autofix/scaffold.py +224 -0
- harness/act/code.py +157 -0
- harness/act/gate.py +229 -0
- harness/act/parse.py +247 -0
- harness/act/patch_fix.py +138 -0
- harness/act/tools.py +244 -0
- harness/agent/__init__.py +11 -0
- harness/agent/dispatch.py +235 -0
- harness/agent/loop.py +699 -0
- harness/agent/options.py +144 -0
- harness/agent/policy.py +856 -0
- harness/agent/prompt.py +170 -0
- harness/cli.py +393 -0
- harness/editor_kit.py +265 -0
- harness/guard/__init__.py +6 -0
- harness/guard/fallbacks.py +6 -0
- harness/guard/loop_guard.py +57 -0
- harness/guard/python_vibe.py +68 -0
- harness/guard/run.py +41 -0
- harness/guard/types.py +19 -0
- harness/locate.py +767 -0
- harness/mcp_stdio.py +306 -0
- harness/memory/__init__.py +5 -0
- harness/memory/conversation.py +104 -0
- harness/model/__init__.py +6 -0
- harness/model/chat_backend.py +100 -0
- harness/model/engine.py +165 -0
- harness/model/ollama_generate.py +60 -0
- harness/model/openai_generate.py +156 -0
- harness/model/outbound.py +83 -0
- harness/model/route.py +90 -0
- harness/observe/__init__.py +6 -0
- harness/observe/eval_gate.py +80 -0
- harness/observe/eval_loop.py +185 -0
- harness/observe/eval_tasks.py +399 -0
- harness/observe/report_md.py +102 -0
- harness/observe/trace_record.py +79 -0
- harness/openai_api.py +81 -0
- harness/paths.py +88 -0
- harness/py.typed +0 -0
- harness/scan/__init__.py +6 -0
- harness/scan/app_spec.py +338 -0
- harness/scan/design.py +112 -0
- harness/scan/existing.py +131 -0
- harness/scan/layout.py +254 -0
- harness/scan/names.py +308 -0
- harness/scan/project_brief.py +287 -0
- harness/scan/project_docs.py +42 -0
- harness/scan/project_scan.py +49 -0
- harness/scan/repo_map.py +101 -0
- harness/secrets.py +39 -0
- harness/server.py +199 -0
- harness/ship/__init__.py +1 -0
- harness/ship/bot_pr.py +221 -0
- harness/ship/git_ship.py +262 -0
- harness/ship/identity.py +62 -0
- harness/ship/ticket.py +251 -0
- harness/skillkit/__init__.py +6 -0
- harness/skillkit/catalog.py +241 -0
- harness/skillkit/refuse_change.py +640 -0
- harness/skillkit/refuse_finish.py +295 -0
- harness/skillkit/target.py +238 -0
- harness/task.py +717 -0
- py_harness_cli-0.3.0.dist-info/METADATA +177 -0
- py_harness_cli-0.3.0.dist-info/RECORD +92 -0
- py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
- py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
- py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
- py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
- py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
"""Held-out execution tasks. None of these prompts are in the 45 train pairs."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True)
|
|
9
|
+
class Task:
|
|
10
|
+
id: str
|
|
11
|
+
prompt: str
|
|
12
|
+
reference: str
|
|
13
|
+
argv: tuple[str, ...] = ()
|
|
14
|
+
stdin: str = ""
|
|
15
|
+
expect_stdout: str = ""
|
|
16
|
+
files: tuple[tuple[str, str], ...] = ()
|
|
17
|
+
timeout: float = 8.0
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
RUN_PREFIX = (
|
|
21
|
+
"Reply with one fenced ```python block only. "
|
|
22
|
+
"Call main() from `if __name__ == '__main__'` so running the file prints. "
|
|
23
|
+
"Stdlib only. Read extra args from sys.argv. "
|
|
24
|
+
"Print exactly what is asked — no extra text.\n\n"
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
REPAIR_PREFIX = (
|
|
28
|
+
"The script failed when I ran it. Fix it.\n"
|
|
29
|
+
"Reply with one complete fenced python block.\n"
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def all_tasks() -> tuple[Task, ...]:
|
|
34
|
+
return _TASKS
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
_TASKS: tuple[Task, ...] = (
|
|
38
|
+
Task(
|
|
39
|
+
id="weekday",
|
|
40
|
+
prompt="Print the English weekday name for a YYYY-MM-DD date in sys.argv[1].",
|
|
41
|
+
argv=("2026-08-29",),
|
|
42
|
+
expect_stdout="Saturday\n",
|
|
43
|
+
reference="""
|
|
44
|
+
import sys
|
|
45
|
+
from datetime import date
|
|
46
|
+
|
|
47
|
+
def main() -> None:
|
|
48
|
+
y, m, d = sys.argv[1].split("-")
|
|
49
|
+
print(date(int(y), int(m), int(d)).strftime("%A"))
|
|
50
|
+
|
|
51
|
+
if __name__ == "__main__":
|
|
52
|
+
main()
|
|
53
|
+
""",
|
|
54
|
+
),
|
|
55
|
+
Task(
|
|
56
|
+
id="count-ext",
|
|
57
|
+
prompt=(
|
|
58
|
+
"Count files in directory sys.argv[1] whose names end with sys.argv[2] "
|
|
59
|
+
"(example: .md). Do not recurse. Print the integer count."
|
|
60
|
+
),
|
|
61
|
+
argv=(".", ".md"),
|
|
62
|
+
files=(("a.md", "x"), ("b.md", "y"), ("c.txt", "z"), ("notes.md.bak", "q")),
|
|
63
|
+
expect_stdout="2\n",
|
|
64
|
+
reference="""
|
|
65
|
+
import sys
|
|
66
|
+
from pathlib import Path
|
|
67
|
+
|
|
68
|
+
def main() -> None:
|
|
69
|
+
root = Path(sys.argv[1])
|
|
70
|
+
suffix = sys.argv[2]
|
|
71
|
+
print(sum(1 for p in root.iterdir() if p.is_file() and p.name.endswith(suffix)))
|
|
72
|
+
|
|
73
|
+
if __name__ == "__main__":
|
|
74
|
+
main()
|
|
75
|
+
""",
|
|
76
|
+
),
|
|
77
|
+
Task(
|
|
78
|
+
id="fizzbuzz",
|
|
79
|
+
prompt=(
|
|
80
|
+
"Print FizzBuzz for 1..N inclusive, N from sys.argv[1]. "
|
|
81
|
+
"Fizz on multiples of 3, Buzz on 5, FizzBuzz on both. One value per line."
|
|
82
|
+
),
|
|
83
|
+
argv=("15",),
|
|
84
|
+
expect_stdout=(
|
|
85
|
+
"1\n2\nFizz\n4\nBuzz\nFizz\n7\n8\nFizz\nBuzz\n11\nFizz\n13\n14\nFizzBuzz\n"
|
|
86
|
+
),
|
|
87
|
+
reference="""
|
|
88
|
+
import sys
|
|
89
|
+
|
|
90
|
+
def main() -> None:
|
|
91
|
+
n = int(sys.argv[1])
|
|
92
|
+
for i in range(1, n + 1):
|
|
93
|
+
out = ""
|
|
94
|
+
if i % 3 == 0:
|
|
95
|
+
out += "Fizz"
|
|
96
|
+
if i % 5 == 0:
|
|
97
|
+
out += "Buzz"
|
|
98
|
+
print(out or i)
|
|
99
|
+
|
|
100
|
+
if __name__ == "__main__":
|
|
101
|
+
main()
|
|
102
|
+
""",
|
|
103
|
+
),
|
|
104
|
+
Task(
|
|
105
|
+
id="clamp",
|
|
106
|
+
prompt="Print clamp(x, lo, hi) for three integers sys.argv[1:4] (x, lo, hi).",
|
|
107
|
+
argv=("12", "0", "10"),
|
|
108
|
+
expect_stdout="10\n",
|
|
109
|
+
reference="""
|
|
110
|
+
import sys
|
|
111
|
+
|
|
112
|
+
def main() -> None:
|
|
113
|
+
x, lo, hi = (int(a) for a in sys.argv[1:4])
|
|
114
|
+
print(min(hi, max(lo, x)))
|
|
115
|
+
|
|
116
|
+
if __name__ == "__main__":
|
|
117
|
+
main()
|
|
118
|
+
""",
|
|
119
|
+
),
|
|
120
|
+
Task(
|
|
121
|
+
id="slugify",
|
|
122
|
+
prompt=(
|
|
123
|
+
"Slugify sys.argv[1]: lowercase, replace each run of non-ascii-letters/"
|
|
124
|
+
"digits with one hyphen, strip leading/trailing hyphens. Print the slug."
|
|
125
|
+
),
|
|
126
|
+
argv=("Hello, World!",),
|
|
127
|
+
expect_stdout="hello-world\n",
|
|
128
|
+
reference="""
|
|
129
|
+
import re
|
|
130
|
+
import sys
|
|
131
|
+
|
|
132
|
+
def main() -> None:
|
|
133
|
+
text = re.sub(r"[^a-z0-9]+", "-", sys.argv[1].lower()).strip("-")
|
|
134
|
+
print(text)
|
|
135
|
+
|
|
136
|
+
if __name__ == "__main__":
|
|
137
|
+
main()
|
|
138
|
+
""",
|
|
139
|
+
),
|
|
140
|
+
Task(
|
|
141
|
+
id="median",
|
|
142
|
+
prompt="Print the median of the integers in sys.argv[1:]. For even n, print the lower middle (integer).",
|
|
143
|
+
argv=("1", "3", "2", "9", "5"),
|
|
144
|
+
expect_stdout="3\n",
|
|
145
|
+
reference="""
|
|
146
|
+
import sys
|
|
147
|
+
|
|
148
|
+
def main() -> None:
|
|
149
|
+
nums = sorted(int(a) for a in sys.argv[1:])
|
|
150
|
+
print(nums[(len(nums) - 1) // 2])
|
|
151
|
+
|
|
152
|
+
if __name__ == "__main__":
|
|
153
|
+
main()
|
|
154
|
+
""",
|
|
155
|
+
),
|
|
156
|
+
Task(
|
|
157
|
+
id="hhmmss",
|
|
158
|
+
prompt="Convert a non-negative integer number of seconds (sys.argv[1]) to HH:MM:SS with zero-padded fields.",
|
|
159
|
+
argv=("3661",),
|
|
160
|
+
expect_stdout="01:01:01\n",
|
|
161
|
+
reference="""
|
|
162
|
+
import sys
|
|
163
|
+
|
|
164
|
+
def main() -> None:
|
|
165
|
+
total = int(sys.argv[1])
|
|
166
|
+
h, rem = divmod(total, 3600)
|
|
167
|
+
m, s = divmod(rem, 60)
|
|
168
|
+
print(f"{h:02d}:{m:02d}:{s:02d}")
|
|
169
|
+
|
|
170
|
+
if __name__ == "__main__":
|
|
171
|
+
main()
|
|
172
|
+
""",
|
|
173
|
+
),
|
|
174
|
+
Task(
|
|
175
|
+
id="rotate",
|
|
176
|
+
prompt=(
|
|
177
|
+
"Left-rotate the words in sys.argv[2:] by k=int(sys.argv[1]) positions. "
|
|
178
|
+
"Print the words joined by a single space."
|
|
179
|
+
),
|
|
180
|
+
argv=("2", "a", "b", "c", "d"),
|
|
181
|
+
expect_stdout="c d a b\n",
|
|
182
|
+
reference="""
|
|
183
|
+
import sys
|
|
184
|
+
|
|
185
|
+
def main() -> None:
|
|
186
|
+
k = int(sys.argv[1])
|
|
187
|
+
words = sys.argv[2:]
|
|
188
|
+
k %= len(words)
|
|
189
|
+
print(" ".join(words[k:] + words[:k]))
|
|
190
|
+
|
|
191
|
+
if __name__ == "__main__":
|
|
192
|
+
main()
|
|
193
|
+
""",
|
|
194
|
+
),
|
|
195
|
+
Task(
|
|
196
|
+
id="unique-order",
|
|
197
|
+
prompt="Print the words in sys.argv[1:] with duplicates removed, first occurrence kept, space-separated.",
|
|
198
|
+
argv=("a", "b", "a", "c", "b"),
|
|
199
|
+
expect_stdout="a b c\n",
|
|
200
|
+
reference="""
|
|
201
|
+
import sys
|
|
202
|
+
|
|
203
|
+
def main() -> None:
|
|
204
|
+
seen: list[str] = []
|
|
205
|
+
for word in sys.argv[1:]:
|
|
206
|
+
if word not in seen:
|
|
207
|
+
seen.append(word)
|
|
208
|
+
print(" ".join(seen))
|
|
209
|
+
|
|
210
|
+
if __name__ == "__main__":
|
|
211
|
+
main()
|
|
212
|
+
""",
|
|
213
|
+
),
|
|
214
|
+
Task(
|
|
215
|
+
id="palindrome",
|
|
216
|
+
prompt=(
|
|
217
|
+
"Print yes or no: is sys.argv[1] a palindrome if you lowercase it and "
|
|
218
|
+
"drop every character that is not a-z or 0-9?"
|
|
219
|
+
),
|
|
220
|
+
argv=("RaceCar",),
|
|
221
|
+
expect_stdout="yes\n",
|
|
222
|
+
reference="""
|
|
223
|
+
import sys
|
|
224
|
+
|
|
225
|
+
def main() -> None:
|
|
226
|
+
chars = [c for c in sys.argv[1].lower() if c.isalnum() and c.isascii()]
|
|
227
|
+
print("yes" if chars == chars[::-1] else "no")
|
|
228
|
+
|
|
229
|
+
if __name__ == "__main__":
|
|
230
|
+
main()
|
|
231
|
+
""",
|
|
232
|
+
),
|
|
233
|
+
Task(
|
|
234
|
+
id="fib",
|
|
235
|
+
prompt="Print F(n) where F(0)=0, F(1)=1, n=int(sys.argv[1]).",
|
|
236
|
+
argv=("10",),
|
|
237
|
+
expect_stdout="55\n",
|
|
238
|
+
reference="""
|
|
239
|
+
import sys
|
|
240
|
+
|
|
241
|
+
def main() -> None:
|
|
242
|
+
n = int(sys.argv[1])
|
|
243
|
+
a, b = 0, 1
|
|
244
|
+
for _ in range(n):
|
|
245
|
+
a, b = b, a + b
|
|
246
|
+
print(a)
|
|
247
|
+
|
|
248
|
+
if __name__ == "__main__":
|
|
249
|
+
main()
|
|
250
|
+
""",
|
|
251
|
+
),
|
|
252
|
+
Task(
|
|
253
|
+
id="sum-even",
|
|
254
|
+
prompt="Print the sum of the even integers in sys.argv[1:].",
|
|
255
|
+
argv=("1", "2", "3", "4", "5", "6"),
|
|
256
|
+
expect_stdout="12\n",
|
|
257
|
+
reference="""
|
|
258
|
+
import sys
|
|
259
|
+
|
|
260
|
+
def main() -> None:
|
|
261
|
+
print(sum(int(a) for a in sys.argv[1:] if int(a) % 2 == 0))
|
|
262
|
+
|
|
263
|
+
if __name__ == "__main__":
|
|
264
|
+
main()
|
|
265
|
+
""",
|
|
266
|
+
),
|
|
267
|
+
Task(
|
|
268
|
+
id="csv-col",
|
|
269
|
+
prompt=(
|
|
270
|
+
"Read CSV from stdin (no quotes). Print 0-based column sys.argv[1], "
|
|
271
|
+
"one cell per line."
|
|
272
|
+
),
|
|
273
|
+
argv=("1",),
|
|
274
|
+
stdin="a,1\nb,2\n",
|
|
275
|
+
expect_stdout="1\n2\n",
|
|
276
|
+
reference="""
|
|
277
|
+
import sys
|
|
278
|
+
|
|
279
|
+
def main() -> None:
|
|
280
|
+
idx = int(sys.argv[1])
|
|
281
|
+
for line in sys.stdin:
|
|
282
|
+
line = line.rstrip("\\n")
|
|
283
|
+
if not line:
|
|
284
|
+
continue
|
|
285
|
+
print(line.split(",")[idx])
|
|
286
|
+
|
|
287
|
+
if __name__ == "__main__":
|
|
288
|
+
main()
|
|
289
|
+
""",
|
|
290
|
+
),
|
|
291
|
+
Task(
|
|
292
|
+
id="indent4",
|
|
293
|
+
prompt="Read stdin and print it again with four spaces prepended to every line. Keep the last newline.",
|
|
294
|
+
stdin="foo\nbar\n",
|
|
295
|
+
expect_stdout=" foo\n bar\n",
|
|
296
|
+
reference="""
|
|
297
|
+
import sys
|
|
298
|
+
|
|
299
|
+
def main() -> None:
|
|
300
|
+
for line in sys.stdin:
|
|
301
|
+
if line.endswith("\\n"):
|
|
302
|
+
print(" " + line[:-1])
|
|
303
|
+
else:
|
|
304
|
+
print(" " + line, end="")
|
|
305
|
+
|
|
306
|
+
if __name__ == "__main__":
|
|
307
|
+
main()
|
|
308
|
+
""",
|
|
309
|
+
),
|
|
310
|
+
Task(
|
|
311
|
+
id="anagram",
|
|
312
|
+
prompt=(
|
|
313
|
+
"Print yes or no: are sys.argv[1] and sys.argv[2] anagrams after "
|
|
314
|
+
"lowercasing and dropping spaces?"
|
|
315
|
+
),
|
|
316
|
+
argv=("listen", "silent"),
|
|
317
|
+
expect_stdout="yes\n",
|
|
318
|
+
reference="""
|
|
319
|
+
import sys
|
|
320
|
+
|
|
321
|
+
def norm(text: str) -> list[str]:
|
|
322
|
+
return sorted(c for c in text.lower() if c != " ")
|
|
323
|
+
|
|
324
|
+
def main() -> None:
|
|
325
|
+
print("yes" if norm(sys.argv[1]) == norm(sys.argv[2]) else "no")
|
|
326
|
+
|
|
327
|
+
if __name__ == "__main__":
|
|
328
|
+
main()
|
|
329
|
+
""",
|
|
330
|
+
),
|
|
331
|
+
Task(
|
|
332
|
+
id="wrap",
|
|
333
|
+
prompt=(
|
|
334
|
+
"Word-wrap stdin to width N=int(sys.argv[1]). Split on spaces. "
|
|
335
|
+
"If a word is longer than N, put it on its own line. Print one wrapped line per row."
|
|
336
|
+
),
|
|
337
|
+
argv=("5",),
|
|
338
|
+
stdin="hello world\n",
|
|
339
|
+
expect_stdout="hello\nworld\n",
|
|
340
|
+
reference="""
|
|
341
|
+
import sys
|
|
342
|
+
|
|
343
|
+
def main() -> None:
|
|
344
|
+
width = int(sys.argv[1])
|
|
345
|
+
words = sys.stdin.read().split()
|
|
346
|
+
line: list[str] = []
|
|
347
|
+
size = 0
|
|
348
|
+
for word in words:
|
|
349
|
+
extra = len(word) if not line else len(word) + 1
|
|
350
|
+
if line and size + extra > width:
|
|
351
|
+
print(" ".join(line))
|
|
352
|
+
line = [word]
|
|
353
|
+
size = len(word)
|
|
354
|
+
else:
|
|
355
|
+
line.append(word)
|
|
356
|
+
size += extra
|
|
357
|
+
if line:
|
|
358
|
+
print(" ".join(line))
|
|
359
|
+
|
|
360
|
+
if __name__ == "__main__":
|
|
361
|
+
main()
|
|
362
|
+
""",
|
|
363
|
+
),
|
|
364
|
+
Task(
|
|
365
|
+
id="iso-date",
|
|
366
|
+
prompt="Print YYYY-MM-DD from an ISO-8601 timestamp in sys.argv[1] (may include time and timezone).",
|
|
367
|
+
argv=("2026-09-05T17:27:00",),
|
|
368
|
+
expect_stdout="2026-09-05\n",
|
|
369
|
+
reference="""
|
|
370
|
+
import sys
|
|
371
|
+
from datetime import datetime
|
|
372
|
+
|
|
373
|
+
def main() -> None:
|
|
374
|
+
raw = sys.argv[1]
|
|
375
|
+
if raw.endswith("Z"):
|
|
376
|
+
raw = raw[:-1] + "+00:00"
|
|
377
|
+
print(datetime.fromisoformat(raw).date().isoformat())
|
|
378
|
+
|
|
379
|
+
if __name__ == "__main__":
|
|
380
|
+
main()
|
|
381
|
+
""",
|
|
382
|
+
),
|
|
383
|
+
Task(
|
|
384
|
+
id="relpath",
|
|
385
|
+
prompt="Print path sys.argv[2] relative to directory sys.argv[1], using forward slashes.",
|
|
386
|
+
argv=("/Users/x/proj", "/Users/x/proj/src/a.py"),
|
|
387
|
+
expect_stdout="src/a.py\n",
|
|
388
|
+
reference="""
|
|
389
|
+
import sys
|
|
390
|
+
from pathlib import Path
|
|
391
|
+
|
|
392
|
+
def main() -> None:
|
|
393
|
+
print(Path(sys.argv[2]).relative_to(sys.argv[1]).as_posix())
|
|
394
|
+
|
|
395
|
+
if __name__ == "__main__":
|
|
396
|
+
main()
|
|
397
|
+
""",
|
|
398
|
+
),
|
|
399
|
+
)
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Turn batch-review JSONL into markdown. No model. Not a second harness."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def load_rows(path: Path) -> list[dict]:
|
|
10
|
+
rows: list[dict] = []
|
|
11
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
12
|
+
line = line.strip()
|
|
13
|
+
if line:
|
|
14
|
+
rows.append(json.loads(line))
|
|
15
|
+
return rows
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _clean_review(text: str) -> str:
|
|
19
|
+
first = (text or "").strip()
|
|
20
|
+
if first.lower().startswith("no issue"):
|
|
21
|
+
return "no issues"
|
|
22
|
+
return first.split("\n", 1)[0].strip() or "(empty)"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def render_markdown(
|
|
26
|
+
rows: list[dict],
|
|
27
|
+
*,
|
|
28
|
+
project: str,
|
|
29
|
+
extra: dict | None = None,
|
|
30
|
+
) -> str:
|
|
31
|
+
extra = extra or {}
|
|
32
|
+
clean = 0
|
|
33
|
+
flagged = 0
|
|
34
|
+
applied = 0
|
|
35
|
+
lines = [
|
|
36
|
+
"# py-harness batch review",
|
|
37
|
+
"",
|
|
38
|
+
f"Project: `{project}`",
|
|
39
|
+
"",
|
|
40
|
+
"This file is **formatted from JSONL**. The 0.5B model reviewed one "
|
|
41
|
+
"small `.py` file at a time. `PythonVibeGuard` only gates the draft "
|
|
42
|
+
"(empty / keys / `curl|sh`). It does not judge review quality. "
|
|
43
|
+
"There is no separate markdown harness.",
|
|
44
|
+
"",
|
|
45
|
+
]
|
|
46
|
+
if extra:
|
|
47
|
+
bits = [f"{k}={v}" for k, v in extra.items()]
|
|
48
|
+
lines.append("Run: " + ", ".join(bits))
|
|
49
|
+
lines.append("")
|
|
50
|
+
body: list[str] = ["| File | Bytes | Review | Applied |", "| --- | ---: | --- | --- |"]
|
|
51
|
+
extras: list[str] = []
|
|
52
|
+
for row in rows:
|
|
53
|
+
review = str(row.get("review") or "")
|
|
54
|
+
label = _clean_review(review)
|
|
55
|
+
if label == "no issues":
|
|
56
|
+
clean += 1
|
|
57
|
+
else:
|
|
58
|
+
flagged += 1
|
|
59
|
+
if row.get("applied"):
|
|
60
|
+
applied += 1
|
|
61
|
+
rel = row.get("file", "")
|
|
62
|
+
body.append(
|
|
63
|
+
f"| `{rel}` | {row.get('bytes', '')} | {label} | "
|
|
64
|
+
f"{'yes' if row.get('applied') else 'no'} |"
|
|
65
|
+
)
|
|
66
|
+
if "```" in review or (label == "no issues" and len(review) > 40):
|
|
67
|
+
extras.append(
|
|
68
|
+
f"### `{rel}`\n\nThe model said no issues but also emitted extra text "
|
|
69
|
+
f"(ignored for apply).\n"
|
|
70
|
+
)
|
|
71
|
+
lines += [
|
|
72
|
+
f"- Files: **{len(rows)}**",
|
|
73
|
+
f"- Said no issues: **{clean}**",
|
|
74
|
+
f"- Other review text: **{flagged}**",
|
|
75
|
+
f"- Applied rewrites: **{applied}**",
|
|
76
|
+
"",
|
|
77
|
+
"## Files",
|
|
78
|
+
"",
|
|
79
|
+
]
|
|
80
|
+
lines.extend(body)
|
|
81
|
+
if extras:
|
|
82
|
+
lines += ["", "## Extra model text (not applied)", ""]
|
|
83
|
+
lines.extend(extras)
|
|
84
|
+
lines += [
|
|
85
|
+
"",
|
|
86
|
+
"## How to read this",
|
|
87
|
+
"",
|
|
88
|
+
"A hundred `no issues` on 200–350 byte `__init__.py` / constants / "
|
|
89
|
+
"verifiers is the expected 0.5B outcome: the files are tiny re-exports. "
|
|
90
|
+
"It is not a sign OpenSRE was audited. For a real review, open a module "
|
|
91
|
+
"over ~1 KB with `--file` or raise `--min-bytes`.",
|
|
92
|
+
"",
|
|
93
|
+
]
|
|
94
|
+
return "\n".join(lines) + "\n"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def write_report(jsonl: Path, dest: Path, *, project: str, extra: dict | None = None) -> Path:
|
|
98
|
+
dest.write_text(
|
|
99
|
+
render_markdown(load_rows(jsonl), project=project, extra=extra),
|
|
100
|
+
encoding="utf-8",
|
|
101
|
+
)
|
|
102
|
+
return dest
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Append redacted agent turns. Never store raw keys or home paths."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from harness.secrets import secret_in
|
|
10
|
+
|
|
11
|
+
# The four shapes are shared with the guard and the outbound check, so
|
|
12
|
+
# a shape learned once is known everywhere. These two are extra to this
|
|
13
|
+
# file: a trace is written to disk and kept, so it redacts wider than a
|
|
14
|
+
# refusal needs to.
|
|
15
|
+
_ALSO_REDACT = re.compile(r"(HF_TOKEN=|-----BEGIN )")
|
|
16
|
+
_HOME = re.compile(r"/(Users|home)/[^/\s]+")
|
|
17
|
+
_URL_HOST = re.compile(r"\b([a-z][a-z0-9+.-]*://)(?:[^/@\s]+@)?([^/\s]+)", re.IGNORECASE)
|
|
18
|
+
# A hostname with no scheme in front of it. `.home` and `.local` are
|
|
19
|
+
# deliberately absent from the list without a port: `Path.home()` is
|
|
20
|
+
# standard Python and `settings.local.json` is a real file name, and a
|
|
21
|
+
# trace is training data, so mangling either teaches the model a
|
|
22
|
+
# mistake. With a port there is no ambiguity — `box.local:8443` is a
|
|
23
|
+
# host and `Path.home()` never carries one.
|
|
24
|
+
_BARE_HOST = re.compile(
|
|
25
|
+
r"\b[A-Za-z0-9-]+(?:\.[A-Za-z0-9-]+)*"
|
|
26
|
+
r"(?:\.(?:lan|internal|corp)(?::\d+)?|\.(?:local|home):\d+)\b"
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def redact(text: str) -> str:
|
|
31
|
+
if secret_in(text) or _ALSO_REDACT.search(text):
|
|
32
|
+
return "[redacted]"
|
|
33
|
+
text = _HOME.sub(lambda match: f"/{match.group(1)}/you", text)
|
|
34
|
+
text = _URL_HOST.sub(lambda match: f"{match.group(1)}[host]", text)
|
|
35
|
+
return _BARE_HOST.sub("[host]", text)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
# Where a run writes its turns when nobody says otherwise. Inside the
|
|
39
|
+
# project, because the traces are about that project's code, and hidden
|
|
40
|
+
# because nobody wants it in a listing.
|
|
41
|
+
TRACE_DIR = ".python-vibe"
|
|
42
|
+
TRACE_FILE = "traces.jsonl"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def default_trace_path(project: Path) -> Path:
|
|
46
|
+
"""Where turns go when no --record is given."""
|
|
47
|
+
return Path(project) / TRACE_DIR / TRACE_FILE
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def render_last(project: Path, *, limit: int = 8) -> str:
|
|
51
|
+
"""The most recent recorded turns, or a line saying there are none."""
|
|
52
|
+
path = default_trace_path(project)
|
|
53
|
+
if not path.is_file():
|
|
54
|
+
return f"no traces at {path}"
|
|
55
|
+
rows = [
|
|
56
|
+
json.loads(line)
|
|
57
|
+
for line in path.read_text(encoding="utf-8").splitlines()
|
|
58
|
+
if line.strip()
|
|
59
|
+
]
|
|
60
|
+
if not rows:
|
|
61
|
+
return f"no traces at {path}"
|
|
62
|
+
lines = [f"{len(rows)} turns in {path}", ""]
|
|
63
|
+
for row in rows[-limit:]:
|
|
64
|
+
action = (row.get("action") or "-").strip() or "-"
|
|
65
|
+
text = (row.get("assistant") or row.get("user") or "").strip()
|
|
66
|
+
text = " ".join(text.split())
|
|
67
|
+
if len(text) > 80:
|
|
68
|
+
text = text[:77] + "..."
|
|
69
|
+
lines.append(f"{action}: {text}" if text else action)
|
|
70
|
+
return "\n".join(lines)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def append_turn(path: Path, row: dict[str, object]) -> None:
|
|
74
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
75
|
+
clean = {key: redact(str(value)) for key, value in row.items()}
|
|
76
|
+
if any(v == "[redacted]" for v in clean.values()):
|
|
77
|
+
return
|
|
78
|
+
with path.open("a", encoding="utf-8") as fh:
|
|
79
|
+
fh.write(json.dumps(clean, ensure_ascii=False) + "\n")
|
harness/openai_api.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""The shapes an OpenAI-compatible client sends and expects.
|
|
2
|
+
|
|
3
|
+
Request parsing and reply payloads for `server.py`, so an editor can
|
|
4
|
+
talk to this project using the API it already speaks.
|
|
5
|
+
|
|
6
|
+
This is the request and reply shape, not the model. It knows what a chat request
|
|
7
|
+
looks like and nothing about weights, which is why it sits beside the
|
|
8
|
+
server rather than inside `model/`: that package is only the code that
|
|
9
|
+
talks to a model, and the CLI and the server do not reach into it.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from finetune.everyday import DEFAULT_EVERYDAY_OLLAMA, is_tiny_model
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def parse_chat_body(raw: bytes) -> dict[str, Any]:
|
|
21
|
+
body = json.loads(raw or b"{}")
|
|
22
|
+
if not isinstance(body, dict):
|
|
23
|
+
raise ValueError("json object required")
|
|
24
|
+
messages = body.get("messages")
|
|
25
|
+
if not isinstance(messages, list) or not messages:
|
|
26
|
+
raise ValueError("messages required")
|
|
27
|
+
model = str(body.get("model") or DEFAULT_EVERYDAY_OLLAMA)
|
|
28
|
+
return {"model": model, "messages": messages, "stream": bool(body.get("stream"))}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def warn_tiny(model: str) -> str | None:
|
|
32
|
+
if is_tiny_model(model):
|
|
33
|
+
return (
|
|
34
|
+
f"{model} is the 0.5B sidecar. Everyday laptop use should be "
|
|
35
|
+
f"{DEFAULT_EVERYDAY_OLLAMA} (or qwen2.5-coder:7b / 14b)."
|
|
36
|
+
)
|
|
37
|
+
return None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def ollama_openai_url(host: str) -> str:
|
|
41
|
+
return host.rstrip("/") + "/v1/chat/completions"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def models_payload(model: str) -> dict[str, Any]:
|
|
45
|
+
return {
|
|
46
|
+
"object": "list",
|
|
47
|
+
"data": [{"id": model, "object": "model", "owned_by": "ollama"}],
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def last_user_text(messages: list[Any]) -> str:
|
|
52
|
+
"""The last user turn, as plain text. Editors send a string or parts."""
|
|
53
|
+
for message in reversed(messages):
|
|
54
|
+
if not isinstance(message, dict) or message.get("role") != "user":
|
|
55
|
+
continue
|
|
56
|
+
content = message.get("content")
|
|
57
|
+
if isinstance(content, str):
|
|
58
|
+
return content.strip()
|
|
59
|
+
if isinstance(content, list):
|
|
60
|
+
parts = [
|
|
61
|
+
str(item.get("text") or "")
|
|
62
|
+
for item in content
|
|
63
|
+
if isinstance(item, dict) and item.get("type") == "text"
|
|
64
|
+
]
|
|
65
|
+
return "\n".join(part for part in parts if part).strip()
|
|
66
|
+
return ""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def chat_completion_payload(content: str, model: str) -> dict[str, Any]:
|
|
70
|
+
return {
|
|
71
|
+
"id": "py-harness",
|
|
72
|
+
"object": "chat.completion",
|
|
73
|
+
"choices": [
|
|
74
|
+
{
|
|
75
|
+
"index": 0,
|
|
76
|
+
"message": {"role": "assistant", "content": content},
|
|
77
|
+
"finish_reason": "stop",
|
|
78
|
+
}
|
|
79
|
+
],
|
|
80
|
+
"model": model,
|
|
81
|
+
}
|