hugpy-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hugpy_agent/__init__.py +10 -0
- hugpy_agent/adapter.py +464 -0
- hugpy_agent/audit.py +105 -0
- hugpy_agent/cli.py +376 -0
- hugpy_agent/comms.py +223 -0
- hugpy_agent/config.py +383 -0
- hugpy_agent/eval.py +510 -0
- hugpy_agent/gateway.py +435 -0
- hugpy_agent/install.py +276 -0
- hugpy_agent/journal.py +292 -0
- hugpy_agent/loop.py +654 -0
- hugpy_agent/memory.py +51 -0
- hugpy_agent/node.py +536 -0
- hugpy_agent/policy.py +61 -0
- hugpy_agent/rag.py +194 -0
- hugpy_agent/serve.py +306 -0
- hugpy_agent/subagent.py +270 -0
- hugpy_agent/tools/__init__.py +297 -0
- hugpy_agent/tools/fleet.py +633 -0
- hugpy_agent/tools/fs.py +114 -0
- hugpy_agent/tools/http.py +41 -0
- hugpy_agent/tools/shell.py +56 -0
- hugpy_agent-0.1.0.dist-info/METADATA +330 -0
- hugpy_agent-0.1.0.dist-info/RECORD +28 -0
- hugpy_agent-0.1.0.dist-info/WHEEL +5 -0
- hugpy_agent-0.1.0.dist-info/entry_points.txt +2 -0
- hugpy_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
- hugpy_agent-0.1.0.dist-info/top_level.txt +1 -0
hugpy_agent/__init__.py
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""hugpy_agent — portable agent runtime on the hugpy fleet (Phase 1 MVP).
|
|
2
|
+
|
|
3
|
+
Layers (thin -> thick, per AGENT-SYSTEM-DESIGN.md §3):
|
|
4
|
+
config -> gateway (OpenAI-compat client) -> adapter (tool-calling)
|
|
5
|
+
journal (SQLite run ledger) loop (assess->act->observe)
|
|
6
|
+
tools (registry + shell/fs/http/fleet) memory (markdown facts)
|
|
7
|
+
cli (run | chat | resume | models)
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
__version__ = "0.1.0"
|
hugpy_agent/adapter.py
ADDED
|
@@ -0,0 +1,464 @@
|
|
|
1
|
+
"""Tool-call adapter — the heart of Phase 1 (design §3.1).
|
|
2
|
+
|
|
3
|
+
The /v1 seam silently ignores `tools` today (live-probed §2.1), so the harness
|
|
4
|
+
owns tool-calling with a three-tier fallback chain:
|
|
5
|
+
|
|
6
|
+
1. NATIVE — pass `tools` through, parse `message.tool_calls`. Dormant
|
|
7
|
+
until a probe shows the seam supports it; kept so it
|
|
8
|
+
activates the day platform P0 lands, with zero code change
|
|
9
|
+
above this module.
|
|
10
|
+
2. PROMPTED — inject tool JSON schemas into the system prompt using the
|
|
11
|
+
Qwen2.5/Hermes convention and parse
|
|
12
|
+
`<tool_call>{...}</tool_call>` blocks. Default tier: the
|
|
13
|
+
fleet's Qwen-family GGUFs were trained on exactly this
|
|
14
|
+
format, so a 3B model can drive it.
|
|
15
|
+
3. CONSTRAINED — plain-JSON answer contract for models with no
|
|
16
|
+
function-calling template at all.
|
|
17
|
+
|
|
18
|
+
Sloppy-small-model posture (design §8 risk #1): schema validation with benign
|
|
19
|
+
type coercion, ONE repair round-trip on invalid output, and errors returned
|
|
20
|
+
as data — the loop decides when repeated failure becomes a structured abort.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
import os
|
|
26
|
+
import re
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
|
|
29
|
+
from .gateway import CONTINUATION_LEAK
|
|
30
|
+
|
|
31
|
+
MODE_NATIVE = "native"
|
|
32
|
+
MODE_PROMPTED = "prompted"
|
|
33
|
+
MODE_CONSTRAINED = "constrained"
|
|
34
|
+
|
|
35
|
+
_TOOL_CALL_RE = re.compile(r"<tool_call>\s*(.*?)\s*</tool_call>", re.DOTALL)
|
|
36
|
+
# Thinking-model reasoning spans. A Qwen3-family brain may emit <think>…</think>
|
|
37
|
+
# even when asked not to; the reasoning must never reach the tool_call parser
|
|
38
|
+
# or a final answer. Two patterns: a closed block, and a DANGLING open tag
|
|
39
|
+
# (budget ran out before the close) — the latter is dropped from the tag to the
|
|
40
|
+
# end of the text. Applied unconditionally (belt-and-suspenders on top of the
|
|
41
|
+
# /no_think wire suffix) so BOTH tiers benefit.
|
|
42
|
+
_THINK_CLOSED_RE = re.compile(r"<think>.*?</think>", re.DOTALL | re.IGNORECASE)
|
|
43
|
+
_THINK_OPEN_RE = re.compile(r"<think>.*\Z", re.DOTALL | re.IGNORECASE)
|
|
44
|
+
# Known junk strings the platform can leak into replies (design §2.3). Scrubbed
|
|
45
|
+
# before parsing because a leak INSIDE a JSON block corrupts structured output.
|
|
46
|
+
_LEAKS = (CONTINUATION_LEAK,)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def strip_think(text: str) -> str:
|
|
50
|
+
"""Remove <think>…</think> reasoning spans, preserving all text outside
|
|
51
|
+
them. Closed blocks first, then any remaining dangling <think> (no close)
|
|
52
|
+
to end-of-text. Whitespace at the seams is collapsed so a stripped block
|
|
53
|
+
doesn't leave a gaping gap the JSON scanner has to step over."""
|
|
54
|
+
if not text or "<think>" not in text.lower():
|
|
55
|
+
return text
|
|
56
|
+
text = _THINK_CLOSED_RE.sub("", text)
|
|
57
|
+
text = _THINK_OPEN_RE.sub("", text)
|
|
58
|
+
return text.strip()
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class ToolCall:
|
|
63
|
+
name: str
|
|
64
|
+
arguments: dict
|
|
65
|
+
raw: str = ""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class ParseOutcome:
|
|
70
|
+
calls: list[ToolCall] = field(default_factory=list)
|
|
71
|
+
errors: list[str] = field(default_factory=list) # human-readable, sent back to the model
|
|
72
|
+
plain_text: str = "" # text with tool_call blocks removed
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def scrub(text: str) -> str:
|
|
76
|
+
"""Remove known platform leak strings. Defense in depth on top of
|
|
77
|
+
max_chunks:1 — a leaked continuation prompt mid-JSON is unparseable."""
|
|
78
|
+
for leak in _LEAKS:
|
|
79
|
+
text = text.replace(leak, "")
|
|
80
|
+
return text
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
# ── prompt rendering ────────────────────────────────────────────────────────
|
|
84
|
+
_PROMPTED_TEMPLATE = """
|
|
85
|
+
# Tools
|
|
86
|
+
|
|
87
|
+
You may call ONE function per reply to assist with the task.
|
|
88
|
+
|
|
89
|
+
You are provided with function signatures within <tools></tools> XML tags:
|
|
90
|
+
<tools>
|
|
91
|
+
{tool_lines}
|
|
92
|
+
</tools>
|
|
93
|
+
|
|
94
|
+
For a function call, return a json object with function name and arguments \
|
|
95
|
+
within <tool_call></tool_call> XML tags, then STOP:
|
|
96
|
+
<tool_call>
|
|
97
|
+
{{"name": "<function-name>", "arguments": {{<args-json-object>}}}}
|
|
98
|
+
</tool_call>
|
|
99
|
+
|
|
100
|
+
The function result will come back inside <tool_response></tool_response> tags.
|
|
101
|
+
Never invent a function result — wait for the real one.
|
|
102
|
+
""".rstrip()
|
|
103
|
+
|
|
104
|
+
_CONSTRAINED_TEMPLATE = """
|
|
105
|
+
# Actions
|
|
106
|
+
|
|
107
|
+
Reply with ONLY a single JSON object (no prose, no markdown fences) choosing
|
|
108
|
+
one action per reply:
|
|
109
|
+
{{"name": "<action-name>", "arguments": {{...}}}}
|
|
110
|
+
|
|
111
|
+
Available actions and their argument schemas:
|
|
112
|
+
{tool_lines}
|
|
113
|
+
|
|
114
|
+
The action result comes back as a JSON message. Never invent a result.
|
|
115
|
+
""".rstrip()
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class Adapter:
|
|
119
|
+
"""Renders tool schemas into the system prompt and extracts/validates
|
|
120
|
+
tool calls from model output, per the active mode."""
|
|
121
|
+
|
|
122
|
+
def __init__(self, mode: str = MODE_PROMPTED):
|
|
123
|
+
self.mode = mode
|
|
124
|
+
|
|
125
|
+
# ── system-prompt injection ──────────────────────────────────────────
|
|
126
|
+
def system_prompt_block(self, tool_specs) -> str:
|
|
127
|
+
"""tool_specs: iterable with .name, .description, .parameters."""
|
|
128
|
+
lines = []
|
|
129
|
+
for t in tool_specs:
|
|
130
|
+
lines.append(json.dumps({
|
|
131
|
+
"type": "function",
|
|
132
|
+
"function": {"name": t.name, "description": t.description,
|
|
133
|
+
"parameters": t.parameters},
|
|
134
|
+
}, separators=(",", ":")))
|
|
135
|
+
joined = "\n".join(lines)
|
|
136
|
+
if self.mode == MODE_NATIVE:
|
|
137
|
+
# Native: the wire carries the schemas; the prompt only sets rules.
|
|
138
|
+
return ("# Tools\nUse the provided function-calling interface. "
|
|
139
|
+
"Call ONE function per reply.")
|
|
140
|
+
if self.mode == MODE_CONSTRAINED:
|
|
141
|
+
return _CONSTRAINED_TEMPLATE.format(tool_lines=joined)
|
|
142
|
+
return _PROMPTED_TEMPLATE.format(tool_lines=joined)
|
|
143
|
+
|
|
144
|
+
def wire_tools(self, tool_specs):
|
|
145
|
+
"""OpenAI `tools` array for the native tier; None otherwise (sending
|
|
146
|
+
it in prompted mode would waste context on a field /v1 ignores)."""
|
|
147
|
+
if self.mode != MODE_NATIVE:
|
|
148
|
+
return None
|
|
149
|
+
return [{"type": "function",
|
|
150
|
+
"function": {"name": t.name, "description": t.description,
|
|
151
|
+
"parameters": t.parameters}}
|
|
152
|
+
for t in tool_specs]
|
|
153
|
+
|
|
154
|
+
# ── extraction ───────────────────────────────────────────────────────
|
|
155
|
+
def extract(self, text: str, native_tool_calls=None) -> ParseOutcome:
|
|
156
|
+
out = ParseOutcome()
|
|
157
|
+
# Strip reasoning spans BEFORE scrubbing/parsing: a <think> block can
|
|
158
|
+
# contain braces and prose that would otherwise derail the JSON scanner
|
|
159
|
+
# or become plain_text/a final answer.
|
|
160
|
+
text = scrub(strip_think(text or ""))
|
|
161
|
+
if self.mode == MODE_NATIVE and native_tool_calls:
|
|
162
|
+
for c in native_tool_calls:
|
|
163
|
+
fn = (c or {}).get("function") or {}
|
|
164
|
+
args_raw = fn.get("arguments")
|
|
165
|
+
try:
|
|
166
|
+
args = (json.loads(args_raw)
|
|
167
|
+
if isinstance(args_raw, str) else (args_raw or {}))
|
|
168
|
+
if not isinstance(args, dict):
|
|
169
|
+
raise ValueError("arguments is not an object")
|
|
170
|
+
out.calls.append(ToolCall(fn.get("name") or "", args,
|
|
171
|
+
raw=json.dumps(c)))
|
|
172
|
+
except (json.JSONDecodeError, ValueError) as exc:
|
|
173
|
+
out.errors.append("invalid native tool_call arguments for %r: %s"
|
|
174
|
+
% (fn.get("name"), exc))
|
|
175
|
+
out.plain_text = text.strip()
|
|
176
|
+
return out
|
|
177
|
+
|
|
178
|
+
remaining = text
|
|
179
|
+
for m in _TOOL_CALL_RE.finditer(text):
|
|
180
|
+
block = m.group(1)
|
|
181
|
+
remaining = remaining.replace(m.group(0), "")
|
|
182
|
+
call, err = self._parse_call_json(block)
|
|
183
|
+
if call:
|
|
184
|
+
out.calls.append(call)
|
|
185
|
+
else:
|
|
186
|
+
out.errors.append(err)
|
|
187
|
+
out.plain_text = remaining.strip()
|
|
188
|
+
|
|
189
|
+
if not out.calls and not out.errors:
|
|
190
|
+
# Tolerance for sloppy small models: an un-fenced bare JSON call
|
|
191
|
+
# object anywhere in the reply (also the constrained tier's
|
|
192
|
+
# primary format). Better to accept a slightly-off format than to
|
|
193
|
+
# burn a repair round-trip on it.
|
|
194
|
+
call, err = self._bare_call(text)
|
|
195
|
+
if call:
|
|
196
|
+
out.calls.append(call)
|
|
197
|
+
out.plain_text = ""
|
|
198
|
+
elif err:
|
|
199
|
+
out.errors.append(err)
|
|
200
|
+
return out
|
|
201
|
+
|
|
202
|
+
def _parse_call_json(self, block: str):
|
|
203
|
+
try:
|
|
204
|
+
data = json.loads(block)
|
|
205
|
+
except json.JSONDecodeError as exc:
|
|
206
|
+
return None, ("tool_call block is not valid JSON (%s). Block was: %s"
|
|
207
|
+
% (exc, block[:300]))
|
|
208
|
+
if not isinstance(data, dict) or not data.get("name"):
|
|
209
|
+
return None, ("tool_call JSON must be an object with 'name' and "
|
|
210
|
+
"'arguments'. Got: %s" % block[:300])
|
|
211
|
+
args = data.get("arguments", {})
|
|
212
|
+
if isinstance(args, str):
|
|
213
|
+
# Models sometimes double-encode arguments; unwrap one level.
|
|
214
|
+
try:
|
|
215
|
+
args = json.loads(args)
|
|
216
|
+
except json.JSONDecodeError:
|
|
217
|
+
return None, ("'arguments' for %r is a string that is not valid "
|
|
218
|
+
"JSON: %s" % (data["name"], args[:200]))
|
|
219
|
+
if not isinstance(args, dict):
|
|
220
|
+
return None, "'arguments' for %r must be a JSON object" % data["name"]
|
|
221
|
+
return ToolCall(str(data["name"]), args, raw=block), ""
|
|
222
|
+
|
|
223
|
+
def _bare_call(self, text: str):
|
|
224
|
+
"""Find the first standalone {"name":..., "arguments":...} object.
|
|
225
|
+
Scans balanced-brace candidates rather than regex-matching JSON, since
|
|
226
|
+
nested braces defeat any regex."""
|
|
227
|
+
idx = 0
|
|
228
|
+
while True:
|
|
229
|
+
start = text.find("{", idx)
|
|
230
|
+
if start < 0:
|
|
231
|
+
return None, ""
|
|
232
|
+
depth = 0
|
|
233
|
+
for i in range(start, len(text)):
|
|
234
|
+
ch = text[i]
|
|
235
|
+
if ch == "{":
|
|
236
|
+
depth += 1
|
|
237
|
+
elif ch == "}":
|
|
238
|
+
depth -= 1
|
|
239
|
+
if depth == 0:
|
|
240
|
+
candidate = text[start:i + 1]
|
|
241
|
+
if '"name"' in candidate:
|
|
242
|
+
call, err = self._parse_call_json(candidate)
|
|
243
|
+
if call:
|
|
244
|
+
return call, ""
|
|
245
|
+
break
|
|
246
|
+
else:
|
|
247
|
+
return None, ""
|
|
248
|
+
idx = start + 1
|
|
249
|
+
|
|
250
|
+
# ── validation ───────────────────────────────────────────────────────
|
|
251
|
+
def validate(self, schema: dict, args: dict):
|
|
252
|
+
"""(errors, normalized_args) against a JSON-schema subset: object
|
|
253
|
+
properties, required, primitive types, enum, array items.
|
|
254
|
+
|
|
255
|
+
Benign coercions (str->int/float/bool where unambiguous) are applied
|
|
256
|
+
instead of rejected: small models quote numbers constantly, and a
|
|
257
|
+
silent fix here saves a whole model round-trip (design §8 risk #1).
|
|
258
|
+
"""
|
|
259
|
+
errors: list[str] = []
|
|
260
|
+
norm = dict(args)
|
|
261
|
+
props = schema.get("properties") or {}
|
|
262
|
+
for req in schema.get("required") or []:
|
|
263
|
+
if req not in args:
|
|
264
|
+
errors.append("missing required argument %r" % req)
|
|
265
|
+
for k, v in args.items():
|
|
266
|
+
spec = props.get(k)
|
|
267
|
+
if spec is None:
|
|
268
|
+
continue # unknown extras tolerated; schemas here are advisory
|
|
269
|
+
ok, coerced, err = _check_type(k, v, spec)
|
|
270
|
+
if not ok:
|
|
271
|
+
errors.append(err)
|
|
272
|
+
else:
|
|
273
|
+
norm[k] = coerced
|
|
274
|
+
return errors, norm
|
|
275
|
+
|
|
276
|
+
# ── messages back to the model ───────────────────────────────────────
|
|
277
|
+
def tool_response_message(self, name: str, result: str) -> dict:
|
|
278
|
+
"""Wrap a tool result for the wire. Prompted/constrained tiers send it
|
|
279
|
+
as a user turn (the /v1 seam has no real 'tool' role today); the Qwen
|
|
280
|
+
convention is <tool_response> tags."""
|
|
281
|
+
if self.mode == MODE_NATIVE:
|
|
282
|
+
return {"role": "tool", "name": name, "content": result}
|
|
283
|
+
if self.mode == MODE_CONSTRAINED:
|
|
284
|
+
return {"role": "user",
|
|
285
|
+
"content": json.dumps({"action_result": {"name": name,
|
|
286
|
+
"result": result}})}
|
|
287
|
+
return {"role": "user",
|
|
288
|
+
"content": "<tool_response>\n%s\n</tool_response>"
|
|
289
|
+
% json.dumps({"name": name, "result": result})}
|
|
290
|
+
|
|
291
|
+
def repair_message(self, errors: list[str]) -> dict:
|
|
292
|
+
"""The ONE repair round-trip: tell the model exactly what was wrong
|
|
293
|
+
and restate the contract. Sent as a user turn so any model sees it."""
|
|
294
|
+
detail = "\n".join("- %s" % e for e in errors)
|
|
295
|
+
if self.mode == MODE_CONSTRAINED:
|
|
296
|
+
fmt = 'a single JSON object {"name": ..., "arguments": {...}}'
|
|
297
|
+
else:
|
|
298
|
+
fmt = ('<tool_call>\n{"name": "<function-name>", "arguments": '
|
|
299
|
+
'{<args>}}\n</tool_call>')
|
|
300
|
+
return {"role": "user", "content":
|
|
301
|
+
"Your last reply had an invalid tool call:\n%s\n\n"
|
|
302
|
+
"Reply again with a corrected call, formatted EXACTLY as:\n%s\n"
|
|
303
|
+
"Output nothing else." % (detail, fmt)}
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def _check_type(key: str, value, spec: dict):
|
|
307
|
+
"""One property check. Returns (ok, coerced_value, error)."""
|
|
308
|
+
t = spec.get("type")
|
|
309
|
+
enum = spec.get("enum")
|
|
310
|
+
if enum is not None and value not in enum:
|
|
311
|
+
return False, value, ("argument %r must be one of %s, got %r"
|
|
312
|
+
% (key, enum, value))
|
|
313
|
+
if t is None:
|
|
314
|
+
return True, value, ""
|
|
315
|
+
if t == "string":
|
|
316
|
+
if isinstance(value, str):
|
|
317
|
+
return True, value, ""
|
|
318
|
+
return False, value, "argument %r must be a string, got %s" % (key, type(value).__name__)
|
|
319
|
+
if t == "integer":
|
|
320
|
+
if isinstance(value, bool):
|
|
321
|
+
return False, value, "argument %r must be an integer, got bool" % key
|
|
322
|
+
if isinstance(value, int):
|
|
323
|
+
return True, value, ""
|
|
324
|
+
if isinstance(value, str):
|
|
325
|
+
try:
|
|
326
|
+
return True, int(value.strip()), ""
|
|
327
|
+
except ValueError:
|
|
328
|
+
pass
|
|
329
|
+
if isinstance(value, float) and value.is_integer():
|
|
330
|
+
return True, int(value), ""
|
|
331
|
+
return False, value, "argument %r must be an integer, got %r" % (key, value)
|
|
332
|
+
if t == "number":
|
|
333
|
+
if isinstance(value, bool):
|
|
334
|
+
return False, value, "argument %r must be a number, got bool" % key
|
|
335
|
+
if isinstance(value, (int, float)):
|
|
336
|
+
return True, value, ""
|
|
337
|
+
if isinstance(value, str):
|
|
338
|
+
try:
|
|
339
|
+
return True, float(value.strip()), ""
|
|
340
|
+
except ValueError:
|
|
341
|
+
pass
|
|
342
|
+
return False, value, "argument %r must be a number, got %r" % (key, value)
|
|
343
|
+
if t == "boolean":
|
|
344
|
+
if isinstance(value, bool):
|
|
345
|
+
return True, value, ""
|
|
346
|
+
if isinstance(value, str) and value.strip().lower() in ("true", "false"):
|
|
347
|
+
return True, value.strip().lower() == "true", ""
|
|
348
|
+
return False, value, "argument %r must be a boolean, got %r" % (key, value)
|
|
349
|
+
if t == "array":
|
|
350
|
+
if not isinstance(value, list):
|
|
351
|
+
return False, value, "argument %r must be an array, got %s" % (key, type(value).__name__)
|
|
352
|
+
items = spec.get("items")
|
|
353
|
+
if items:
|
|
354
|
+
coerced = []
|
|
355
|
+
for i, item in enumerate(value):
|
|
356
|
+
ok, c, err = _check_type("%s[%d]" % (key, i), item, items)
|
|
357
|
+
if not ok:
|
|
358
|
+
return False, value, err
|
|
359
|
+
coerced.append(c)
|
|
360
|
+
return True, coerced, ""
|
|
361
|
+
return True, value, ""
|
|
362
|
+
if t == "object":
|
|
363
|
+
if isinstance(value, dict):
|
|
364
|
+
return True, value, ""
|
|
365
|
+
return False, value, "argument %r must be an object, got %s" % (key, type(value).__name__)
|
|
366
|
+
return True, value, "" # unknown schema type: fail open, schemas are ours
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def probe_native(gateway, model: str) -> bool:
|
|
370
|
+
"""One tiny live call: does the seam pass `tools` through? If the model
|
|
371
|
+
comes back with native tool_calls for a trivial forced call, the native
|
|
372
|
+
tier is real. Any error means 'no' — fail closed to prompted, which
|
|
373
|
+
always works.
|
|
374
|
+
|
|
375
|
+
THIS PROBE IS EXPENSIVE SERVER-SIDE TODAY — never call it unasked, and
|
|
376
|
+
only ever through cached_probe_native(). The hugpy central keeper's
|
|
377
|
+
2026-07-14 packet capture (tracing this very request as a "mystery
|
|
378
|
+
caller" incident) showed the /v1 shim does NOT forward `max_chunks` to
|
|
379
|
+
the worker: the central->worker hop carried max_chunks:null despite our
|
|
380
|
+
max_chunks:1, so a model that never saw the tools rambles and central's
|
|
381
|
+
continuation machinery extends the generation — ~58s of GPU per probe.
|
|
382
|
+
Hence max_tokens=16 (the smallest budget that still fits a tool_calls
|
|
383
|
+
reply) and a short dedicated 25s timeout so a stalled request cannot
|
|
384
|
+
hold a deployment for the full chat timeout. That same non-forwarding
|
|
385
|
+
bug is why tools-carrying requests appear to 'stall' at the seam."""
|
|
386
|
+
tools = [{"type": "function",
|
|
387
|
+
"function": {"name": "ping",
|
|
388
|
+
"description": "Reply check. Call this.",
|
|
389
|
+
"parameters": {"type": "object", "properties": {},
|
|
390
|
+
"required": []}}}]
|
|
391
|
+
try:
|
|
392
|
+
res = gateway.chat(
|
|
393
|
+
[{"role": "user", "content": "Call the ping function now."}],
|
|
394
|
+
model=model, max_tokens=16, stream=False, tools=tools, retries=0,
|
|
395
|
+
timeout=25)
|
|
396
|
+
return bool(res.ok and res.native_tool_calls)
|
|
397
|
+
except Exception:
|
|
398
|
+
return False
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
# ── probe result cache (per box, NOT per workspace) ─────────────────────────
|
|
402
|
+
# Incident root cause #1 (2026-07-14): the probe result was cached in the
|
|
403
|
+
# per-workspace journal, so every fresh workspace/test re-probed dev. The
|
|
404
|
+
# cache now lives in one per-user file so a box probes a (base, model) pair
|
|
405
|
+
# at most once per TTL across all workspaces.
|
|
406
|
+
PROBE_TTL = 7 * 24 * 3600 # seconds; seam capabilities change rarely
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def probe_cache_path() -> str:
|
|
410
|
+
"""~/.cache/hugpy_agent/probe.json, honoring XDG_CACHE_HOME. Resolved on
|
|
411
|
+
every call (not at import) so tests and sandboxes can redirect it via
|
|
412
|
+
the environment."""
|
|
413
|
+
base = os.environ.get("XDG_CACHE_HOME") or os.path.join(
|
|
414
|
+
os.path.expanduser("~"), ".cache")
|
|
415
|
+
return os.path.join(base, "hugpy_agent", "probe.json")
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _load_probe_cache(path: str) -> dict:
|
|
419
|
+
"""Corruption-tolerant read: an unreadable/invalid file means an empty
|
|
420
|
+
cache (worst case: ONE extra probe, then it is rewritten valid), never a
|
|
421
|
+
crash and never a probe loop."""
|
|
422
|
+
try:
|
|
423
|
+
with open(path, encoding="utf-8") as fh:
|
|
424
|
+
data = json.load(fh)
|
|
425
|
+
return data if isinstance(data, dict) else {}
|
|
426
|
+
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
|
427
|
+
return {}
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _store_probe_cache(path: str, data: dict) -> None:
|
|
431
|
+
"""Atomic write (tmp file + os.replace): concurrent agents on one box
|
|
432
|
+
must never observe a half-written cache — a torn file would read as
|
|
433
|
+
'empty' and trigger avoidable re-probes."""
|
|
434
|
+
import tempfile
|
|
435
|
+
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
436
|
+
fd, tmp = tempfile.mkstemp(dir=os.path.dirname(path), suffix=".tmp")
|
|
437
|
+
try:
|
|
438
|
+
with os.fdopen(fd, "w", encoding="utf-8") as fh:
|
|
439
|
+
json.dump(data, fh)
|
|
440
|
+
os.replace(tmp, path)
|
|
441
|
+
except OSError:
|
|
442
|
+
try:
|
|
443
|
+
os.unlink(tmp)
|
|
444
|
+
except OSError:
|
|
445
|
+
pass
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def cached_probe_native(gateway, model: str, ttl: int = PROBE_TTL,
|
|
449
|
+
cache_path: str | None = None) -> bool:
|
|
450
|
+
"""probe_native() behind the per-box file cache — the ONLY entry point
|
|
451
|
+
callers may use. Guarantees at most one live probe per (base, model) per
|
|
452
|
+
TTL on this machine, regardless of how many workspaces exist."""
|
|
453
|
+
import time
|
|
454
|
+
path = cache_path or probe_cache_path()
|
|
455
|
+
key = "%s|%s" % (getattr(gateway, "base", "?"), model)
|
|
456
|
+
cache = _load_probe_cache(path)
|
|
457
|
+
entry = cache.get(key)
|
|
458
|
+
if (isinstance(entry, dict) and isinstance(entry.get("ts"), (int, float))
|
|
459
|
+
and "native" in entry and time.time() - entry["ts"] < ttl):
|
|
460
|
+
return bool(entry["native"])
|
|
461
|
+
result = probe_native(gateway, model)
|
|
462
|
+
cache[key] = {"native": result, "ts": time.time()}
|
|
463
|
+
_store_probe_cache(path, cache)
|
|
464
|
+
return result
|
hugpy_agent/audit.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Audit trail — append-only JSONL, one line per tool call (P2.2).
|
|
2
|
+
|
|
3
|
+
The SQLite journal is RUN STATE (it exists so resume works); this file is
|
|
4
|
+
the AUDIT TRAIL (it exists so an operator can answer "what did the agent
|
|
5
|
+
do, when, and was it allowed?"). Keeping them separate means the audit log
|
|
6
|
+
survives journal compaction/deletion and can be shipped off-box.
|
|
7
|
+
|
|
8
|
+
Line schema (one JSON object per line, keys always present):
|
|
9
|
+
|
|
10
|
+
{ts_iso, run_id, step, tool, risk, decision,
|
|
11
|
+
args_sha256, result_sha256, result_len, duration_ms, error_bool}
|
|
12
|
+
|
|
13
|
+
Doctrines:
|
|
14
|
+
* Args/results are HASHED, never stored — an audit trail must not become
|
|
15
|
+
a data leak (secrets ride through tool args). `verbose=True`
|
|
16
|
+
(HUGPY_AUDIT_VERBOSE=1 / --audit-verbose) opts into TRUNCATED plaintext
|
|
17
|
+
for debugging; the hashes remain so lines stay correlatable.
|
|
18
|
+
* The writer NEVER raises into the run: any failure (unwritable path,
|
|
19
|
+
full disk, bad payload) is reported via on_event("audit_error", ...)
|
|
20
|
+
and the run continues. Losing an audit line is bad; killing the run
|
|
21
|
+
over it is worse.
|
|
22
|
+
* Clock-free: the caller passes `ts` (the loop sends
|
|
23
|
+
`datetime.now(timezone.utc)`), so tests are deterministic.
|
|
24
|
+
* Append is a single open('a') + write + flush per line — O_APPEND makes
|
|
25
|
+
concurrent writers line-atomic for our line sizes on POSIX.
|
|
26
|
+
"""
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import hashlib
|
|
30
|
+
import json
|
|
31
|
+
import os
|
|
32
|
+
|
|
33
|
+
# Chars of plaintext kept per field in verbose mode. Enough to debug a call;
|
|
34
|
+
# small enough that a huge fs_read result cannot bloat the log.
|
|
35
|
+
VERBOSE_MAX = 500
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def sha256_of(value) -> str:
|
|
39
|
+
"""Stable content hash. Non-strings (arg dicts) are canonicalized —
|
|
40
|
+
sorted keys, no whitespace — so the same logical args hash identically
|
|
41
|
+
across runs, processes, and dict insertion orders."""
|
|
42
|
+
if not isinstance(value, str):
|
|
43
|
+
value = json.dumps(value, sort_keys=True, separators=(",", ":"))
|
|
44
|
+
return hashlib.sha256(value.encode("utf-8")).hexdigest()
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def default_audit_path(workspace: str) -> str:
|
|
48
|
+
return os.path.join(os.path.realpath(workspace), ".hugpy_agent",
|
|
49
|
+
"audit.jsonl")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class AuditLog:
|
|
53
|
+
"""Append-only JSONL writer. `path` empty/None => auditing disabled
|
|
54
|
+
(every record() is a no-op). `on_event` receives ("audit_error", msg)
|
|
55
|
+
on write failure — the only failure signal this class ever emits."""
|
|
56
|
+
|
|
57
|
+
def __init__(self, path: str | None, verbose: bool = False,
|
|
58
|
+
on_event=None):
|
|
59
|
+
self.path = path or ""
|
|
60
|
+
self.verbose = bool(verbose)
|
|
61
|
+
self.on_event = on_event or (lambda *a, **k: None)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def enabled(self) -> bool:
|
|
65
|
+
return bool(self.path)
|
|
66
|
+
|
|
67
|
+
def record(self, ts, *, run_id: str, step: int, tool: str, risk: str,
|
|
68
|
+
decision: str, args: dict, result: str, duration_ms: int,
|
|
69
|
+
error: bool, args_sha256: str | None = None) -> None:
|
|
70
|
+
"""Append one line for a resolved tool call. `ts` is an aware
|
|
71
|
+
datetime supplied by the caller (injectable clock). Never raises.
|
|
72
|
+
`args_sha256` lets a caller that already hashed the args (the loop
|
|
73
|
+
computes it once per call for the P2.4 loop-guard) pass it in
|
|
74
|
+
instead of hashing twice; omitted, it is computed here."""
|
|
75
|
+
if not self.path:
|
|
76
|
+
return
|
|
77
|
+
try:
|
|
78
|
+
entry = {
|
|
79
|
+
"ts_iso": ts.isoformat(),
|
|
80
|
+
"run_id": run_id,
|
|
81
|
+
"step": step,
|
|
82
|
+
"tool": tool,
|
|
83
|
+
"risk": risk,
|
|
84
|
+
"decision": decision,
|
|
85
|
+
"args_sha256": args_sha256 or sha256_of(args),
|
|
86
|
+
"result_sha256": sha256_of(result),
|
|
87
|
+
"result_len": len(result),
|
|
88
|
+
"duration_ms": duration_ms,
|
|
89
|
+
"error_bool": bool(error),
|
|
90
|
+
}
|
|
91
|
+
if self.verbose:
|
|
92
|
+
# Debug opt-in: truncated plaintext ALONGSIDE the hashes
|
|
93
|
+
# (hashes stay the correlation key across modes).
|
|
94
|
+
entry["args_text"] = json.dumps(
|
|
95
|
+
args, sort_keys=True, separators=(",", ":"))[:VERBOSE_MAX]
|
|
96
|
+
entry["result_text"] = result[:VERBOSE_MAX]
|
|
97
|
+
parent = os.path.dirname(self.path)
|
|
98
|
+
if parent:
|
|
99
|
+
os.makedirs(parent, exist_ok=True)
|
|
100
|
+
with open(self.path, "a", encoding="utf-8") as fh:
|
|
101
|
+
fh.write(json.dumps(entry) + "\n")
|
|
102
|
+
fh.flush()
|
|
103
|
+
except Exception as exc: # audit must NEVER break a run (doctrine)
|
|
104
|
+
self.on_event("audit_error",
|
|
105
|
+
"%s: %s" % (type(exc).__name__, exc))
|