premanmcp 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/cli.js +36 -1
- package/bin/eval.js +1200 -0
- package/bin/eval_harness.py +231 -0
- package/bin/eval_target.js +530 -0
- package/bin/hook.js +61 -5
- package/bin/link.js +81 -6
- package/bin/runner.js +193 -7
- package/bin/shared.js +37 -2
- package/package.json +2 -2
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Run assert-ai against Anthropic, working around four upstream gaps.
|
|
2
|
+
|
|
3
|
+
All four workarounds live here rather than in the installed package so they
|
|
4
|
+
survive `uv sync`, which re-downloads the pinned ASSERT rev and discards any
|
|
5
|
+
edit made under .venv/.
|
|
6
|
+
|
|
7
|
+
1. Workspace header. Identity-linked Anthropic API keys carry no workspace
|
|
8
|
+
binding, so the Messages API rejects them unless the request names the
|
|
9
|
+
workspace it acts in. assert-ai has no header passthrough, so set LiteLLM's
|
|
10
|
+
global header dict before any stage runs.
|
|
11
|
+
|
|
12
|
+
2. summary_items validator. assert-ai's systematize prompt was migrated to a
|
|
13
|
+
JSON contract that no longer emits `summary_items`, but
|
|
14
|
+
`_validate_summary_items` still rejects an empty list -- while the only
|
|
15
|
+
downstream consumer, systematization_convert, treats them as optional.
|
|
16
|
+
Anthropic returns the empty list the prompt implies and the stage dies.
|
|
17
|
+
|
|
18
|
+
3. extra_body on the multi-turn tester. inference.py hardcodes
|
|
19
|
+
`extra_kwargs={"extra_body": {"store": True}}` on every tester call --
|
|
20
|
+
an OpenAI response-retention flag. Anthropic rejects the unknown field with
|
|
21
|
+
`invalid_request_error: extra_body: Extra inputs are not permitted`, which
|
|
22
|
+
assert-ai catches as LLMInputError and mislabels `tester_input_refused`, as
|
|
23
|
+
if the model had refused on content grounds. It had not. Every scenario case
|
|
24
|
+
dies this way, deterministically, on any Anthropic tester -- silently
|
|
25
|
+
removing the multi-turn half of every test bank while blaming the model.
|
|
26
|
+
Only the chat path needs this: the responses path is OpenAI-only.
|
|
27
|
+
|
|
28
|
+
4. The systematization response schema, which contradicts its own prompt.
|
|
29
|
+
`systematization_single.md` tells the model to return a behavior spec --
|
|
30
|
+
`scope`, `concept_spec.patterns` with slot components, `impact_analysis`,
|
|
31
|
+
`references` -- and says "no additional fields". The code then forces a
|
|
32
|
+
different schema through structured output: `{systematization: str,
|
|
33
|
+
summary_items: [...]}`, `extra="forbid"`. Structured output wins, so the
|
|
34
|
+
model cannot return what it was asked for. It puts something in the one
|
|
35
|
+
string field it was given and the spec is never produced.
|
|
36
|
+
|
|
37
|
+
Observed on three OpenAI configurations: the field came back as `""`, as
|
|
38
|
+
the literal word `"systematization"`, and as the behavior's own name. Two
|
|
39
|
+
of those pass `_validate_systematization`, which only checks
|
|
40
|
+
non-emptiness -- so the pipeline proceeds on a hollow taxonomy and scores
|
|
41
|
+
an agent against nothing. The gpt-4o run that burned its full 16,000-token
|
|
42
|
+
budget was the same fault wearing a different hat: the model trying to
|
|
43
|
+
write the whole document into the single string it had.
|
|
44
|
+
|
|
45
|
+
It is a leftover from the pre-migration markdown era, and the same
|
|
46
|
+
migration workaround 2 exists for. `systematization_convert` reads
|
|
47
|
+
`data["systematization"]` as text and its documented input contract is
|
|
48
|
+
exactly the object the prompt describes -- so the value was always meant to
|
|
49
|
+
be that document, serialised.
|
|
50
|
+
|
|
51
|
+
All four go away upstream: (1) with a workspace-scoped API key, (2), (3) and
|
|
52
|
+
(4) with a fix in AleWang16/ASSERT plus a bumped rev in pyproject.toml.
|
|
53
|
+
|
|
54
|
+
This file exists twice. ``preman-mcp/bin/eval_harness.py`` is a byte-identical
|
|
55
|
+
copy, shipped in the npm package so a run leased to a paired device goes through
|
|
56
|
+
the same three workarounds a server-side run does -- otherwise an Anthropic
|
|
57
|
+
tester would silently lose the multi-turn half of every bank on a customer's
|
|
58
|
+
machine and keep it on ours, and the two runs would not be comparable while
|
|
59
|
+
looking like they were. ``tests/test_eval_harness_shim.py`` fails if the copies
|
|
60
|
+
drift, and byte-identical is what makes that check possible: edit this file and
|
|
61
|
+
copy it over, do not maintain two.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
import os
|
|
65
|
+
import sys
|
|
66
|
+
from importlib.metadata import entry_points
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _config_mentions_anthropic() -> bool:
|
|
70
|
+
"""Whether this invocation routes any stage to Anthropic.
|
|
71
|
+
|
|
72
|
+
Workaround 1 is Anthropic-only, and `litellm.headers` is global -- it would
|
|
73
|
+
otherwise attach a vendor header to every provider's requests. Reading the
|
|
74
|
+
config is cheap and keeps a DeepSeek or OpenAI run from being blocked by a
|
|
75
|
+
requirement that does not apply to it.
|
|
76
|
+
"""
|
|
77
|
+
argv = sys.argv
|
|
78
|
+
if "--config" not in argv:
|
|
79
|
+
return True # cannot tell; keep the stricter behaviour
|
|
80
|
+
try:
|
|
81
|
+
path = argv[argv.index("--config") + 1]
|
|
82
|
+
return "anthropic/" in open(path, encoding="utf-8").read()
|
|
83
|
+
except (IndexError, OSError):
|
|
84
|
+
return True
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
WORKSPACE_ID = os.environ.get("ANTHROPIC_WORKSPACE_ID")
|
|
88
|
+
if _config_mentions_anthropic() and not WORKSPACE_ID:
|
|
89
|
+
sys.exit("ANTHROPIC_WORKSPACE_ID is not set (expected a wrkspc_... value)")
|
|
90
|
+
|
|
91
|
+
import litellm
|
|
92
|
+
|
|
93
|
+
if WORKSPACE_ID and _config_mentions_anthropic():
|
|
94
|
+
litellm.headers = {"anthropic-workspace-id": WORKSPACE_ID}
|
|
95
|
+
|
|
96
|
+
from pydantic import BaseModel, ConfigDict
|
|
97
|
+
|
|
98
|
+
from assert_ai.core import model_client
|
|
99
|
+
from assert_ai.stages import systematization
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _validate_summary_items(summary_items) -> None:
|
|
103
|
+
for item in summary_items:
|
|
104
|
+
if not item.description.strip():
|
|
105
|
+
raise ValueError("systematization summary_items.description must be non-empty")
|
|
106
|
+
if not item.example.strip():
|
|
107
|
+
raise ValueError("systematization summary_items.example must be non-empty")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
systematization._validate_summary_items = _validate_summary_items
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class _KeyTerm(BaseModel):
|
|
114
|
+
term: str
|
|
115
|
+
definition: str
|
|
116
|
+
|
|
117
|
+
model_config = ConfigDict(extra="forbid")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class _SlotValue(BaseModel):
|
|
121
|
+
slot_value: str
|
|
122
|
+
definition: str
|
|
123
|
+
example_phrase: str
|
|
124
|
+
|
|
125
|
+
model_config = ConfigDict(extra="forbid")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class _SlotComponent(BaseModel):
|
|
129
|
+
component: str
|
|
130
|
+
slot_values: list[_SlotValue]
|
|
131
|
+
|
|
132
|
+
model_config = ConfigDict(extra="forbid")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class _Pattern(BaseModel):
|
|
136
|
+
pattern: str
|
|
137
|
+
pattern_role: str
|
|
138
|
+
primary_theory: str
|
|
139
|
+
related_theory: str
|
|
140
|
+
key_terms: list[_KeyTerm]
|
|
141
|
+
slot_components: list[_SlotComponent]
|
|
142
|
+
|
|
143
|
+
model_config = ConfigDict(extra="forbid")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class _ConceptSpec(BaseModel):
|
|
147
|
+
behavior: str
|
|
148
|
+
patterns: list[_Pattern]
|
|
149
|
+
|
|
150
|
+
model_config = ConfigDict(extra="forbid")
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class _StakeholderLens(BaseModel):
|
|
154
|
+
label: str
|
|
155
|
+
expertise: str
|
|
156
|
+
|
|
157
|
+
model_config = ConfigDict(extra="forbid")
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
class _SystematizationDocument(BaseModel):
|
|
161
|
+
"""The output contract ``systematization_single.md`` actually asks for.
|
|
162
|
+
|
|
163
|
+
Stands in for assert-ai's ``SystematizationResponse``, which asks for
|
|
164
|
+
something else -- see workaround 4. The call site reads
|
|
165
|
+
``.model_json_schema()`` to build the request and then validates the reply
|
|
166
|
+
with the same class, so replacing the class replaces both halves at once
|
|
167
|
+
and the two cannot drift apart the way the schema and the prompt did.
|
|
168
|
+
|
|
169
|
+
``nested_slot_components`` is the one contract field left out. OpenAI's
|
|
170
|
+
strict structured output allows no unconstrained objects and no more than
|
|
171
|
+
five levels of nesting, and this schema already reaches five at
|
|
172
|
+
``slot_values``. The convert prompt treats nested slots as conditional --
|
|
173
|
+
"when a slot component contains nested_slot_components" -- so their absence
|
|
174
|
+
costs a refinement rather than breaking the stage, which an invalid schema
|
|
175
|
+
would.
|
|
176
|
+
"""
|
|
177
|
+
|
|
178
|
+
behavior: str
|
|
179
|
+
scope: str
|
|
180
|
+
impact_analysis: str
|
|
181
|
+
alternative_systematizations: str
|
|
182
|
+
references: list[str]
|
|
183
|
+
stakeholder_lenses: list[_StakeholderLens]
|
|
184
|
+
reasoning_summary: str
|
|
185
|
+
concept_spec: _ConceptSpec
|
|
186
|
+
|
|
187
|
+
model_config = ConfigDict(extra="forbid")
|
|
188
|
+
|
|
189
|
+
@property
|
|
190
|
+
def systematization(self) -> str:
|
|
191
|
+
"""The document as the text ``systematization_convert`` expects.
|
|
192
|
+
|
|
193
|
+
Serialised rather than summarised: the convert prompt's input contract
|
|
194
|
+
names these fields and reads them out of this string, so anything
|
|
195
|
+
lossy here is a taxonomy built from less than was researched.
|
|
196
|
+
"""
|
|
197
|
+
return self.model_dump_json(indent=2)
|
|
198
|
+
|
|
199
|
+
@property
|
|
200
|
+
def summary_items(self) -> list:
|
|
201
|
+
"""Always empty, and legitimately so.
|
|
202
|
+
|
|
203
|
+
The migrated prompt does not ask for them and the convert stage treats
|
|
204
|
+
them as optional. Workaround 2 is what stops the validator upstream of
|
|
205
|
+
it from rejecting the empty list.
|
|
206
|
+
"""
|
|
207
|
+
return []
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
systematization.SystematizationResponse = _SystematizationDocument
|
|
211
|
+
|
|
212
|
+
# Providers that understand `extra_body`. Everything else rejects it outright.
|
|
213
|
+
_EXTRA_BODY_PROVIDERS = ("openai/", "azure/", "azure_ai/")
|
|
214
|
+
_build_chat_payload = model_client._build_chat_payload
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _build_chat_payload_without_extra_body(model, messages, options):
|
|
218
|
+
payload = _build_chat_payload(model, messages, options)
|
|
219
|
+
if not str(model).startswith(_EXTRA_BODY_PROVIDERS):
|
|
220
|
+
payload.pop("extra_body", None)
|
|
221
|
+
return payload
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# Patched as a module attribute because the call sites resolve it as a global
|
|
225
|
+
# at call time, so rebinding here reaches all of them.
|
|
226
|
+
model_client._build_chat_payload = _build_chat_payload_without_extra_body
|
|
227
|
+
|
|
228
|
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
229
|
+
|
|
230
|
+
_cli = next(e for e in entry_points(group="console_scripts") if e.name == "assert-ai")
|
|
231
|
+
sys.exit(_cli.load()())
|