premanmcp 1.0.4 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,231 @@
1
+ """Run assert-ai against Anthropic, working around four upstream gaps.
2
+
3
+ All four workarounds live here rather than in the installed package so they
4
+ survive `uv sync`, which re-downloads the pinned ASSERT rev and discards any
5
+ edit made under .venv/.
6
+
7
+ 1. Workspace header. Identity-linked Anthropic API keys carry no workspace
8
+ binding, so the Messages API rejects them unless the request names the
9
+ workspace it acts in. assert-ai has no header passthrough, so set LiteLLM's
10
+ global header dict before any stage runs.
11
+
12
+ 2. summary_items validator. assert-ai's systematize prompt was migrated to a
13
+ JSON contract that no longer emits `summary_items`, but
14
+ `_validate_summary_items` still rejects an empty list -- while the only
15
+ downstream consumer, systematization_convert, treats them as optional.
16
+ Anthropic returns the empty list the prompt implies and the stage dies.
17
+
18
+ 3. extra_body on the multi-turn tester. inference.py hardcodes
19
+ `extra_kwargs={"extra_body": {"store": True}}` on every tester call --
20
+ an OpenAI response-retention flag. Anthropic rejects the unknown field with
21
+ `invalid_request_error: extra_body: Extra inputs are not permitted`, which
22
+ assert-ai catches as LLMInputError and mislabels `tester_input_refused`, as
23
+ if the model had refused on content grounds. It had not. Every scenario case
24
+ dies this way, deterministically, on any Anthropic tester -- silently
25
+ removing the multi-turn half of every test bank while blaming the model.
26
+ Only the chat path needs this: the responses path is OpenAI-only.
27
+
28
+ 4. The systematization response schema, which contradicts its own prompt.
29
+ `systematization_single.md` tells the model to return a behavior spec --
30
+ `scope`, `concept_spec.patterns` with slot components, `impact_analysis`,
31
+ `references` -- and says "no additional fields". The code then forces a
32
+ different schema through structured output: `{systematization: str,
33
+ summary_items: [...]}`, `extra="forbid"`. Structured output wins, so the
34
+ model cannot return what it was asked for. It puts something in the one
35
+ string field it was given and the spec is never produced.
36
+
37
+ Observed on three OpenAI configurations: the field came back as `""`, as
38
+ the literal word `"systematization"`, and as the behavior's own name. Two
39
+ of those pass `_validate_systematization`, which only checks
40
+ non-emptiness -- so the pipeline proceeds on a hollow taxonomy and scores
41
+ an agent against nothing. The gpt-4o run that burned its full 16,000-token
42
+ budget was the same fault wearing a different hat: the model trying to
43
+ write the whole document into the single string it had.
44
+
45
+ It is a leftover from the pre-migration markdown era, and the same
46
+ migration workaround 2 exists for. `systematization_convert` reads
47
+ `data["systematization"]` as text and its documented input contract is
48
+ exactly the object the prompt describes -- so the value was always meant to
49
+ be that document, serialised.
50
+
51
+ All four go away upstream: (1) with a workspace-scoped API key, (2), (3) and
52
+ (4) with a fix in AleWang16/ASSERT plus a bumped rev in pyproject.toml.
53
+
54
+ This file exists twice. ``preman-mcp/bin/eval_harness.py`` is a byte-identical
55
+ copy, shipped in the npm package so a run leased to a paired device goes through
56
+ the same three workarounds a server-side run does -- otherwise an Anthropic
57
+ tester would silently lose the multi-turn half of every bank on a customer's
58
+ machine and keep it on ours, and the two runs would not be comparable while
59
+ looking like they were. ``tests/test_eval_harness_shim.py`` fails if the copies
60
+ drift, and byte-identical is what makes that check possible: edit this file and
61
+ copy it over, do not maintain two.
62
+ """
63
+
64
+ import os
65
+ import sys
66
+ from importlib.metadata import entry_points
67
+
68
+
69
+ def _config_mentions_anthropic() -> bool:
70
+ """Whether this invocation routes any stage to Anthropic.
71
+
72
+ Workaround 1 is Anthropic-only, and `litellm.headers` is global -- it would
73
+ otherwise attach a vendor header to every provider's requests. Reading the
74
+ config is cheap and keeps a DeepSeek or OpenAI run from being blocked by a
75
+ requirement that does not apply to it.
76
+ """
77
+ argv = sys.argv
78
+ if "--config" not in argv:
79
+ return True # cannot tell; keep the stricter behaviour
80
+ try:
81
+ path = argv[argv.index("--config") + 1]
82
+ return "anthropic/" in open(path, encoding="utf-8").read()
83
+ except (IndexError, OSError):
84
+ return True
85
+
86
+
87
+ WORKSPACE_ID = os.environ.get("ANTHROPIC_WORKSPACE_ID")
88
+ if _config_mentions_anthropic() and not WORKSPACE_ID:
89
+ sys.exit("ANTHROPIC_WORKSPACE_ID is not set (expected a wrkspc_... value)")
90
+
91
+ import litellm
92
+
93
+ if WORKSPACE_ID and _config_mentions_anthropic():
94
+ litellm.headers = {"anthropic-workspace-id": WORKSPACE_ID}
95
+
96
+ from pydantic import BaseModel, ConfigDict
97
+
98
+ from assert_ai.core import model_client
99
+ from assert_ai.stages import systematization
100
+
101
+
102
+ def _validate_summary_items(summary_items) -> None:
103
+ for item in summary_items:
104
+ if not item.description.strip():
105
+ raise ValueError("systematization summary_items.description must be non-empty")
106
+ if not item.example.strip():
107
+ raise ValueError("systematization summary_items.example must be non-empty")
108
+
109
+
110
+ systematization._validate_summary_items = _validate_summary_items
111
+
112
+
113
+ class _KeyTerm(BaseModel):
114
+ term: str
115
+ definition: str
116
+
117
+ model_config = ConfigDict(extra="forbid")
118
+
119
+
120
+ class _SlotValue(BaseModel):
121
+ slot_value: str
122
+ definition: str
123
+ example_phrase: str
124
+
125
+ model_config = ConfigDict(extra="forbid")
126
+
127
+
128
+ class _SlotComponent(BaseModel):
129
+ component: str
130
+ slot_values: list[_SlotValue]
131
+
132
+ model_config = ConfigDict(extra="forbid")
133
+
134
+
135
+ class _Pattern(BaseModel):
136
+ pattern: str
137
+ pattern_role: str
138
+ primary_theory: str
139
+ related_theory: str
140
+ key_terms: list[_KeyTerm]
141
+ slot_components: list[_SlotComponent]
142
+
143
+ model_config = ConfigDict(extra="forbid")
144
+
145
+
146
+ class _ConceptSpec(BaseModel):
147
+ behavior: str
148
+ patterns: list[_Pattern]
149
+
150
+ model_config = ConfigDict(extra="forbid")
151
+
152
+
153
+ class _StakeholderLens(BaseModel):
154
+ label: str
155
+ expertise: str
156
+
157
+ model_config = ConfigDict(extra="forbid")
158
+
159
+
160
+ class _SystematizationDocument(BaseModel):
161
+ """The output contract ``systematization_single.md`` actually asks for.
162
+
163
+ Stands in for assert-ai's ``SystematizationResponse``, which asks for
164
+ something else -- see workaround 4. The call site reads
165
+ ``.model_json_schema()`` to build the request and then validates the reply
166
+ with the same class, so replacing the class replaces both halves at once
167
+ and the two cannot drift apart the way the schema and the prompt did.
168
+
169
+ ``nested_slot_components`` is the one contract field left out. OpenAI's
170
+ strict structured output allows no unconstrained objects and no more than
171
+ five levels of nesting, and this schema already reaches five at
172
+ ``slot_values``. The convert prompt treats nested slots as conditional --
173
+ "when a slot component contains nested_slot_components" -- so their absence
174
+ costs a refinement rather than breaking the stage, which an invalid schema
175
+ would.
176
+ """
177
+
178
+ behavior: str
179
+ scope: str
180
+ impact_analysis: str
181
+ alternative_systematizations: str
182
+ references: list[str]
183
+ stakeholder_lenses: list[_StakeholderLens]
184
+ reasoning_summary: str
185
+ concept_spec: _ConceptSpec
186
+
187
+ model_config = ConfigDict(extra="forbid")
188
+
189
+ @property
190
+ def systematization(self) -> str:
191
+ """The document as the text ``systematization_convert`` expects.
192
+
193
+ Serialised rather than summarised: the convert prompt's input contract
194
+ names these fields and reads them out of this string, so anything
195
+ lossy here is a taxonomy built from less than was researched.
196
+ """
197
+ return self.model_dump_json(indent=2)
198
+
199
+ @property
200
+ def summary_items(self) -> list:
201
+ """Always empty, and legitimately so.
202
+
203
+ The migrated prompt does not ask for them and the convert stage treats
204
+ them as optional. Workaround 2 is what stops the validator upstream of
205
+ it from rejecting the empty list.
206
+ """
207
+ return []
208
+
209
+
210
+ systematization.SystematizationResponse = _SystematizationDocument
211
+
212
+ # Providers that understand `extra_body`. Everything else rejects it outright.
213
+ _EXTRA_BODY_PROVIDERS = ("openai/", "azure/", "azure_ai/")
214
+ _build_chat_payload = model_client._build_chat_payload
215
+
216
+
217
+ def _build_chat_payload_without_extra_body(model, messages, options):
218
+ payload = _build_chat_payload(model, messages, options)
219
+ if not str(model).startswith(_EXTRA_BODY_PROVIDERS):
220
+ payload.pop("extra_body", None)
221
+ return payload
222
+
223
+
224
+ # Patched as a module attribute because the call sites resolve it as a global
225
+ # at call time, so rebinding here reaches all of them.
226
+ model_client._build_chat_payload = _build_chat_payload_without_extra_body
227
+
228
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
229
+
230
+ _cli = next(e for e in entry_points(group="console_scripts") if e.name == "assert-ai")
231
+ sys.exit(_cli.load()())