agent-loss-map 0.18.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_loss_map/__init__.py +47 -0
- agent_loss_map/adapters.py +272 -0
- agent_loss_map/cli.py +134 -0
- agent_loss_map/corpus.py +121 -0
- agent_loss_map/extra_corpus.py +218 -0
- agent_loss_map/impl_audit.py +325 -0
- agent_loss_map/report.py +188 -0
- agent_loss_map/schemas/PROVENANCE.json +14 -0
- agent_loss_map/schemas/agent-descriptor.schema.json +64 -0
- agent_loss_map/schemas/capability.schema.json +45 -0
- agent_loss_map/schemas/envelope.schema.json +159 -0
- agent_loss_map/schemas/error.schema.json +46 -0
- agent_loss_map/schemas/handshake.schema.json +79 -0
- agent_loss_map/schemas/registry.schema.json +86 -0
- agent_loss_map/schemas.py +44 -0
- agent_loss_map/spec_audit.py +307 -0
- agent_loss_map/types.py +94 -0
- agent_loss_map-0.18.0.dist-info/METADATA +283 -0
- agent_loss_map-0.18.0.dist-info/RECORD +23 -0
- agent_loss_map-0.18.0.dist-info/WHEEL +5 -0
- agent_loss_map-0.18.0.dist-info/entry_points.txt +2 -0
- agent_loss_map-0.18.0.dist-info/licenses/LICENSE +21 -0
- agent_loss_map-0.18.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""UACP interop conformance harness.
|
|
2
|
+
|
|
3
|
+
Purpose: find where a UACP agent definition and a framework-native agent
|
|
4
|
+
definition actually diverge, by computing the divergences rather than asserting
|
|
5
|
+
them, and by machine-checking UACP's own JSON Schemas against its normative text.
|
|
6
|
+
|
|
7
|
+
Provenance discipline: every foreign-format entry in `corpus` records where it
|
|
8
|
+
came from and how confident we are. Nothing here is presented as a wire format
|
|
9
|
+
unless a real serialization was observed.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _detect_version() -> str:
|
|
18
|
+
"""Single source of truth: the installed distribution's version.
|
|
19
|
+
|
|
20
|
+
This used to be a hardcoded string here alongside the one in pyproject.toml,
|
|
21
|
+
and the two drifted: 0.2.0 was published while the package reported itself as
|
|
22
|
+
0.1.0, and nothing caught it because the release gate only compares the git
|
|
23
|
+
tag against pyproject.toml. An installed package is the normal case, so ask
|
|
24
|
+
the installed metadata; a bare source checkout falls back to pyproject.
|
|
25
|
+
"""
|
|
26
|
+
try:
|
|
27
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
28
|
+
try:
|
|
29
|
+
return version("agent-loss-map")
|
|
30
|
+
except PackageNotFoundError:
|
|
31
|
+
pass
|
|
32
|
+
except ImportError: # pragma: no cover - importlib.metadata is stdlib >=3.8
|
|
33
|
+
pass
|
|
34
|
+
|
|
35
|
+
for candidate in (Path(__file__).resolve().parent.parent / "pyproject.toml",
|
|
36
|
+
Path(__file__).resolve().parent / "pyproject.toml"):
|
|
37
|
+
try:
|
|
38
|
+
text = candidate.read_text(encoding="utf-8")
|
|
39
|
+
except OSError:
|
|
40
|
+
continue
|
|
41
|
+
match = re.search(r'^version\s*=\s*"([^"]+)"', text, re.MULTILINE)
|
|
42
|
+
if match:
|
|
43
|
+
return match.group(1)
|
|
44
|
+
return "0.0.0+unknown"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
__version__ = _detect_version()
|
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""Adapters: framework-native agent description -> UACP agent descriptor.
|
|
2
|
+
|
|
3
|
+
Each adapter returns (descriptor, divergences). Divergences are *computed* from
|
|
4
|
+
the structure of the source, and the produced descriptor is then validated
|
|
5
|
+
against UACP's real JSON Schema, so a SCHEMA_VIOLATION is an observed fact
|
|
6
|
+
rather than a prediction.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from .schemas import load_schema, validator
|
|
13
|
+
from .types import Divergence, ForeignAgent
|
|
14
|
+
|
|
15
|
+
# Keywords permitted in an OpenAI function-calling `parameters` schema.
|
|
16
|
+
# Anything outside this set has no representation in that format.
|
|
17
|
+
OPENAI_PARAM_SUBSET = {
|
|
18
|
+
"type", "properties", "required", "additionalProperties",
|
|
19
|
+
"description", "enum", "default", "items",
|
|
20
|
+
}
|
|
21
|
+
OPENAI_DISALLOWED = {
|
|
22
|
+
"oneOf", "anyOf", "allOf", "not", "$ref", "$defs", "definitions",
|
|
23
|
+
"if", "then", "else", "patternProperties", "dependentSchemas",
|
|
24
|
+
"dependentRequired", "propertyNames", "const", "pattern", "format",
|
|
25
|
+
"minimum", "maximum", "minLength", "maxLength", "multipleOf",
|
|
26
|
+
"minItems", "maxItems", "uniqueItems", "contentEncoding", "prefixItems",
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
# Deliberately absent: a hand-copied copy of the id and capability-name
|
|
30
|
+
# patterns from the JSON Schemas. Predicting a SCHEMA_VIOLATION from a regex
|
|
31
|
+
# here duplicated what validate_descriptor() already observes from the real
|
|
32
|
+
# validator, so every such finding was counted twice in the totals -- once
|
|
33
|
+
# predicted, once observed. A copied pattern is also a second source of truth
|
|
34
|
+
# that can drift from the schema it was copied from, and CI only pins the
|
|
35
|
+
# schemas, not this file. The schemas are the authority. See
|
|
36
|
+
# validate_descriptor(), which is the sole source of SCHEMA_VIOLATION.
|
|
37
|
+
UACP_NAMESPACES = ("csv", "llm", "web", "deploy", "data")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def collect_disallowed(node: Any, path: str = "$", found: dict[str, str] | None = None) -> dict[str, str]:
|
|
41
|
+
"""Walk a JSON Schema and record every keyword OpenAI's subset rejects."""
|
|
42
|
+
if found is None:
|
|
43
|
+
found = {}
|
|
44
|
+
if isinstance(node, dict):
|
|
45
|
+
for key, value in node.items():
|
|
46
|
+
here = f"{path}.{key}"
|
|
47
|
+
if key in OPENAI_DISALLOWED:
|
|
48
|
+
found.setdefault(key, here)
|
|
49
|
+
collect_disallowed(value, here, found)
|
|
50
|
+
elif isinstance(node, list):
|
|
51
|
+
for i, value in enumerate(node):
|
|
52
|
+
collect_disallowed(value, f"{path}[{i}]", found)
|
|
53
|
+
return found
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def to_uacp_descriptor(agent: ForeignAgent) -> tuple[dict, list[Divergence]]:
|
|
57
|
+
divs: list[Divergence] = []
|
|
58
|
+
capabilities: list[dict] = []
|
|
59
|
+
|
|
60
|
+
if not agent.has_machine_readable_capabilities:
|
|
61
|
+
divs.append(Divergence(
|
|
62
|
+
"UNVERIFIABLE_CLAIM", "blocker", "capabilities",
|
|
63
|
+
"This framework exposes no machine-readable capability list. A UACP "
|
|
64
|
+
"registry consumer asking 'who can do csv.analyze' can only "
|
|
65
|
+
"substring-match free text, so a capability answer derived from this "
|
|
66
|
+
"agent is a guess, not a verifiable claim.",
|
|
67
|
+
f"UACP SPEC.md section 5 (Registry / Discovery) depends on "
|
|
68
|
+
f"{agent.framework} exposing structured capabilities; "
|
|
69
|
+
f"source shape: {agent.source_shape}",
|
|
70
|
+
))
|
|
71
|
+
|
|
72
|
+
for tool in agent.capabilities:
|
|
73
|
+
name = tool.get("name", "")
|
|
74
|
+
params = tool.get("parameters") or tool.get("inputSchema")
|
|
75
|
+
out = tool.get("outputSchema")
|
|
76
|
+
|
|
77
|
+
# 1. output schema: UACP requires it on every capability
|
|
78
|
+
if out is None:
|
|
79
|
+
divs.append(Divergence(
|
|
80
|
+
"LOSSY", "major", f"capabilities[{name}].outputSchema",
|
|
81
|
+
"UACP requires outputSchema on every capability. The source "
|
|
82
|
+
"declares no output contract, so the adapter must fabricate a "
|
|
83
|
+
"permissive one. Consumers will believe a guarantee the source "
|
|
84
|
+
"never made.",
|
|
85
|
+
"UACP capability.schema.json requires "
|
|
86
|
+
"['name','description','inputSchema','outputSchema']; "
|
|
87
|
+
f"{agent.framework} exposes no output schema for this tool",
|
|
88
|
+
))
|
|
89
|
+
|
|
90
|
+
# 2. OpenAI 'strict' has no JSON Schema equivalent
|
|
91
|
+
if "strict" in tool:
|
|
92
|
+
divs.append(Divergence(
|
|
93
|
+
"LOSSY", "minor", f"capabilities[{name}].strict",
|
|
94
|
+
f"The source sets strict={tool['strict']!r}. JSON Schema has no "
|
|
95
|
+
"equivalent of the OpenAI 'strict' flag, so the guarantee is "
|
|
96
|
+
"dropped in either direction and cannot be re-expressed as a "
|
|
97
|
+
"UACP consumer check.",
|
|
98
|
+
"OpenAI function-calling tool shape vs JSON Schema (draft 2020-12); "
|
|
99
|
+
"no 'strict' keyword exists in JSON Schema",
|
|
100
|
+
))
|
|
101
|
+
|
|
102
|
+
# 3. dot-notation namespace convention
|
|
103
|
+
if "." not in name:
|
|
104
|
+
divs.append(Divergence(
|
|
105
|
+
"LOSSY", "minor", f"capabilities[{name}].name",
|
|
106
|
+
f"Capability {name!r} carries no namespace prefix, so the "
|
|
107
|
+
"adapter must rename it. Registry queries scoped to a "
|
|
108
|
+
"namespace (for example 'data.*') will not match the original "
|
|
109
|
+
"name.",
|
|
110
|
+
f"UACP SPEC.md section 2 requires dot-notation namespaces "
|
|
111
|
+
f"({', '.join(UACP_NAMESPACES)}.*, custom allowed); "
|
|
112
|
+
f"source name {name!r} is unnamespaced",
|
|
113
|
+
))
|
|
114
|
+
|
|
115
|
+
capabilities.append({
|
|
116
|
+
"name": name,
|
|
117
|
+
"description": tool.get("description", ""),
|
|
118
|
+
"inputSchema": params if params is not None else {"type": "object"},
|
|
119
|
+
"outputSchema": out if out is not None else {"type": "object"},
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
# 4. statefulness has no UACP representation
|
|
123
|
+
if agent.is_stateful:
|
|
124
|
+
divs.append(Divergence(
|
|
125
|
+
"SEMANTIC_MISMATCH", "major", "lifecycle",
|
|
126
|
+
"The source agent is stateful: run() mutates internal history and is "
|
|
127
|
+
"documented as being called with new messages rather than complete "
|
|
128
|
+
"history. A UACP descriptor has no lifecycle or state field, and the "
|
|
129
|
+
"UACP envelope is stateless message passing, so two calls that look "
|
|
130
|
+
"identical on the wire mean different things on either side.",
|
|
131
|
+
f"{agent.framework} documents run() as stateful "
|
|
132
|
+
"(microsoft.github.io/autogen AgentChat agents tutorial); "
|
|
133
|
+
"UACP SPEC.md section 1 agent descriptor has no state field",
|
|
134
|
+
))
|
|
135
|
+
|
|
136
|
+
# 5. token streaming vs result streaming
|
|
137
|
+
if agent.streams_tokens:
|
|
138
|
+
divs.append(Divergence(
|
|
139
|
+
"SEMANTIC_MISMATCH", "major", "streaming",
|
|
140
|
+
"The source streams per-token chunks (ModelClientStreamingChunkEvent) "
|
|
141
|
+
"that are part of the agent's own message trace. UACP stream messages "
|
|
142
|
+
"are partial *results* correlated to a request via replyTo and "
|
|
143
|
+
"sequence numbers. Mapping one onto the other conflates 'the agent's "
|
|
144
|
+
"reasoning trace' with 'the result stream', so a UACP consumer would "
|
|
145
|
+
"receive internal reasoning it never asked for and could not tell the "
|
|
146
|
+
"two apart.",
|
|
147
|
+
f"{agent.framework} emits per-token chunks in its output stream; "
|
|
148
|
+
"UACP SPEC.md section 13 defines stream/stream.end as correlated "
|
|
149
|
+
"partial results",
|
|
150
|
+
))
|
|
151
|
+
|
|
152
|
+
# 6. multimodal content parts
|
|
153
|
+
if agent.supports_multimodal:
|
|
154
|
+
divs.append(Divergence(
|
|
155
|
+
"UNREPRESENTABLE", "major", "payload",
|
|
156
|
+
"The source accepts heterogeneous content parts (for example "
|
|
157
|
+
"MultiModalMessage(content=[str, Image])). The UACP envelope's payload "
|
|
158
|
+
"is untyped, so a list would validate, but the spec defines no "
|
|
159
|
+
"modality, media type or part ordering, and an in-memory Image object "
|
|
160
|
+
"has no wire form. Nothing in UACP distinguishes this from an "
|
|
161
|
+
"ordinary array payload.",
|
|
162
|
+
f"{agent.framework} MultiModalMessage; UACP SPEC.md section 3 "
|
|
163
|
+
"envelope fields define no content-part or modality construct",
|
|
164
|
+
))
|
|
165
|
+
|
|
166
|
+
# The agent id pattern is deliberately *not* checked here. validate_descriptor()
|
|
167
|
+
# runs the real validator over the descriptor built below and reports whatever
|
|
168
|
+
# it rejects, which is an observation rather than a prediction. Predicting here
|
|
169
|
+
# as well double-counted every such finding in the report totals.
|
|
170
|
+
|
|
171
|
+
descriptor = {
|
|
172
|
+
"id": agent.agent_id,
|
|
173
|
+
"name": agent.agent_id,
|
|
174
|
+
"version": "0.0.0-probe",
|
|
175
|
+
"description": agent.description,
|
|
176
|
+
"author": "agent-loss-map-probe",
|
|
177
|
+
"capabilities": capabilities,
|
|
178
|
+
"transports": ["stdio"],
|
|
179
|
+
"metadata": {
|
|
180
|
+
"framework": agent.framework,
|
|
181
|
+
"probeConfidence": agent.confidence,
|
|
182
|
+
},
|
|
183
|
+
}
|
|
184
|
+
return descriptor, divs
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def validate_descriptor(descriptor: dict) -> list[Divergence]:
|
|
188
|
+
"""Validate against UACP's real schema. Failures are observed, not predicted."""
|
|
189
|
+
divs: list[Divergence] = []
|
|
190
|
+
schema = load_schema("agent-descriptor")
|
|
191
|
+
for err in sorted(validator(schema).iter_errors(descriptor), key=lambda e: e.path):
|
|
192
|
+
loc = "$" + "".join(f"[{p!r}]" for p in err.absolute_path)
|
|
193
|
+
divs.append(Divergence(
|
|
194
|
+
"SCHEMA_VIOLATION", "blocker", loc,
|
|
195
|
+
f"UACP agent-descriptor.schema.json rejected the produced descriptor: "
|
|
196
|
+
f"{err.message}",
|
|
197
|
+
"agent_loss_map/schemas/agent-descriptor.schema.json",
|
|
198
|
+
))
|
|
199
|
+
return divs
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def tool_parameter_completeness(agent: ForeignAgent) -> list[Divergence]:
|
|
203
|
+
"""Report capability schemas that a strict tool-calling client will refuse.
|
|
204
|
+
|
|
205
|
+
An OpenAI-compatible `function` block is `parameters: JSONSchema`, and the
|
|
206
|
+
clients that validate it strictly require `parameters.properties` to be
|
|
207
|
+
present. A schema of exactly `{"type": "object"}` has no `properties` key, so
|
|
208
|
+
those clients reject the whole tool list with a 400 before the request runs.
|
|
209
|
+
This was not reachable from the existing checks: the capability schema
|
|
210
|
+
itself accepts `{"type": "object"}` perfectly well, so the constraint lives in
|
|
211
|
+
the *target format's* supported subset rather than in JSON Schema.
|
|
212
|
+
|
|
213
|
+
What this check can and cannot claim matters, so:
|
|
214
|
+
|
|
215
|
+
- It OBSERVES that the schema is rejected. That is computed against the
|
|
216
|
+
target format's rule, not predicted from a copied regex.
|
|
217
|
+
- It CANNOT observe how many arguments the tool actually accepts. A schema
|
|
218
|
+
that declares no properties is indistinguishable from a tool that genuinely
|
|
219
|
+
takes none. If the tool does take arguments, they are unreachable, but
|
|
220
|
+
that needs the server's source to establish and is not asserted here.
|
|
221
|
+
The detail says so rather than guessing, which is the same rule the corpus
|
|
222
|
+
provenance tiers follow.
|
|
223
|
+
"""
|
|
224
|
+
divs: list[Divergence] = []
|
|
225
|
+
for cap in agent.capabilities:
|
|
226
|
+
schema = cap.get("inputSchema") or cap.get("parameters")
|
|
227
|
+
if not isinstance(schema, dict):
|
|
228
|
+
continue
|
|
229
|
+
if schema.get("type") != "object":
|
|
230
|
+
continue
|
|
231
|
+
if "properties" in schema:
|
|
232
|
+
continue
|
|
233
|
+
divs.append(Divergence(
|
|
234
|
+
"UNREPRESENTABLE", "blocker", f"capabilities[{cap.get('name', '?')}].inputSchema",
|
|
235
|
+
"This capability declares an object schema with no `properties` key. "
|
|
236
|
+
"Clients that validate an OpenAI-compatible function block strictly "
|
|
237
|
+
"require `parameters.properties` to be present and reject the entire "
|
|
238
|
+
"tool list with a 400 before any request runs, so the capability "
|
|
239
|
+
"cannot be offered to them at all. Whether the tool actually accepts "
|
|
240
|
+
"arguments is not decidable from the schema: a tool that genuinely "
|
|
241
|
+
"takes none publishes the same thing, and in that case the fix is a "
|
|
242
|
+
"present-but-empty `properties` rather than a different schema.",
|
|
243
|
+
"OpenAI function-calling `parameters` block requires the `properties` "
|
|
244
|
+
"key for object-typed schemas; this one has keys "
|
|
245
|
+
f"{sorted(schema) or 'none'}",
|
|
246
|
+
))
|
|
247
|
+
return divs
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def uacp_to_openai_probes(agent: ForeignAgent) -> list[Divergence]:
|
|
251
|
+
"""Test the UACP -> foreign direction, where the unrepresentable losses live."""
|
|
252
|
+
divs: list[Divergence] = []
|
|
253
|
+
if agent.framework != "uacp-native":
|
|
254
|
+
return divs
|
|
255
|
+
for cap in agent.capabilities:
|
|
256
|
+
for field_name in ("inputSchema", "outputSchema"):
|
|
257
|
+
schema = cap.get(field_name)
|
|
258
|
+
if not schema:
|
|
259
|
+
continue
|
|
260
|
+
bad = collect_disallowed(schema)
|
|
261
|
+
for keyword, where in sorted(bad.items()):
|
|
262
|
+
divs.append(Divergence(
|
|
263
|
+
"UNREPRESENTABLE", "blocker", f"{cap['name']}.{field_name}.{keyword}",
|
|
264
|
+
f"UACP capability uses JSON Schema keyword {keyword!r} at "
|
|
265
|
+
f"{where}, which has no representation in an OpenAI "
|
|
266
|
+
"function-calling `parameters` block. The capability cannot be "
|
|
267
|
+
"exposed to an OpenAI-style tool consumer without either "
|
|
268
|
+
"dropping the constraint or flattening it into prose.",
|
|
269
|
+
f"{where} in the UACP capability schema vs the OpenAI "
|
|
270
|
+
"function-calling parameter subset",
|
|
271
|
+
))
|
|
272
|
+
return divs
|
agent_loss_map/cli.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""CLI: run the spec audit and the cross-format divergence probe, print a report."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
from . import __version__, impl_audit, report, spec_audit
|
|
9
|
+
from .adapters import (
|
|
10
|
+
load_schema, to_uacp_descriptor, tool_parameter_completeness,
|
|
11
|
+
uacp_to_openai_probes, validate_descriptor,
|
|
12
|
+
)
|
|
13
|
+
from . import extra_corpus
|
|
14
|
+
from .corpus import CORPUS
|
|
15
|
+
from .types import Divergence, sort_divergences
|
|
16
|
+
|
|
17
|
+
SEV_MARK = {"blocker": "[BLOCKER]", "major": "[major] ", "minor": "[minor] "}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def render(divs, title: str) -> None:
|
|
21
|
+
print()
|
|
22
|
+
print("=" * 78)
|
|
23
|
+
print(title)
|
|
24
|
+
print("=" * 78)
|
|
25
|
+
if not divs:
|
|
26
|
+
print(" no divergences found")
|
|
27
|
+
return
|
|
28
|
+
for d in sort_divergences(divs):
|
|
29
|
+
print(f" {SEV_MARK[d.severity]} {d.code:<18} {d.field}")
|
|
30
|
+
print(f" {d.detail}")
|
|
31
|
+
print(f" evidence: {d.evidence}")
|
|
32
|
+
print()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def main(argv=None) -> int:
|
|
36
|
+
ap = argparse.ArgumentParser(prog="agent-loss-map")
|
|
37
|
+
ap.add_argument("--json", action="store_true", help="emit machine-readable JSON")
|
|
38
|
+
ap.add_argument("--markdown", action="store_true",
|
|
39
|
+
help="emit the full report as Markdown, for an issue or a PR")
|
|
40
|
+
ap.add_argument("--scorecard", action="store_true",
|
|
41
|
+
help="emit the one-glance per-section table only")
|
|
42
|
+
ap.add_argument("--badge", action="store_true",
|
|
43
|
+
help="emit badge JSON for the worst finding against UACP itself")
|
|
44
|
+
ap.add_argument("--corpus", action="append", default=[], metavar="PATH",
|
|
45
|
+
help="extra corpus entry file, or a directory of them; "
|
|
46
|
+
"repeatable. Lets a format be measured without editing "
|
|
47
|
+
"the package.")
|
|
48
|
+
ap.add_argument("--fail-on-blocker", action="store_true",
|
|
49
|
+
help="exit 1 if any blocker is found")
|
|
50
|
+
args = ap.parse_args(argv)
|
|
51
|
+
|
|
52
|
+
try:
|
|
53
|
+
active_corpus = extra_corpus.merge_with_builtin(
|
|
54
|
+
extra_corpus.load_extra_corpus(args.corpus)
|
|
55
|
+
) if args.corpus else list(CORPUS)
|
|
56
|
+
except extra_corpus.CorpusError as e:
|
|
57
|
+
print(f"agent-loss-map: {e}", file=sys.stderr)
|
|
58
|
+
return 2
|
|
59
|
+
|
|
60
|
+
envelope_schema = load_schema("envelope")
|
|
61
|
+
capability_schema = load_schema("capability")
|
|
62
|
+
descriptor_schema = load_schema("agent-descriptor")
|
|
63
|
+
|
|
64
|
+
spec_divs = spec_audit.run_all(capability_schema, envelope_schema)
|
|
65
|
+
spec_divs += impl_audit.run_all(envelope_schema)
|
|
66
|
+
|
|
67
|
+
per_agent = {}
|
|
68
|
+
all_divs = []
|
|
69
|
+
for agent in active_corpus:
|
|
70
|
+
descriptor, divs = to_uacp_descriptor(agent)
|
|
71
|
+
# observed, not predicted: validate what we actually produced
|
|
72
|
+
divs += validate_descriptor(descriptor)
|
|
73
|
+
divs += uacp_to_openai_probes(agent)
|
|
74
|
+
divs += tool_parameter_completeness(agent)
|
|
75
|
+
per_agent[agent.framework] = {
|
|
76
|
+
"descriptor": descriptor,
|
|
77
|
+
"divergences": [d.as_dict() for d in sort_divergences(divs)],
|
|
78
|
+
"provenance": {
|
|
79
|
+
"confidence": agent.confidence,
|
|
80
|
+
"source_shape": agent.source_shape,
|
|
81
|
+
"provenance": agent.provenance,
|
|
82
|
+
},
|
|
83
|
+
}
|
|
84
|
+
all_divs += divs
|
|
85
|
+
|
|
86
|
+
if args.badge:
|
|
87
|
+
print(json.dumps(report.badge(spec_divs, per_agent)))
|
|
88
|
+
elif args.json:
|
|
89
|
+
print(json.dumps({
|
|
90
|
+
"harness": f"agent-loss-map {__version__}",
|
|
91
|
+
"spec_audit": [d.as_dict() for d in sort_divergences(spec_divs)],
|
|
92
|
+
"cross_format": {k: {"divergences": v["divergences"],
|
|
93
|
+
"provenance": v["provenance"]}
|
|
94
|
+
for k, v in per_agent.items()},
|
|
95
|
+
}, indent=2))
|
|
96
|
+
elif args.markdown:
|
|
97
|
+
print(report.markdown(spec_divs, per_agent, __version__))
|
|
98
|
+
elif args.scorecard:
|
|
99
|
+
print(report.scorecard(spec_divs, per_agent, __version__))
|
|
100
|
+
else:
|
|
101
|
+
print(report.scorecard(spec_divs, per_agent, __version__))
|
|
102
|
+
print("Provenance of corpus entries:")
|
|
103
|
+
for agent in active_corpus:
|
|
104
|
+
print(f" {agent.framework:<24} confidence={agent.confidence}")
|
|
105
|
+
render(spec_divs, "PART 1 UACP schema vs normative text, and schema vs the two reference implementations")
|
|
106
|
+
for framework, data in per_agent.items():
|
|
107
|
+
render(
|
|
108
|
+
[d for d in (Divergence(**x) for x in data["divergences"])],
|
|
109
|
+
f"PART 2 {framework} (confidence: {data['provenance']['confidence']})",
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
counts = report.tally(all_divs + spec_divs)
|
|
113
|
+
print()
|
|
114
|
+
print("=" * 78)
|
|
115
|
+
print("TOTALS")
|
|
116
|
+
print("=" * 78)
|
|
117
|
+
print(f" spec/schema conflicts : {len(spec_divs)}")
|
|
118
|
+
print(f" cross-format findings : {len(all_divs)}")
|
|
119
|
+
print(f" {'-' * 76}")
|
|
120
|
+
print(f" all findings : {len(all_divs) + len(spec_divs)}")
|
|
121
|
+
for sev in ("blocker", "major", "minor"):
|
|
122
|
+
print(f" {sev:<8} {counts.get(sev, 0)}")
|
|
123
|
+
print()
|
|
124
|
+
print("Caveat: the AutoGen entry models a documented constructor surface, not an")
|
|
125
|
+
print("observed serialization. Findings derived from it are weaker evidence than")
|
|
126
|
+
print("findings derived from the UACP-native entry, which is quoted from SPEC.md.")
|
|
127
|
+
|
|
128
|
+
if args.fail_on_blocker and any(d.severity == "blocker" for d in all_divs + spec_divs):
|
|
129
|
+
return 1
|
|
130
|
+
return 0
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
if __name__ == "__main__":
|
|
134
|
+
sys.exit(main())
|
agent_loss_map/corpus.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Corpus of framework-native agent descriptions.
|
|
2
|
+
|
|
3
|
+
PROVENANCE RULES (learned the hard way once already):
|
|
4
|
+
observed-serialization we read a real serialized artifact
|
|
5
|
+
documented-api taken from official documentation of the public API
|
|
6
|
+
inferred our modelling choice, not a documented shape
|
|
7
|
+
|
|
8
|
+
Where a framework has no documented portable serialization of an agent
|
|
9
|
+
definition, we say so in `source_shape` rather than implying one exists. The
|
|
10
|
+
adapter then models the constructor/API surface, and every finding derived from
|
|
11
|
+
such an entry is weaker evidence than one derived from an observed artifact.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from .types import ForeignAgent
|
|
16
|
+
|
|
17
|
+
# AutoGen AgentChat AssistantAgent.
|
|
18
|
+
# Source: microsoft.github.io/autogen/stable AgentChat "Agents" tutorial.
|
|
19
|
+
# The FunctionTool.schema output below is quoted verbatim from that page.
|
|
20
|
+
# NOTE: this models the documented constructor surface. We did NOT observe an
|
|
21
|
+
# AutoGen agent serialized to a portable artifact, so anything about *wire*
|
|
22
|
+
# portability for AutoGen is unverified.
|
|
23
|
+
AUTOGEN_ASSISTANT = ForeignAgent(
|
|
24
|
+
framework="autogen-agentchat",
|
|
25
|
+
agent_id="assistant",
|
|
26
|
+
description="Use tools to solve tasks.",
|
|
27
|
+
capabilities=[
|
|
28
|
+
{
|
|
29
|
+
# verbatim from the docs' FunctionTool.schema output
|
|
30
|
+
"name": "web_search_func",
|
|
31
|
+
"description": "web_search_func",
|
|
32
|
+
"parameters": {
|
|
33
|
+
"type": "object",
|
|
34
|
+
"properties": {"query": {"description": "query", "type": "string"}},
|
|
35
|
+
"required": ["query"],
|
|
36
|
+
"additionalProperties": False,
|
|
37
|
+
},
|
|
38
|
+
"strict": False,
|
|
39
|
+
}
|
|
40
|
+
],
|
|
41
|
+
prose_claims=["Use tools to solve tasks."],
|
|
42
|
+
is_stateful=True,
|
|
43
|
+
streams_tokens=True,
|
|
44
|
+
supports_multimodal=True,
|
|
45
|
+
provenance="Official AutoGen AgentChat agents tutorial, read 2026-09-27",
|
|
46
|
+
confidence="documented-api",
|
|
47
|
+
source_shape=(
|
|
48
|
+
"Constructor surface only: name, description, system_message, tools, "
|
|
49
|
+
"workbench, model_client. No observed portable serialization of an agent "
|
|
50
|
+
"definition."
|
|
51
|
+
),
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
# Same tool shape reached through the generic OpenAI function-calling convention.
|
|
55
|
+
# This is the lowest common denominator that most frameworks agree on, so it is
|
|
56
|
+
# the most likely thing a UACP capability has to land on in practice.
|
|
57
|
+
OPENAI_FUNCTION_TOOL = ForeignAgent(
|
|
58
|
+
framework="openai-function-calling",
|
|
59
|
+
agent_id="tool-caller",
|
|
60
|
+
description="Calls OpenAI-style function tools.",
|
|
61
|
+
capabilities=[
|
|
62
|
+
{
|
|
63
|
+
"name": "get_weather",
|
|
64
|
+
"description": "Get current weather for a city.",
|
|
65
|
+
"parameters": {
|
|
66
|
+
"type": "object",
|
|
67
|
+
"properties": {"city": {"type": "string"}},
|
|
68
|
+
"required": ["city"],
|
|
69
|
+
"additionalProperties": False,
|
|
70
|
+
},
|
|
71
|
+
"strict": True,
|
|
72
|
+
}
|
|
73
|
+
],
|
|
74
|
+
prose_claims=["Calls OpenAI-style function tools."],
|
|
75
|
+
is_stateful=None,
|
|
76
|
+
provenance="OpenAI function-calling tool shape, as emitted by AutoGen FunctionTool.schema",
|
|
77
|
+
confidence="documented-api",
|
|
78
|
+
source_shape="Function/tool declaration only. No agent identity, no lifecycle.",
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
# A hypothetical-but-realistic case: a UACP capability that uses JSON Schema
|
|
82
|
+
# features outside the OpenAI function-calling subset. Used to test the
|
|
83
|
+
# UACP -> foreign direction, which is where the hard losses live.
|
|
84
|
+
UACP_RICH_CAPABILITY = ForeignAgent(
|
|
85
|
+
framework="uacp-native",
|
|
86
|
+
agent_id="csv_analyzer_01",
|
|
87
|
+
description="Analyzes CSV files and outputs structured JSON schemas.",
|
|
88
|
+
capabilities=[
|
|
89
|
+
{
|
|
90
|
+
"name": "csv.analyze",
|
|
91
|
+
"description": "Parse a CSV and return column types and statistics.",
|
|
92
|
+
"inputSchema": {
|
|
93
|
+
"type": "object",
|
|
94
|
+
"properties": {
|
|
95
|
+
"csv": {"type": "string"},
|
|
96
|
+
"mode": {"enum": ["strict", "lenient"]},
|
|
97
|
+
},
|
|
98
|
+
"required": ["csv"],
|
|
99
|
+
"oneOf": [
|
|
100
|
+
{"properties": {"mode": {"const": "strict"}}},
|
|
101
|
+
{"properties": {"mode": {"const": "lenient"}}},
|
|
102
|
+
],
|
|
103
|
+
},
|
|
104
|
+
"outputSchema": {
|
|
105
|
+
"type": "object",
|
|
106
|
+
"properties": {"rowCount": {"type": "integer"}},
|
|
107
|
+
"required": ["rowCount"],
|
|
108
|
+
},
|
|
109
|
+
"async": True,
|
|
110
|
+
}
|
|
111
|
+
],
|
|
112
|
+
provenance="UACP SPEC.md section 1 and 2, agent descriptor example",
|
|
113
|
+
confidence="observed-serialization",
|
|
114
|
+
source_shape="UACP agent descriptor, quoted from SPEC.md",
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
CORPUS: list[ForeignAgent] = [
|
|
118
|
+
AUTOGEN_ASSISTANT,
|
|
119
|
+
OPENAI_FUNCTION_TOOL,
|
|
120
|
+
UACP_RICH_CAPABILITY,
|
|
121
|
+
]
|