bantamkit 0.27.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bantamkit/__init__.py +32 -0
- bantamkit/agent.py +458 -0
- bantamkit/assets/contracts/default.yaml +90 -0
- bantamkit/assets/evals/devteam/manifest.yaml +351 -0
- bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
- bantamkit/assets/evals/devteam/repo/README.md +12 -0
- bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
- bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
- bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
- bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
- bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
- bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
- bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
- bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
- bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
- bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
- bantamkit/assets/evals/fixtures/.gitkeep +0 -0
- bantamkit/assets/evals/fixtures/catalog.json +6 -0
- bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
- bantamkit/assets/evals/tasks/.gitkeep +0 -0
- bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
- bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
- bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
- bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
- bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
- bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
- bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
- bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
- bantamkit/assets/profiles/default.yaml +31 -0
- bantamkit/assets/profiles/patient.yaml +31 -0
- bantamkit/assets/rubrics/.gitkeep +0 -0
- bantamkit/assets/rubrics/code-quality.yaml +20 -0
- bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
- bantamkit/assets/rubrics/task-completion.yaml +28 -0
- bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
- bantamkit/assets/skills/.gitkeep +0 -0
- bantamkit/assets/skills/file-graph.md +7 -0
- bantamkit/assets/skills/memory.md +35 -0
- bantamkit/assets/tools/.gitkeep +0 -0
- bantamkit/assets/tools/bantamkit_read.json +48 -0
- bantamkit/assets/tools/bantamkit_status.json +25 -0
- bantamkit/assets/tools/build_identity.json +17 -0
- bantamkit/assets/tools/document_list.json +12 -0
- bantamkit/assets/tools/document_read.json +31 -0
- bantamkit/assets/tools/file_graph.json +12 -0
- bantamkit/assets/tools/memory_compact.json +31 -0
- bantamkit/assets/tools/memory_recall.json +38 -0
- bantamkit/assets/tools/memory_save.json +61 -0
- bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
- bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
- bantamkit/assets/tools/shiftwork_status.json +25 -0
- bantamkit/assets/tools/skill_audit.json +70 -0
- bantamkit/assets/tools/validate_json.json +31 -0
- bantamkit/assets.py +67 -0
- bantamkit/budget.py +114 -0
- bantamkit/client.py +329 -0
- bantamkit/contract.py +522 -0
- bantamkit/criticreplay.py +3241 -0
- bantamkit/critique.py +301 -0
- bantamkit/docread.py +1744 -0
- bantamkit/evalrun.py +2003 -0
- bantamkit/eventlog.py +282 -0
- bantamkit/filegraph.py +218 -0
- bantamkit/loopguard.py +101 -0
- bantamkit/mcpreport.py +763 -0
- bantamkit/mcpserver.py +1334 -0
- bantamkit/memory/__init__.py +28 -0
- bantamkit/memory/__main__.py +291 -0
- bantamkit/memory/component.py +569 -0
- bantamkit/memory/divergence.py +744 -0
- bantamkit/memory/layers.py +257 -0
- bantamkit/memory/store.py +940 -0
- bantamkit/pdfread.py +1402 -0
- bantamkit/profile.py +46 -0
- bantamkit/shiftwork.py +212 -0
- bantamkit/skillaudit.py +853 -0
- bantamkit/statusline.py +313 -0
- bantamkit/structured.py +125 -0
- bantamkit/textutil.py +30 -0
- bantamkit-0.27.0.dist-info/METADATA +207 -0
- bantamkit-0.27.0.dist-info/RECORD +119 -0
- bantamkit-0.27.0.dist-info/WHEEL +4 -0
- bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "skill_audit",
|
|
3
|
+
"description": "Price the skill catalogue every session pays for, and name the collisions in it. A skill's `description:` frontmatter is loaded into the agent's context in EVERY session; its body is read only when the skill is invoked, so the descriptions are the standing bill. Walks `root` for `<marketplace>/<plugin>/<version>/skills/<name>/SKILL.md` and answers one JSON document: `roots` (what was scanned), `skills` (how many were counted), `catalogue_bytes` (UTF-8 bytes of their `description:` values), `findings` and `omissions`. Counted skills plus omissions account for every SKILL.md found — an omission is `{subject, count, size, what}` with subject one of `plugin-not-enabled`, `duplicate-skill`, `stale-version`, `unreadable-file`, `unparsed-frontmatter`, and `what` naming the paths. Pass `enabled` with the plugin ids the host has switched on (`<plugin>@<marketplace>`, as settings.json spells them): the plugin cache also holds disabled plugins and stale version directories, and counting those inflated a real audit to 41 skills / 19,065 B against a true 34 / 14,515. Identity comes from the path: `<marketplace>/<plugin>/<version>/skills/<name>/SKILL.md` relative to `root`; any other shape is a skill outside a plugin, identified by its directory name alone and never excluded by `enabled`. Exactly ONE version directory is resolved per (marketplace, plugin) BEFORE anything is counted, and its skills are that plugin's skills: a name absent from the resolved version is absent, never merged in from another version directory. The winner is the version directory name that sorts LAST in byte order — a byte order and deliberately not a semver comparison, so two runtimes cannot disagree about which directory was read, and so that names which are not versions at all still have a total order (a real cache spells nine of one plugin's directories as content hashes plus the literal `unknown`). What byte order gets wrong is stated rather than hidden: `10.0.0` loses to `9.0.0`, and among non-version names the winner is deterministic but arbitrary; both losers are named in an omission record. Pass `versions` with the version directory the host actually serves, per `<plugin>@<marketplace>` — the `installPath` recorded in the same `installed_plugins.json` the caller reads `enabled` out of — and that plugin is resolved to the named directory whatever byte order would have said; a plugin absent from `versions` falls back to byte order. It is host truth the caller supplies, exactly as `enabled` is, and it is not refused when it names a plugin or a directory this `root` does not hold: a plugin with no version directory here is not resolved at all, and a directory that is not here means every directory that IS here loses and is named in a `stale-version` record. Measured 2026-09-05 on a real cache, byte order picks `unknown` for a plugin whose host serves `1dd995193ba2`, so the installed copy is discarded as a duplicate. The candidates are the version DIRECTORIES on disk — every path spelled `<marketplace>/<plugin>/<version>/skills` — and not the skills that survived reading: a version directory holding no `SKILL.md`, or only files that will not decode or will not parse, still exists and the host still serves it, so an older directory must not win by default. Every SKILL.md under a version directory that did not win is omitted, under one of two subjects: `duplicate-skill` when the resolved version has a skill of the same name standing in for it, `stale-version` when it does not — a skill the resolved version DROPPED, which is on disk, in nobody's bill, and would otherwise be resurrected by the audit. Deduping by the (marketplace, plugin, name) triple instead merges versions rather than choosing between them, and measured 2026-09-05 over a real cache holding a plugin's `0.4.0` (14 skills) and `0.5.0` (5, nine having moved to another plugin) it answered 31 skills / 9,280 bytes where the host serves 22 / 3,396. Inside the resolved version `duplicate-skill` still covers two files claiming the same (marketplace, plugin, name), which is reachable only for skills outside a plugin; the last in scan order wins. Every omission carries the bytes it would have added. A skill is identified as `<plugin>:<name>`, the spelling the host uses and the key `usage` is looked up by, or as `<name>` when it has no plugin. A `description:` folded over several lines is joined with single spaces before it is counted or scanned, and a value written as a whole-value quoted YAML scalar is UNWRAPPED after that fold and before either — the host's parser strips those quotes before a session sees the description, so they are not `catalogue_bytes`, and only quotes INSIDE the value delimit a trigger phrase. The unwrap applies to every frontmatter key, not just `description:`, so `name: \"s\"` is the name `s` and `router: \"true\"` is a router. A value is a whole-value scalar only when it begins with `\"` or `'` AND that opening quote's own closing quote is the value's LAST character, judged by YAML's two escape rules and no others: inside a `\"` scalar a backslash escapes the character after it, so `\\\"` does not close it and the sequences `\\\"` and `\\\\` resolve to `\"` and `\\` while every other `\\x` is left exactly as written; inside a `'` scalar a doubled `''` is one literal apostrophe, so it does not close it and resolves to `'`. Every other value is left as written and scanned literally: `\"a\" and \"b\"` begins and ends with `\"` and is NOT one scalar, because its opening quote closes at index 2, and stripping it would weld the value into one junk phrase and destroy both real ones. A value that OPENS with a quote and never closes one — including one whose last quote is escaped — is a LITERAL: it is not unwrapped, its quote character is counted, its unpaired quote opens no phrase, and it is not a `frontmatter-malformed` finding, which names failures of the BLOCK and not of one value. Measured 2026-09-05 over a real plugin cache: 17 of 31 enabled skills write `description: \"...\"`, so a reader that takes those quotes literally over-prices the majority of a catalogue and buries any phrase quoted inside them. Five findings, each `{kind, severity, skills, detail}`, severity fixed per kind: `shared-trigger-phrase` (high) — two or more skills quote the same literal phrase, so which one fires is a coin flip; `catalogue-over-budget` (high) — `catalogue_bytes` exceeds `budget`; `never-invoked` (low) — a counted skill has zero calls in `usage`; `frontmatter-malformed` (medium) — no frontmatter block, an unterminated one, or no `description:` key, where a block that cannot be parsed at all is also omitted as `unparsed-frontmatter` and adds nothing to `skills` while a parsed block with no description is a skill costing zero bytes; `name-mismatch` (medium) — the frontmatter `name:` is not the skill's directory name. Collisions are EXACT matches on literal quoted phrases and never a similarity score: Jaccard over description tokens was measured across 820 real pairs and refuted, topping out at 0.239 and ranking the two real collisions 118th and 791st of 820. Phrases are the text of the UNWRAPPED value between `\"` pairs, or between `'` pairs where the quote is not flanked by letters — `'don't'` is a contraction, not a phrase — and a phrase with no letter or digit in it is ignored. A router, a skill whose frontmatter carries `router: true`, quotes its siblings' phrases by design and is left out of the phrase index entirely. Four argument failures are refused with their own sentences and never as a partial answer: an unknown `check`, a negative `budget`, a `root` that is missing or is a file, and an EMPTY `root` — which passes JSON-schema `string` and would otherwise resolve to the server's own working directory. Nothing about the CONTENT of the tree refuses. Deterministic: the tool reads nothing but `root`. The three facts a directory of files cannot answer arrive from the caller — call counts in `usage` (`node tools/ledger/tool-usage.mjs --group skill --json`), switched-on plugins in `enabled`, served version directories in `versions` — never by reading the host's transcripts, settings or plugin registry.",
|
|
4
|
+
"surfaces": [
|
|
5
|
+
"mcp"
|
|
6
|
+
],
|
|
7
|
+
"parameters": {
|
|
8
|
+
"type": "object",
|
|
9
|
+
"required": [
|
|
10
|
+
"root"
|
|
11
|
+
],
|
|
12
|
+
"properties": {
|
|
13
|
+
"root": {
|
|
14
|
+
"type": "string",
|
|
15
|
+
"description": "Directory to scan for SKILL.md files, absolute or relative to the server's working directory. It must name a directory: the empty string is refused rather than resolved, because resolving it would audit whatever directory the server happens to be standing in"
|
|
16
|
+
},
|
|
17
|
+
"enabled": {
|
|
18
|
+
"type": "array",
|
|
19
|
+
"items": {
|
|
20
|
+
"type": "string"
|
|
21
|
+
},
|
|
22
|
+
"description": "Plugin ids that are switched on, `<plugin>@<marketplace>`; when given only skills under those plugins are counted, when omitted every skill under `root` counts"
|
|
23
|
+
},
|
|
24
|
+
"usage": {
|
|
25
|
+
"type": "object",
|
|
26
|
+
"additionalProperties": {
|
|
27
|
+
"type": "integer",
|
|
28
|
+
"minimum": 0
|
|
29
|
+
},
|
|
30
|
+
"description": "Skill id (`<plugin>:<name>`) to call count, the caller's own measurement; a counted skill absent from a supplied map is treated as zero"
|
|
31
|
+
},
|
|
32
|
+
"check": {
|
|
33
|
+
"type": "string",
|
|
34
|
+
"enum": [
|
|
35
|
+
"all",
|
|
36
|
+
"phrase",
|
|
37
|
+
"budget",
|
|
38
|
+
"frontmatter"
|
|
39
|
+
],
|
|
40
|
+
"description": "Which family of findings to report; default `all`. `skills`, `catalogue_bytes` and `omissions` are reported whatever this says"
|
|
41
|
+
},
|
|
42
|
+
"budget": {
|
|
43
|
+
"type": "integer",
|
|
44
|
+
"minimum": 0,
|
|
45
|
+
"maximum": 9007199254740991,
|
|
46
|
+
"description": "Catalogue byte budget; `catalogue-over-budget` is only reported when this is given"
|
|
47
|
+
},
|
|
48
|
+
"versions": {
|
|
49
|
+
"type": "object",
|
|
50
|
+
"additionalProperties": {
|
|
51
|
+
"type": "string"
|
|
52
|
+
},
|
|
53
|
+
"description": "Plugin id (`<plugin>@<marketplace>`) to the version directory name the host actually serves; when a plugin is named here that directory is resolved whatever byte order would have said, and a plugin absent from it falls back to byte order"
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
},
|
|
57
|
+
"output_schema": {
|
|
58
|
+
"properties": {
|
|
59
|
+
"result": {
|
|
60
|
+
"title": "Result",
|
|
61
|
+
"type": "string"
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
"required": [
|
|
65
|
+
"result"
|
|
66
|
+
],
|
|
67
|
+
"type": "object",
|
|
68
|
+
"title": "skill_auditOutput"
|
|
69
|
+
}
|
|
70
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "validate_json",
|
|
3
|
+
"description": "Validate candidate output text against a JSON Schema. Returns {valid, feedback}; when invalid, feed the feedback back to the model and retry.",
|
|
4
|
+
"surfaces": [
|
|
5
|
+
"mcp"
|
|
6
|
+
],
|
|
7
|
+
"parameters": {
|
|
8
|
+
"properties": {
|
|
9
|
+
"output": {
|
|
10
|
+
"title": "Output",
|
|
11
|
+
"type": "string"
|
|
12
|
+
},
|
|
13
|
+
"schema": {
|
|
14
|
+
"additionalProperties": true,
|
|
15
|
+
"title": "Schema",
|
|
16
|
+
"type": "object"
|
|
17
|
+
}
|
|
18
|
+
},
|
|
19
|
+
"required": [
|
|
20
|
+
"output",
|
|
21
|
+
"schema"
|
|
22
|
+
],
|
|
23
|
+
"type": "object",
|
|
24
|
+
"title": "validate_jsonArguments"
|
|
25
|
+
},
|
|
26
|
+
"output_schema": {
|
|
27
|
+
"type": "object",
|
|
28
|
+
"additionalProperties": true,
|
|
29
|
+
"title": "validate_jsonDictOutput"
|
|
30
|
+
}
|
|
31
|
+
}
|
bantamkit/assets.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Locate and load the language-agnostic asset pack."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from bantamkit.client import BantamError, Tool
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class AssetNotFound(BantamError):
|
|
13
|
+
pass
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def assets_root() -> Path:
|
|
17
|
+
env = os.environ.get("BANTAMKIT_ASSETS")
|
|
18
|
+
if env:
|
|
19
|
+
return Path(env)
|
|
20
|
+
packaged = Path(__file__).parent / "assets"
|
|
21
|
+
if packaged.exists():
|
|
22
|
+
return packaged
|
|
23
|
+
repo = Path(__file__).resolve().parents[3] / "assets"
|
|
24
|
+
if repo.exists():
|
|
25
|
+
return repo
|
|
26
|
+
raise AssetNotFound("no assets directory found; set BANTAMKIT_ASSETS")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def load_tool_asset(name: str) -> dict:
|
|
30
|
+
"""Return one tool asset VERBATIM — every key, including the ones no surface uses.
|
|
31
|
+
|
|
32
|
+
`load_tool` below narrows the same file to the three fields a model is shown. This
|
|
33
|
+
returns the whole manifest entry, because a server has to read `surfaces` (may I
|
|
34
|
+
register this at all?) and `output_schema` (what do I advertise as the return shape?)
|
|
35
|
+
and neither belongs on the agent-facing `Tool`.
|
|
36
|
+
"""
|
|
37
|
+
path = assets_root() / "tools" / f"{name}.json"
|
|
38
|
+
if not path.exists():
|
|
39
|
+
raise AssetNotFound(f"tool asset not found: {path}")
|
|
40
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def load_tool(name: str) -> Tool:
|
|
44
|
+
"""The agent-facing view of a tool asset: name, description, input schema.
|
|
45
|
+
|
|
46
|
+
The three keys are named EXPLICITLY rather than splatted. `assets/tools/` is shared by
|
|
47
|
+
two tool surfaces and carries fields that only the MCP one consumes; a loader that
|
|
48
|
+
passed the dict through would add them to every tool definition an eval-run model is
|
|
49
|
+
shown, changing the agent surface every time the manifest grows.
|
|
50
|
+
"""
|
|
51
|
+
data = load_tool_asset(name)
|
|
52
|
+
return Tool(name=data["name"], description=data["description"], parameters=data["parameters"])
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def load_schema(name: str) -> dict:
|
|
56
|
+
"""Return a JSON Schema asset (parsed) — e.g. the shift-work checkpoint contract."""
|
|
57
|
+
path = assets_root() / "schemas" / f"{name}.json"
|
|
58
|
+
if not path.exists():
|
|
59
|
+
raise AssetNotFound(f"schema asset not found: {path}")
|
|
60
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def load_skill(name: str) -> str:
|
|
64
|
+
path = assets_root() / "skills" / f"{name}.md"
|
|
65
|
+
if not path.exists():
|
|
66
|
+
raise AssetNotFound(f"skill asset not found: {path}")
|
|
67
|
+
return path.read_text(encoding="utf-8")
|
bantamkit/budget.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Layer 1 — a global spend governor for one agent run: `TokenBudget`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from bantamkit.client import Usage
|
|
6
|
+
from bantamkit.profile import default as profile_default
|
|
7
|
+
from bantamkit.profile import default_float as profile_default_float
|
|
8
|
+
|
|
9
|
+
__all__ = ["TokenBudget"]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class _BudgetedClient:
|
|
13
|
+
"""The recording point: every chat through the agent's client books its usage.
|
|
14
|
+
|
|
15
|
+
A proxy rather than a subclass, because the thing being wrapped is whatever
|
|
16
|
+
the caller handed the agent (a raw adapter, the eval harness's tracking
|
|
17
|
+
client, a fake). `__getattr__` passes everything else through, so the
|
|
18
|
+
duck-typed contracts callers test for — `seed`, `model`,
|
|
19
|
+
`_response_format_unsupported` — read the inner client unchanged.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
def __init__(self, inner, budget: TokenBudget):
|
|
23
|
+
self.inner = inner
|
|
24
|
+
self.budget = budget
|
|
25
|
+
|
|
26
|
+
def chat(self, *args, **kwargs):
|
|
27
|
+
resp = self.inner.chat(*args, **kwargs)
|
|
28
|
+
self.budget.record(resp.usage)
|
|
29
|
+
return resp
|
|
30
|
+
|
|
31
|
+
def __getattr__(self, name: str):
|
|
32
|
+
return getattr(self.inner, name)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class TokenBudget:
|
|
36
|
+
"""A governor, not an odometer: it decides what a run may still spend.
|
|
37
|
+
|
|
38
|
+
Every gate already caps its own retries; composition multiplies those caps
|
|
39
|
+
and nothing bounded the total (3b `full`: 214,698 tokens for 15/66). This
|
|
40
|
+
component bounds the run and degrades it in two steps instead of one:
|
|
41
|
+
|
|
42
|
+
- past ``ceiling * optional_cutoff`` spent, ``allow("optional")`` is denied,
|
|
43
|
+
so optional work is skipped and a `full` run degrades toward `lean`/`bare`;
|
|
44
|
+
- past ``ceiling``, everything is denied and `Agent.run` stops at the top of
|
|
45
|
+
the next turn, returning the last assistant content as a best-effort result.
|
|
46
|
+
|
|
47
|
+
The gap between the cutoff and the ceiling **is** the reserve: the final
|
|
48
|
+
answer emission always fits.
|
|
49
|
+
|
|
50
|
+
Priority classification (v1, deliberately coarse — no cost estimation, which
|
|
51
|
+
would be guesswork until a measured need exists):
|
|
52
|
+
|
|
53
|
+
- ``"optional"`` — critique rounds, blind and grounded. Denied means "accept
|
|
54
|
+
the answer", never an exception.
|
|
55
|
+
- ``"required"`` — the agent's own turns and the schema / json-answer retries.
|
|
56
|
+
Cheap, high-value, and denied only past the hard ceiling.
|
|
57
|
+
|
|
58
|
+
Spend is booked at the client boundary: `setup` wraps `agent.client` in a
|
|
59
|
+
`_BudgetedClient`, so every call made through the agent's client moves
|
|
60
|
+
`spent` — the agent's own turns *and* the critic calls the gates issue
|
|
61
|
+
through `structured()` on that same client. That is the point: the critique
|
|
62
|
+
rounds are the dominant optional spend the cutoff exists to govern, and
|
|
63
|
+
while `Agent.run` was the only recording point they were invisible to the
|
|
64
|
+
governor.
|
|
65
|
+
|
|
66
|
+
`structured()` is still not budgeted as a *loop*: it never asks `allow()`
|
|
67
|
+
between its own retries, which stay bounded by `max_retries`. What changed
|
|
68
|
+
is visibility, not control.
|
|
69
|
+
|
|
70
|
+
Attach with `Agent.use(...)`; `setup` resets per-run state, so one instance
|
|
71
|
+
may not measure two runs at once but can be reused across sequential ones.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
def __init__(self, ceiling: int | None = None, optional_cutoff: float | None = None):
|
|
75
|
+
self.ceiling = (
|
|
76
|
+
ceiling if ceiling is not None else profile_default("token_budget", "ceiling")
|
|
77
|
+
)
|
|
78
|
+
self.optional_cutoff = (
|
|
79
|
+
optional_cutoff
|
|
80
|
+
if optional_cutoff is not None
|
|
81
|
+
else profile_default_float("token_budget", "optional_cutoff")
|
|
82
|
+
)
|
|
83
|
+
self.spent = 0
|
|
84
|
+
self.exhausted = False
|
|
85
|
+
|
|
86
|
+
def setup(self, agent) -> None:
|
|
87
|
+
self.spent = 0
|
|
88
|
+
self.exhausted = False
|
|
89
|
+
# Unwrap first: setting up twice on the same agent (or reusing an agent
|
|
90
|
+
# across budgets) must leave exactly one wrapper, or every call would be
|
|
91
|
+
# booked once per layer.
|
|
92
|
+
inner = agent.client
|
|
93
|
+
if isinstance(inner, _BudgetedClient):
|
|
94
|
+
inner = inner.inner
|
|
95
|
+
agent.client = _BudgetedClient(inner, self)
|
|
96
|
+
agent.budget = self
|
|
97
|
+
|
|
98
|
+
def record(self, usage: Usage) -> None:
|
|
99
|
+
"""Book one chat response against the budget. Called by `_BudgetedClient`."""
|
|
100
|
+
self.spent += usage.total
|
|
101
|
+
|
|
102
|
+
def allow(self, priority: str) -> bool:
|
|
103
|
+
"""Ask before spending. `priority` is ``"required"`` or ``"optional"``."""
|
|
104
|
+
if priority not in ("required", "optional"):
|
|
105
|
+
raise ValueError(f"unknown budget priority {priority!r}")
|
|
106
|
+
if self.spent >= self.ceiling:
|
|
107
|
+
# Latched for the run: the hard ceiling denied work, which is what
|
|
108
|
+
# `budget-exhausted` reports. Cutoff-band denials are degradation,
|
|
109
|
+
# not exhaustion, and deliberately do not set this.
|
|
110
|
+
self.exhausted = True
|
|
111
|
+
return False
|
|
112
|
+
if priority == "optional":
|
|
113
|
+
return self.spent < self.ceiling * self.optional_cutoff
|
|
114
|
+
return True
|
bantamkit/client.py
ADDED
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
"""Core types + ModelClient protocol + OpenAI-compatible adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import time
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from typing import Protocol
|
|
9
|
+
|
|
10
|
+
import httpx
|
|
11
|
+
|
|
12
|
+
BODY_SNIPPET = 200
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class BantamError(Exception):
|
|
16
|
+
"""Base for all bantamkit errors."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class TransportError(BantamError):
|
|
20
|
+
"""HTTP-level failure after bounded retries."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class APIError(BantamError):
|
|
24
|
+
"""Non-retryable, non-success response from the model endpoint."""
|
|
25
|
+
|
|
26
|
+
def __init__(self, status_code: int, body: str):
|
|
27
|
+
self.status_code = status_code
|
|
28
|
+
self.body = body[:BODY_SNIPPET]
|
|
29
|
+
super().__init__(f"API error {status_code}: {self.body!r}")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class ToolCall:
|
|
34
|
+
id: str
|
|
35
|
+
name: str
|
|
36
|
+
arguments: dict
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class Message:
|
|
41
|
+
role: str # "system" | "user" | "assistant" | "tool"
|
|
42
|
+
content: str | None = None
|
|
43
|
+
tool_calls: list[ToolCall] = field(default_factory=list)
|
|
44
|
+
tool_call_id: str | None = None
|
|
45
|
+
|
|
46
|
+
def to_wire(self) -> dict:
|
|
47
|
+
wire: dict = {"role": self.role, "content": self.content}
|
|
48
|
+
if self.tool_calls:
|
|
49
|
+
wire["tool_calls"] = [
|
|
50
|
+
{
|
|
51
|
+
"id": tc.id,
|
|
52
|
+
"type": "function",
|
|
53
|
+
"function": {"name": tc.name, "arguments": json.dumps(tc.arguments)},
|
|
54
|
+
}
|
|
55
|
+
for tc in self.tool_calls
|
|
56
|
+
]
|
|
57
|
+
if self.tool_call_id is not None:
|
|
58
|
+
wire["tool_call_id"] = self.tool_call_id
|
|
59
|
+
return wire
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class Tool:
|
|
64
|
+
name: str
|
|
65
|
+
description: str
|
|
66
|
+
parameters: dict # JSON Schema
|
|
67
|
+
|
|
68
|
+
def to_wire(self) -> dict:
|
|
69
|
+
return {
|
|
70
|
+
"type": "function",
|
|
71
|
+
"function": {
|
|
72
|
+
"name": self.name,
|
|
73
|
+
"description": self.description,
|
|
74
|
+
"parameters": self.parameters,
|
|
75
|
+
},
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
# RB-P53's class, site S4. The four states `prompt_tokens` can be in, as data, because
|
|
80
|
+
# a check has to be able to name one. MEASURED is the only one that licenses arithmetic.
|
|
81
|
+
MEASURED = "MEASURED"
|
|
82
|
+
#: The response carried no usable `usage.prompt_tokens` at all — no `usage` object, an
|
|
83
|
+
#: empty or null one, or one that omits the field. The 0 in `Usage.prompt_tokens` is then
|
|
84
|
+
#: THIS PROGRAM'S DEFAULT and not anything the endpoint said. It needs its own state
|
|
85
|
+
#: because a fabricated zero compares unequal to any declared window, so without this
|
|
86
|
+
#: branch the emptiest possible response read `MEASURED` — RB-P51's defect inside the
|
|
87
|
+
#: code written to honour RB-P51.
|
|
88
|
+
UNREPORTED = "UNREPORTED"
|
|
89
|
+
#: `prompt_tokens` equals the window the caller declared. RB-P53 measured ollama's `/v1`
|
|
90
|
+
#: endpoint reporting the CONTEXT WINDOW in that field when it silently clamps a prompt,
|
|
91
|
+
#: so at the window the number is the window and not a measurement of anything.
|
|
92
|
+
VOID = "VOID"
|
|
93
|
+
#: No window was declared, so nothing could be compared. RB-P51: unmeasured is a verdict,
|
|
94
|
+
#: and it is a different verdict from MEASURED.
|
|
95
|
+
UNCHECKED = "UNCHECKED"
|
|
96
|
+
|
|
97
|
+
# Worst first. A sum containing a non-measurement is a non-measurement, and a sum
|
|
98
|
+
# containing an unchecked component is not a measured total either. UNREPORTED outranks
|
|
99
|
+
# VOID: a VOID call at least returned a number the endpoint chose, while an UNREPORTED one
|
|
100
|
+
# contributed a zero this module invented, so a total containing one is short by an
|
|
101
|
+
# unknown amount rather than merely untrustworthy.
|
|
102
|
+
_VERDICT_SEVERITY = (UNREPORTED, VOID, UNCHECKED, MEASURED)
|
|
103
|
+
|
|
104
|
+
#: Stop reasons that mean the generation was CUT rather than finished. A completion that
|
|
105
|
+
#: hit the cap is not a shorter answer, it is an unfinished one, and the difference is
|
|
106
|
+
#: invisible unless something carries it.
|
|
107
|
+
TRUNCATING_FINISH_REASONS = frozenset({"length"})
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _worse_verdict(a: str | None, b: str | None) -> str | None:
|
|
111
|
+
"""The worse of two `prompt_tokens` verdicts; `None` contributes nothing.
|
|
112
|
+
|
|
113
|
+
`None` is the zero value's state and it is NOT the same as `UNCHECKED`: `Usage()` is
|
|
114
|
+
the accumulator seed in the agent loop and in `evalrun`, it carries no tokens and no
|
|
115
|
+
claim, and it must not poison a fold. `UNCHECKED` means a real response arrived and
|
|
116
|
+
nobody could check it, which is a claim and does survive the fold.
|
|
117
|
+
"""
|
|
118
|
+
for verdict in _VERDICT_SEVERITY:
|
|
119
|
+
if a == verdict or b == verdict:
|
|
120
|
+
return verdict
|
|
121
|
+
return a if a is not None else b
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _worse_finish_reason(a: str | None, b: str | None) -> str | None:
|
|
125
|
+
"""An aggregate has no single finish reason — but a truncation must not be summed away.
|
|
126
|
+
|
|
127
|
+
S5's shape, one repository over: a length stop absorbed into an outcome instead of
|
|
128
|
+
being reported as an instrument verdict. So a cut anywhere in the fold survives it,
|
|
129
|
+
and two different non-cut reasons collapse to `None` rather than to whichever came
|
|
130
|
+
first.
|
|
131
|
+
"""
|
|
132
|
+
for reason in (a, b):
|
|
133
|
+
if reason in TRUNCATING_FINISH_REASONS:
|
|
134
|
+
return reason
|
|
135
|
+
if a == b:
|
|
136
|
+
return a
|
|
137
|
+
return a if b is None else (b if a is None else None)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@dataclass
|
|
141
|
+
class Usage:
|
|
142
|
+
prompt_tokens: int = 0
|
|
143
|
+
completion_tokens: int = 0
|
|
144
|
+
#: What the endpoint said stopped the generation, verbatim, or `None` when it said
|
|
145
|
+
#: nothing. Read back rather than assumed: before this field existed a length-stopped
|
|
146
|
+
#: completion and a finished one were the same object.
|
|
147
|
+
finish_reason: str | None = None
|
|
148
|
+
#: `MEASURED` / `VOID` / `UNCHECKED` / `UNREPORTED`, or `None` for a `Usage` no
|
|
149
|
+
#: response produced.
|
|
150
|
+
prompt_tokens_verdict: str | None = None
|
|
151
|
+
|
|
152
|
+
def __add__(self, other: Usage) -> Usage:
|
|
153
|
+
return Usage(
|
|
154
|
+
self.prompt_tokens + other.prompt_tokens,
|
|
155
|
+
self.completion_tokens + other.completion_tokens,
|
|
156
|
+
_worse_finish_reason(self.finish_reason, other.finish_reason),
|
|
157
|
+
_worse_verdict(self.prompt_tokens_verdict, other.prompt_tokens_verdict),
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
@property
|
|
161
|
+
def total(self) -> int:
|
|
162
|
+
return self.prompt_tokens + self.completion_tokens
|
|
163
|
+
|
|
164
|
+
@property
|
|
165
|
+
def truncated(self) -> bool:
|
|
166
|
+
"""The generation was cut by a cap rather than finished."""
|
|
167
|
+
return self.finish_reason in TRUNCATING_FINISH_REASONS
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
@dataclass
|
|
171
|
+
class Response:
|
|
172
|
+
message: Message
|
|
173
|
+
usage: Usage
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class ModelClient(Protocol):
|
|
177
|
+
def chat(self, messages: list[Message], tools: list[Tool] | None = None) -> Response: ...
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class OpenAICompatible:
|
|
181
|
+
"""One adapter covers Ollama / vLLM / LM Studio / llama.cpp server / OpenRouter."""
|
|
182
|
+
|
|
183
|
+
def __init__(
|
|
184
|
+
self,
|
|
185
|
+
base_url: str,
|
|
186
|
+
model: str,
|
|
187
|
+
api_key: str = "none",
|
|
188
|
+
timeout: float = 60.0,
|
|
189
|
+
max_retries: int = 3,
|
|
190
|
+
transport: httpx.BaseTransport | None = None,
|
|
191
|
+
seed: int | None = None,
|
|
192
|
+
context_window: int | None = None,
|
|
193
|
+
):
|
|
194
|
+
self.base_url = base_url.rstrip("/")
|
|
195
|
+
self.model = model
|
|
196
|
+
self.api_key = api_key
|
|
197
|
+
self.max_retries = max_retries
|
|
198
|
+
# The context window this endpoint is configured with, DECLARED by the caller.
|
|
199
|
+
# It is never sent: this adapter sets no context length and this field changes no
|
|
200
|
+
# request. It exists so that `usage.prompt_tokens == context_window` can be given
|
|
201
|
+
# a verdict, which is RB-P53's shape — a silent clamp reports the window in the
|
|
202
|
+
# field a reader takes for a prompt size. Left unset, the comparison cannot be
|
|
203
|
+
# made and every response says `UNCHECKED` rather than `MEASURED`.
|
|
204
|
+
#
|
|
205
|
+
# THIS IS A DETECTOR AND NOT A PREVENTER, and calling it protection would be the
|
|
206
|
+
# failure this job is about. It cannot stop a clamp, it cannot detect one below
|
|
207
|
+
# the window, and it cannot tell a clamped prompt from a genuine prompt that
|
|
208
|
+
# happens to be exactly `context_window` tokens long.
|
|
209
|
+
self.context_window = context_window
|
|
210
|
+
# Sampling seed, sent only when set. OpenAI chat-completions and Ollama's
|
|
211
|
+
# OpenAI-compat endpoint both accept it; servers that ignore it degrade to
|
|
212
|
+
# unseeded sampling. Writable per run — the eval harness pins it per task.
|
|
213
|
+
self.seed = seed
|
|
214
|
+
# Capability memo for the constrained-decoding tier, flipped by the first
|
|
215
|
+
# HTTP 400 on a request that carried `response_format`. Its presence is also
|
|
216
|
+
# the duck-typed signal callers test before sending the kwarg at all.
|
|
217
|
+
self._response_format_unsupported = False
|
|
218
|
+
self._http = httpx.Client(timeout=timeout, transport=transport)
|
|
219
|
+
|
|
220
|
+
def close(self) -> None:
|
|
221
|
+
"""Release the underlying HTTP connection pool. Idempotent."""
|
|
222
|
+
self._http.close()
|
|
223
|
+
|
|
224
|
+
def __enter__(self) -> OpenAICompatible:
|
|
225
|
+
return self
|
|
226
|
+
|
|
227
|
+
def __exit__(self, *exc_info: object) -> None:
|
|
228
|
+
self.close()
|
|
229
|
+
|
|
230
|
+
def _post(self, payload: dict) -> httpx.Response:
|
|
231
|
+
return self._http.post(
|
|
232
|
+
f"{self.base_url}/chat/completions",
|
|
233
|
+
headers={"Authorization": f"Bearer {self.api_key}"},
|
|
234
|
+
json=payload,
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
def chat(
|
|
238
|
+
self,
|
|
239
|
+
messages: list[Message],
|
|
240
|
+
tools: list[Tool] | None = None,
|
|
241
|
+
response_format: dict | None = None,
|
|
242
|
+
) -> Response:
|
|
243
|
+
payload: dict = {"model": self.model, "messages": [m.to_wire() for m in messages]}
|
|
244
|
+
if tools:
|
|
245
|
+
payload["tools"] = [t.to_wire() for t in tools]
|
|
246
|
+
if self.seed is not None:
|
|
247
|
+
payload["seed"] = self.seed
|
|
248
|
+
if response_format is not None:
|
|
249
|
+
payload["response_format"] = response_format
|
|
250
|
+
last_err: Exception | None = None
|
|
251
|
+
for attempt in range(self.max_retries):
|
|
252
|
+
try:
|
|
253
|
+
r = self._post(payload)
|
|
254
|
+
if r.status_code == 400 and "response_format" in payload:
|
|
255
|
+
# That single 400 IS the capability detection — no probe request,
|
|
256
|
+
# no server allowlist. Drop the field, remember it for every later
|
|
257
|
+
# call on this client, and let this call still succeed. A 400 that
|
|
258
|
+
# survives the drop was never about response_format and raises below.
|
|
259
|
+
self._response_format_unsupported = True
|
|
260
|
+
del payload["response_format"]
|
|
261
|
+
r = self._post(payload)
|
|
262
|
+
if r.status_code == 429 or r.status_code >= 500:
|
|
263
|
+
raise TransportError(f"server error {r.status_code}: {r.text[:BODY_SNIPPET]}")
|
|
264
|
+
if not r.is_success:
|
|
265
|
+
# Non-retryable. Covers 1xx/3xx too: redirects are not followed, so
|
|
266
|
+
# anything but 2xx would otherwise reach _parse as non-JSON.
|
|
267
|
+
raise APIError(r.status_code, r.text)
|
|
268
|
+
return self._parse(r.json())
|
|
269
|
+
except (httpx.TransportError, TransportError) as e:
|
|
270
|
+
last_err = e
|
|
271
|
+
if attempt < self.max_retries - 1:
|
|
272
|
+
# No backoff after the final attempt — nothing follows it but the raise.
|
|
273
|
+
time.sleep(0.5 * (2**attempt))
|
|
274
|
+
raise TransportError(f"chat failed after {self.max_retries} attempts: {last_err}")
|
|
275
|
+
|
|
276
|
+
def _parse(self, data: dict) -> Response:
|
|
277
|
+
try:
|
|
278
|
+
# `finish_reason` lives on the CHOICE, not on the message inside it, which is
|
|
279
|
+
# why reading it back needs the enclosing object and not just `["message"]`.
|
|
280
|
+
raw_choice = data["choices"][0]
|
|
281
|
+
choice = raw_choice["message"]
|
|
282
|
+
except (KeyError, IndexError) as e:
|
|
283
|
+
raise BantamError(f"malformed chat response: {data!r:.200}") from e
|
|
284
|
+
|
|
285
|
+
tool_calls = []
|
|
286
|
+
for tc in choice.get("tool_calls") or []:
|
|
287
|
+
try:
|
|
288
|
+
name = tc["function"]["name"]
|
|
289
|
+
raw_args = tc["function"]["arguments"]
|
|
290
|
+
# Support dict-form arguments (already a dict) or JSON string
|
|
291
|
+
if isinstance(raw_args, dict):
|
|
292
|
+
arguments = raw_args
|
|
293
|
+
else:
|
|
294
|
+
arguments = json.loads(raw_args)
|
|
295
|
+
tool_calls.append(ToolCall(id=tc["id"], name=name, arguments=arguments))
|
|
296
|
+
except (json.JSONDecodeError, ValueError) as e:
|
|
297
|
+
raw = tc["function"]["arguments"]
|
|
298
|
+
name = tc["function"]["name"]
|
|
299
|
+
raise BantamError(
|
|
300
|
+
f"tool call '{name}' has malformed JSON arguments: {raw!r:.200}"
|
|
301
|
+
) from e
|
|
302
|
+
|
|
303
|
+
raw_usage = data.get("usage")
|
|
304
|
+
usage = raw_usage if isinstance(raw_usage, dict) else {}
|
|
305
|
+
reported = usage.get("prompt_tokens")
|
|
306
|
+
# `bool` is an `int` and `True` is not a token count.
|
|
307
|
+
usable = isinstance(reported, int) and not isinstance(reported, bool)
|
|
308
|
+
prompt_tokens = reported if usable else 0
|
|
309
|
+
# The classifier. Four states, one branch each, and no branch is a message: a
|
|
310
|
+
# mutation that rewrites the docstrings above cannot move any of them (N-12).
|
|
311
|
+
# UNREPORTED is tested FIRST because it is a fact about what arrived, and the
|
|
312
|
+
# window question does not arise for a number that was never sent.
|
|
313
|
+
if not usable:
|
|
314
|
+
verdict = UNREPORTED
|
|
315
|
+
elif self.context_window is None:
|
|
316
|
+
verdict = UNCHECKED
|
|
317
|
+
elif prompt_tokens == self.context_window:
|
|
318
|
+
verdict = VOID
|
|
319
|
+
else:
|
|
320
|
+
verdict = MEASURED
|
|
321
|
+
return Response(
|
|
322
|
+
message=Message(role="assistant", content=choice.get("content"), tool_calls=tool_calls),
|
|
323
|
+
usage=Usage(
|
|
324
|
+
prompt_tokens,
|
|
325
|
+
usage.get("completion_tokens", 0),
|
|
326
|
+
raw_choice.get("finish_reason"),
|
|
327
|
+
verdict,
|
|
328
|
+
),
|
|
329
|
+
)
|