bantamkit 0.27.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. bantamkit/__init__.py +32 -0
  2. bantamkit/agent.py +458 -0
  3. bantamkit/assets/contracts/default.yaml +90 -0
  4. bantamkit/assets/evals/devteam/manifest.yaml +351 -0
  5. bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
  6. bantamkit/assets/evals/devteam/repo/README.md +12 -0
  7. bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
  8. bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
  9. bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
  10. bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
  11. bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
  12. bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
  13. bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
  14. bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
  15. bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
  16. bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
  17. bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
  18. bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
  19. bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
  20. bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
  21. bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
  22. bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
  23. bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
  24. bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
  25. bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
  26. bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
  27. bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
  28. bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
  29. bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
  30. bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
  31. bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
  32. bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
  33. bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
  34. bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
  35. bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
  36. bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
  37. bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
  38. bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
  39. bantamkit/assets/evals/fixtures/.gitkeep +0 -0
  40. bantamkit/assets/evals/fixtures/catalog.json +6 -0
  41. bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
  42. bantamkit/assets/evals/tasks/.gitkeep +0 -0
  43. bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
  44. bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
  45. bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
  46. bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
  47. bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
  48. bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
  49. bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
  50. bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
  51. bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
  52. bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
  53. bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
  54. bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
  55. bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
  56. bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
  57. bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
  58. bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
  59. bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
  60. bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
  61. bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
  62. bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
  63. bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
  64. bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
  65. bantamkit/assets/profiles/default.yaml +31 -0
  66. bantamkit/assets/profiles/patient.yaml +31 -0
  67. bantamkit/assets/rubrics/.gitkeep +0 -0
  68. bantamkit/assets/rubrics/code-quality.yaml +20 -0
  69. bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
  70. bantamkit/assets/rubrics/task-completion.yaml +28 -0
  71. bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
  72. bantamkit/assets/skills/.gitkeep +0 -0
  73. bantamkit/assets/skills/file-graph.md +7 -0
  74. bantamkit/assets/skills/memory.md +35 -0
  75. bantamkit/assets/tools/.gitkeep +0 -0
  76. bantamkit/assets/tools/bantamkit_read.json +48 -0
  77. bantamkit/assets/tools/bantamkit_status.json +25 -0
  78. bantamkit/assets/tools/build_identity.json +17 -0
  79. bantamkit/assets/tools/document_list.json +12 -0
  80. bantamkit/assets/tools/document_read.json +31 -0
  81. bantamkit/assets/tools/file_graph.json +12 -0
  82. bantamkit/assets/tools/memory_compact.json +31 -0
  83. bantamkit/assets/tools/memory_recall.json +38 -0
  84. bantamkit/assets/tools/memory_save.json +61 -0
  85. bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
  86. bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
  87. bantamkit/assets/tools/shiftwork_status.json +25 -0
  88. bantamkit/assets/tools/skill_audit.json +70 -0
  89. bantamkit/assets/tools/validate_json.json +31 -0
  90. bantamkit/assets.py +67 -0
  91. bantamkit/budget.py +114 -0
  92. bantamkit/client.py +329 -0
  93. bantamkit/contract.py +522 -0
  94. bantamkit/criticreplay.py +3241 -0
  95. bantamkit/critique.py +301 -0
  96. bantamkit/docread.py +1744 -0
  97. bantamkit/evalrun.py +2003 -0
  98. bantamkit/eventlog.py +282 -0
  99. bantamkit/filegraph.py +218 -0
  100. bantamkit/loopguard.py +101 -0
  101. bantamkit/mcpreport.py +763 -0
  102. bantamkit/mcpserver.py +1334 -0
  103. bantamkit/memory/__init__.py +28 -0
  104. bantamkit/memory/__main__.py +291 -0
  105. bantamkit/memory/component.py +569 -0
  106. bantamkit/memory/divergence.py +744 -0
  107. bantamkit/memory/layers.py +257 -0
  108. bantamkit/memory/store.py +940 -0
  109. bantamkit/pdfread.py +1402 -0
  110. bantamkit/profile.py +46 -0
  111. bantamkit/shiftwork.py +212 -0
  112. bantamkit/skillaudit.py +853 -0
  113. bantamkit/statusline.py +313 -0
  114. bantamkit/structured.py +125 -0
  115. bantamkit/textutil.py +30 -0
  116. bantamkit-0.27.0.dist-info/METADATA +207 -0
  117. bantamkit-0.27.0.dist-info/RECORD +119 -0
  118. bantamkit-0.27.0.dist-info/WHEEL +4 -0
  119. bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,70 @@
1
+ {
2
+ "name": "skill_audit",
3
+ "description": "Price the skill catalogue every session pays for, and name the collisions in it. A skill's `description:` frontmatter is loaded into the agent's context in EVERY session; its body is read only when the skill is invoked, so the descriptions are the standing bill. Walks `root` for `<marketplace>/<plugin>/<version>/skills/<name>/SKILL.md` and answers one JSON document: `roots` (what was scanned), `skills` (how many were counted), `catalogue_bytes` (UTF-8 bytes of their `description:` values), `findings` and `omissions`. Counted skills plus omissions account for every SKILL.md found — an omission is `{subject, count, size, what}` with subject one of `plugin-not-enabled`, `duplicate-skill`, `stale-version`, `unreadable-file`, `unparsed-frontmatter`, and `what` naming the paths. Pass `enabled` with the plugin ids the host has switched on (`<plugin>@<marketplace>`, as settings.json spells them): the plugin cache also holds disabled plugins and stale version directories, and counting those inflated a real audit to 41 skills / 19,065 B against a true 34 / 14,515. Identity comes from the path: `<marketplace>/<plugin>/<version>/skills/<name>/SKILL.md` relative to `root`; any other shape is a skill outside a plugin, identified by its directory name alone and never excluded by `enabled`. Exactly ONE version directory is resolved per (marketplace, plugin) BEFORE anything is counted, and its skills are that plugin's skills: a name absent from the resolved version is absent, never merged in from another version directory. The winner is the version directory name that sorts LAST in byte order — a byte order and deliberately not a semver comparison, so two runtimes cannot disagree about which directory was read, and so that names which are not versions at all still have a total order (a real cache spells nine of one plugin's directories as content hashes plus the literal `unknown`). What byte order gets wrong is stated rather than hidden: `10.0.0` loses to `9.0.0`, and among non-version names the winner is deterministic but arbitrary; both losers are named in an omission record. Pass `versions` with the version directory the host actually serves, per `<plugin>@<marketplace>` — the `installPath` recorded in the same `installed_plugins.json` the caller reads `enabled` out of — and that plugin is resolved to the named directory whatever byte order would have said; a plugin absent from `versions` falls back to byte order. It is host truth the caller supplies, exactly as `enabled` is, and it is not refused when it names a plugin or a directory this `root` does not hold: a plugin with no version directory here is not resolved at all, and a directory that is not here means every directory that IS here loses and is named in a `stale-version` record. Measured 2026-09-05 on a real cache, byte order picks `unknown` for a plugin whose host serves `1dd995193ba2`, so the installed copy is discarded as a duplicate. The candidates are the version DIRECTORIES on disk — every path spelled `<marketplace>/<plugin>/<version>/skills` — and not the skills that survived reading: a version directory holding no `SKILL.md`, or only files that will not decode or will not parse, still exists and the host still serves it, so an older directory must not win by default. Every SKILL.md under a version directory that did not win is omitted, under one of two subjects: `duplicate-skill` when the resolved version has a skill of the same name standing in for it, `stale-version` when it does not — a skill the resolved version DROPPED, which is on disk, in nobody's bill, and would otherwise be resurrected by the audit. Deduping by the (marketplace, plugin, name) triple instead merges versions rather than choosing between them, and measured 2026-09-05 over a real cache holding a plugin's `0.4.0` (14 skills) and `0.5.0` (5, nine having moved to another plugin) it answered 31 skills / 9,280 bytes where the host serves 22 / 3,396. Inside the resolved version `duplicate-skill` still covers two files claiming the same (marketplace, plugin, name), which is reachable only for skills outside a plugin; the last in scan order wins. Every omission carries the bytes it would have added. A skill is identified as `<plugin>:<name>`, the spelling the host uses and the key `usage` is looked up by, or as `<name>` when it has no plugin. A `description:` folded over several lines is joined with single spaces before it is counted or scanned, and a value written as a whole-value quoted YAML scalar is UNWRAPPED after that fold and before either — the host's parser strips those quotes before a session sees the description, so they are not `catalogue_bytes`, and only quotes INSIDE the value delimit a trigger phrase. The unwrap applies to every frontmatter key, not just `description:`, so `name: \"s\"` is the name `s` and `router: \"true\"` is a router. A value is a whole-value scalar only when it begins with `\"` or `'` AND that opening quote's own closing quote is the value's LAST character, judged by YAML's two escape rules and no others: inside a `\"` scalar a backslash escapes the character after it, so `\\\"` does not close it and the sequences `\\\"` and `\\\\` resolve to `\"` and `\\` while every other `\\x` is left exactly as written; inside a `'` scalar a doubled `''` is one literal apostrophe, so it does not close it and resolves to `'`. Every other value is left as written and scanned literally: `\"a\" and \"b\"` begins and ends with `\"` and is NOT one scalar, because its opening quote closes at index 2, and stripping it would weld the value into one junk phrase and destroy both real ones. A value that OPENS with a quote and never closes one — including one whose last quote is escaped — is a LITERAL: it is not unwrapped, its quote character is counted, its unpaired quote opens no phrase, and it is not a `frontmatter-malformed` finding, which names failures of the BLOCK and not of one value. Measured 2026-09-05 over a real plugin cache: 17 of 31 enabled skills write `description: \"...\"`, so a reader that takes those quotes literally over-prices the majority of a catalogue and buries any phrase quoted inside them. Five findings, each `{kind, severity, skills, detail}`, severity fixed per kind: `shared-trigger-phrase` (high) — two or more skills quote the same literal phrase, so which one fires is a coin flip; `catalogue-over-budget` (high) — `catalogue_bytes` exceeds `budget`; `never-invoked` (low) — a counted skill has zero calls in `usage`; `frontmatter-malformed` (medium) — no frontmatter block, an unterminated one, or no `description:` key, where a block that cannot be parsed at all is also omitted as `unparsed-frontmatter` and adds nothing to `skills` while a parsed block with no description is a skill costing zero bytes; `name-mismatch` (medium) — the frontmatter `name:` is not the skill's directory name. Collisions are EXACT matches on literal quoted phrases and never a similarity score: Jaccard over description tokens was measured across 820 real pairs and refuted, topping out at 0.239 and ranking the two real collisions 118th and 791st of 820. Phrases are the text of the UNWRAPPED value between `\"` pairs, or between `'` pairs where the quote is not flanked by letters — `'don't'` is a contraction, not a phrase — and a phrase with no letter or digit in it is ignored. A router, a skill whose frontmatter carries `router: true`, quotes its siblings' phrases by design and is left out of the phrase index entirely. Four argument failures are refused with their own sentences and never as a partial answer: an unknown `check`, a negative `budget`, a `root` that is missing or is a file, and an EMPTY `root` — which passes JSON-schema `string` and would otherwise resolve to the server's own working directory. Nothing about the CONTENT of the tree refuses. Deterministic: the tool reads nothing but `root`. The three facts a directory of files cannot answer arrive from the caller — call counts in `usage` (`node tools/ledger/tool-usage.mjs --group skill --json`), switched-on plugins in `enabled`, served version directories in `versions` — never by reading the host's transcripts, settings or plugin registry.",
4
+ "surfaces": [
5
+ "mcp"
6
+ ],
7
+ "parameters": {
8
+ "type": "object",
9
+ "required": [
10
+ "root"
11
+ ],
12
+ "properties": {
13
+ "root": {
14
+ "type": "string",
15
+ "description": "Directory to scan for SKILL.md files, absolute or relative to the server's working directory. It must name a directory: the empty string is refused rather than resolved, because resolving it would audit whatever directory the server happens to be standing in"
16
+ },
17
+ "enabled": {
18
+ "type": "array",
19
+ "items": {
20
+ "type": "string"
21
+ },
22
+ "description": "Plugin ids that are switched on, `<plugin>@<marketplace>`; when given only skills under those plugins are counted, when omitted every skill under `root` counts"
23
+ },
24
+ "usage": {
25
+ "type": "object",
26
+ "additionalProperties": {
27
+ "type": "integer",
28
+ "minimum": 0
29
+ },
30
+ "description": "Skill id (`<plugin>:<name>`) to call count, the caller's own measurement; a counted skill absent from a supplied map is treated as zero"
31
+ },
32
+ "check": {
33
+ "type": "string",
34
+ "enum": [
35
+ "all",
36
+ "phrase",
37
+ "budget",
38
+ "frontmatter"
39
+ ],
40
+ "description": "Which family of findings to report; default `all`. `skills`, `catalogue_bytes` and `omissions` are reported whatever this says"
41
+ },
42
+ "budget": {
43
+ "type": "integer",
44
+ "minimum": 0,
45
+ "maximum": 9007199254740991,
46
+ "description": "Catalogue byte budget; `catalogue-over-budget` is only reported when this is given"
47
+ },
48
+ "versions": {
49
+ "type": "object",
50
+ "additionalProperties": {
51
+ "type": "string"
52
+ },
53
+ "description": "Plugin id (`<plugin>@<marketplace>`) to the version directory name the host actually serves; when a plugin is named here that directory is resolved whatever byte order would have said, and a plugin absent from it falls back to byte order"
54
+ }
55
+ }
56
+ },
57
+ "output_schema": {
58
+ "properties": {
59
+ "result": {
60
+ "title": "Result",
61
+ "type": "string"
62
+ }
63
+ },
64
+ "required": [
65
+ "result"
66
+ ],
67
+ "type": "object",
68
+ "title": "skill_auditOutput"
69
+ }
70
+ }
@@ -0,0 +1,31 @@
1
+ {
2
+ "name": "validate_json",
3
+ "description": "Validate candidate output text against a JSON Schema. Returns {valid, feedback}; when invalid, feed the feedback back to the model and retry.",
4
+ "surfaces": [
5
+ "mcp"
6
+ ],
7
+ "parameters": {
8
+ "properties": {
9
+ "output": {
10
+ "title": "Output",
11
+ "type": "string"
12
+ },
13
+ "schema": {
14
+ "additionalProperties": true,
15
+ "title": "Schema",
16
+ "type": "object"
17
+ }
18
+ },
19
+ "required": [
20
+ "output",
21
+ "schema"
22
+ ],
23
+ "type": "object",
24
+ "title": "validate_jsonArguments"
25
+ },
26
+ "output_schema": {
27
+ "type": "object",
28
+ "additionalProperties": true,
29
+ "title": "validate_jsonDictOutput"
30
+ }
31
+ }
bantamkit/assets.py ADDED
@@ -0,0 +1,67 @@
1
+ """Locate and load the language-agnostic asset pack."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ from pathlib import Path
8
+
9
+ from bantamkit.client import BantamError, Tool
10
+
11
+
12
+ class AssetNotFound(BantamError):
13
+ pass
14
+
15
+
16
+ def assets_root() -> Path:
17
+ env = os.environ.get("BANTAMKIT_ASSETS")
18
+ if env:
19
+ return Path(env)
20
+ packaged = Path(__file__).parent / "assets"
21
+ if packaged.exists():
22
+ return packaged
23
+ repo = Path(__file__).resolve().parents[3] / "assets"
24
+ if repo.exists():
25
+ return repo
26
+ raise AssetNotFound("no assets directory found; set BANTAMKIT_ASSETS")
27
+
28
+
29
+ def load_tool_asset(name: str) -> dict:
30
+ """Return one tool asset VERBATIM — every key, including the ones no surface uses.
31
+
32
+ `load_tool` below narrows the same file to the three fields a model is shown. This
33
+ returns the whole manifest entry, because a server has to read `surfaces` (may I
34
+ register this at all?) and `output_schema` (what do I advertise as the return shape?)
35
+ and neither belongs on the agent-facing `Tool`.
36
+ """
37
+ path = assets_root() / "tools" / f"{name}.json"
38
+ if not path.exists():
39
+ raise AssetNotFound(f"tool asset not found: {path}")
40
+ return json.loads(path.read_text(encoding="utf-8"))
41
+
42
+
43
+ def load_tool(name: str) -> Tool:
44
+ """The agent-facing view of a tool asset: name, description, input schema.
45
+
46
+ The three keys are named EXPLICITLY rather than splatted. `assets/tools/` is shared by
47
+ two tool surfaces and carries fields that only the MCP one consumes; a loader that
48
+ passed the dict through would add them to every tool definition an eval-run model is
49
+ shown, changing the agent surface every time the manifest grows.
50
+ """
51
+ data = load_tool_asset(name)
52
+ return Tool(name=data["name"], description=data["description"], parameters=data["parameters"])
53
+
54
+
55
+ def load_schema(name: str) -> dict:
56
+ """Return a JSON Schema asset (parsed) — e.g. the shift-work checkpoint contract."""
57
+ path = assets_root() / "schemas" / f"{name}.json"
58
+ if not path.exists():
59
+ raise AssetNotFound(f"schema asset not found: {path}")
60
+ return json.loads(path.read_text(encoding="utf-8"))
61
+
62
+
63
+ def load_skill(name: str) -> str:
64
+ path = assets_root() / "skills" / f"{name}.md"
65
+ if not path.exists():
66
+ raise AssetNotFound(f"skill asset not found: {path}")
67
+ return path.read_text(encoding="utf-8")
bantamkit/budget.py ADDED
@@ -0,0 +1,114 @@
1
+ """Layer 1 — a global spend governor for one agent run: `TokenBudget`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from bantamkit.client import Usage
6
+ from bantamkit.profile import default as profile_default
7
+ from bantamkit.profile import default_float as profile_default_float
8
+
9
+ __all__ = ["TokenBudget"]
10
+
11
+
12
+ class _BudgetedClient:
13
+ """The recording point: every chat through the agent's client books its usage.
14
+
15
+ A proxy rather than a subclass, because the thing being wrapped is whatever
16
+ the caller handed the agent (a raw adapter, the eval harness's tracking
17
+ client, a fake). `__getattr__` passes everything else through, so the
18
+ duck-typed contracts callers test for — `seed`, `model`,
19
+ `_response_format_unsupported` — read the inner client unchanged.
20
+ """
21
+
22
+ def __init__(self, inner, budget: TokenBudget):
23
+ self.inner = inner
24
+ self.budget = budget
25
+
26
+ def chat(self, *args, **kwargs):
27
+ resp = self.inner.chat(*args, **kwargs)
28
+ self.budget.record(resp.usage)
29
+ return resp
30
+
31
+ def __getattr__(self, name: str):
32
+ return getattr(self.inner, name)
33
+
34
+
35
+ class TokenBudget:
36
+ """A governor, not an odometer: it decides what a run may still spend.
37
+
38
+ Every gate already caps its own retries; composition multiplies those caps
39
+ and nothing bounded the total (3b `full`: 214,698 tokens for 15/66). This
40
+ component bounds the run and degrades it in two steps instead of one:
41
+
42
+ - past ``ceiling * optional_cutoff`` spent, ``allow("optional")`` is denied,
43
+ so optional work is skipped and a `full` run degrades toward `lean`/`bare`;
44
+ - past ``ceiling``, everything is denied and `Agent.run` stops at the top of
45
+ the next turn, returning the last assistant content as a best-effort result.
46
+
47
+ The gap between the cutoff and the ceiling **is** the reserve: the final
48
+ answer emission always fits.
49
+
50
+ Priority classification (v1, deliberately coarse — no cost estimation, which
51
+ would be guesswork until a measured need exists):
52
+
53
+ - ``"optional"`` — critique rounds, blind and grounded. Denied means "accept
54
+ the answer", never an exception.
55
+ - ``"required"`` — the agent's own turns and the schema / json-answer retries.
56
+ Cheap, high-value, and denied only past the hard ceiling.
57
+
58
+ Spend is booked at the client boundary: `setup` wraps `agent.client` in a
59
+ `_BudgetedClient`, so every call made through the agent's client moves
60
+ `spent` — the agent's own turns *and* the critic calls the gates issue
61
+ through `structured()` on that same client. That is the point: the critique
62
+ rounds are the dominant optional spend the cutoff exists to govern, and
63
+ while `Agent.run` was the only recording point they were invisible to the
64
+ governor.
65
+
66
+ `structured()` is still not budgeted as a *loop*: it never asks `allow()`
67
+ between its own retries, which stay bounded by `max_retries`. What changed
68
+ is visibility, not control.
69
+
70
+ Attach with `Agent.use(...)`; `setup` resets per-run state, so one instance
71
+ may not measure two runs at once but can be reused across sequential ones.
72
+ """
73
+
74
+ def __init__(self, ceiling: int | None = None, optional_cutoff: float | None = None):
75
+ self.ceiling = (
76
+ ceiling if ceiling is not None else profile_default("token_budget", "ceiling")
77
+ )
78
+ self.optional_cutoff = (
79
+ optional_cutoff
80
+ if optional_cutoff is not None
81
+ else profile_default_float("token_budget", "optional_cutoff")
82
+ )
83
+ self.spent = 0
84
+ self.exhausted = False
85
+
86
+ def setup(self, agent) -> None:
87
+ self.spent = 0
88
+ self.exhausted = False
89
+ # Unwrap first: setting up twice on the same agent (or reusing an agent
90
+ # across budgets) must leave exactly one wrapper, or every call would be
91
+ # booked once per layer.
92
+ inner = agent.client
93
+ if isinstance(inner, _BudgetedClient):
94
+ inner = inner.inner
95
+ agent.client = _BudgetedClient(inner, self)
96
+ agent.budget = self
97
+
98
+ def record(self, usage: Usage) -> None:
99
+ """Book one chat response against the budget. Called by `_BudgetedClient`."""
100
+ self.spent += usage.total
101
+
102
+ def allow(self, priority: str) -> bool:
103
+ """Ask before spending. `priority` is ``"required"`` or ``"optional"``."""
104
+ if priority not in ("required", "optional"):
105
+ raise ValueError(f"unknown budget priority {priority!r}")
106
+ if self.spent >= self.ceiling:
107
+ # Latched for the run: the hard ceiling denied work, which is what
108
+ # `budget-exhausted` reports. Cutoff-band denials are degradation,
109
+ # not exhaustion, and deliberately do not set this.
110
+ self.exhausted = True
111
+ return False
112
+ if priority == "optional":
113
+ return self.spent < self.ceiling * self.optional_cutoff
114
+ return True
bantamkit/client.py ADDED
@@ -0,0 +1,329 @@
1
+ """Core types + ModelClient protocol + OpenAI-compatible adapter."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import time
7
+ from dataclasses import dataclass, field
8
+ from typing import Protocol
9
+
10
+ import httpx
11
+
12
+ BODY_SNIPPET = 200
13
+
14
+
15
+ class BantamError(Exception):
16
+ """Base for all bantamkit errors."""
17
+
18
+
19
+ class TransportError(BantamError):
20
+ """HTTP-level failure after bounded retries."""
21
+
22
+
23
+ class APIError(BantamError):
24
+ """Non-retryable, non-success response from the model endpoint."""
25
+
26
+ def __init__(self, status_code: int, body: str):
27
+ self.status_code = status_code
28
+ self.body = body[:BODY_SNIPPET]
29
+ super().__init__(f"API error {status_code}: {self.body!r}")
30
+
31
+
32
+ @dataclass
33
+ class ToolCall:
34
+ id: str
35
+ name: str
36
+ arguments: dict
37
+
38
+
39
+ @dataclass
40
+ class Message:
41
+ role: str # "system" | "user" | "assistant" | "tool"
42
+ content: str | None = None
43
+ tool_calls: list[ToolCall] = field(default_factory=list)
44
+ tool_call_id: str | None = None
45
+
46
+ def to_wire(self) -> dict:
47
+ wire: dict = {"role": self.role, "content": self.content}
48
+ if self.tool_calls:
49
+ wire["tool_calls"] = [
50
+ {
51
+ "id": tc.id,
52
+ "type": "function",
53
+ "function": {"name": tc.name, "arguments": json.dumps(tc.arguments)},
54
+ }
55
+ for tc in self.tool_calls
56
+ ]
57
+ if self.tool_call_id is not None:
58
+ wire["tool_call_id"] = self.tool_call_id
59
+ return wire
60
+
61
+
62
+ @dataclass
63
+ class Tool:
64
+ name: str
65
+ description: str
66
+ parameters: dict # JSON Schema
67
+
68
+ def to_wire(self) -> dict:
69
+ return {
70
+ "type": "function",
71
+ "function": {
72
+ "name": self.name,
73
+ "description": self.description,
74
+ "parameters": self.parameters,
75
+ },
76
+ }
77
+
78
+
79
+ # RB-P53's class, site S4. The four states `prompt_tokens` can be in, as data, because
80
+ # a check has to be able to name one. MEASURED is the only one that licenses arithmetic.
81
+ MEASURED = "MEASURED"
82
+ #: The response carried no usable `usage.prompt_tokens` at all — no `usage` object, an
83
+ #: empty or null one, or one that omits the field. The 0 in `Usage.prompt_tokens` is then
84
+ #: THIS PROGRAM'S DEFAULT and not anything the endpoint said. It needs its own state
85
+ #: because a fabricated zero compares unequal to any declared window, so without this
86
+ #: branch the emptiest possible response read `MEASURED` — RB-P51's defect inside the
87
+ #: code written to honour RB-P51.
88
+ UNREPORTED = "UNREPORTED"
89
+ #: `prompt_tokens` equals the window the caller declared. RB-P53 measured ollama's `/v1`
90
+ #: endpoint reporting the CONTEXT WINDOW in that field when it silently clamps a prompt,
91
+ #: so at the window the number is the window and not a measurement of anything.
92
+ VOID = "VOID"
93
+ #: No window was declared, so nothing could be compared. RB-P51: unmeasured is a verdict,
94
+ #: and it is a different verdict from MEASURED.
95
+ UNCHECKED = "UNCHECKED"
96
+
97
+ # Worst first. A sum containing a non-measurement is a non-measurement, and a sum
98
+ # containing an unchecked component is not a measured total either. UNREPORTED outranks
99
+ # VOID: a VOID call at least returned a number the endpoint chose, while an UNREPORTED one
100
+ # contributed a zero this module invented, so a total containing one is short by an
101
+ # unknown amount rather than merely untrustworthy.
102
+ _VERDICT_SEVERITY = (UNREPORTED, VOID, UNCHECKED, MEASURED)
103
+
104
+ #: Stop reasons that mean the generation was CUT rather than finished. A completion that
105
+ #: hit the cap is not a shorter answer, it is an unfinished one, and the difference is
106
+ #: invisible unless something carries it.
107
+ TRUNCATING_FINISH_REASONS = frozenset({"length"})
108
+
109
+
110
+ def _worse_verdict(a: str | None, b: str | None) -> str | None:
111
+ """The worse of two `prompt_tokens` verdicts; `None` contributes nothing.
112
+
113
+ `None` is the zero value's state and it is NOT the same as `UNCHECKED`: `Usage()` is
114
+ the accumulator seed in the agent loop and in `evalrun`, it carries no tokens and no
115
+ claim, and it must not poison a fold. `UNCHECKED` means a real response arrived and
116
+ nobody could check it, which is a claim and does survive the fold.
117
+ """
118
+ for verdict in _VERDICT_SEVERITY:
119
+ if a == verdict or b == verdict:
120
+ return verdict
121
+ return a if a is not None else b
122
+
123
+
124
+ def _worse_finish_reason(a: str | None, b: str | None) -> str | None:
125
+ """An aggregate has no single finish reason — but a truncation must not be summed away.
126
+
127
+ S5's shape, one repository over: a length stop absorbed into an outcome instead of
128
+ being reported as an instrument verdict. So a cut anywhere in the fold survives it,
129
+ and two different non-cut reasons collapse to `None` rather than to whichever came
130
+ first.
131
+ """
132
+ for reason in (a, b):
133
+ if reason in TRUNCATING_FINISH_REASONS:
134
+ return reason
135
+ if a == b:
136
+ return a
137
+ return a if b is None else (b if a is None else None)
138
+
139
+
140
+ @dataclass
141
+ class Usage:
142
+ prompt_tokens: int = 0
143
+ completion_tokens: int = 0
144
+ #: What the endpoint said stopped the generation, verbatim, or `None` when it said
145
+ #: nothing. Read back rather than assumed: before this field existed a length-stopped
146
+ #: completion and a finished one were the same object.
147
+ finish_reason: str | None = None
148
+ #: `MEASURED` / `VOID` / `UNCHECKED` / `UNREPORTED`, or `None` for a `Usage` no
149
+ #: response produced.
150
+ prompt_tokens_verdict: str | None = None
151
+
152
+ def __add__(self, other: Usage) -> Usage:
153
+ return Usage(
154
+ self.prompt_tokens + other.prompt_tokens,
155
+ self.completion_tokens + other.completion_tokens,
156
+ _worse_finish_reason(self.finish_reason, other.finish_reason),
157
+ _worse_verdict(self.prompt_tokens_verdict, other.prompt_tokens_verdict),
158
+ )
159
+
160
+ @property
161
+ def total(self) -> int:
162
+ return self.prompt_tokens + self.completion_tokens
163
+
164
+ @property
165
+ def truncated(self) -> bool:
166
+ """The generation was cut by a cap rather than finished."""
167
+ return self.finish_reason in TRUNCATING_FINISH_REASONS
168
+
169
+
170
+ @dataclass
171
+ class Response:
172
+ message: Message
173
+ usage: Usage
174
+
175
+
176
+ class ModelClient(Protocol):
177
+ def chat(self, messages: list[Message], tools: list[Tool] | None = None) -> Response: ...
178
+
179
+
180
+ class OpenAICompatible:
181
+ """One adapter covers Ollama / vLLM / LM Studio / llama.cpp server / OpenRouter."""
182
+
183
+ def __init__(
184
+ self,
185
+ base_url: str,
186
+ model: str,
187
+ api_key: str = "none",
188
+ timeout: float = 60.0,
189
+ max_retries: int = 3,
190
+ transport: httpx.BaseTransport | None = None,
191
+ seed: int | None = None,
192
+ context_window: int | None = None,
193
+ ):
194
+ self.base_url = base_url.rstrip("/")
195
+ self.model = model
196
+ self.api_key = api_key
197
+ self.max_retries = max_retries
198
+ # The context window this endpoint is configured with, DECLARED by the caller.
199
+ # It is never sent: this adapter sets no context length and this field changes no
200
+ # request. It exists so that `usage.prompt_tokens == context_window` can be given
201
+ # a verdict, which is RB-P53's shape — a silent clamp reports the window in the
202
+ # field a reader takes for a prompt size. Left unset, the comparison cannot be
203
+ # made and every response says `UNCHECKED` rather than `MEASURED`.
204
+ #
205
+ # THIS IS A DETECTOR AND NOT A PREVENTER, and calling it protection would be the
206
+ # failure this job is about. It cannot stop a clamp, it cannot detect one below
207
+ # the window, and it cannot tell a clamped prompt from a genuine prompt that
208
+ # happens to be exactly `context_window` tokens long.
209
+ self.context_window = context_window
210
+ # Sampling seed, sent only when set. OpenAI chat-completions and Ollama's
211
+ # OpenAI-compat endpoint both accept it; servers that ignore it degrade to
212
+ # unseeded sampling. Writable per run — the eval harness pins it per task.
213
+ self.seed = seed
214
+ # Capability memo for the constrained-decoding tier, flipped by the first
215
+ # HTTP 400 on a request that carried `response_format`. Its presence is also
216
+ # the duck-typed signal callers test before sending the kwarg at all.
217
+ self._response_format_unsupported = False
218
+ self._http = httpx.Client(timeout=timeout, transport=transport)
219
+
220
+ def close(self) -> None:
221
+ """Release the underlying HTTP connection pool. Idempotent."""
222
+ self._http.close()
223
+
224
+ def __enter__(self) -> OpenAICompatible:
225
+ return self
226
+
227
+ def __exit__(self, *exc_info: object) -> None:
228
+ self.close()
229
+
230
+ def _post(self, payload: dict) -> httpx.Response:
231
+ return self._http.post(
232
+ f"{self.base_url}/chat/completions",
233
+ headers={"Authorization": f"Bearer {self.api_key}"},
234
+ json=payload,
235
+ )
236
+
237
+ def chat(
238
+ self,
239
+ messages: list[Message],
240
+ tools: list[Tool] | None = None,
241
+ response_format: dict | None = None,
242
+ ) -> Response:
243
+ payload: dict = {"model": self.model, "messages": [m.to_wire() for m in messages]}
244
+ if tools:
245
+ payload["tools"] = [t.to_wire() for t in tools]
246
+ if self.seed is not None:
247
+ payload["seed"] = self.seed
248
+ if response_format is not None:
249
+ payload["response_format"] = response_format
250
+ last_err: Exception | None = None
251
+ for attempt in range(self.max_retries):
252
+ try:
253
+ r = self._post(payload)
254
+ if r.status_code == 400 and "response_format" in payload:
255
+ # That single 400 IS the capability detection — no probe request,
256
+ # no server allowlist. Drop the field, remember it for every later
257
+ # call on this client, and let this call still succeed. A 400 that
258
+ # survives the drop was never about response_format and raises below.
259
+ self._response_format_unsupported = True
260
+ del payload["response_format"]
261
+ r = self._post(payload)
262
+ if r.status_code == 429 or r.status_code >= 500:
263
+ raise TransportError(f"server error {r.status_code}: {r.text[:BODY_SNIPPET]}")
264
+ if not r.is_success:
265
+ # Non-retryable. Covers 1xx/3xx too: redirects are not followed, so
266
+ # anything but 2xx would otherwise reach _parse as non-JSON.
267
+ raise APIError(r.status_code, r.text)
268
+ return self._parse(r.json())
269
+ except (httpx.TransportError, TransportError) as e:
270
+ last_err = e
271
+ if attempt < self.max_retries - 1:
272
+ # No backoff after the final attempt — nothing follows it but the raise.
273
+ time.sleep(0.5 * (2**attempt))
274
+ raise TransportError(f"chat failed after {self.max_retries} attempts: {last_err}")
275
+
276
+ def _parse(self, data: dict) -> Response:
277
+ try:
278
+ # `finish_reason` lives on the CHOICE, not on the message inside it, which is
279
+ # why reading it back needs the enclosing object and not just `["message"]`.
280
+ raw_choice = data["choices"][0]
281
+ choice = raw_choice["message"]
282
+ except (KeyError, IndexError) as e:
283
+ raise BantamError(f"malformed chat response: {data!r:.200}") from e
284
+
285
+ tool_calls = []
286
+ for tc in choice.get("tool_calls") or []:
287
+ try:
288
+ name = tc["function"]["name"]
289
+ raw_args = tc["function"]["arguments"]
290
+ # Support dict-form arguments (already a dict) or JSON string
291
+ if isinstance(raw_args, dict):
292
+ arguments = raw_args
293
+ else:
294
+ arguments = json.loads(raw_args)
295
+ tool_calls.append(ToolCall(id=tc["id"], name=name, arguments=arguments))
296
+ except (json.JSONDecodeError, ValueError) as e:
297
+ raw = tc["function"]["arguments"]
298
+ name = tc["function"]["name"]
299
+ raise BantamError(
300
+ f"tool call '{name}' has malformed JSON arguments: {raw!r:.200}"
301
+ ) from e
302
+
303
+ raw_usage = data.get("usage")
304
+ usage = raw_usage if isinstance(raw_usage, dict) else {}
305
+ reported = usage.get("prompt_tokens")
306
+ # `bool` is an `int` and `True` is not a token count.
307
+ usable = isinstance(reported, int) and not isinstance(reported, bool)
308
+ prompt_tokens = reported if usable else 0
309
+ # The classifier. Four states, one branch each, and no branch is a message: a
310
+ # mutation that rewrites the docstrings above cannot move any of them (N-12).
311
+ # UNREPORTED is tested FIRST because it is a fact about what arrived, and the
312
+ # window question does not arise for a number that was never sent.
313
+ if not usable:
314
+ verdict = UNREPORTED
315
+ elif self.context_window is None:
316
+ verdict = UNCHECKED
317
+ elif prompt_tokens == self.context_window:
318
+ verdict = VOID
319
+ else:
320
+ verdict = MEASURED
321
+ return Response(
322
+ message=Message(role="assistant", content=choice.get("content"), tool_calls=tool_calls),
323
+ usage=Usage(
324
+ prompt_tokens,
325
+ usage.get("completion_tokens", 0),
326
+ raw_choice.get("finish_reason"),
327
+ verdict,
328
+ ),
329
+ )