bantamkit 0.27.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. bantamkit/__init__.py +32 -0
  2. bantamkit/agent.py +458 -0
  3. bantamkit/assets/contracts/default.yaml +90 -0
  4. bantamkit/assets/evals/devteam/manifest.yaml +351 -0
  5. bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
  6. bantamkit/assets/evals/devteam/repo/README.md +12 -0
  7. bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
  8. bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
  9. bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
  10. bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
  11. bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
  12. bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
  13. bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
  14. bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
  15. bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
  16. bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
  17. bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
  18. bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
  19. bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
  20. bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
  21. bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
  22. bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
  23. bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
  24. bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
  25. bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
  26. bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
  27. bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
  28. bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
  29. bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
  30. bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
  31. bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
  32. bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
  33. bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
  34. bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
  35. bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
  36. bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
  37. bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
  38. bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
  39. bantamkit/assets/evals/fixtures/.gitkeep +0 -0
  40. bantamkit/assets/evals/fixtures/catalog.json +6 -0
  41. bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
  42. bantamkit/assets/evals/tasks/.gitkeep +0 -0
  43. bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
  44. bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
  45. bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
  46. bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
  47. bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
  48. bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
  49. bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
  50. bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
  51. bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
  52. bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
  53. bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
  54. bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
  55. bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
  56. bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
  57. bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
  58. bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
  59. bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
  60. bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
  61. bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
  62. bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
  63. bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
  64. bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
  65. bantamkit/assets/profiles/default.yaml +31 -0
  66. bantamkit/assets/profiles/patient.yaml +31 -0
  67. bantamkit/assets/rubrics/.gitkeep +0 -0
  68. bantamkit/assets/rubrics/code-quality.yaml +20 -0
  69. bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
  70. bantamkit/assets/rubrics/task-completion.yaml +28 -0
  71. bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
  72. bantamkit/assets/skills/.gitkeep +0 -0
  73. bantamkit/assets/skills/file-graph.md +7 -0
  74. bantamkit/assets/skills/memory.md +35 -0
  75. bantamkit/assets/tools/.gitkeep +0 -0
  76. bantamkit/assets/tools/bantamkit_read.json +48 -0
  77. bantamkit/assets/tools/bantamkit_status.json +25 -0
  78. bantamkit/assets/tools/build_identity.json +17 -0
  79. bantamkit/assets/tools/document_list.json +12 -0
  80. bantamkit/assets/tools/document_read.json +31 -0
  81. bantamkit/assets/tools/file_graph.json +12 -0
  82. bantamkit/assets/tools/memory_compact.json +31 -0
  83. bantamkit/assets/tools/memory_recall.json +38 -0
  84. bantamkit/assets/tools/memory_save.json +61 -0
  85. bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
  86. bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
  87. bantamkit/assets/tools/shiftwork_status.json +25 -0
  88. bantamkit/assets/tools/skill_audit.json +70 -0
  89. bantamkit/assets/tools/validate_json.json +31 -0
  90. bantamkit/assets.py +67 -0
  91. bantamkit/budget.py +114 -0
  92. bantamkit/client.py +329 -0
  93. bantamkit/contract.py +522 -0
  94. bantamkit/criticreplay.py +3241 -0
  95. bantamkit/critique.py +301 -0
  96. bantamkit/docread.py +1744 -0
  97. bantamkit/evalrun.py +2003 -0
  98. bantamkit/eventlog.py +282 -0
  99. bantamkit/filegraph.py +218 -0
  100. bantamkit/loopguard.py +101 -0
  101. bantamkit/mcpreport.py +763 -0
  102. bantamkit/mcpserver.py +1334 -0
  103. bantamkit/memory/__init__.py +28 -0
  104. bantamkit/memory/__main__.py +291 -0
  105. bantamkit/memory/component.py +569 -0
  106. bantamkit/memory/divergence.py +744 -0
  107. bantamkit/memory/layers.py +257 -0
  108. bantamkit/memory/store.py +940 -0
  109. bantamkit/pdfread.py +1402 -0
  110. bantamkit/profile.py +46 -0
  111. bantamkit/shiftwork.py +212 -0
  112. bantamkit/skillaudit.py +853 -0
  113. bantamkit/statusline.py +313 -0
  114. bantamkit/structured.py +125 -0
  115. bantamkit/textutil.py +30 -0
  116. bantamkit-0.27.0.dist-info/METADATA +207 -0
  117. bantamkit-0.27.0.dist-info/RECORD +119 -0
  118. bantamkit-0.27.0.dist-info/WHEEL +4 -0
  119. bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,14 @@
1
+ name: extract-invoice
2
+ family: structured-extraction
3
+ prompt: |
4
+ Extract the invoice as JSON with keys "number" (string) and "total" (integer).
5
+ Text: "Invoice INV-42 came to 199 USD, paid by card."
6
+ schema:
7
+ type: object
8
+ required: [number, total]
9
+ properties:
10
+ number: {type: string}
11
+ total: {type: integer}
12
+ scoring:
13
+ kind: json_equal
14
+ expected: {number: "INV-42", total: 199}
@@ -0,0 +1,15 @@
1
+ name: extract-order
2
+ family: structured-extraction
3
+ prompt: |
4
+ Extract the order as JSON with keys "item" (singular, lowercase) and
5
+ "quantity" (integer).
6
+ Text: "Customer wants three Widgets shipped by Friday."
7
+ schema:
8
+ type: object
9
+ required: [item, quantity]
10
+ properties:
11
+ item: {type: string}
12
+ quantity: {type: integer}
13
+ scoring:
14
+ kind: json_equal
15
+ expected: {item: "widget", quantity: 3}
@@ -0,0 +1,14 @@
1
+ name: extract-schedule
2
+ family: structured-extraction
3
+ prompt: |
4
+ Extract the schedule as JSON with keys "day" (lowercase) and "time" (HH:MM).
5
+ Text: "Standup happens every Tuesday at 09:30 sharp."
6
+ schema:
7
+ type: object
8
+ required: [day, time]
9
+ properties:
10
+ day: {type: string}
11
+ time: {type: string}
12
+ scoring:
13
+ kind: json_equal
14
+ expected: {day: "tuesday", time: "09:30"}
@@ -0,0 +1,17 @@
1
+ name: extract-versions
2
+ family: structured-extraction
3
+ prompt: |
4
+ Extract as JSON with keys "package" (string) and "versions" (array of strings,
5
+ in the order mentioned).
6
+ Text: "Package foo supports versions 1.2, 1.3 and 2.0."
7
+ schema:
8
+ type: object
9
+ required: [package, versions]
10
+ properties:
11
+ package: {type: string}
12
+ versions:
13
+ type: array
14
+ items: {type: string}
15
+ scoring:
16
+ kind: json_equal
17
+ expected: {package: "foo", versions: ["1.2", "1.3", "2.0"]}
@@ -0,0 +1,84 @@
1
+ name: nav-prod-port
2
+ family: file-nav
3
+ tools:
4
+ - read_file
5
+ - list_files
6
+ workspace:
7
+ README.md: '# billing-svc
8
+
9
+
10
+ Billing service. Deployment details live in docs/deploy.md.
11
+
12
+ Development setup is in docs/dev.md (local only).
13
+
14
+
15
+ For coding conventions see style.md.'
16
+ docs/deploy.md: '# Deploying billing-svc
17
+
18
+
19
+ Production settings are NOT here. Every environment value
20
+
21
+ (host, port, pool sizes) is read from config/prod.yaml at boot.
22
+
23
+ Staging mirrors production except where config/staging.yaml overrides.'
24
+ config/prod.yaml: "host: billing.internal.example.com\nport: 9443\npool:\n min: 4\n max: 32\ntls: required\n"
25
+ config/staging.yaml: "host: billing.staging.example.com\nport: 8443\npool:\n min: 1\n max: 8\ntls: optional\n"
26
+ style.md: '# Style guide (excerpt)
27
+
28
+
29
+ Prefer small modules with one clear responsibility. Public functions carry
30
+
31
+ docstrings describing behavior, not implementation. Errors are values where
32
+
33
+ practical; exceptions for programmer mistakes. Log lines are structured
34
+
35
+ key=value pairs, lowercase keys. Feature flags default off and are removed
36
+
37
+ within two releases of full rollout. Database migrations are forward-only;
38
+
39
+ rollbacks happen by writing a new migration. Review comments prefixed nit:
40
+
41
+ are optional. Vendored code lives under third_party/ and is never edited in
42
+
43
+ place — patch files only. Commit subjects use the imperative mood and stay
44
+
45
+ under 68 characters. Squash-merge is the default; merge commits are for
46
+
47
+ release branches only. Generated files carry a do-not-edit header and are
48
+
49
+ regenerated in CI, never by hand.
50
+
51
+ '
52
+ ci.md: '# CI pipeline notes
53
+
54
+
55
+ The pipeline runs on every push: lint, unit tests, integration tests, and a
56
+
57
+ packaging step. Caching of the dependency layer cut average build time from
58
+
59
+ 11 minutes to 6. Flaky tests are quarantined under tests/flaky/ and re-run
60
+
61
+ three times before being reported as failures. The nightly job additionally
62
+
63
+ runs the slow property-based suite and publishes coverage to the internal
64
+
65
+ dashboard. Artifacts older than 30 days are pruned automatically. If a
66
+
67
+ release tag is pushed, the pipeline also builds the signed bundle and
68
+
69
+ uploads it to the artifact store with a 90-day retention override. Secrets
70
+
71
+ are injected via the runner''s OIDC exchange; never commit tokens. When the
72
+
73
+ integration environment is unavailable the pipeline degrades to unit tests
74
+
75
+ only and marks the run amber, not red. Amber runs do not block merges but
76
+
77
+ three consecutive ambers page the on-call.
78
+
79
+ '
80
+ prompt: 'Find the production port of billing-svc from the workspace files. Start with list_files, then follow the documentation to the right file. Use list_files first, then read_file on every file you need — follow each pointer the files give you; do not answer from memory or guess. Answer with ONLY this JSON, nothing else: {"port": <number>}'
81
+ scoring:
82
+ kind: json_equal
83
+ expected:
84
+ port: 9443
@@ -0,0 +1,87 @@
1
+ name: nav-release-bundle
2
+ family: file-nav
3
+ tools:
4
+ - read_file
5
+ - list_files
6
+ workspace:
7
+ README.md: '# imgproc
8
+
9
+
10
+ Image processing CLI. Release process: docs/release.md.'
11
+ docs/release.md: '# Releasing imgproc
12
+
13
+
14
+ The bundle name is assembled as <name>-<version>.tar.gz where <name>
15
+
16
+ comes from package.cfg and <version> comes from VERSION. Both files
17
+
18
+ live at the repo root. Never hardcode either value.'
19
+ package.cfg: '[package]
20
+
21
+ name = imgproc-cli
22
+
23
+ license = apache-2.0
24
+
25
+ '
26
+ VERSION: '2.9.1
27
+
28
+ '
29
+ docs/ci.md: '# CI pipeline notes
30
+
31
+
32
+ The pipeline runs on every push: lint, unit tests, integration tests, and a
33
+
34
+ packaging step. Caching of the dependency layer cut average build time from
35
+
36
+ 11 minutes to 6. Flaky tests are quarantined under tests/flaky/ and re-run
37
+
38
+ three times before being reported as failures. The nightly job additionally
39
+
40
+ runs the slow property-based suite and publishes coverage to the internal
41
+
42
+ dashboard. Artifacts older than 30 days are pruned automatically. If a
43
+
44
+ release tag is pushed, the pipeline also builds the signed bundle and
45
+
46
+ uploads it to the artifact store with a 90-day retention override. Secrets
47
+
48
+ are injected via the runner''s OIDC exchange; never commit tokens. When the
49
+
50
+ integration environment is unavailable the pipeline degrades to unit tests
51
+
52
+ only and marks the run amber, not red. Amber runs do not block merges but
53
+
54
+ three consecutive ambers page the on-call.
55
+
56
+ '
57
+ docs/style.md: '# Style guide (excerpt)
58
+
59
+
60
+ Prefer small modules with one clear responsibility. Public functions carry
61
+
62
+ docstrings describing behavior, not implementation. Errors are values where
63
+
64
+ practical; exceptions for programmer mistakes. Log lines are structured
65
+
66
+ key=value pairs, lowercase keys. Feature flags default off and are removed
67
+
68
+ within two releases of full rollout. Database migrations are forward-only;
69
+
70
+ rollbacks happen by writing a new migration. Review comments prefixed nit:
71
+
72
+ are optional. Vendored code lives under third_party/ and is never edited in
73
+
74
+ place — patch files only. Commit subjects use the imperative mood and stay
75
+
76
+ under 68 characters. Squash-merge is the default; merge commits are for
77
+
78
+ release branches only. Generated files carry a do-not-edit header and are
79
+
80
+ regenerated in CI, never by hand.
81
+
82
+ '
83
+ prompt: 'What is the exact release bundle filename for the current version? Follow the release documentation and assemble it from the files it names. Use list_files first, then read_file on every file you need — follow each pointer the files give you; do not answer from memory or guess. Answer with ONLY this JSON, nothing else: {"bundle": "<filename>"}'
84
+ scoring:
85
+ kind: json_equal
86
+ expected:
87
+ bundle: imgproc-cli-2.9.1.tar.gz
@@ -0,0 +1,17 @@
1
+ name: recall-audit-retention
2
+ family: memory-recall
3
+ prompt: 'How many days do we keep AUDIT logs? If you have a memory tool, check memory
4
+ first. Answer with ONLY this JSON, nothing else: {"days": <number>}'
5
+ memory_setup:
6
+ - type: project
7
+ name: log-retention-default
8
+ description: how long ordinary application logs are kept
9
+ body: Application logs are kept for 45 days.
10
+ - type: project
11
+ name: log-retention-audit
12
+ description: retention exception for the audit trail
13
+ body: Audit logs are the exception and are kept for 400 days.
14
+ scoring:
15
+ kind: json_equal
16
+ expected:
17
+ days: 400
@@ -0,0 +1,13 @@
1
+ name: recall-cache-ttl
2
+ family: memory-recall
3
+ prompt: 'What is the cache TTL for the pricing service? If you have a memory tool,
4
+ check memory first. Answer with ONLY this JSON, nothing else: {"seconds": <number>}'
5
+ memory_setup:
6
+ - type: project
7
+ name: pricing-cache-ttl
8
+ description: cache ttl seconds for the pricing service
9
+ body: The pricing service caches responses for 240 seconds.
10
+ scoring:
11
+ kind: json_equal
12
+ expected:
13
+ seconds: 240
@@ -0,0 +1,17 @@
1
+ name: recall-db-port
2
+ family: memory-recall
3
+ prompt: 'What port does the STAGING database listen on? If you have a memory tool,
4
+ check memory first. Answer with ONLY this JSON, nothing else: {"port": <number>}'
5
+ memory_setup:
6
+ - type: project
7
+ name: staging-db-port
8
+ description: staging database port number
9
+ body: The staging database listens on port 5433.
10
+ - type: project
11
+ name: prod-db-connection
12
+ description: production postgres connection endpoint
13
+ body: Production postgres is at db.prod.internal on port 5432.
14
+ scoring:
15
+ kind: json_equal
16
+ expected:
17
+ port: 5433
@@ -0,0 +1,13 @@
1
+ name: recall-deploy
2
+ family: memory-recall
3
+ prompt: 'How do we deploy this project to production? If you have a memory tool,
4
+ check memory first. Answer with ONLY this JSON, nothing else: {"command": "<command>"}'
5
+ memory_setup:
6
+ - type: project
7
+ name: deploy-command
8
+ description: how we deploy this project to production
9
+ body: Deploy with `make ship-prod` from the repo root.
10
+ scoring:
11
+ kind: json_equal
12
+ expected:
13
+ command: make ship-prod
@@ -0,0 +1,18 @@
1
+ name: recall-env-endpoint
2
+ family: memory-recall
3
+ prompt: 'What is the full base URL for the reports API in PRODUCTION, including
4
+ the version prefix? If you have a memory tool, check memory first. Answer with
5
+ ONLY this JSON, nothing else: {"url": "<url>"}'
6
+ memory_setup:
7
+ - type: project
8
+ name: prod-api-host
9
+ description: hostname serving production traffic
10
+ body: Production traffic is served from https://api.example-prod.io.
11
+ - type: project
12
+ name: reports-version-prefix
13
+ description: version path segment used by the reports service
14
+ body: The reports service is mounted under /v3/reports on every host.
15
+ scoring:
16
+ kind: json_equal
17
+ expected:
18
+ url: https://api.example-prod.io/v3/reports
@@ -0,0 +1,21 @@
1
+ name: recall-oncall-rotation
2
+ family: memory-recall
3
+ prompt: 'Who is on call for the PAYMENTS service this week? If you have a memory
4
+ tool, check memory first. Answer with ONLY this JSON, nothing else: {"name": "<name>"}'
5
+ memory_setup:
6
+ - type: project
7
+ name: oncall-payments
8
+ description: current pager duty for the payments service
9
+ body: Payments on-call this week is Priya.
10
+ - type: project
11
+ name: oncall-search
12
+ description: rotation owner covering search infrastructure
13
+ body: Search on-call this week is Marcus.
14
+ - type: project
15
+ name: oncall-ingest
16
+ description: escalation contact for the ingest pipeline
17
+ body: Ingest on-call this week is Dana.
18
+ scoring:
19
+ kind: json_equal
20
+ expected:
21
+ name: Priya
@@ -0,0 +1,13 @@
1
+ name: recall-oncall
2
+ family: memory-recall
3
+ prompt: 'Who is on-call for infrastructure this quarter? If you have a memory tool,
4
+ check memory first. Answer with ONLY this JSON, nothing else: {"name": "<name>"}'
5
+ memory_setup:
6
+ - type: project
7
+ name: infra-oncall
8
+ description: who is on-call for infrastructure this quarter
9
+ body: Nadia is on-call for infrastructure until end of Q3.
10
+ scoring:
11
+ kind: json_equal
12
+ expected:
13
+ name: Nadia
@@ -0,0 +1,18 @@
1
+ name: recall-org-quota
2
+ family: memory-recall
3
+ prompt: 'What is the TOTAL requests-per-minute quota for one full org on the api
4
+ gateway? If you have a memory tool, check memory first, then compute the answer.
5
+ Answer with ONLY this JSON, nothing else: {"total": <number>}'
6
+ memory_setup:
7
+ - type: project
8
+ name: gateway-user-quota
9
+ description: per-user rate limit on the api gateway
10
+ body: The api gateway allows each user 40 requests per minute.
11
+ - type: project
12
+ name: org-seat-count
13
+ description: how many seats one org licence includes
14
+ body: Every org licence includes exactly 5 user seats.
15
+ scoring:
16
+ kind: json_equal
17
+ expected:
18
+ total: 200
@@ -0,0 +1,13 @@
1
+ name: recall-owner
2
+ family: memory-recall
3
+ prompt: 'Which team owns the payments API? If you have a memory tool, check memory
4
+ first. Answer with ONLY this JSON, nothing else: {"team": "<team name>"}'
5
+ memory_setup:
6
+ - type: project
7
+ name: payments-api-owner
8
+ description: which team owns the payments api
9
+ body: The payments API is owned by team Atlas.
10
+ scoring:
11
+ kind: json_equal
12
+ expected:
13
+ team: Atlas
@@ -0,0 +1,10 @@
1
+ name: shop-basket-total
2
+ family: tool-use
3
+ prompt: |
4
+ A customer orders 2 widgets, 3 doohickeys and 1 gadget. Use the tools to
5
+ look up unit prices, then answer with ONLY this JSON, nothing else:
6
+ {"total": <number>}
7
+ tools: [price_lookup]
8
+ scoring:
9
+ kind: json_equal
10
+ expected: {total: 131}
@@ -0,0 +1,9 @@
1
+ name: shop-cheapest
2
+ family: tool-use
3
+ prompt: |
4
+ Use the tools to check the unit prices of "widget" and "gadget".
5
+ Answer with ONLY this JSON, nothing else: {"cheaper": "<item name>"}
6
+ tools: [price_lookup]
7
+ scoring:
8
+ kind: json_equal
9
+ expected: {cheaper: "widget"}
@@ -0,0 +1,9 @@
1
+ name: shop-compare
2
+ family: tool-use
3
+ prompt: |
4
+ Use the tools to check the unit prices of "widget" and "gadget",
5
+ and answer with the name of the more expensive item.
6
+ tools: [price_lookup]
7
+ scoring:
8
+ kind: tool_trace
9
+ expected: [price_lookup, price_lookup]
@@ -0,0 +1,9 @@
1
+ name: shop-gadget-value
2
+ family: tool-use
3
+ prompt: |
4
+ Use the tools to find the unit price and stock count of "gadget",
5
+ then answer with the total value of the stock (price times stock) as a number.
6
+ tools: [price_lookup, stock_lookup]
7
+ scoring:
8
+ kind: contains
9
+ expected: ["540"]
@@ -0,0 +1,9 @@
1
+ name: shop-stock-total
2
+ family: tool-use
3
+ prompt: |
4
+ Use the tools to find the stock counts of "widget" and "gadget",
5
+ then answer with the combined total stock as a number.
6
+ tools: [stock_lookup]
7
+ scoring:
8
+ kind: contains
9
+ expected: ["13"]
@@ -0,0 +1,9 @@
1
+ name: shop-total
2
+ family: tool-use
3
+ prompt: |
4
+ Use the tools to find the unit price and stock count of "widget",
5
+ then answer with the total value of the stock (price times stock) as a number.
6
+ tools: [price_lookup, stock_lookup]
7
+ scoring:
8
+ kind: contains
9
+ expected: ["100"]
@@ -0,0 +1,31 @@
1
+ # The conservative default profile. Every number here was calibrated on
2
+ # qwen3:4b-instruct (the reference model) — the cross-model sweep measured
3
+ # that honestly; per-model profiles are future work (P6).
4
+ name: default
5
+ agent:
6
+ max_turns: 10
7
+ observation_budget: 4096
8
+ structured:
9
+ max_retries: 3
10
+ schema_gate:
11
+ max_attempts: 3
12
+ # One retry only: the gate rescues a missing-JSON restatement, and a second
13
+ # miss is format abandonment that more nagging does not fix (spec §2.4).
14
+ json_answer:
15
+ max_attempts: 1
16
+ critique:
17
+ max_rounds: 3
18
+ evidence_budget: 4096
19
+ # Global spend governor (P3), applied only when a TokenBudget is attached.
20
+ # `ceiling` is the hard stop for one run; past `ceiling * optional_cutoff`
21
+ # optional work (critique rounds) is denied. The gap between the two IS the
22
+ # reserve that keeps the final answer emission affordable.
23
+ token_budget:
24
+ ceiling: 6000
25
+ optional_cutoff: 0.75
26
+ # Loop-detection thresholds, applied only when a LoopGuard is attached. Probe-
27
+ # calibrated: max observation-repeat streak in every passing run was 2, looping
28
+ # runs burned 6-16 — so the note fires at 3 and the hard warning at 5.
29
+ loop_guard:
30
+ inject_at: 3
31
+ warn_at: 5
@@ -0,0 +1,31 @@
1
+ # Default, but with a longer turn budget: for models that need more steps than
2
+ # the 4b-calibrated default (measured: 3b file-nav and 7b recall died of turn
3
+ # exhaustion under max_turns 10). Calibration-only until a bar says otherwise.
4
+ name: patient
5
+ agent:
6
+ max_turns: 16
7
+ observation_budget: 4096
8
+ structured:
9
+ max_retries: 3
10
+ schema_gate:
11
+ max_attempts: 3
12
+ # One retry only: the gate rescues a missing-JSON restatement, and a second
13
+ # miss is format abandonment that more nagging does not fix (spec §2.4).
14
+ json_answer:
15
+ max_attempts: 1
16
+ critique:
17
+ max_rounds: 3
18
+ evidence_budget: 4096
19
+ # Global spend governor (P3), applied only when a TokenBudget is attached.
20
+ # `ceiling` is the hard stop for one run; past `ceiling * optional_cutoff`
21
+ # optional work (critique rounds) is denied. The gap between the two IS the
22
+ # reserve that keeps the final answer emission affordable.
23
+ token_budget:
24
+ ceiling: 6000
25
+ optional_cutoff: 0.75
26
+ # Loop-detection thresholds, applied only when a LoopGuard is attached. Probe-
27
+ # calibrated: max observation-repeat streak in every passing run was 2, looping
28
+ # runs burned 6-16 — so the note fires at 3 and the hard warning at 5.
29
+ loop_guard:
30
+ inject_at: 3
31
+ warn_at: 5
File without changes
@@ -0,0 +1,20 @@
1
+ name: code-quality
2
+ threshold: 7
3
+ schema:
4
+ type: object
5
+ required: [score, feedback]
6
+ properties:
7
+ score: {type: integer, minimum: 0, maximum: 10}
8
+ feedback: {type: string}
9
+ prompt: |
10
+ You are a strict code reviewer. Judge the code below.
11
+
12
+ Task:
13
+ {task}
14
+
15
+ Code:
16
+ {output}
17
+
18
+ Score 0-10 on: correctness for the task, handling of error cases, and
19
+ absence of dead or needless code. 10 = ship as-is.
20
+ Return ONLY JSON: {{"score": <int>, "feedback": "<specific defects to fix>"}}
@@ -0,0 +1,37 @@
1
+ name: grounded-completion
2
+ threshold: 7
3
+ schema:
4
+ type: object
5
+ required: [reasoning, score, feedback]
6
+ properties:
7
+ reasoning: {type: string}
8
+ score: {type: integer, minimum: 0, maximum: 10}
9
+ feedback: {type: string}
10
+ prompt: |
11
+ You are a reviewer checking whether the answer contains the correct content.
12
+ The tool evidence below is the ground truth: it lists every tool call the
13
+ answerer made and what the tool returned. Verify the answer against it and
14
+ recompute any numbers yourself from the evidence. If the answer states a
15
+ fact or number that contradicts the evidence, score it 0-4 and put the
16
+ correct values from the evidence in your feedback.
17
+ In the reasoning field, first work out what the correct answer to the task
18
+ is, step by step, using ONLY the values in the evidence — do the arithmetic
19
+ and comparisons yourself. Then compare the given answer against your result.
20
+ Judge ONLY content. Do NOT deduct points for formatting, phrasing, extra
21
+ surrounding text, hedging, or verbosity. An answer that refuses or declines
22
+ to provide what the task asks for is missing the required content — score
23
+ it 0-4, even when the refusal is polite or explains itself.
24
+
25
+ Task:
26
+ {task}
27
+
28
+ Tool evidence:
29
+ {evidence}
30
+
31
+ Answer:
32
+ {output}
33
+
34
+ Score 0-10: 9-10 = required content present, correct, and consistent with
35
+ the evidence; 5-8 = partially correct or missing pieces; 0-4 = wrong,
36
+ absent, or contradicted by the evidence.
37
+ Return ONLY JSON: {{"reasoning": "<derive the correct answer from the evidence, then compare the given answer to it>", "score": <int>, "feedback": "<what is wrong or missing, and the correct values from the evidence>"}}
@@ -0,0 +1,28 @@
1
+ name: task-completion
2
+ threshold: 7
3
+ schema:
4
+ type: object
5
+ required: [score, feedback]
6
+ properties:
7
+ score: {type: integer, minimum: 0, maximum: 10}
8
+ feedback: {type: string}
9
+ prompt: |
10
+ You are a reviewer checking whether the answer contains the correct content.
11
+ Judge ONLY whether the information the task asks for is present and correct.
12
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
13
+ hedging, or verbosity. If the required facts are present and right, the
14
+ answer completes the task.
15
+ An answer that refuses or declines to provide what the task asks for is
16
+ missing the required content — score it 0-4, even when the refusal is
17
+ polite or explains itself. Hedging around a real answer is fine; hedging
18
+ instead of an answer is not.
19
+
20
+ Task:
21
+ {task}
22
+
23
+ Answer:
24
+ {output}
25
+
26
+ Score 0-10: 9-10 = required content present and correct; 5-8 = partially
27
+ correct or missing pieces; 0-4 = wrong or absent.
28
+ Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}