bantamkit 0.27.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bantamkit/__init__.py +32 -0
- bantamkit/agent.py +458 -0
- bantamkit/assets/contracts/default.yaml +90 -0
- bantamkit/assets/evals/devteam/manifest.yaml +351 -0
- bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
- bantamkit/assets/evals/devteam/repo/README.md +12 -0
- bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
- bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
- bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
- bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
- bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
- bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
- bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
- bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
- bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
- bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
- bantamkit/assets/evals/fixtures/.gitkeep +0 -0
- bantamkit/assets/evals/fixtures/catalog.json +6 -0
- bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
- bantamkit/assets/evals/tasks/.gitkeep +0 -0
- bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
- bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
- bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
- bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
- bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
- bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
- bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
- bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
- bantamkit/assets/profiles/default.yaml +31 -0
- bantamkit/assets/profiles/patient.yaml +31 -0
- bantamkit/assets/rubrics/.gitkeep +0 -0
- bantamkit/assets/rubrics/code-quality.yaml +20 -0
- bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
- bantamkit/assets/rubrics/task-completion.yaml +28 -0
- bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
- bantamkit/assets/skills/.gitkeep +0 -0
- bantamkit/assets/skills/file-graph.md +7 -0
- bantamkit/assets/skills/memory.md +35 -0
- bantamkit/assets/tools/.gitkeep +0 -0
- bantamkit/assets/tools/bantamkit_read.json +48 -0
- bantamkit/assets/tools/bantamkit_status.json +25 -0
- bantamkit/assets/tools/build_identity.json +17 -0
- bantamkit/assets/tools/document_list.json +12 -0
- bantamkit/assets/tools/document_read.json +31 -0
- bantamkit/assets/tools/file_graph.json +12 -0
- bantamkit/assets/tools/memory_compact.json +31 -0
- bantamkit/assets/tools/memory_recall.json +38 -0
- bantamkit/assets/tools/memory_save.json +61 -0
- bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
- bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
- bantamkit/assets/tools/shiftwork_status.json +25 -0
- bantamkit/assets/tools/skill_audit.json +70 -0
- bantamkit/assets/tools/validate_json.json +31 -0
- bantamkit/assets.py +67 -0
- bantamkit/budget.py +114 -0
- bantamkit/client.py +329 -0
- bantamkit/contract.py +522 -0
- bantamkit/criticreplay.py +3241 -0
- bantamkit/critique.py +301 -0
- bantamkit/docread.py +1744 -0
- bantamkit/evalrun.py +2003 -0
- bantamkit/eventlog.py +282 -0
- bantamkit/filegraph.py +218 -0
- bantamkit/loopguard.py +101 -0
- bantamkit/mcpreport.py +763 -0
- bantamkit/mcpserver.py +1334 -0
- bantamkit/memory/__init__.py +28 -0
- bantamkit/memory/__main__.py +291 -0
- bantamkit/memory/component.py +569 -0
- bantamkit/memory/divergence.py +744 -0
- bantamkit/memory/layers.py +257 -0
- bantamkit/memory/store.py +940 -0
- bantamkit/pdfread.py +1402 -0
- bantamkit/profile.py +46 -0
- bantamkit/shiftwork.py +212 -0
- bantamkit/skillaudit.py +853 -0
- bantamkit/statusline.py +313 -0
- bantamkit/structured.py +125 -0
- bantamkit/textutil.py +30 -0
- bantamkit-0.27.0.dist-info/METADATA +207 -0
- bantamkit-0.27.0.dist-info/RECORD +119 -0
- bantamkit-0.27.0.dist-info/WHEEL +4 -0
- bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
name: extract-invoice
|
|
2
|
+
family: structured-extraction
|
|
3
|
+
prompt: |
|
|
4
|
+
Extract the invoice as JSON with keys "number" (string) and "total" (integer).
|
|
5
|
+
Text: "Invoice INV-42 came to 199 USD, paid by card."
|
|
6
|
+
schema:
|
|
7
|
+
type: object
|
|
8
|
+
required: [number, total]
|
|
9
|
+
properties:
|
|
10
|
+
number: {type: string}
|
|
11
|
+
total: {type: integer}
|
|
12
|
+
scoring:
|
|
13
|
+
kind: json_equal
|
|
14
|
+
expected: {number: "INV-42", total: 199}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
name: extract-order
|
|
2
|
+
family: structured-extraction
|
|
3
|
+
prompt: |
|
|
4
|
+
Extract the order as JSON with keys "item" (singular, lowercase) and
|
|
5
|
+
"quantity" (integer).
|
|
6
|
+
Text: "Customer wants three Widgets shipped by Friday."
|
|
7
|
+
schema:
|
|
8
|
+
type: object
|
|
9
|
+
required: [item, quantity]
|
|
10
|
+
properties:
|
|
11
|
+
item: {type: string}
|
|
12
|
+
quantity: {type: integer}
|
|
13
|
+
scoring:
|
|
14
|
+
kind: json_equal
|
|
15
|
+
expected: {item: "widget", quantity: 3}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
name: extract-schedule
|
|
2
|
+
family: structured-extraction
|
|
3
|
+
prompt: |
|
|
4
|
+
Extract the schedule as JSON with keys "day" (lowercase) and "time" (HH:MM).
|
|
5
|
+
Text: "Standup happens every Tuesday at 09:30 sharp."
|
|
6
|
+
schema:
|
|
7
|
+
type: object
|
|
8
|
+
required: [day, time]
|
|
9
|
+
properties:
|
|
10
|
+
day: {type: string}
|
|
11
|
+
time: {type: string}
|
|
12
|
+
scoring:
|
|
13
|
+
kind: json_equal
|
|
14
|
+
expected: {day: "tuesday", time: "09:30"}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
name: extract-versions
|
|
2
|
+
family: structured-extraction
|
|
3
|
+
prompt: |
|
|
4
|
+
Extract as JSON with keys "package" (string) and "versions" (array of strings,
|
|
5
|
+
in the order mentioned).
|
|
6
|
+
Text: "Package foo supports versions 1.2, 1.3 and 2.0."
|
|
7
|
+
schema:
|
|
8
|
+
type: object
|
|
9
|
+
required: [package, versions]
|
|
10
|
+
properties:
|
|
11
|
+
package: {type: string}
|
|
12
|
+
versions:
|
|
13
|
+
type: array
|
|
14
|
+
items: {type: string}
|
|
15
|
+
scoring:
|
|
16
|
+
kind: json_equal
|
|
17
|
+
expected: {package: "foo", versions: ["1.2", "1.3", "2.0"]}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
name: nav-prod-port
|
|
2
|
+
family: file-nav
|
|
3
|
+
tools:
|
|
4
|
+
- read_file
|
|
5
|
+
- list_files
|
|
6
|
+
workspace:
|
|
7
|
+
README.md: '# billing-svc
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
Billing service. Deployment details live in docs/deploy.md.
|
|
11
|
+
|
|
12
|
+
Development setup is in docs/dev.md (local only).
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
For coding conventions see style.md.'
|
|
16
|
+
docs/deploy.md: '# Deploying billing-svc
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
Production settings are NOT here. Every environment value
|
|
20
|
+
|
|
21
|
+
(host, port, pool sizes) is read from config/prod.yaml at boot.
|
|
22
|
+
|
|
23
|
+
Staging mirrors production except where config/staging.yaml overrides.'
|
|
24
|
+
config/prod.yaml: "host: billing.internal.example.com\nport: 9443\npool:\n min: 4\n max: 32\ntls: required\n"
|
|
25
|
+
config/staging.yaml: "host: billing.staging.example.com\nport: 8443\npool:\n min: 1\n max: 8\ntls: optional\n"
|
|
26
|
+
style.md: '# Style guide (excerpt)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
Prefer small modules with one clear responsibility. Public functions carry
|
|
30
|
+
|
|
31
|
+
docstrings describing behavior, not implementation. Errors are values where
|
|
32
|
+
|
|
33
|
+
practical; exceptions for programmer mistakes. Log lines are structured
|
|
34
|
+
|
|
35
|
+
key=value pairs, lowercase keys. Feature flags default off and are removed
|
|
36
|
+
|
|
37
|
+
within two releases of full rollout. Database migrations are forward-only;
|
|
38
|
+
|
|
39
|
+
rollbacks happen by writing a new migration. Review comments prefixed nit:
|
|
40
|
+
|
|
41
|
+
are optional. Vendored code lives under third_party/ and is never edited in
|
|
42
|
+
|
|
43
|
+
place — patch files only. Commit subjects use the imperative mood and stay
|
|
44
|
+
|
|
45
|
+
under 68 characters. Squash-merge is the default; merge commits are for
|
|
46
|
+
|
|
47
|
+
release branches only. Generated files carry a do-not-edit header and are
|
|
48
|
+
|
|
49
|
+
regenerated in CI, never by hand.
|
|
50
|
+
|
|
51
|
+
'
|
|
52
|
+
ci.md: '# CI pipeline notes
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
The pipeline runs on every push: lint, unit tests, integration tests, and a
|
|
56
|
+
|
|
57
|
+
packaging step. Caching of the dependency layer cut average build time from
|
|
58
|
+
|
|
59
|
+
11 minutes to 6. Flaky tests are quarantined under tests/flaky/ and re-run
|
|
60
|
+
|
|
61
|
+
three times before being reported as failures. The nightly job additionally
|
|
62
|
+
|
|
63
|
+
runs the slow property-based suite and publishes coverage to the internal
|
|
64
|
+
|
|
65
|
+
dashboard. Artifacts older than 30 days are pruned automatically. If a
|
|
66
|
+
|
|
67
|
+
release tag is pushed, the pipeline also builds the signed bundle and
|
|
68
|
+
|
|
69
|
+
uploads it to the artifact store with a 90-day retention override. Secrets
|
|
70
|
+
|
|
71
|
+
are injected via the runner''s OIDC exchange; never commit tokens. When the
|
|
72
|
+
|
|
73
|
+
integration environment is unavailable the pipeline degrades to unit tests
|
|
74
|
+
|
|
75
|
+
only and marks the run amber, not red. Amber runs do not block merges but
|
|
76
|
+
|
|
77
|
+
three consecutive ambers page the on-call.
|
|
78
|
+
|
|
79
|
+
'
|
|
80
|
+
prompt: 'Find the production port of billing-svc from the workspace files. Start with list_files, then follow the documentation to the right file. Use list_files first, then read_file on every file you need — follow each pointer the files give you; do not answer from memory or guess. Answer with ONLY this JSON, nothing else: {"port": <number>}'
|
|
81
|
+
scoring:
|
|
82
|
+
kind: json_equal
|
|
83
|
+
expected:
|
|
84
|
+
port: 9443
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
name: nav-release-bundle
|
|
2
|
+
family: file-nav
|
|
3
|
+
tools:
|
|
4
|
+
- read_file
|
|
5
|
+
- list_files
|
|
6
|
+
workspace:
|
|
7
|
+
README.md: '# imgproc
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
Image processing CLI. Release process: docs/release.md.'
|
|
11
|
+
docs/release.md: '# Releasing imgproc
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
The bundle name is assembled as <name>-<version>.tar.gz where <name>
|
|
15
|
+
|
|
16
|
+
comes from package.cfg and <version> comes from VERSION. Both files
|
|
17
|
+
|
|
18
|
+
live at the repo root. Never hardcode either value.'
|
|
19
|
+
package.cfg: '[package]
|
|
20
|
+
|
|
21
|
+
name = imgproc-cli
|
|
22
|
+
|
|
23
|
+
license = apache-2.0
|
|
24
|
+
|
|
25
|
+
'
|
|
26
|
+
VERSION: '2.9.1
|
|
27
|
+
|
|
28
|
+
'
|
|
29
|
+
docs/ci.md: '# CI pipeline notes
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
The pipeline runs on every push: lint, unit tests, integration tests, and a
|
|
33
|
+
|
|
34
|
+
packaging step. Caching of the dependency layer cut average build time from
|
|
35
|
+
|
|
36
|
+
11 minutes to 6. Flaky tests are quarantined under tests/flaky/ and re-run
|
|
37
|
+
|
|
38
|
+
three times before being reported as failures. The nightly job additionally
|
|
39
|
+
|
|
40
|
+
runs the slow property-based suite and publishes coverage to the internal
|
|
41
|
+
|
|
42
|
+
dashboard. Artifacts older than 30 days are pruned automatically. If a
|
|
43
|
+
|
|
44
|
+
release tag is pushed, the pipeline also builds the signed bundle and
|
|
45
|
+
|
|
46
|
+
uploads it to the artifact store with a 90-day retention override. Secrets
|
|
47
|
+
|
|
48
|
+
are injected via the runner''s OIDC exchange; never commit tokens. When the
|
|
49
|
+
|
|
50
|
+
integration environment is unavailable the pipeline degrades to unit tests
|
|
51
|
+
|
|
52
|
+
only and marks the run amber, not red. Amber runs do not block merges but
|
|
53
|
+
|
|
54
|
+
three consecutive ambers page the on-call.
|
|
55
|
+
|
|
56
|
+
'
|
|
57
|
+
docs/style.md: '# Style guide (excerpt)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
Prefer small modules with one clear responsibility. Public functions carry
|
|
61
|
+
|
|
62
|
+
docstrings describing behavior, not implementation. Errors are values where
|
|
63
|
+
|
|
64
|
+
practical; exceptions for programmer mistakes. Log lines are structured
|
|
65
|
+
|
|
66
|
+
key=value pairs, lowercase keys. Feature flags default off and are removed
|
|
67
|
+
|
|
68
|
+
within two releases of full rollout. Database migrations are forward-only;
|
|
69
|
+
|
|
70
|
+
rollbacks happen by writing a new migration. Review comments prefixed nit:
|
|
71
|
+
|
|
72
|
+
are optional. Vendored code lives under third_party/ and is never edited in
|
|
73
|
+
|
|
74
|
+
place — patch files only. Commit subjects use the imperative mood and stay
|
|
75
|
+
|
|
76
|
+
under 68 characters. Squash-merge is the default; merge commits are for
|
|
77
|
+
|
|
78
|
+
release branches only. Generated files carry a do-not-edit header and are
|
|
79
|
+
|
|
80
|
+
regenerated in CI, never by hand.
|
|
81
|
+
|
|
82
|
+
'
|
|
83
|
+
prompt: 'What is the exact release bundle filename for the current version? Follow the release documentation and assemble it from the files it names. Use list_files first, then read_file on every file you need — follow each pointer the files give you; do not answer from memory or guess. Answer with ONLY this JSON, nothing else: {"bundle": "<filename>"}'
|
|
84
|
+
scoring:
|
|
85
|
+
kind: json_equal
|
|
86
|
+
expected:
|
|
87
|
+
bundle: imgproc-cli-2.9.1.tar.gz
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
name: recall-audit-retention
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'How many days do we keep AUDIT logs? If you have a memory tool, check memory
|
|
4
|
+
first. Answer with ONLY this JSON, nothing else: {"days": <number>}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: log-retention-default
|
|
8
|
+
description: how long ordinary application logs are kept
|
|
9
|
+
body: Application logs are kept for 45 days.
|
|
10
|
+
- type: project
|
|
11
|
+
name: log-retention-audit
|
|
12
|
+
description: retention exception for the audit trail
|
|
13
|
+
body: Audit logs are the exception and are kept for 400 days.
|
|
14
|
+
scoring:
|
|
15
|
+
kind: json_equal
|
|
16
|
+
expected:
|
|
17
|
+
days: 400
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
name: recall-cache-ttl
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'What is the cache TTL for the pricing service? If you have a memory tool,
|
|
4
|
+
check memory first. Answer with ONLY this JSON, nothing else: {"seconds": <number>}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: pricing-cache-ttl
|
|
8
|
+
description: cache ttl seconds for the pricing service
|
|
9
|
+
body: The pricing service caches responses for 240 seconds.
|
|
10
|
+
scoring:
|
|
11
|
+
kind: json_equal
|
|
12
|
+
expected:
|
|
13
|
+
seconds: 240
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
name: recall-db-port
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'What port does the STAGING database listen on? If you have a memory tool,
|
|
4
|
+
check memory first. Answer with ONLY this JSON, nothing else: {"port": <number>}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: staging-db-port
|
|
8
|
+
description: staging database port number
|
|
9
|
+
body: The staging database listens on port 5433.
|
|
10
|
+
- type: project
|
|
11
|
+
name: prod-db-connection
|
|
12
|
+
description: production postgres connection endpoint
|
|
13
|
+
body: Production postgres is at db.prod.internal on port 5432.
|
|
14
|
+
scoring:
|
|
15
|
+
kind: json_equal
|
|
16
|
+
expected:
|
|
17
|
+
port: 5433
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
name: recall-deploy
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'How do we deploy this project to production? If you have a memory tool,
|
|
4
|
+
check memory first. Answer with ONLY this JSON, nothing else: {"command": "<command>"}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: deploy-command
|
|
8
|
+
description: how we deploy this project to production
|
|
9
|
+
body: Deploy with `make ship-prod` from the repo root.
|
|
10
|
+
scoring:
|
|
11
|
+
kind: json_equal
|
|
12
|
+
expected:
|
|
13
|
+
command: make ship-prod
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
name: recall-env-endpoint
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'What is the full base URL for the reports API in PRODUCTION, including
|
|
4
|
+
the version prefix? If you have a memory tool, check memory first. Answer with
|
|
5
|
+
ONLY this JSON, nothing else: {"url": "<url>"}'
|
|
6
|
+
memory_setup:
|
|
7
|
+
- type: project
|
|
8
|
+
name: prod-api-host
|
|
9
|
+
description: hostname serving production traffic
|
|
10
|
+
body: Production traffic is served from https://api.example-prod.io.
|
|
11
|
+
- type: project
|
|
12
|
+
name: reports-version-prefix
|
|
13
|
+
description: version path segment used by the reports service
|
|
14
|
+
body: The reports service is mounted under /v3/reports on every host.
|
|
15
|
+
scoring:
|
|
16
|
+
kind: json_equal
|
|
17
|
+
expected:
|
|
18
|
+
url: https://api.example-prod.io/v3/reports
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: recall-oncall-rotation
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'Who is on call for the PAYMENTS service this week? If you have a memory
|
|
4
|
+
tool, check memory first. Answer with ONLY this JSON, nothing else: {"name": "<name>"}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: oncall-payments
|
|
8
|
+
description: current pager duty for the payments service
|
|
9
|
+
body: Payments on-call this week is Priya.
|
|
10
|
+
- type: project
|
|
11
|
+
name: oncall-search
|
|
12
|
+
description: rotation owner covering search infrastructure
|
|
13
|
+
body: Search on-call this week is Marcus.
|
|
14
|
+
- type: project
|
|
15
|
+
name: oncall-ingest
|
|
16
|
+
description: escalation contact for the ingest pipeline
|
|
17
|
+
body: Ingest on-call this week is Dana.
|
|
18
|
+
scoring:
|
|
19
|
+
kind: json_equal
|
|
20
|
+
expected:
|
|
21
|
+
name: Priya
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
name: recall-oncall
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'Who is on-call for infrastructure this quarter? If you have a memory tool,
|
|
4
|
+
check memory first. Answer with ONLY this JSON, nothing else: {"name": "<name>"}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: infra-oncall
|
|
8
|
+
description: who is on-call for infrastructure this quarter
|
|
9
|
+
body: Nadia is on-call for infrastructure until end of Q3.
|
|
10
|
+
scoring:
|
|
11
|
+
kind: json_equal
|
|
12
|
+
expected:
|
|
13
|
+
name: Nadia
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
name: recall-org-quota
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'What is the TOTAL requests-per-minute quota for one full org on the api
|
|
4
|
+
gateway? If you have a memory tool, check memory first, then compute the answer.
|
|
5
|
+
Answer with ONLY this JSON, nothing else: {"total": <number>}'
|
|
6
|
+
memory_setup:
|
|
7
|
+
- type: project
|
|
8
|
+
name: gateway-user-quota
|
|
9
|
+
description: per-user rate limit on the api gateway
|
|
10
|
+
body: The api gateway allows each user 40 requests per minute.
|
|
11
|
+
- type: project
|
|
12
|
+
name: org-seat-count
|
|
13
|
+
description: how many seats one org licence includes
|
|
14
|
+
body: Every org licence includes exactly 5 user seats.
|
|
15
|
+
scoring:
|
|
16
|
+
kind: json_equal
|
|
17
|
+
expected:
|
|
18
|
+
total: 200
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
name: recall-owner
|
|
2
|
+
family: memory-recall
|
|
3
|
+
prompt: 'Which team owns the payments API? If you have a memory tool, check memory
|
|
4
|
+
first. Answer with ONLY this JSON, nothing else: {"team": "<team name>"}'
|
|
5
|
+
memory_setup:
|
|
6
|
+
- type: project
|
|
7
|
+
name: payments-api-owner
|
|
8
|
+
description: which team owns the payments api
|
|
9
|
+
body: The payments API is owned by team Atlas.
|
|
10
|
+
scoring:
|
|
11
|
+
kind: json_equal
|
|
12
|
+
expected:
|
|
13
|
+
team: Atlas
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
name: shop-basket-total
|
|
2
|
+
family: tool-use
|
|
3
|
+
prompt: |
|
|
4
|
+
A customer orders 2 widgets, 3 doohickeys and 1 gadget. Use the tools to
|
|
5
|
+
look up unit prices, then answer with ONLY this JSON, nothing else:
|
|
6
|
+
{"total": <number>}
|
|
7
|
+
tools: [price_lookup]
|
|
8
|
+
scoring:
|
|
9
|
+
kind: json_equal
|
|
10
|
+
expected: {total: 131}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
name: shop-cheapest
|
|
2
|
+
family: tool-use
|
|
3
|
+
prompt: |
|
|
4
|
+
Use the tools to check the unit prices of "widget" and "gadget".
|
|
5
|
+
Answer with ONLY this JSON, nothing else: {"cheaper": "<item name>"}
|
|
6
|
+
tools: [price_lookup]
|
|
7
|
+
scoring:
|
|
8
|
+
kind: json_equal
|
|
9
|
+
expected: {cheaper: "widget"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
name: shop-compare
|
|
2
|
+
family: tool-use
|
|
3
|
+
prompt: |
|
|
4
|
+
Use the tools to check the unit prices of "widget" and "gadget",
|
|
5
|
+
and answer with the name of the more expensive item.
|
|
6
|
+
tools: [price_lookup]
|
|
7
|
+
scoring:
|
|
8
|
+
kind: tool_trace
|
|
9
|
+
expected: [price_lookup, price_lookup]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
name: shop-gadget-value
|
|
2
|
+
family: tool-use
|
|
3
|
+
prompt: |
|
|
4
|
+
Use the tools to find the unit price and stock count of "gadget",
|
|
5
|
+
then answer with the total value of the stock (price times stock) as a number.
|
|
6
|
+
tools: [price_lookup, stock_lookup]
|
|
7
|
+
scoring:
|
|
8
|
+
kind: contains
|
|
9
|
+
expected: ["540"]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
name: shop-total
|
|
2
|
+
family: tool-use
|
|
3
|
+
prompt: |
|
|
4
|
+
Use the tools to find the unit price and stock count of "widget",
|
|
5
|
+
then answer with the total value of the stock (price times stock) as a number.
|
|
6
|
+
tools: [price_lookup, stock_lookup]
|
|
7
|
+
scoring:
|
|
8
|
+
kind: contains
|
|
9
|
+
expected: ["100"]
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# The conservative default profile. Every number here was calibrated on
|
|
2
|
+
# qwen3:4b-instruct (the reference model) — the cross-model sweep measured
|
|
3
|
+
# that honestly; per-model profiles are future work (P6).
|
|
4
|
+
name: default
|
|
5
|
+
agent:
|
|
6
|
+
max_turns: 10
|
|
7
|
+
observation_budget: 4096
|
|
8
|
+
structured:
|
|
9
|
+
max_retries: 3
|
|
10
|
+
schema_gate:
|
|
11
|
+
max_attempts: 3
|
|
12
|
+
# One retry only: the gate rescues a missing-JSON restatement, and a second
|
|
13
|
+
# miss is format abandonment that more nagging does not fix (spec §2.4).
|
|
14
|
+
json_answer:
|
|
15
|
+
max_attempts: 1
|
|
16
|
+
critique:
|
|
17
|
+
max_rounds: 3
|
|
18
|
+
evidence_budget: 4096
|
|
19
|
+
# Global spend governor (P3), applied only when a TokenBudget is attached.
|
|
20
|
+
# `ceiling` is the hard stop for one run; past `ceiling * optional_cutoff`
|
|
21
|
+
# optional work (critique rounds) is denied. The gap between the two IS the
|
|
22
|
+
# reserve that keeps the final answer emission affordable.
|
|
23
|
+
token_budget:
|
|
24
|
+
ceiling: 6000
|
|
25
|
+
optional_cutoff: 0.75
|
|
26
|
+
# Loop-detection thresholds, applied only when a LoopGuard is attached. Probe-
|
|
27
|
+
# calibrated: max observation-repeat streak in every passing run was 2, looping
|
|
28
|
+
# runs burned 6-16 — so the note fires at 3 and the hard warning at 5.
|
|
29
|
+
loop_guard:
|
|
30
|
+
inject_at: 3
|
|
31
|
+
warn_at: 5
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Default, but with a longer turn budget: for models that need more steps than
|
|
2
|
+
# the 4b-calibrated default (measured: 3b file-nav and 7b recall died of turn
|
|
3
|
+
# exhaustion under max_turns 10). Calibration-only until a bar says otherwise.
|
|
4
|
+
name: patient
|
|
5
|
+
agent:
|
|
6
|
+
max_turns: 16
|
|
7
|
+
observation_budget: 4096
|
|
8
|
+
structured:
|
|
9
|
+
max_retries: 3
|
|
10
|
+
schema_gate:
|
|
11
|
+
max_attempts: 3
|
|
12
|
+
# One retry only: the gate rescues a missing-JSON restatement, and a second
|
|
13
|
+
# miss is format abandonment that more nagging does not fix (spec §2.4).
|
|
14
|
+
json_answer:
|
|
15
|
+
max_attempts: 1
|
|
16
|
+
critique:
|
|
17
|
+
max_rounds: 3
|
|
18
|
+
evidence_budget: 4096
|
|
19
|
+
# Global spend governor (P3), applied only when a TokenBudget is attached.
|
|
20
|
+
# `ceiling` is the hard stop for one run; past `ceiling * optional_cutoff`
|
|
21
|
+
# optional work (critique rounds) is denied. The gap between the two IS the
|
|
22
|
+
# reserve that keeps the final answer emission affordable.
|
|
23
|
+
token_budget:
|
|
24
|
+
ceiling: 6000
|
|
25
|
+
optional_cutoff: 0.75
|
|
26
|
+
# Loop-detection thresholds, applied only when a LoopGuard is attached. Probe-
|
|
27
|
+
# calibrated: max observation-repeat streak in every passing run was 2, looping
|
|
28
|
+
# runs burned 6-16 — so the note fires at 3 and the hard warning at 5.
|
|
29
|
+
loop_guard:
|
|
30
|
+
inject_at: 3
|
|
31
|
+
warn_at: 5
|
|
File without changes
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
name: code-quality
|
|
2
|
+
threshold: 7
|
|
3
|
+
schema:
|
|
4
|
+
type: object
|
|
5
|
+
required: [score, feedback]
|
|
6
|
+
properties:
|
|
7
|
+
score: {type: integer, minimum: 0, maximum: 10}
|
|
8
|
+
feedback: {type: string}
|
|
9
|
+
prompt: |
|
|
10
|
+
You are a strict code reviewer. Judge the code below.
|
|
11
|
+
|
|
12
|
+
Task:
|
|
13
|
+
{task}
|
|
14
|
+
|
|
15
|
+
Code:
|
|
16
|
+
{output}
|
|
17
|
+
|
|
18
|
+
Score 0-10 on: correctness for the task, handling of error cases, and
|
|
19
|
+
absence of dead or needless code. 10 = ship as-is.
|
|
20
|
+
Return ONLY JSON: {{"score": <int>, "feedback": "<specific defects to fix>"}}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
name: grounded-completion
|
|
2
|
+
threshold: 7
|
|
3
|
+
schema:
|
|
4
|
+
type: object
|
|
5
|
+
required: [reasoning, score, feedback]
|
|
6
|
+
properties:
|
|
7
|
+
reasoning: {type: string}
|
|
8
|
+
score: {type: integer, minimum: 0, maximum: 10}
|
|
9
|
+
feedback: {type: string}
|
|
10
|
+
prompt: |
|
|
11
|
+
You are a reviewer checking whether the answer contains the correct content.
|
|
12
|
+
The tool evidence below is the ground truth: it lists every tool call the
|
|
13
|
+
answerer made and what the tool returned. Verify the answer against it and
|
|
14
|
+
recompute any numbers yourself from the evidence. If the answer states a
|
|
15
|
+
fact or number that contradicts the evidence, score it 0-4 and put the
|
|
16
|
+
correct values from the evidence in your feedback.
|
|
17
|
+
In the reasoning field, first work out what the correct answer to the task
|
|
18
|
+
is, step by step, using ONLY the values in the evidence — do the arithmetic
|
|
19
|
+
and comparisons yourself. Then compare the given answer against your result.
|
|
20
|
+
Judge ONLY content. Do NOT deduct points for formatting, phrasing, extra
|
|
21
|
+
surrounding text, hedging, or verbosity. An answer that refuses or declines
|
|
22
|
+
to provide what the task asks for is missing the required content — score
|
|
23
|
+
it 0-4, even when the refusal is polite or explains itself.
|
|
24
|
+
|
|
25
|
+
Task:
|
|
26
|
+
{task}
|
|
27
|
+
|
|
28
|
+
Tool evidence:
|
|
29
|
+
{evidence}
|
|
30
|
+
|
|
31
|
+
Answer:
|
|
32
|
+
{output}
|
|
33
|
+
|
|
34
|
+
Score 0-10: 9-10 = required content present, correct, and consistent with
|
|
35
|
+
the evidence; 5-8 = partially correct or missing pieces; 0-4 = wrong,
|
|
36
|
+
absent, or contradicted by the evidence.
|
|
37
|
+
Return ONLY JSON: {{"reasoning": "<derive the correct answer from the evidence, then compare the given answer to it>", "score": <int>, "feedback": "<what is wrong or missing, and the correct values from the evidence>"}}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
name: task-completion
|
|
2
|
+
threshold: 7
|
|
3
|
+
schema:
|
|
4
|
+
type: object
|
|
5
|
+
required: [score, feedback]
|
|
6
|
+
properties:
|
|
7
|
+
score: {type: integer, minimum: 0, maximum: 10}
|
|
8
|
+
feedback: {type: string}
|
|
9
|
+
prompt: |
|
|
10
|
+
You are a reviewer checking whether the answer contains the correct content.
|
|
11
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
12
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
13
|
+
hedging, or verbosity. If the required facts are present and right, the
|
|
14
|
+
answer completes the task.
|
|
15
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
16
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
17
|
+
polite or explains itself. Hedging around a real answer is fine; hedging
|
|
18
|
+
instead of an answer is not.
|
|
19
|
+
|
|
20
|
+
Task:
|
|
21
|
+
{task}
|
|
22
|
+
|
|
23
|
+
Answer:
|
|
24
|
+
{output}
|
|
25
|
+
|
|
26
|
+
Score 0-10: 9-10 = required content present and correct; 5-8 = partially
|
|
27
|
+
correct or missing pieces; 0-4 = wrong or absent.
|
|
28
|
+
Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
|