bantamkit 0.27.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. bantamkit/__init__.py +32 -0
  2. bantamkit/agent.py +458 -0
  3. bantamkit/assets/contracts/default.yaml +90 -0
  4. bantamkit/assets/evals/devteam/manifest.yaml +351 -0
  5. bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
  6. bantamkit/assets/evals/devteam/repo/README.md +12 -0
  7. bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
  8. bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
  9. bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
  10. bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
  11. bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
  12. bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
  13. bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
  14. bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
  15. bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
  16. bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
  17. bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
  18. bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
  19. bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
  20. bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
  21. bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
  22. bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
  23. bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
  24. bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
  25. bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
  26. bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
  27. bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
  28. bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
  29. bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
  30. bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
  31. bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
  32. bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
  33. bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
  34. bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
  35. bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
  36. bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
  37. bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
  38. bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
  39. bantamkit/assets/evals/fixtures/.gitkeep +0 -0
  40. bantamkit/assets/evals/fixtures/catalog.json +6 -0
  41. bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
  42. bantamkit/assets/evals/tasks/.gitkeep +0 -0
  43. bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
  44. bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
  45. bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
  46. bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
  47. bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
  48. bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
  49. bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
  50. bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
  51. bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
  52. bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
  53. bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
  54. bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
  55. bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
  56. bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
  57. bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
  58. bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
  59. bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
  60. bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
  61. bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
  62. bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
  63. bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
  64. bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
  65. bantamkit/assets/profiles/default.yaml +31 -0
  66. bantamkit/assets/profiles/patient.yaml +31 -0
  67. bantamkit/assets/rubrics/.gitkeep +0 -0
  68. bantamkit/assets/rubrics/code-quality.yaml +20 -0
  69. bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
  70. bantamkit/assets/rubrics/task-completion.yaml +28 -0
  71. bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
  72. bantamkit/assets/skills/.gitkeep +0 -0
  73. bantamkit/assets/skills/file-graph.md +7 -0
  74. bantamkit/assets/skills/memory.md +35 -0
  75. bantamkit/assets/tools/.gitkeep +0 -0
  76. bantamkit/assets/tools/bantamkit_read.json +48 -0
  77. bantamkit/assets/tools/bantamkit_status.json +25 -0
  78. bantamkit/assets/tools/build_identity.json +17 -0
  79. bantamkit/assets/tools/document_list.json +12 -0
  80. bantamkit/assets/tools/document_read.json +31 -0
  81. bantamkit/assets/tools/file_graph.json +12 -0
  82. bantamkit/assets/tools/memory_compact.json +31 -0
  83. bantamkit/assets/tools/memory_recall.json +38 -0
  84. bantamkit/assets/tools/memory_save.json +61 -0
  85. bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
  86. bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
  87. bantamkit/assets/tools/shiftwork_status.json +25 -0
  88. bantamkit/assets/tools/skill_audit.json +70 -0
  89. bantamkit/assets/tools/validate_json.json +31 -0
  90. bantamkit/assets.py +67 -0
  91. bantamkit/budget.py +114 -0
  92. bantamkit/client.py +329 -0
  93. bantamkit/contract.py +522 -0
  94. bantamkit/criticreplay.py +3241 -0
  95. bantamkit/critique.py +301 -0
  96. bantamkit/docread.py +1744 -0
  97. bantamkit/evalrun.py +2003 -0
  98. bantamkit/eventlog.py +282 -0
  99. bantamkit/filegraph.py +218 -0
  100. bantamkit/loopguard.py +101 -0
  101. bantamkit/mcpreport.py +763 -0
  102. bantamkit/mcpserver.py +1334 -0
  103. bantamkit/memory/__init__.py +28 -0
  104. bantamkit/memory/__main__.py +291 -0
  105. bantamkit/memory/component.py +569 -0
  106. bantamkit/memory/divergence.py +744 -0
  107. bantamkit/memory/layers.py +257 -0
  108. bantamkit/memory/store.py +940 -0
  109. bantamkit/pdfread.py +1402 -0
  110. bantamkit/profile.py +46 -0
  111. bantamkit/shiftwork.py +212 -0
  112. bantamkit/skillaudit.py +853 -0
  113. bantamkit/statusline.py +313 -0
  114. bantamkit/structured.py +125 -0
  115. bantamkit/textutil.py +30 -0
  116. bantamkit-0.27.0.dist-info/METADATA +207 -0
  117. bantamkit-0.27.0.dist-info/RECORD +119 -0
  118. bantamkit-0.27.0.dist-info/WHEEL +4 -0
  119. bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,37 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell SMALL-IN. Corpus `inventory-small.xlsx`, 400 data rows, 8,620 extracted bytes
6
+ # (~2,155 est. tokens, 0.07x WORKER_NUM_CTX). At PASTE_MAX_BYTES = 8,621 (Amendment 1,
7
+ # 2026-08-20; pre-registration read 12,288) the
8
+ # WHOLE corpus fits, so the `paste` arm here is a COMPLETE paste. This is the cell that
9
+ # separates "the reader works" from "paging works".
10
+ name: doc-small-261
11
+ family: document-read
12
+ tools: []
13
+ document_setup:
14
+ - path: inventory-small.xlsx
15
+ seed: 4021
16
+ sheets:
17
+ - name: stock
18
+ rows: 400
19
+ columns:
20
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
21
+ - {name: region, kind: choice, values: [north, south, east, west]}
22
+ - {name: units, kind: int, low: 1000, high: 9999}
23
+ answers:
24
+ question_sku: stock!A262
25
+ expected_region: stock!B262
26
+ expected_units: stock!C262
27
+ prompt: >-
28
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
29
+ and units, one row per sku. Find the single row whose sku is exactly SKU-000261 and
30
+ report that row's region and units. Do not compute anything and do not summarise the
31
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
32
+ {"region": "<region>", "units": <integer>}
33
+ scoring:
34
+ kind: json_equal
35
+ expected:
36
+ region: south
37
+ units: 3788
@@ -0,0 +1,37 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell SMALL-IN. Corpus `inventory-small.xlsx`, 400 data rows, 8,620 extracted bytes
6
+ # (~2,155 est. tokens, 0.07x WORKER_NUM_CTX). At PASTE_MAX_BYTES = 8,621 (Amendment 1,
7
+ # 2026-08-20; pre-registration read 12,288) the
8
+ # WHOLE corpus fits, so the `paste` arm here is a COMPLETE paste. This is the cell that
9
+ # separates "the reader works" from "paging works".
10
+ name: doc-small-388
11
+ family: document-read
12
+ tools: []
13
+ document_setup:
14
+ - path: inventory-small.xlsx
15
+ seed: 4021
16
+ sheets:
17
+ - name: stock
18
+ rows: 400
19
+ columns:
20
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
21
+ - {name: region, kind: choice, values: [north, south, east, west]}
22
+ - {name: units, kind: int, low: 1000, high: 9999}
23
+ answers:
24
+ question_sku: stock!A389
25
+ expected_region: stock!B389
26
+ expected_units: stock!C389
27
+ prompt: >-
28
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
29
+ and units, one row per sku. Find the single row whose sku is exactly SKU-000388 and
30
+ report that row's region and units. Do not compute anything and do not summarise the
31
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
32
+ {"region": "<region>", "units": <integer>}
33
+ scoring:
34
+ kind: json_equal
35
+ expected:
36
+ region: east
37
+ units: 5208
File without changes
@@ -0,0 +1,6 @@
1
+ {
2
+ "widget": {"price": 25, "stock": 4},
3
+ "gadget": {"price": 60, "stock": 9},
4
+ "doohickey": {"price": 7, "stock": 12},
5
+ "sprocket": {"price": 14, "stock": 3}
6
+ }
@@ -0,0 +1,576 @@
1
+ rubric: task-completion
2
+ spec: docs/superpowers/specs/2026-08-11-perturbation-bar-spec.md
3
+ note: 'Perturbation family for the RB-P14 bar. Measurement input, NOT a Contract asset: no product code
4
+ path loads it, and criticreplay.py is its only reader. Every point is a declared, text-anchored transformation
5
+ with a stable rule id, applicable to any template that contains its anchor, so the same family lands
6
+ on every rubric variant under comparison. A point whose anchor is absent from a variant is dropped from
7
+ BOTH families of a pairwise comparison and named in the summary.'
8
+ requirement_inventory_note: 'The requirement inventory of the AS-FILED task-completion rubric (spec section
9
+ 3.3 step 1), committed before any paraphrase was authored. Admissibility test for every P point: re-derive
10
+ this list from the perturbed text alone; it must come out the same - same count, same directive per
11
+ item, same bands, same key names. A variant that changes the inventory (C-attempted adds a required
12
+ `reasoning` key) is a different rubric under test, not a perturbation.'
13
+ requirement_inventory:
14
+ - Judge whether the information the task asks for is present and correct.
15
+ - Do not deduct for formatting.
16
+ - Do not deduct for phrasing.
17
+ - Do not deduct for extra surrounding text.
18
+ - Do not deduct for hedging.
19
+ - Do not deduct for verbosity.
20
+ - Facts present and right => the task is complete.
21
+ - A refusal or decline => 0-4, however polite.
22
+ - Hedging instead of an answer => missing content; hedging around an answer => fine.
23
+ - 'Bands: 9-10 present and correct; 5-8 partial or missing pieces; 0-4 wrong or absent.'
24
+ - Output only JSON with keys score (int) and feedback (string).
25
+ materialized_variants:
26
+ A-asfiled:
27
+ source: git:d2f78b7:assets/rubrics/task-completion.yaml
28
+ note: byte-identical to the shipped assets/rubrics/task-completion.yaml
29
+ base_sha256: e018854368c1b675e7cff5109a3dce715d59c86d871cd3d1d83b9083188a065e
30
+ B-nonewline:
31
+ source: A-asfiled with the template's single trailing newline removed
32
+ note: 'materialize as the same YAML with `prompt: |-`'
33
+ base_sha256: d1f32ad2947b4d6f6079833847eae96c79fddeb2683ceda322940bdc8cbf13a6
34
+ C-attempted:
35
+ source: git:e57f1a6:assets/rubrics/task-completion.yaml
36
+ note: the withdrawn derive-before-score rubric; requires `reasoning` on the wire
37
+ base_sha256: 5f588e4a07a6fa29572ed2e3bdbd39132c0a8f9908b0ce47408ad72c015d2298
38
+ points:
39
+ - id: identity
40
+ class: identity
41
+ rule: identity
42
+ op: identity
43
+ note: Mandatory member, not a perturbation class (spec section 3.4). Zero edits. Its replays give the
44
+ pure within-cell replay spread RB-P15 asks for, and anchor the family against the committed SA3 replay
45
+ record.
46
+ variants:
47
+ A-asfiled:
48
+ applicable: true
49
+ sha256: e018854368c1b675e7cff5109a3dce715d59c86d871cd3d1d83b9083188a065e
50
+ diff: ''
51
+ B-nonewline:
52
+ applicable: true
53
+ sha256: d1f32ad2947b4d6f6079833847eae96c79fddeb2683ceda322940bdc8cbf13a6
54
+ diff: ''
55
+ C-attempted:
56
+ applicable: true
57
+ sha256: 5f588e4a07a6fa29572ed2e3bdbd39132c0a8f9908b0ce47408ad72c015d2298
58
+ diff: ''
59
+ - id: W1-trailing-newline
60
+ class: whitespace
61
+ rule: strip-trailing-newline
62
+ op: strip-trailing-newline
63
+ note: THE NULL CONTROL, and a mandatory member. This is the edit that reproduced the entire RB-P4 pass
64
+ signature while changing no word of the rubric. Shortening.
65
+ variants:
66
+ A-asfiled:
67
+ applicable: true
68
+ sha256: d1f32ad2947b4d6f6079833847eae96c79fddeb2683ceda322940bdc8cbf13a6
69
+ diff: |-
70
+ --- A-asfiled
71
+ +++ W1-trailing-newline
72
+ @@ -19 +19,2 @@
73
+ Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
74
+ +
75
+ B-nonewline:
76
+ applicable: false
77
+ C-attempted:
78
+ applicable: true
79
+ sha256: 4054b3b6c5c328a4bcfc0f7191958f10b6f9ad4fd2f3f0f4dbdbd0c231c4d94f
80
+ diff: |-
81
+ --- C-attempted
82
+ +++ W1-trailing-newline
83
+ @@ -27 +27,2 @@
84
+ Return ONLY JSON: {{"reasoning": "<the fact the task asks for, then the value the answer supplies for it>", "score": <int>, "feedback": "<what content is wrong or missing>"}}
85
+ +
86
+ - id: W2-double-trailing
87
+ class: whitespace
88
+ rule: append-trailing-newline
89
+ op: append-trailing-newline
90
+ note: Lengthening. Paired with W1 so the family is not biased in one direction.
91
+ variants:
92
+ A-asfiled:
93
+ applicable: true
94
+ sha256: 447e5be27613ef5cd090f12e8e5d0616ff838bc22ef43c0001b625743b88b303
95
+ diff: |-
96
+ --- A-asfiled
97
+ +++ W2-double-trailing
98
+ @@ -19 +19,2 @@
99
+ Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
100
+ +
101
+ B-nonewline:
102
+ applicable: true
103
+ sha256: e018854368c1b675e7cff5109a3dce715d59c86d871cd3d1d83b9083188a065e
104
+ diff: |-
105
+ --- B-nonewline
106
+ +++ W2-double-trailing
107
+ @@ -19,2 +19 @@
108
+ Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
109
+ -
110
+ C-attempted:
111
+ applicable: true
112
+ sha256: 03bfdcb7b224a03e3e9d832e8b6bb0a0b77f375062498cc766084504a0e46a95
113
+ diff: |-
114
+ --- C-attempted
115
+ +++ W2-double-trailing
116
+ @@ -27 +27,2 @@
117
+ Return ONLY JSON: {{"reasoning": "<the fact the task asks for, then the value the answer supplies for it>", "score": <int>, "feedback": "<what content is wrong or missing>"}}
118
+ +
119
+ - id: W3-unwrap-opening
120
+ class: whitespace
121
+ rule: unwrap-hard-wrapped-run
122
+ op: replace
123
+ replace:
124
+ - from: |-
125
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
126
+ hedging, or verbosity. If the required facts are present and right, the
127
+ answer completes the task.
128
+ to: Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity.
129
+ If the required facts are present and right, the answer completes the task.
130
+ note: 'Anchored to the first hard-wrapped run of the SHARED opening paragraph rather than to the whole
131
+ opening paragraph the spec''s table names, because the whole paragraph differs between A-asfiled and
132
+ C-attempted and an anchor absent from a variant would drop this rule from the very comparison the
133
+ bar exists for (spec section 4.1). Length-neutral: newline -> space.'
134
+ variants:
135
+ A-asfiled:
136
+ applicable: true
137
+ sha256: 0a41447a57d8b987e170275280eafe6e2568c7cc3d3bdbd9583ffcddef29e067
138
+ diff: |-
139
+ --- A-asfiled
140
+ +++ W3-unwrap-opening
141
+ @@ -2,5 +2,3 @@
142
+ Judge ONLY whether the information the task asks for is present and correct.
143
+ -Do NOT deduct points for formatting, phrasing, extra surrounding text,
144
+ -hedging, or verbosity. If the required facts are present and right, the
145
+ -answer completes the task.
146
+ +Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity. If the required facts are present and right, the answer completes the task.
147
+ An answer that refuses or declines to provide what the task asks for is
148
+ B-nonewline:
149
+ applicable: true
150
+ sha256: dc6d35d1e4570a53a33a8b3c3bbacc6020fb305ecb3dd3c3fb94c27b891c0384
151
+ diff: |-
152
+ --- B-nonewline
153
+ +++ W3-unwrap-opening
154
+ @@ -2,5 +2,3 @@
155
+ Judge ONLY whether the information the task asks for is present and correct.
156
+ -Do NOT deduct points for formatting, phrasing, extra surrounding text,
157
+ -hedging, or verbosity. If the required facts are present and right, the
158
+ -answer completes the task.
159
+ +Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity. If the required facts are present and right, the answer completes the task.
160
+ An answer that refuses or declines to provide what the task asks for is
161
+ C-attempted:
162
+ applicable: true
163
+ sha256: c137757ae33652de54184d36e9e02a704150903952d8a53ef54fdc4ef3273309
164
+ diff: |-
165
+ --- C-attempted
166
+ +++ W3-unwrap-opening
167
+ @@ -10,5 +10,3 @@
168
+ Judge ONLY whether the information the task asks for is present and correct.
169
+ -Do NOT deduct points for formatting, phrasing, extra surrounding text,
170
+ -hedging, or verbosity. If the required facts are present and right, the
171
+ -answer completes the task.
172
+ +Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity. If the required facts are present and right, the answer completes the task.
173
+ An answer that refuses or declines to provide what the task asks for is
174
+ - id: W4-double-space
175
+ class: whitespace
176
+ rule: double-space-after-period
177
+ op: replace
178
+ replace:
179
+ - from: '. '
180
+ to: '. '
181
+ occurrences: all
182
+ note: Position-independent. Lengthening. Every sentence-terminating period.
183
+ variants:
184
+ A-asfiled:
185
+ applicable: true
186
+ sha256: 4c54e4d49cf74a0f5ed3bfabf347c405cd0a54eb2dd10c693b6a19e5dee23792
187
+ diff: |-
188
+ --- A-asfiled
189
+ +++ W4-double-space
190
+ @@ -3,3 +3,3 @@
191
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
192
+ -hedging, or verbosity. If the required facts are present and right, the
193
+ +hedging, or verbosity. If the required facts are present and right, the
194
+ answer completes the task.
195
+ @@ -7,3 +7,3 @@
196
+ missing the required content — score it 0-4, even when the refusal is
197
+ -polite or explains itself. Hedging around a real answer is fine; hedging
198
+ +polite or explains itself. Hedging around a real answer is fine; hedging
199
+ instead of an answer is not.
200
+ B-nonewline:
201
+ applicable: true
202
+ sha256: 82309ca57ce8d5ae23ce3ee1f7788eec0eabf1c7e8d0b91e8557052ec9b3a55a
203
+ diff: |-
204
+ --- B-nonewline
205
+ +++ W4-double-space
206
+ @@ -3,3 +3,3 @@
207
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
208
+ -hedging, or verbosity. If the required facts are present and right, the
209
+ +hedging, or verbosity. If the required facts are present and right, the
210
+ answer completes the task.
211
+ @@ -7,3 +7,3 @@
212
+ missing the required content — score it 0-4, even when the refusal is
213
+ -polite or explains itself. Hedging around a real answer is fine; hedging
214
+ +polite or explains itself. Hedging around a real answer is fine; hedging
215
+ instead of an answer is not.
216
+ C-attempted:
217
+ applicable: true
218
+ sha256: 44458e5ed6b1848464940598900f9956c841ae1cad72cc0f10df2759233b1338
219
+ diff: |-
220
+ --- C-attempted
221
+ +++ W4-double-space
222
+ @@ -4,6 +4,6 @@
223
+ written in; second, the value the answer supplies for that fact, quoted from
224
+ -the answer. Then say whether that value is present and responsive to what was
225
+ -asked. You cannot check the value against records: you have no files, no
226
+ +the answer. Then say whether that value is present and responsive to what was
227
+ +asked. You cannot check the value against records: you have no files, no
228
+ tools, and no evidence here, so do not try to work out what the value should
229
+ -have been — a value you invented would be a guess. You are checking that the
230
+ +have been — a value you invented would be a guess. You are checking that the
231
+ answer supplies the fact that was asked for.
232
+ @@ -11,3 +11,3 @@
233
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
234
+ -hedging, or verbosity. If the required facts are present and right, the
235
+ +hedging, or verbosity. If the required facts are present and right, the
236
+ answer completes the task.
237
+ @@ -15,3 +15,3 @@
238
+ missing the required content — score it 0-4, even when the refusal is
239
+ -polite or explains itself. Hedging around a real answer is fine; hedging
240
+ +polite or explains itself. Hedging around a real answer is fine; hedging
241
+ instead of an answer is not.
242
+ - id: W5-blank-line-before-bands
243
+ class: whitespace
244
+ rule: extra-blank-line-before-anchor
245
+ op: replace
246
+ replace:
247
+ - from: |2-
248
+
249
+
250
+ Score 0-10:
251
+ to: |2-
252
+
253
+
254
+
255
+ Score 0-10:
256
+ note: Anchored. Lengthening. Does not move any structural block.
257
+ variants:
258
+ A-asfiled:
259
+ applicable: true
260
+ sha256: f55493678f7d4f418291b885b16899251c0ad58fad78705eb915eb38393234f5
261
+ diff: "--- A-asfiled\n+++ W5-blank-line-before-bands\n@@ -16,2 +16,3 @@\n \n+\n Score 0-10: 9-10\
262
+ \ = required content present and correct; 5-8 = partially"
263
+ B-nonewline:
264
+ applicable: true
265
+ sha256: 92dbe03ae12ba4d56c3808e8e51190f1ba6d2b0a7a9256395eece3df351d44f9
266
+ diff: "--- B-nonewline\n+++ W5-blank-line-before-bands\n@@ -16,2 +16,3 @@\n \n+\n Score 0-10: 9-10\
267
+ \ = required content present and correct; 5-8 = partially"
268
+ C-attempted:
269
+ applicable: true
270
+ sha256: bab6a5c4f792037d4c0df2d335c7e298b51bb676afc2d0bc416293be17ecfcb7
271
+ diff: "--- C-attempted\n+++ W5-blank-line-before-bands\n@@ -24,2 +24,3 @@\n \n+\n Score 0-10: 9-10\
272
+ \ = required content present and correct; 5-8 = partially"
273
+ - id: O1-swap-format-refusal
274
+ class: order
275
+ rule: swap-sentences
276
+ op: swap
277
+ swap:
278
+ a: |-
279
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
280
+ hedging, or verbosity.
281
+ b: |-
282
+ An answer that refuses or declines to provide what the task asks for is
283
+ missing the required content — score it 0-4, even when the refusal is
284
+ polite or explains itself.
285
+ note: 'Both units are whole sentences inside ONE contiguous prose paragraph and neither carries an anaphor
286
+ or discourse connective pointing outside itself. No structural block moves: Task:/Answer:/Return ONLY
287
+ JSON: keep their positions.'
288
+ variants:
289
+ A-asfiled:
290
+ applicable: true
291
+ sha256: b6a0693eacbd525df6e53b8021c39af9d6c7b15504fe3d975c0d0e17d5e847c0
292
+ diff: |-
293
+ --- A-asfiled
294
+ +++ O1-swap-format-refusal
295
+ @@ -2,8 +2,8 @@
296
+ Judge ONLY whether the information the task asks for is present and correct.
297
+ -Do NOT deduct points for formatting, phrasing, extra surrounding text,
298
+ -hedging, or verbosity. If the required facts are present and right, the
299
+ -answer completes the task.
300
+ An answer that refuses or declines to provide what the task asks for is
301
+ missing the required content — score it 0-4, even when the refusal is
302
+ -polite or explains itself. Hedging around a real answer is fine; hedging
303
+ +polite or explains itself. If the required facts are present and right, the
304
+ +answer completes the task.
305
+ +Do NOT deduct points for formatting, phrasing, extra surrounding text,
306
+ +hedging, or verbosity. Hedging around a real answer is fine; hedging
307
+ instead of an answer is not.
308
+ B-nonewline:
309
+ applicable: true
310
+ sha256: f8fecf8b2315bbe7b0505a6642e9dcc65dee259b18eac146302ebbe6b305fd5d
311
+ diff: |-
312
+ --- B-nonewline
313
+ +++ O1-swap-format-refusal
314
+ @@ -2,8 +2,8 @@
315
+ Judge ONLY whether the information the task asks for is present and correct.
316
+ -Do NOT deduct points for formatting, phrasing, extra surrounding text,
317
+ -hedging, or verbosity. If the required facts are present and right, the
318
+ -answer completes the task.
319
+ An answer that refuses or declines to provide what the task asks for is
320
+ missing the required content — score it 0-4, even when the refusal is
321
+ -polite or explains itself. Hedging around a real answer is fine; hedging
322
+ +polite or explains itself. If the required facts are present and right, the
323
+ +answer completes the task.
324
+ +Do NOT deduct points for formatting, phrasing, extra surrounding text,
325
+ +hedging, or verbosity. Hedging around a real answer is fine; hedging
326
+ instead of an answer is not.
327
+ C-attempted:
328
+ applicable: true
329
+ sha256: dc8f94e075f52a59437b63c6a70dc6b888e7fe24ea86f6165dfe9274dda37519
330
+ diff: |-
331
+ --- C-attempted
332
+ +++ O1-swap-format-refusal
333
+ @@ -10,8 +10,8 @@
334
+ Judge ONLY whether the information the task asks for is present and correct.
335
+ -Do NOT deduct points for formatting, phrasing, extra surrounding text,
336
+ -hedging, or verbosity. If the required facts are present and right, the
337
+ -answer completes the task.
338
+ An answer that refuses or declines to provide what the task asks for is
339
+ missing the required content — score it 0-4, even when the refusal is
340
+ -polite or explains itself. Hedging around a real answer is fine; hedging
341
+ +polite or explains itself. If the required facts are present and right, the
342
+ +answer completes the task.
343
+ +Do NOT deduct points for formatting, phrasing, extra surrounding text,
344
+ +hedging, or verbosity. Hedging around a real answer is fine; hedging
345
+ instead of an answer is not.
346
+ - id: O2-bands-ascending
347
+ class: order
348
+ rule: reorder-list-items
349
+ op: replace
350
+ replace:
351
+ - from: |-
352
+ 9-10 = required content present and correct; 5-8 = partially
353
+ correct or missing pieces; 0-4 = wrong or absent.
354
+ to: |-
355
+ 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =
356
+ required content present and correct.
357
+ note: Each band states its own range and its own criterion, so each is a self-contained semicolon-delimited
358
+ list item. The band digits 9-10 / 5-8 / 0-4 are byte-frozen.
359
+ variants:
360
+ A-asfiled:
361
+ applicable: true
362
+ sha256: 740cbc7c7289238ef717bd25e8e846ed70200e9d45d7e0c0e3ecf7e5d4474830
363
+ diff: "--- A-asfiled\n+++ O2-bands-ascending\n@@ -16,4 +16,4 @@\n \n-Score 0-10: 9-10 = required\
364
+ \ content present and correct; 5-8 = partially\n-correct or missing pieces; 0-4 = wrong or absent.\n\
365
+ +Score 0-10: 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =\n+required\
366
+ \ content present and correct.\n Return ONLY JSON: {{\"score\": <int>, \"feedback\": \"<what content\
367
+ \ is wrong or missing>\"}}"
368
+ B-nonewline:
369
+ applicable: true
370
+ sha256: 1aa55cc1112f4f4491c77abdbce76a2b4c1a7f8af4e410095a1d704e2c9f876e
371
+ diff: "--- B-nonewline\n+++ O2-bands-ascending\n@@ -16,4 +16,4 @@\n \n-Score 0-10: 9-10 = required\
372
+ \ content present and correct; 5-8 = partially\n-correct or missing pieces; 0-4 = wrong or absent.\n\
373
+ +Score 0-10: 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =\n+required\
374
+ \ content present and correct.\n Return ONLY JSON: {{\"score\": <int>, \"feedback\": \"<what content\
375
+ \ is wrong or missing>\"}}"
376
+ C-attempted:
377
+ applicable: true
378
+ sha256: 69eed0c1e3ef2dae255e1ceefcac162cc0996a4fecaffacecb63c8c28fe41a47
379
+ diff: "--- C-attempted\n+++ O2-bands-ascending\n@@ -24,4 +24,4 @@\n \n-Score 0-10: 9-10 = required\
380
+ \ content present and correct; 5-8 = partially\n-correct or missing pieces; 0-4 = wrong or absent.\n\
381
+ +Score 0-10: 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =\n+required\
382
+ \ content present and correct.\n Return ONLY JSON: {{\"reasoning\": \"<the fact the task asks\
383
+ \ for, then the value the answer supplies for it>\", \"score\": <int>, \"feedback\": \"<what content\
384
+ \ is wrong or missing>\"}}"
385
+ - id: O3-swap-judge-only
386
+ class: order
387
+ rule: swap-sentences
388
+ op: swap
389
+ swap:
390
+ a: Judge ONLY whether the information the task asks for is present and correct.
391
+ b: |-
392
+ If the required facts are present and right, the
393
+ answer completes the task.
394
+ note: Same paragraph, same admissibility test as O1.
395
+ variants:
396
+ A-asfiled:
397
+ applicable: true
398
+ sha256: 70511e55353278c7d14697db9cc43a519e2b5e979f00e33b0fd305b013b273f1
399
+ diff: |-
400
+ --- A-asfiled
401
+ +++ O3-swap-judge-only
402
+ @@ -1,6 +1,6 @@
403
+ You are a reviewer checking whether the answer contains the correct content.
404
+ -Judge ONLY whether the information the task asks for is present and correct.
405
+ +If the required facts are present and right, the
406
+ +answer completes the task.
407
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
408
+ -hedging, or verbosity. If the required facts are present and right, the
409
+ -answer completes the task.
410
+ +hedging, or verbosity. Judge ONLY whether the information the task asks for is present and correct.
411
+ An answer that refuses or declines to provide what the task asks for is
412
+ B-nonewline:
413
+ applicable: true
414
+ sha256: 54f9016476447a98d09a89a334efab7cca195b10254dfe17e00ff9de492a1421
415
+ diff: |-
416
+ --- B-nonewline
417
+ +++ O3-swap-judge-only
418
+ @@ -1,6 +1,6 @@
419
+ You are a reviewer checking whether the answer contains the correct content.
420
+ -Judge ONLY whether the information the task asks for is present and correct.
421
+ +If the required facts are present and right, the
422
+ +answer completes the task.
423
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
424
+ -hedging, or verbosity. If the required facts are present and right, the
425
+ -answer completes the task.
426
+ +hedging, or verbosity. Judge ONLY whether the information the task asks for is present and correct.
427
+ An answer that refuses or declines to provide what the task asks for is
428
+ C-attempted:
429
+ applicable: true
430
+ sha256: 5190061b1ae48f7fb2752208d8fca31a177b35e50e6f781a3f19e8d58c84f536
431
+ diff: |-
432
+ --- C-attempted
433
+ +++ O3-swap-judge-only
434
+ @@ -9,6 +9,6 @@
435
+ answer supplies the fact that was asked for.
436
+ -Judge ONLY whether the information the task asks for is present and correct.
437
+ +If the required facts are present and right, the
438
+ +answer completes the task.
439
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
440
+ -hedging, or verbosity. If the required facts are present and right, the
441
+ -answer completes the task.
442
+ +hedging, or verbosity. Judge ONLY whether the information the task asks for is present and correct.
443
+ An answer that refuses or declines to provide what the task asks for is
444
+ - id: P1-reviewer-relative
445
+ class: paraphrase
446
+ rule: reword-frame-clause
447
+ op: replace
448
+ replace:
449
+ - from: You are a reviewer checking whether
450
+ to: You are a reviewer who checks whether
451
+ justification: Frame only; touches no inventory item. Re-deriving the inventory from the perturbed text
452
+ alone yields the same 11 items, the same bands and the same key names. No frozen keyword, no schema
453
+ literal and no band digit moves. The changed words (checking / who / checks) are absent from the acceptance
454
+ cell's task prompt.
455
+ variants:
456
+ A-asfiled:
457
+ applicable: true
458
+ sha256: 0db59b358d4f96c15ddde5ee916fe7dd76e3e2075d9a2d9d208625d096ad855d
459
+ diff: |-
460
+ --- A-asfiled
461
+ +++ P1-reviewer-relative
462
+ @@ -1,2 +1,2 @@
463
+ -You are a reviewer checking whether the answer contains the correct content.
464
+ +You are a reviewer who checks whether the answer contains the correct content.
465
+ Judge ONLY whether the information the task asks for is present and correct.
466
+ B-nonewline:
467
+ applicable: true
468
+ sha256: 348e162113f777d79eae2c0b5bc2d6069eaa4181314a124b8e5c6a6c9b9ecf17
469
+ diff: |-
470
+ --- B-nonewline
471
+ +++ P1-reviewer-relative
472
+ @@ -1,2 +1,2 @@
473
+ -You are a reviewer checking whether the answer contains the correct content.
474
+ +You are a reviewer who checks whether the answer contains the correct content.
475
+ Judge ONLY whether the information the task asks for is present and correct.
476
+ C-attempted:
477
+ applicable: true
478
+ sha256: 574d76ea9a0c0edc7cd69d2092f454a3da334b3e4809f31b414445bcb51e575a
479
+ diff: |-
480
+ --- C-attempted
481
+ +++ P1-reviewer-relative
482
+ @@ -1,2 +1,2 @@
483
+ -You are a reviewer checking whether the answer contains the correct content.
484
+ +You are a reviewer who checks whether the answer contains the correct content.
485
+ In the reasoning field, before you score, write down two things: first, the
486
+ - id: P2-asks-requests
487
+ class: paraphrase
488
+ rule: reword-verb
489
+ op: replace
490
+ replace:
491
+ - from: the information the task asks for
492
+ to: the information the task requests
493
+ justification: 'Inventory item 1, same directive: what is judged is still the information the task calls
494
+ for. No item added, dropped, widened, narrowed or made conditional. The changed words (asks / for
495
+ / requests) are absent from the acceptance cell''s task prompt.'
496
+ variants:
497
+ A-asfiled:
498
+ applicable: true
499
+ sha256: 018859a1c9b11378ffebcaee6dcb199dd6e181e2049fe8cae8a6abe03a280b3c
500
+ diff: |-
501
+ --- A-asfiled
502
+ +++ P2-asks-requests
503
+ @@ -1,3 +1,3 @@
504
+ You are a reviewer checking whether the answer contains the correct content.
505
+ -Judge ONLY whether the information the task asks for is present and correct.
506
+ +Judge ONLY whether the information the task requests is present and correct.
507
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
508
+ B-nonewline:
509
+ applicable: true
510
+ sha256: 333fecc3fd53a7caa194f414e866c3a217b5b7342f0da138e52d1fe552bdf7ff
511
+ diff: |-
512
+ --- B-nonewline
513
+ +++ P2-asks-requests
514
+ @@ -1,3 +1,3 @@
515
+ You are a reviewer checking whether the answer contains the correct content.
516
+ -Judge ONLY whether the information the task asks for is present and correct.
517
+ +Judge ONLY whether the information the task requests is present and correct.
518
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
519
+ C-attempted:
520
+ applicable: true
521
+ sha256: f99e624712f0e2878f75f6a2217098deb0a69d226481f892d56f51c8d0a6e7c9
522
+ diff: |-
523
+ --- C-attempted
524
+ +++ P2-asks-requests
525
+ @@ -9,3 +9,3 @@
526
+ answer supplies the fact that was asked for.
527
+ -Judge ONLY whether the information the task asks for is present and correct.
528
+ +Judge ONLY whether the information the task requests is present and correct.
529
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
530
+ - id: P3-right-correct
531
+ class: paraphrase
532
+ rule: reword-adjective
533
+ op: replace
534
+ replace:
535
+ - from: present and right
536
+ to: present and correct
537
+ justification: 'Inventory item 7, same directive: facts present and right still means the task is complete.
538
+ `correct` already appears in the rubric. KNOWN GUARD VIOLATION, shipped as the spec specifies it and
539
+ pinned by an offline test: the REMOVED word `right` does appear in the nav-prod-port task prompt ("follow
540
+ the documentation to the right file"), so this point fails the section 3.3 shared-token guard on the
541
+ very cell the acceptance run uses. The spec''s own justification checks only the added word. Treat
542
+ any P3 result on that cell as suspect.'
543
+ variants:
544
+ A-asfiled:
545
+ applicable: true
546
+ sha256: 24ea0c22c8889820cf4a16d77010655a6070e05c5abf0cf1de5c3888f671c240
547
+ diff: |-
548
+ --- A-asfiled
549
+ +++ P3-right-correct
550
+ @@ -3,3 +3,3 @@
551
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
552
+ -hedging, or verbosity. If the required facts are present and right, the
553
+ +hedging, or verbosity. If the required facts are present and correct, the
554
+ answer completes the task.
555
+ B-nonewline:
556
+ applicable: true
557
+ sha256: f9404eb022558566b0284fa44e915b2775dc380d52642c5ba3c97310d77e61f0
558
+ diff: |-
559
+ --- B-nonewline
560
+ +++ P3-right-correct
561
+ @@ -3,3 +3,3 @@
562
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
563
+ -hedging, or verbosity. If the required facts are present and right, the
564
+ +hedging, or verbosity. If the required facts are present and correct, the
565
+ answer completes the task.
566
+ C-attempted:
567
+ applicable: true
568
+ sha256: ef825a6459dc820ce8c065dfc5f90bf008fdd2c25ab7aa4043fff03d1c7d450a
569
+ diff: |-
570
+ --- C-attempted
571
+ +++ P3-right-correct
572
+ @@ -11,3 +11,3 @@
573
+ Do NOT deduct points for formatting, phrasing, extra surrounding text,
574
+ -hedging, or verbosity. If the required facts are present and right, the
575
+ +hedging, or verbosity. If the required facts are present and correct, the
576
+ answer completes the task.
File without changes
@@ -0,0 +1,14 @@
1
+ name: extract-contact
2
+ family: structured-extraction
3
+ prompt: |
4
+ Extract the contact as JSON with keys "name" and "email".
5
+ Text: "Reach out to Ann Chen, she is at ann.chen@example.com, usually after 2pm."
6
+ schema:
7
+ type: object
8
+ required: [name, email]
9
+ properties:
10
+ name: {type: string}
11
+ email: {type: string}
12
+ scoring:
13
+ kind: json_equal
14
+ expected: {name: "Ann Chen", email: "ann.chen@example.com"}