bantamkit 0.27.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bantamkit/__init__.py +32 -0
- bantamkit/agent.py +458 -0
- bantamkit/assets/contracts/default.yaml +90 -0
- bantamkit/assets/evals/devteam/manifest.yaml +351 -0
- bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
- bantamkit/assets/evals/devteam/repo/README.md +12 -0
- bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
- bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
- bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
- bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
- bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
- bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
- bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
- bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
- bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
- bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
- bantamkit/assets/evals/fixtures/.gitkeep +0 -0
- bantamkit/assets/evals/fixtures/catalog.json +6 -0
- bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
- bantamkit/assets/evals/tasks/.gitkeep +0 -0
- bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
- bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
- bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
- bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
- bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
- bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
- bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
- bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
- bantamkit/assets/profiles/default.yaml +31 -0
- bantamkit/assets/profiles/patient.yaml +31 -0
- bantamkit/assets/rubrics/.gitkeep +0 -0
- bantamkit/assets/rubrics/code-quality.yaml +20 -0
- bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
- bantamkit/assets/rubrics/task-completion.yaml +28 -0
- bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
- bantamkit/assets/skills/.gitkeep +0 -0
- bantamkit/assets/skills/file-graph.md +7 -0
- bantamkit/assets/skills/memory.md +35 -0
- bantamkit/assets/tools/.gitkeep +0 -0
- bantamkit/assets/tools/bantamkit_read.json +48 -0
- bantamkit/assets/tools/bantamkit_status.json +25 -0
- bantamkit/assets/tools/build_identity.json +17 -0
- bantamkit/assets/tools/document_list.json +12 -0
- bantamkit/assets/tools/document_read.json +31 -0
- bantamkit/assets/tools/file_graph.json +12 -0
- bantamkit/assets/tools/memory_compact.json +31 -0
- bantamkit/assets/tools/memory_recall.json +38 -0
- bantamkit/assets/tools/memory_save.json +61 -0
- bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
- bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
- bantamkit/assets/tools/shiftwork_status.json +25 -0
- bantamkit/assets/tools/skill_audit.json +70 -0
- bantamkit/assets/tools/validate_json.json +31 -0
- bantamkit/assets.py +67 -0
- bantamkit/budget.py +114 -0
- bantamkit/client.py +329 -0
- bantamkit/contract.py +522 -0
- bantamkit/criticreplay.py +3241 -0
- bantamkit/critique.py +301 -0
- bantamkit/docread.py +1744 -0
- bantamkit/evalrun.py +2003 -0
- bantamkit/eventlog.py +282 -0
- bantamkit/filegraph.py +218 -0
- bantamkit/loopguard.py +101 -0
- bantamkit/mcpreport.py +763 -0
- bantamkit/mcpserver.py +1334 -0
- bantamkit/memory/__init__.py +28 -0
- bantamkit/memory/__main__.py +291 -0
- bantamkit/memory/component.py +569 -0
- bantamkit/memory/divergence.py +744 -0
- bantamkit/memory/layers.py +257 -0
- bantamkit/memory/store.py +940 -0
- bantamkit/pdfread.py +1402 -0
- bantamkit/profile.py +46 -0
- bantamkit/shiftwork.py +212 -0
- bantamkit/skillaudit.py +853 -0
- bantamkit/statusline.py +313 -0
- bantamkit/structured.py +125 -0
- bantamkit/textutil.py +30 -0
- bantamkit-0.27.0.dist-info/METADATA +207 -0
- bantamkit-0.27.0.dist-info/RECORD +119 -0
- bantamkit-0.27.0.dist-info/WHEEL +4 -0
- bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
|
|
2
|
+
# `expected` was READ OUT OF the generated document (the same slice `answers:` names),
|
|
3
|
+
# never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
|
|
4
|
+
# two ever disagree.
|
|
5
|
+
# Cell SMALL-IN. Corpus `inventory-small.xlsx`, 400 data rows, 8,620 extracted bytes
|
|
6
|
+
# (~2,155 est. tokens, 0.07x WORKER_NUM_CTX). At PASTE_MAX_BYTES = 8,621 (Amendment 1,
|
|
7
|
+
# 2026-08-20; pre-registration read 12,288) the
|
|
8
|
+
# WHOLE corpus fits, so the `paste` arm here is a COMPLETE paste. This is the cell that
|
|
9
|
+
# separates "the reader works" from "paging works".
|
|
10
|
+
name: doc-small-261
|
|
11
|
+
family: document-read
|
|
12
|
+
tools: []
|
|
13
|
+
document_setup:
|
|
14
|
+
- path: inventory-small.xlsx
|
|
15
|
+
seed: 4021
|
|
16
|
+
sheets:
|
|
17
|
+
- name: stock
|
|
18
|
+
rows: 400
|
|
19
|
+
columns:
|
|
20
|
+
- {name: sku, kind: key, prefix: 'SKU-', width: 6}
|
|
21
|
+
- {name: region, kind: choice, values: [north, south, east, west]}
|
|
22
|
+
- {name: units, kind: int, low: 1000, high: 9999}
|
|
23
|
+
answers:
|
|
24
|
+
question_sku: stock!A262
|
|
25
|
+
expected_region: stock!B262
|
|
26
|
+
expected_units: stock!C262
|
|
27
|
+
prompt: >-
|
|
28
|
+
A workbook is attached to this task. Its sheet `stock` has the columns sku, region
|
|
29
|
+
and units, one row per sku. Find the single row whose sku is exactly SKU-000261 and
|
|
30
|
+
report that row's region and units. Do not compute anything and do not summarise the
|
|
31
|
+
sheet; read the one row. Answer with ONLY this JSON, nothing else:
|
|
32
|
+
{"region": "<region>", "units": <integer>}
|
|
33
|
+
scoring:
|
|
34
|
+
kind: json_equal
|
|
35
|
+
expected:
|
|
36
|
+
region: south
|
|
37
|
+
units: 3788
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
|
|
2
|
+
# `expected` was READ OUT OF the generated document (the same slice `answers:` names),
|
|
3
|
+
# never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
|
|
4
|
+
# two ever disagree.
|
|
5
|
+
# Cell SMALL-IN. Corpus `inventory-small.xlsx`, 400 data rows, 8,620 extracted bytes
|
|
6
|
+
# (~2,155 est. tokens, 0.07x WORKER_NUM_CTX). At PASTE_MAX_BYTES = 8,621 (Amendment 1,
|
|
7
|
+
# 2026-08-20; pre-registration read 12,288) the
|
|
8
|
+
# WHOLE corpus fits, so the `paste` arm here is a COMPLETE paste. This is the cell that
|
|
9
|
+
# separates "the reader works" from "paging works".
|
|
10
|
+
name: doc-small-388
|
|
11
|
+
family: document-read
|
|
12
|
+
tools: []
|
|
13
|
+
document_setup:
|
|
14
|
+
- path: inventory-small.xlsx
|
|
15
|
+
seed: 4021
|
|
16
|
+
sheets:
|
|
17
|
+
- name: stock
|
|
18
|
+
rows: 400
|
|
19
|
+
columns:
|
|
20
|
+
- {name: sku, kind: key, prefix: 'SKU-', width: 6}
|
|
21
|
+
- {name: region, kind: choice, values: [north, south, east, west]}
|
|
22
|
+
- {name: units, kind: int, low: 1000, high: 9999}
|
|
23
|
+
answers:
|
|
24
|
+
question_sku: stock!A389
|
|
25
|
+
expected_region: stock!B389
|
|
26
|
+
expected_units: stock!C389
|
|
27
|
+
prompt: >-
|
|
28
|
+
A workbook is attached to this task. Its sheet `stock` has the columns sku, region
|
|
29
|
+
and units, one row per sku. Find the single row whose sku is exactly SKU-000388 and
|
|
30
|
+
report that row's region and units. Do not compute anything and do not summarise the
|
|
31
|
+
sheet; read the one row. Answer with ONLY this JSON, nothing else:
|
|
32
|
+
{"region": "<region>", "units": <integer>}
|
|
33
|
+
scoring:
|
|
34
|
+
kind: json_equal
|
|
35
|
+
expected:
|
|
36
|
+
region: east
|
|
37
|
+
units: 5208
|
|
File without changes
|
|
@@ -0,0 +1,576 @@
|
|
|
1
|
+
rubric: task-completion
|
|
2
|
+
spec: docs/superpowers/specs/2026-08-11-perturbation-bar-spec.md
|
|
3
|
+
note: 'Perturbation family for the RB-P14 bar. Measurement input, NOT a Contract asset: no product code
|
|
4
|
+
path loads it, and criticreplay.py is its only reader. Every point is a declared, text-anchored transformation
|
|
5
|
+
with a stable rule id, applicable to any template that contains its anchor, so the same family lands
|
|
6
|
+
on every rubric variant under comparison. A point whose anchor is absent from a variant is dropped from
|
|
7
|
+
BOTH families of a pairwise comparison and named in the summary.'
|
|
8
|
+
requirement_inventory_note: 'The requirement inventory of the AS-FILED task-completion rubric (spec section
|
|
9
|
+
3.3 step 1), committed before any paraphrase was authored. Admissibility test for every P point: re-derive
|
|
10
|
+
this list from the perturbed text alone; it must come out the same - same count, same directive per
|
|
11
|
+
item, same bands, same key names. A variant that changes the inventory (C-attempted adds a required
|
|
12
|
+
`reasoning` key) is a different rubric under test, not a perturbation.'
|
|
13
|
+
requirement_inventory:
|
|
14
|
+
- Judge whether the information the task asks for is present and correct.
|
|
15
|
+
- Do not deduct for formatting.
|
|
16
|
+
- Do not deduct for phrasing.
|
|
17
|
+
- Do not deduct for extra surrounding text.
|
|
18
|
+
- Do not deduct for hedging.
|
|
19
|
+
- Do not deduct for verbosity.
|
|
20
|
+
- Facts present and right => the task is complete.
|
|
21
|
+
- A refusal or decline => 0-4, however polite.
|
|
22
|
+
- Hedging instead of an answer => missing content; hedging around an answer => fine.
|
|
23
|
+
- 'Bands: 9-10 present and correct; 5-8 partial or missing pieces; 0-4 wrong or absent.'
|
|
24
|
+
- Output only JSON with keys score (int) and feedback (string).
|
|
25
|
+
materialized_variants:
|
|
26
|
+
A-asfiled:
|
|
27
|
+
source: git:d2f78b7:assets/rubrics/task-completion.yaml
|
|
28
|
+
note: byte-identical to the shipped assets/rubrics/task-completion.yaml
|
|
29
|
+
base_sha256: e018854368c1b675e7cff5109a3dce715d59c86d871cd3d1d83b9083188a065e
|
|
30
|
+
B-nonewline:
|
|
31
|
+
source: A-asfiled with the template's single trailing newline removed
|
|
32
|
+
note: 'materialize as the same YAML with `prompt: |-`'
|
|
33
|
+
base_sha256: d1f32ad2947b4d6f6079833847eae96c79fddeb2683ceda322940bdc8cbf13a6
|
|
34
|
+
C-attempted:
|
|
35
|
+
source: git:e57f1a6:assets/rubrics/task-completion.yaml
|
|
36
|
+
note: the withdrawn derive-before-score rubric; requires `reasoning` on the wire
|
|
37
|
+
base_sha256: 5f588e4a07a6fa29572ed2e3bdbd39132c0a8f9908b0ce47408ad72c015d2298
|
|
38
|
+
points:
|
|
39
|
+
- id: identity
|
|
40
|
+
class: identity
|
|
41
|
+
rule: identity
|
|
42
|
+
op: identity
|
|
43
|
+
note: Mandatory member, not a perturbation class (spec section 3.4). Zero edits. Its replays give the
|
|
44
|
+
pure within-cell replay spread RB-P15 asks for, and anchor the family against the committed SA3 replay
|
|
45
|
+
record.
|
|
46
|
+
variants:
|
|
47
|
+
A-asfiled:
|
|
48
|
+
applicable: true
|
|
49
|
+
sha256: e018854368c1b675e7cff5109a3dce715d59c86d871cd3d1d83b9083188a065e
|
|
50
|
+
diff: ''
|
|
51
|
+
B-nonewline:
|
|
52
|
+
applicable: true
|
|
53
|
+
sha256: d1f32ad2947b4d6f6079833847eae96c79fddeb2683ceda322940bdc8cbf13a6
|
|
54
|
+
diff: ''
|
|
55
|
+
C-attempted:
|
|
56
|
+
applicable: true
|
|
57
|
+
sha256: 5f588e4a07a6fa29572ed2e3bdbd39132c0a8f9908b0ce47408ad72c015d2298
|
|
58
|
+
diff: ''
|
|
59
|
+
- id: W1-trailing-newline
|
|
60
|
+
class: whitespace
|
|
61
|
+
rule: strip-trailing-newline
|
|
62
|
+
op: strip-trailing-newline
|
|
63
|
+
note: THE NULL CONTROL, and a mandatory member. This is the edit that reproduced the entire RB-P4 pass
|
|
64
|
+
signature while changing no word of the rubric. Shortening.
|
|
65
|
+
variants:
|
|
66
|
+
A-asfiled:
|
|
67
|
+
applicable: true
|
|
68
|
+
sha256: d1f32ad2947b4d6f6079833847eae96c79fddeb2683ceda322940bdc8cbf13a6
|
|
69
|
+
diff: |-
|
|
70
|
+
--- A-asfiled
|
|
71
|
+
+++ W1-trailing-newline
|
|
72
|
+
@@ -19 +19,2 @@
|
|
73
|
+
Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
|
|
74
|
+
+
|
|
75
|
+
B-nonewline:
|
|
76
|
+
applicable: false
|
|
77
|
+
C-attempted:
|
|
78
|
+
applicable: true
|
|
79
|
+
sha256: 4054b3b6c5c328a4bcfc0f7191958f10b6f9ad4fd2f3f0f4dbdbd0c231c4d94f
|
|
80
|
+
diff: |-
|
|
81
|
+
--- C-attempted
|
|
82
|
+
+++ W1-trailing-newline
|
|
83
|
+
@@ -27 +27,2 @@
|
|
84
|
+
Return ONLY JSON: {{"reasoning": "<the fact the task asks for, then the value the answer supplies for it>", "score": <int>, "feedback": "<what content is wrong or missing>"}}
|
|
85
|
+
+
|
|
86
|
+
- id: W2-double-trailing
|
|
87
|
+
class: whitespace
|
|
88
|
+
rule: append-trailing-newline
|
|
89
|
+
op: append-trailing-newline
|
|
90
|
+
note: Lengthening. Paired with W1 so the family is not biased in one direction.
|
|
91
|
+
variants:
|
|
92
|
+
A-asfiled:
|
|
93
|
+
applicable: true
|
|
94
|
+
sha256: 447e5be27613ef5cd090f12e8e5d0616ff838bc22ef43c0001b625743b88b303
|
|
95
|
+
diff: |-
|
|
96
|
+
--- A-asfiled
|
|
97
|
+
+++ W2-double-trailing
|
|
98
|
+
@@ -19 +19,2 @@
|
|
99
|
+
Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
|
|
100
|
+
+
|
|
101
|
+
B-nonewline:
|
|
102
|
+
applicable: true
|
|
103
|
+
sha256: e018854368c1b675e7cff5109a3dce715d59c86d871cd3d1d83b9083188a065e
|
|
104
|
+
diff: |-
|
|
105
|
+
--- B-nonewline
|
|
106
|
+
+++ W2-double-trailing
|
|
107
|
+
@@ -19,2 +19 @@
|
|
108
|
+
Return ONLY JSON: {{"score": <int>, "feedback": "<what content is wrong or missing>"}}
|
|
109
|
+
-
|
|
110
|
+
C-attempted:
|
|
111
|
+
applicable: true
|
|
112
|
+
sha256: 03bfdcb7b224a03e3e9d832e8b6bb0a0b77f375062498cc766084504a0e46a95
|
|
113
|
+
diff: |-
|
|
114
|
+
--- C-attempted
|
|
115
|
+
+++ W2-double-trailing
|
|
116
|
+
@@ -27 +27,2 @@
|
|
117
|
+
Return ONLY JSON: {{"reasoning": "<the fact the task asks for, then the value the answer supplies for it>", "score": <int>, "feedback": "<what content is wrong or missing>"}}
|
|
118
|
+
+
|
|
119
|
+
- id: W3-unwrap-opening
|
|
120
|
+
class: whitespace
|
|
121
|
+
rule: unwrap-hard-wrapped-run
|
|
122
|
+
op: replace
|
|
123
|
+
replace:
|
|
124
|
+
- from: |-
|
|
125
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
126
|
+
hedging, or verbosity. If the required facts are present and right, the
|
|
127
|
+
answer completes the task.
|
|
128
|
+
to: Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity.
|
|
129
|
+
If the required facts are present and right, the answer completes the task.
|
|
130
|
+
note: 'Anchored to the first hard-wrapped run of the SHARED opening paragraph rather than to the whole
|
|
131
|
+
opening paragraph the spec''s table names, because the whole paragraph differs between A-asfiled and
|
|
132
|
+
C-attempted and an anchor absent from a variant would drop this rule from the very comparison the
|
|
133
|
+
bar exists for (spec section 4.1). Length-neutral: newline -> space.'
|
|
134
|
+
variants:
|
|
135
|
+
A-asfiled:
|
|
136
|
+
applicable: true
|
|
137
|
+
sha256: 0a41447a57d8b987e170275280eafe6e2568c7cc3d3bdbd9583ffcddef29e067
|
|
138
|
+
diff: |-
|
|
139
|
+
--- A-asfiled
|
|
140
|
+
+++ W3-unwrap-opening
|
|
141
|
+
@@ -2,5 +2,3 @@
|
|
142
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
143
|
+
-Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
144
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
145
|
+
-answer completes the task.
|
|
146
|
+
+Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity. If the required facts are present and right, the answer completes the task.
|
|
147
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
148
|
+
B-nonewline:
|
|
149
|
+
applicable: true
|
|
150
|
+
sha256: dc6d35d1e4570a53a33a8b3c3bbacc6020fb305ecb3dd3c3fb94c27b891c0384
|
|
151
|
+
diff: |-
|
|
152
|
+
--- B-nonewline
|
|
153
|
+
+++ W3-unwrap-opening
|
|
154
|
+
@@ -2,5 +2,3 @@
|
|
155
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
156
|
+
-Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
157
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
158
|
+
-answer completes the task.
|
|
159
|
+
+Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity. If the required facts are present and right, the answer completes the task.
|
|
160
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
161
|
+
C-attempted:
|
|
162
|
+
applicable: true
|
|
163
|
+
sha256: c137757ae33652de54184d36e9e02a704150903952d8a53ef54fdc4ef3273309
|
|
164
|
+
diff: |-
|
|
165
|
+
--- C-attempted
|
|
166
|
+
+++ W3-unwrap-opening
|
|
167
|
+
@@ -10,5 +10,3 @@
|
|
168
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
169
|
+
-Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
170
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
171
|
+
-answer completes the task.
|
|
172
|
+
+Do NOT deduct points for formatting, phrasing, extra surrounding text, hedging, or verbosity. If the required facts are present and right, the answer completes the task.
|
|
173
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
174
|
+
- id: W4-double-space
|
|
175
|
+
class: whitespace
|
|
176
|
+
rule: double-space-after-period
|
|
177
|
+
op: replace
|
|
178
|
+
replace:
|
|
179
|
+
- from: '. '
|
|
180
|
+
to: '. '
|
|
181
|
+
occurrences: all
|
|
182
|
+
note: Position-independent. Lengthening. Every sentence-terminating period.
|
|
183
|
+
variants:
|
|
184
|
+
A-asfiled:
|
|
185
|
+
applicable: true
|
|
186
|
+
sha256: 4c54e4d49cf74a0f5ed3bfabf347c405cd0a54eb2dd10c693b6a19e5dee23792
|
|
187
|
+
diff: |-
|
|
188
|
+
--- A-asfiled
|
|
189
|
+
+++ W4-double-space
|
|
190
|
+
@@ -3,3 +3,3 @@
|
|
191
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
192
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
193
|
+
+hedging, or verbosity. If the required facts are present and right, the
|
|
194
|
+
answer completes the task.
|
|
195
|
+
@@ -7,3 +7,3 @@
|
|
196
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
197
|
+
-polite or explains itself. Hedging around a real answer is fine; hedging
|
|
198
|
+
+polite or explains itself. Hedging around a real answer is fine; hedging
|
|
199
|
+
instead of an answer is not.
|
|
200
|
+
B-nonewline:
|
|
201
|
+
applicable: true
|
|
202
|
+
sha256: 82309ca57ce8d5ae23ce3ee1f7788eec0eabf1c7e8d0b91e8557052ec9b3a55a
|
|
203
|
+
diff: |-
|
|
204
|
+
--- B-nonewline
|
|
205
|
+
+++ W4-double-space
|
|
206
|
+
@@ -3,3 +3,3 @@
|
|
207
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
208
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
209
|
+
+hedging, or verbosity. If the required facts are present and right, the
|
|
210
|
+
answer completes the task.
|
|
211
|
+
@@ -7,3 +7,3 @@
|
|
212
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
213
|
+
-polite or explains itself. Hedging around a real answer is fine; hedging
|
|
214
|
+
+polite or explains itself. Hedging around a real answer is fine; hedging
|
|
215
|
+
instead of an answer is not.
|
|
216
|
+
C-attempted:
|
|
217
|
+
applicable: true
|
|
218
|
+
sha256: 44458e5ed6b1848464940598900f9956c841ae1cad72cc0f10df2759233b1338
|
|
219
|
+
diff: |-
|
|
220
|
+
--- C-attempted
|
|
221
|
+
+++ W4-double-space
|
|
222
|
+
@@ -4,6 +4,6 @@
|
|
223
|
+
written in; second, the value the answer supplies for that fact, quoted from
|
|
224
|
+
-the answer. Then say whether that value is present and responsive to what was
|
|
225
|
+
-asked. You cannot check the value against records: you have no files, no
|
|
226
|
+
+the answer. Then say whether that value is present and responsive to what was
|
|
227
|
+
+asked. You cannot check the value against records: you have no files, no
|
|
228
|
+
tools, and no evidence here, so do not try to work out what the value should
|
|
229
|
+
-have been — a value you invented would be a guess. You are checking that the
|
|
230
|
+
+have been — a value you invented would be a guess. You are checking that the
|
|
231
|
+
answer supplies the fact that was asked for.
|
|
232
|
+
@@ -11,3 +11,3 @@
|
|
233
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
234
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
235
|
+
+hedging, or verbosity. If the required facts are present and right, the
|
|
236
|
+
answer completes the task.
|
|
237
|
+
@@ -15,3 +15,3 @@
|
|
238
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
239
|
+
-polite or explains itself. Hedging around a real answer is fine; hedging
|
|
240
|
+
+polite or explains itself. Hedging around a real answer is fine; hedging
|
|
241
|
+
instead of an answer is not.
|
|
242
|
+
- id: W5-blank-line-before-bands
|
|
243
|
+
class: whitespace
|
|
244
|
+
rule: extra-blank-line-before-anchor
|
|
245
|
+
op: replace
|
|
246
|
+
replace:
|
|
247
|
+
- from: |2-
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
Score 0-10:
|
|
251
|
+
to: |2-
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
Score 0-10:
|
|
256
|
+
note: Anchored. Lengthening. Does not move any structural block.
|
|
257
|
+
variants:
|
|
258
|
+
A-asfiled:
|
|
259
|
+
applicable: true
|
|
260
|
+
sha256: f55493678f7d4f418291b885b16899251c0ad58fad78705eb915eb38393234f5
|
|
261
|
+
diff: "--- A-asfiled\n+++ W5-blank-line-before-bands\n@@ -16,2 +16,3 @@\n \n+\n Score 0-10: 9-10\
|
|
262
|
+
\ = required content present and correct; 5-8 = partially"
|
|
263
|
+
B-nonewline:
|
|
264
|
+
applicable: true
|
|
265
|
+
sha256: 92dbe03ae12ba4d56c3808e8e51190f1ba6d2b0a7a9256395eece3df351d44f9
|
|
266
|
+
diff: "--- B-nonewline\n+++ W5-blank-line-before-bands\n@@ -16,2 +16,3 @@\n \n+\n Score 0-10: 9-10\
|
|
267
|
+
\ = required content present and correct; 5-8 = partially"
|
|
268
|
+
C-attempted:
|
|
269
|
+
applicable: true
|
|
270
|
+
sha256: bab6a5c4f792037d4c0df2d335c7e298b51bb676afc2d0bc416293be17ecfcb7
|
|
271
|
+
diff: "--- C-attempted\n+++ W5-blank-line-before-bands\n@@ -24,2 +24,3 @@\n \n+\n Score 0-10: 9-10\
|
|
272
|
+
\ = required content present and correct; 5-8 = partially"
|
|
273
|
+
- id: O1-swap-format-refusal
|
|
274
|
+
class: order
|
|
275
|
+
rule: swap-sentences
|
|
276
|
+
op: swap
|
|
277
|
+
swap:
|
|
278
|
+
a: |-
|
|
279
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
280
|
+
hedging, or verbosity.
|
|
281
|
+
b: |-
|
|
282
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
283
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
284
|
+
polite or explains itself.
|
|
285
|
+
note: 'Both units are whole sentences inside ONE contiguous prose paragraph and neither carries an anaphor
|
|
286
|
+
or discourse connective pointing outside itself. No structural block moves: Task:/Answer:/Return ONLY
|
|
287
|
+
JSON: keep their positions.'
|
|
288
|
+
variants:
|
|
289
|
+
A-asfiled:
|
|
290
|
+
applicable: true
|
|
291
|
+
sha256: b6a0693eacbd525df6e53b8021c39af9d6c7b15504fe3d975c0d0e17d5e847c0
|
|
292
|
+
diff: |-
|
|
293
|
+
--- A-asfiled
|
|
294
|
+
+++ O1-swap-format-refusal
|
|
295
|
+
@@ -2,8 +2,8 @@
|
|
296
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
297
|
+
-Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
298
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
299
|
+
-answer completes the task.
|
|
300
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
301
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
302
|
+
-polite or explains itself. Hedging around a real answer is fine; hedging
|
|
303
|
+
+polite or explains itself. If the required facts are present and right, the
|
|
304
|
+
+answer completes the task.
|
|
305
|
+
+Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
306
|
+
+hedging, or verbosity. Hedging around a real answer is fine; hedging
|
|
307
|
+
instead of an answer is not.
|
|
308
|
+
B-nonewline:
|
|
309
|
+
applicable: true
|
|
310
|
+
sha256: f8fecf8b2315bbe7b0505a6642e9dcc65dee259b18eac146302ebbe6b305fd5d
|
|
311
|
+
diff: |-
|
|
312
|
+
--- B-nonewline
|
|
313
|
+
+++ O1-swap-format-refusal
|
|
314
|
+
@@ -2,8 +2,8 @@
|
|
315
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
316
|
+
-Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
317
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
318
|
+
-answer completes the task.
|
|
319
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
320
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
321
|
+
-polite or explains itself. Hedging around a real answer is fine; hedging
|
|
322
|
+
+polite or explains itself. If the required facts are present and right, the
|
|
323
|
+
+answer completes the task.
|
|
324
|
+
+Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
325
|
+
+hedging, or verbosity. Hedging around a real answer is fine; hedging
|
|
326
|
+
instead of an answer is not.
|
|
327
|
+
C-attempted:
|
|
328
|
+
applicable: true
|
|
329
|
+
sha256: dc8f94e075f52a59437b63c6a70dc6b888e7fe24ea86f6165dfe9274dda37519
|
|
330
|
+
diff: |-
|
|
331
|
+
--- C-attempted
|
|
332
|
+
+++ O1-swap-format-refusal
|
|
333
|
+
@@ -10,8 +10,8 @@
|
|
334
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
335
|
+
-Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
336
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
337
|
+
-answer completes the task.
|
|
338
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
339
|
+
missing the required content — score it 0-4, even when the refusal is
|
|
340
|
+
-polite or explains itself. Hedging around a real answer is fine; hedging
|
|
341
|
+
+polite or explains itself. If the required facts are present and right, the
|
|
342
|
+
+answer completes the task.
|
|
343
|
+
+Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
344
|
+
+hedging, or verbosity. Hedging around a real answer is fine; hedging
|
|
345
|
+
instead of an answer is not.
|
|
346
|
+
- id: O2-bands-ascending
|
|
347
|
+
class: order
|
|
348
|
+
rule: reorder-list-items
|
|
349
|
+
op: replace
|
|
350
|
+
replace:
|
|
351
|
+
- from: |-
|
|
352
|
+
9-10 = required content present and correct; 5-8 = partially
|
|
353
|
+
correct or missing pieces; 0-4 = wrong or absent.
|
|
354
|
+
to: |-
|
|
355
|
+
0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =
|
|
356
|
+
required content present and correct.
|
|
357
|
+
note: Each band states its own range and its own criterion, so each is a self-contained semicolon-delimited
|
|
358
|
+
list item. The band digits 9-10 / 5-8 / 0-4 are byte-frozen.
|
|
359
|
+
variants:
|
|
360
|
+
A-asfiled:
|
|
361
|
+
applicable: true
|
|
362
|
+
sha256: 740cbc7c7289238ef717bd25e8e846ed70200e9d45d7e0c0e3ecf7e5d4474830
|
|
363
|
+
diff: "--- A-asfiled\n+++ O2-bands-ascending\n@@ -16,4 +16,4 @@\n \n-Score 0-10: 9-10 = required\
|
|
364
|
+
\ content present and correct; 5-8 = partially\n-correct or missing pieces; 0-4 = wrong or absent.\n\
|
|
365
|
+
+Score 0-10: 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =\n+required\
|
|
366
|
+
\ content present and correct.\n Return ONLY JSON: {{\"score\": <int>, \"feedback\": \"<what content\
|
|
367
|
+
\ is wrong or missing>\"}}"
|
|
368
|
+
B-nonewline:
|
|
369
|
+
applicable: true
|
|
370
|
+
sha256: 1aa55cc1112f4f4491c77abdbce76a2b4c1a7f8af4e410095a1d704e2c9f876e
|
|
371
|
+
diff: "--- B-nonewline\n+++ O2-bands-ascending\n@@ -16,4 +16,4 @@\n \n-Score 0-10: 9-10 = required\
|
|
372
|
+
\ content present and correct; 5-8 = partially\n-correct or missing pieces; 0-4 = wrong or absent.\n\
|
|
373
|
+
+Score 0-10: 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =\n+required\
|
|
374
|
+
\ content present and correct.\n Return ONLY JSON: {{\"score\": <int>, \"feedback\": \"<what content\
|
|
375
|
+
\ is wrong or missing>\"}}"
|
|
376
|
+
C-attempted:
|
|
377
|
+
applicable: true
|
|
378
|
+
sha256: 69eed0c1e3ef2dae255e1ceefcac162cc0996a4fecaffacecb63c8c28fe41a47
|
|
379
|
+
diff: "--- C-attempted\n+++ O2-bands-ascending\n@@ -24,4 +24,4 @@\n \n-Score 0-10: 9-10 = required\
|
|
380
|
+
\ content present and correct; 5-8 = partially\n-correct or missing pieces; 0-4 = wrong or absent.\n\
|
|
381
|
+
+Score 0-10: 0-4 = wrong or absent; 5-8 = partially correct or missing pieces; 9-10 =\n+required\
|
|
382
|
+
\ content present and correct.\n Return ONLY JSON: {{\"reasoning\": \"<the fact the task asks\
|
|
383
|
+
\ for, then the value the answer supplies for it>\", \"score\": <int>, \"feedback\": \"<what content\
|
|
384
|
+
\ is wrong or missing>\"}}"
|
|
385
|
+
- id: O3-swap-judge-only
|
|
386
|
+
class: order
|
|
387
|
+
rule: swap-sentences
|
|
388
|
+
op: swap
|
|
389
|
+
swap:
|
|
390
|
+
a: Judge ONLY whether the information the task asks for is present and correct.
|
|
391
|
+
b: |-
|
|
392
|
+
If the required facts are present and right, the
|
|
393
|
+
answer completes the task.
|
|
394
|
+
note: Same paragraph, same admissibility test as O1.
|
|
395
|
+
variants:
|
|
396
|
+
A-asfiled:
|
|
397
|
+
applicable: true
|
|
398
|
+
sha256: 70511e55353278c7d14697db9cc43a519e2b5e979f00e33b0fd305b013b273f1
|
|
399
|
+
diff: |-
|
|
400
|
+
--- A-asfiled
|
|
401
|
+
+++ O3-swap-judge-only
|
|
402
|
+
@@ -1,6 +1,6 @@
|
|
403
|
+
You are a reviewer checking whether the answer contains the correct content.
|
|
404
|
+
-Judge ONLY whether the information the task asks for is present and correct.
|
|
405
|
+
+If the required facts are present and right, the
|
|
406
|
+
+answer completes the task.
|
|
407
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
408
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
409
|
+
-answer completes the task.
|
|
410
|
+
+hedging, or verbosity. Judge ONLY whether the information the task asks for is present and correct.
|
|
411
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
412
|
+
B-nonewline:
|
|
413
|
+
applicable: true
|
|
414
|
+
sha256: 54f9016476447a98d09a89a334efab7cca195b10254dfe17e00ff9de492a1421
|
|
415
|
+
diff: |-
|
|
416
|
+
--- B-nonewline
|
|
417
|
+
+++ O3-swap-judge-only
|
|
418
|
+
@@ -1,6 +1,6 @@
|
|
419
|
+
You are a reviewer checking whether the answer contains the correct content.
|
|
420
|
+
-Judge ONLY whether the information the task asks for is present and correct.
|
|
421
|
+
+If the required facts are present and right, the
|
|
422
|
+
+answer completes the task.
|
|
423
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
424
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
425
|
+
-answer completes the task.
|
|
426
|
+
+hedging, or verbosity. Judge ONLY whether the information the task asks for is present and correct.
|
|
427
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
428
|
+
C-attempted:
|
|
429
|
+
applicable: true
|
|
430
|
+
sha256: 5190061b1ae48f7fb2752208d8fca31a177b35e50e6f781a3f19e8d58c84f536
|
|
431
|
+
diff: |-
|
|
432
|
+
--- C-attempted
|
|
433
|
+
+++ O3-swap-judge-only
|
|
434
|
+
@@ -9,6 +9,6 @@
|
|
435
|
+
answer supplies the fact that was asked for.
|
|
436
|
+
-Judge ONLY whether the information the task asks for is present and correct.
|
|
437
|
+
+If the required facts are present and right, the
|
|
438
|
+
+answer completes the task.
|
|
439
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
440
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
441
|
+
-answer completes the task.
|
|
442
|
+
+hedging, or verbosity. Judge ONLY whether the information the task asks for is present and correct.
|
|
443
|
+
An answer that refuses or declines to provide what the task asks for is
|
|
444
|
+
- id: P1-reviewer-relative
|
|
445
|
+
class: paraphrase
|
|
446
|
+
rule: reword-frame-clause
|
|
447
|
+
op: replace
|
|
448
|
+
replace:
|
|
449
|
+
- from: You are a reviewer checking whether
|
|
450
|
+
to: You are a reviewer who checks whether
|
|
451
|
+
justification: Frame only; touches no inventory item. Re-deriving the inventory from the perturbed text
|
|
452
|
+
alone yields the same 11 items, the same bands and the same key names. No frozen keyword, no schema
|
|
453
|
+
literal and no band digit moves. The changed words (checking / who / checks) are absent from the acceptance
|
|
454
|
+
cell's task prompt.
|
|
455
|
+
variants:
|
|
456
|
+
A-asfiled:
|
|
457
|
+
applicable: true
|
|
458
|
+
sha256: 0db59b358d4f96c15ddde5ee916fe7dd76e3e2075d9a2d9d208625d096ad855d
|
|
459
|
+
diff: |-
|
|
460
|
+
--- A-asfiled
|
|
461
|
+
+++ P1-reviewer-relative
|
|
462
|
+
@@ -1,2 +1,2 @@
|
|
463
|
+
-You are a reviewer checking whether the answer contains the correct content.
|
|
464
|
+
+You are a reviewer who checks whether the answer contains the correct content.
|
|
465
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
466
|
+
B-nonewline:
|
|
467
|
+
applicable: true
|
|
468
|
+
sha256: 348e162113f777d79eae2c0b5bc2d6069eaa4181314a124b8e5c6a6c9b9ecf17
|
|
469
|
+
diff: |-
|
|
470
|
+
--- B-nonewline
|
|
471
|
+
+++ P1-reviewer-relative
|
|
472
|
+
@@ -1,2 +1,2 @@
|
|
473
|
+
-You are a reviewer checking whether the answer contains the correct content.
|
|
474
|
+
+You are a reviewer who checks whether the answer contains the correct content.
|
|
475
|
+
Judge ONLY whether the information the task asks for is present and correct.
|
|
476
|
+
C-attempted:
|
|
477
|
+
applicable: true
|
|
478
|
+
sha256: 574d76ea9a0c0edc7cd69d2092f454a3da334b3e4809f31b414445bcb51e575a
|
|
479
|
+
diff: |-
|
|
480
|
+
--- C-attempted
|
|
481
|
+
+++ P1-reviewer-relative
|
|
482
|
+
@@ -1,2 +1,2 @@
|
|
483
|
+
-You are a reviewer checking whether the answer contains the correct content.
|
|
484
|
+
+You are a reviewer who checks whether the answer contains the correct content.
|
|
485
|
+
In the reasoning field, before you score, write down two things: first, the
|
|
486
|
+
- id: P2-asks-requests
|
|
487
|
+
class: paraphrase
|
|
488
|
+
rule: reword-verb
|
|
489
|
+
op: replace
|
|
490
|
+
replace:
|
|
491
|
+
- from: the information the task asks for
|
|
492
|
+
to: the information the task requests
|
|
493
|
+
justification: 'Inventory item 1, same directive: what is judged is still the information the task calls
|
|
494
|
+
for. No item added, dropped, widened, narrowed or made conditional. The changed words (asks / for
|
|
495
|
+
/ requests) are absent from the acceptance cell''s task prompt.'
|
|
496
|
+
variants:
|
|
497
|
+
A-asfiled:
|
|
498
|
+
applicable: true
|
|
499
|
+
sha256: 018859a1c9b11378ffebcaee6dcb199dd6e181e2049fe8cae8a6abe03a280b3c
|
|
500
|
+
diff: |-
|
|
501
|
+
--- A-asfiled
|
|
502
|
+
+++ P2-asks-requests
|
|
503
|
+
@@ -1,3 +1,3 @@
|
|
504
|
+
You are a reviewer checking whether the answer contains the correct content.
|
|
505
|
+
-Judge ONLY whether the information the task asks for is present and correct.
|
|
506
|
+
+Judge ONLY whether the information the task requests is present and correct.
|
|
507
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
508
|
+
B-nonewline:
|
|
509
|
+
applicable: true
|
|
510
|
+
sha256: 333fecc3fd53a7caa194f414e866c3a217b5b7342f0da138e52d1fe552bdf7ff
|
|
511
|
+
diff: |-
|
|
512
|
+
--- B-nonewline
|
|
513
|
+
+++ P2-asks-requests
|
|
514
|
+
@@ -1,3 +1,3 @@
|
|
515
|
+
You are a reviewer checking whether the answer contains the correct content.
|
|
516
|
+
-Judge ONLY whether the information the task asks for is present and correct.
|
|
517
|
+
+Judge ONLY whether the information the task requests is present and correct.
|
|
518
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
519
|
+
C-attempted:
|
|
520
|
+
applicable: true
|
|
521
|
+
sha256: f99e624712f0e2878f75f6a2217098deb0a69d226481f892d56f51c8d0a6e7c9
|
|
522
|
+
diff: |-
|
|
523
|
+
--- C-attempted
|
|
524
|
+
+++ P2-asks-requests
|
|
525
|
+
@@ -9,3 +9,3 @@
|
|
526
|
+
answer supplies the fact that was asked for.
|
|
527
|
+
-Judge ONLY whether the information the task asks for is present and correct.
|
|
528
|
+
+Judge ONLY whether the information the task requests is present and correct.
|
|
529
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
530
|
+
- id: P3-right-correct
|
|
531
|
+
class: paraphrase
|
|
532
|
+
rule: reword-adjective
|
|
533
|
+
op: replace
|
|
534
|
+
replace:
|
|
535
|
+
- from: present and right
|
|
536
|
+
to: present and correct
|
|
537
|
+
justification: 'Inventory item 7, same directive: facts present and right still means the task is complete.
|
|
538
|
+
`correct` already appears in the rubric. KNOWN GUARD VIOLATION, shipped as the spec specifies it and
|
|
539
|
+
pinned by an offline test: the REMOVED word `right` does appear in the nav-prod-port task prompt ("follow
|
|
540
|
+
the documentation to the right file"), so this point fails the section 3.3 shared-token guard on the
|
|
541
|
+
very cell the acceptance run uses. The spec''s own justification checks only the added word. Treat
|
|
542
|
+
any P3 result on that cell as suspect.'
|
|
543
|
+
variants:
|
|
544
|
+
A-asfiled:
|
|
545
|
+
applicable: true
|
|
546
|
+
sha256: 24ea0c22c8889820cf4a16d77010655a6070e05c5abf0cf1de5c3888f671c240
|
|
547
|
+
diff: |-
|
|
548
|
+
--- A-asfiled
|
|
549
|
+
+++ P3-right-correct
|
|
550
|
+
@@ -3,3 +3,3 @@
|
|
551
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
552
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
553
|
+
+hedging, or verbosity. If the required facts are present and correct, the
|
|
554
|
+
answer completes the task.
|
|
555
|
+
B-nonewline:
|
|
556
|
+
applicable: true
|
|
557
|
+
sha256: f9404eb022558566b0284fa44e915b2775dc380d52642c5ba3c97310d77e61f0
|
|
558
|
+
diff: |-
|
|
559
|
+
--- B-nonewline
|
|
560
|
+
+++ P3-right-correct
|
|
561
|
+
@@ -3,3 +3,3 @@
|
|
562
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
563
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
564
|
+
+hedging, or verbosity. If the required facts are present and correct, the
|
|
565
|
+
answer completes the task.
|
|
566
|
+
C-attempted:
|
|
567
|
+
applicable: true
|
|
568
|
+
sha256: ef825a6459dc820ce8c065dfc5f90bf008fdd2c25ab7aa4043fff03d1c7d450a
|
|
569
|
+
diff: |-
|
|
570
|
+
--- C-attempted
|
|
571
|
+
+++ P3-right-correct
|
|
572
|
+
@@ -11,3 +11,3 @@
|
|
573
|
+
Do NOT deduct points for formatting, phrasing, extra surrounding text,
|
|
574
|
+
-hedging, or verbosity. If the required facts are present and right, the
|
|
575
|
+
+hedging, or verbosity. If the required facts are present and correct, the
|
|
576
|
+
answer completes the task.
|
|
File without changes
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
name: extract-contact
|
|
2
|
+
family: structured-extraction
|
|
3
|
+
prompt: |
|
|
4
|
+
Extract the contact as JSON with keys "name" and "email".
|
|
5
|
+
Text: "Reach out to Ann Chen, she is at ann.chen@example.com, usually after 2pm."
|
|
6
|
+
schema:
|
|
7
|
+
type: object
|
|
8
|
+
required: [name, email]
|
|
9
|
+
properties:
|
|
10
|
+
name: {type: string}
|
|
11
|
+
email: {type: string}
|
|
12
|
+
scoring:
|
|
13
|
+
kind: json_equal
|
|
14
|
+
expected: {name: "Ann Chen", email: "ann.chen@example.com"}
|