bantamkit 0.27.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. bantamkit/__init__.py +32 -0
  2. bantamkit/agent.py +458 -0
  3. bantamkit/assets/contracts/default.yaml +90 -0
  4. bantamkit/assets/evals/devteam/manifest.yaml +351 -0
  5. bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
  6. bantamkit/assets/evals/devteam/repo/README.md +12 -0
  7. bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
  8. bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
  9. bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
  10. bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
  11. bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
  12. bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
  13. bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
  14. bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
  15. bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
  16. bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
  17. bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
  18. bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
  19. bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
  20. bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
  21. bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
  22. bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
  23. bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
  24. bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
  25. bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
  26. bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
  27. bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
  28. bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
  29. bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
  30. bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
  31. bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
  32. bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
  33. bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
  34. bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
  35. bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
  36. bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
  37. bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
  38. bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
  39. bantamkit/assets/evals/fixtures/.gitkeep +0 -0
  40. bantamkit/assets/evals/fixtures/catalog.json +6 -0
  41. bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
  42. bantamkit/assets/evals/tasks/.gitkeep +0 -0
  43. bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
  44. bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
  45. bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
  46. bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
  47. bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
  48. bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
  49. bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
  50. bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
  51. bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
  52. bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
  53. bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
  54. bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
  55. bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
  56. bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
  57. bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
  58. bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
  59. bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
  60. bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
  61. bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
  62. bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
  63. bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
  64. bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
  65. bantamkit/assets/profiles/default.yaml +31 -0
  66. bantamkit/assets/profiles/patient.yaml +31 -0
  67. bantamkit/assets/rubrics/.gitkeep +0 -0
  68. bantamkit/assets/rubrics/code-quality.yaml +20 -0
  69. bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
  70. bantamkit/assets/rubrics/task-completion.yaml +28 -0
  71. bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
  72. bantamkit/assets/skills/.gitkeep +0 -0
  73. bantamkit/assets/skills/file-graph.md +7 -0
  74. bantamkit/assets/skills/memory.md +35 -0
  75. bantamkit/assets/tools/.gitkeep +0 -0
  76. bantamkit/assets/tools/bantamkit_read.json +48 -0
  77. bantamkit/assets/tools/bantamkit_status.json +25 -0
  78. bantamkit/assets/tools/build_identity.json +17 -0
  79. bantamkit/assets/tools/document_list.json +12 -0
  80. bantamkit/assets/tools/document_read.json +31 -0
  81. bantamkit/assets/tools/file_graph.json +12 -0
  82. bantamkit/assets/tools/memory_compact.json +31 -0
  83. bantamkit/assets/tools/memory_recall.json +38 -0
  84. bantamkit/assets/tools/memory_save.json +61 -0
  85. bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
  86. bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
  87. bantamkit/assets/tools/shiftwork_status.json +25 -0
  88. bantamkit/assets/tools/skill_audit.json +70 -0
  89. bantamkit/assets/tools/validate_json.json +31 -0
  90. bantamkit/assets.py +67 -0
  91. bantamkit/budget.py +114 -0
  92. bantamkit/client.py +329 -0
  93. bantamkit/contract.py +522 -0
  94. bantamkit/criticreplay.py +3241 -0
  95. bantamkit/critique.py +301 -0
  96. bantamkit/docread.py +1744 -0
  97. bantamkit/evalrun.py +2003 -0
  98. bantamkit/eventlog.py +282 -0
  99. bantamkit/filegraph.py +218 -0
  100. bantamkit/loopguard.py +101 -0
  101. bantamkit/mcpreport.py +763 -0
  102. bantamkit/mcpserver.py +1334 -0
  103. bantamkit/memory/__init__.py +28 -0
  104. bantamkit/memory/__main__.py +291 -0
  105. bantamkit/memory/component.py +569 -0
  106. bantamkit/memory/divergence.py +744 -0
  107. bantamkit/memory/layers.py +257 -0
  108. bantamkit/memory/store.py +940 -0
  109. bantamkit/pdfread.py +1402 -0
  110. bantamkit/profile.py +46 -0
  111. bantamkit/shiftwork.py +212 -0
  112. bantamkit/skillaudit.py +853 -0
  113. bantamkit/statusline.py +313 -0
  114. bantamkit/structured.py +125 -0
  115. bantamkit/textutil.py +30 -0
  116. bantamkit-0.27.0.dist-info/METADATA +207 -0
  117. bantamkit-0.27.0.dist-info/RECORD +119 -0
  118. bantamkit-0.27.0.dist-info/WHEEL +4 -0
  119. bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,182 @@
1
+ # GENERATED by tools/devteam/build_tasks.py from assets/evals/devteam/manifest.yaml.
2
+ # Do not hand-edit: `build_tasks.py check` fails if you do. The task's selection
3
+ # rationale, reference walk and exclusion record live in the manifest.
4
+ name: dt-trace-blame
5
+ family: dev-repo-history
6
+ tools:
7
+ - read_file
8
+ - list_files
9
+ workspace:
10
+ HISTORY.md: '# Release history
11
+
12
+
13
+ `git log --oneline --date=short --pretty=''%h %ad %s''`, newest first. Patch files
14
+
15
+ for selected commits are under `patches/`.
16
+
17
+
18
+ ```
19
+
20
+ 9f2c1ab 2026-07-29 fix(retry): read retry_max_attempts from settings, not the constant
21
+
22
+ 7d4e88c 2026-07-22 feat(settle): batch settlement with per-entry retry
23
+
24
+ 6b1a903 2026-07-15 refactor(config): move every tunable into CONFIG_KEYS
25
+
26
+ 5c0de41 2026-07-08 fix(validate): reject zero-amount entries
27
+
28
+ 4a9bb27 2026-07-01 feat(report): render a per-handler summary
29
+
30
+ 3e88f10 2026-06-24 feat(posting): post_entry writes one entry
31
+
32
+ 2d5a6c9 2026-06-17 feat(errors): ValidationError and SettlementError
33
+
34
+ 1c4f0b8 2026-06-10 chore: initial layout
35
+
36
+ ```
37
+
38
+
39
+ Branch `main` is at `9f2c1ab`. Tags: `v0.9.0` on `9f2c1ab`, `v0.8.0` on
40
+
41
+ `5c0de41`.
42
+
43
+ '
44
+ README.md: '# svc-ledger
45
+
46
+
47
+ Double-entry ledger service. Three handlers, one settings module, no globals.
48
+
49
+
50
+ - Module map and the call order between modules: `docs/architecture.md`
51
+
52
+ - Operating procedure, escalation, and where the tunables live: `docs/runbook.md`
53
+
54
+ - Release history, newest first, with commit ids: `HISTORY.md`
55
+
56
+ - Patch files for selected commits: `patches/`
57
+
58
+ - Open incidents, including captured tracebacks: `issues/`
59
+
60
+
61
+ Handlers are resolved at import time from `src/ledger/registry.py`. Nothing else
62
+
63
+ in the service knows the handler names.
64
+
65
+ '
66
+ docs/architecture.md: "# Architecture\n\nThe three entry points are the handlers listed in `src/ledger/registry.py`.\n\
67
+ \n- `settle.py` drives a batch. It wraps `posting.post_entry` in\n `retry.with_retry`, and raises\
68
+ \ its own error when a batch is too large.\n- `posting.py` writes one entry. It validates first, via\n\
69
+ \ `validate.validate_entry`, and defines no exception of its own.\n- `report.py` renders a summary.\
70
+ \ It reads the registry and imports no handler.\n- `validate.py` is the only module that decides an\
71
+ \ entry is malformed.\n- `errors.py` defines every exception class in the service. No other module\n\
72
+ \ declares one.\n\nEvery tunable is a key in `src/ledger/config.py`; that module is the only one\n\
73
+ that touches `os.environ`. Modules ask `load_settings()` for values, so an\nattribute read on a `Settings`\
74
+ \ object is how you tell that a module consumes a\nconfig key.\n"
75
+ docs/runbook.md: '# Runbook
76
+
77
+
78
+ Attempt counts, backoff intervals and batch sizes are **not written on this
79
+
80
+ page** and never will be. Every one of them is a key with its default in
81
+
82
+ `src/ledger/config.py` — read that file for the current value. Do not answer an
83
+
84
+ operational question about a number from this page; it will be stale.
85
+
86
+
87
+ Escalation: three consecutive settlement failures page the on-call. Foreign
88
+
89
+ currency entries are converted, not rejected. A malformed entry is a caller bug
90
+
91
+ and is never retried past the wrapper''s own attempt budget.
92
+
93
+ '
94
+ issues/142-settlement-timeout.md: "# 142 — settlement aborts on a zero-amount entry\n\nReported by ops\
95
+ \ on 2026-08-02 against `main` (`9f2c1ab`). One entry in a batch\nof 40 carried `amount: 0`; the whole\
96
+ \ batch aborted after the wrapper exhausted\nits attempts. Captured traceback, verbatim from the worker\
97
+ \ log:\n\n```\nTraceback (most recent call last):\n File \"src/ledger/settle.py\", line 15, in settle_batch\n\
98
+ \ posted.append(with_retry(post_entry, entry))\n File \"src/ledger/retry.py\", line 17, in with_retry\n\
99
+ \ raise last\n File \"src/ledger/posting.py\", line 9, in post_entry\n validate_entry(entry)\n\
100
+ \ File \"src/ledger/validate.py\", line 13, in validate_entry\n raise ValidationError(f\"amount\
101
+ \ must be non-zero: {entry!r}\")\nledger.errors.ValidationError: amount must be non-zero: {'amount':\
102
+ \ 0, 'ccy': 'EUR'}\n```\n\nOps question: which change put this check in, and should a caller bug really\n\
103
+ burn the whole attempt budget? Runbook says a malformed entry is never retried\npast the wrapper's\
104
+ \ budget, so behaviour matches the runbook; the argument is\nabout whether the budget should apply\
105
+ \ at all.\n"
106
+ patches/0009-retry-budget.patch: "From 9f2c1ab5e2d84c1f0a7b6935cc41d0e2a8f37b41 Mon Sep 17 00:00:00\
107
+ \ 2001\nDate: Wed, 29 Jul 2026 09:14:02 +0000\nSubject: [PATCH] fix(retry): read retry_max_attempts\
108
+ \ from settings, not the\n constant\n\nThe wrapper had its own hard-coded attempt count, so raising\
109
+ \ the deployed\nbudget changed nothing. It now asks load_settings() like every other module.\n---\n\
110
+ \ src/ledger/retry.py | 8 +++++---\n 1 file changed, 5 insertions(+), 3 deletions(-)\n\ndiff --git\
111
+ \ a/src/ledger/retry.py b/src/ledger/retry.py\n--- a/src/ledger/retry.py\n+++ b/src/ledger/retry.py\n\
112
+ @@ -1,14 +1,15 @@\n-\"\"\"Retry wrapper. Attempts are fixed at the module constant below.\"\"\"\n\
113
+ +\"\"\"Retry wrapper. Reads its attempt count from settings, never from a constant.\"\"\"\n\n import\
114
+ \ time\n\n-_MAX_ATTEMPTS = 3\n+from ledger.config import load_settings\n\n\n def with_retry(fn, *args):\n\
115
+ + settings = load_settings()\n last = None\n- for _attempt in range(_MAX_ATTEMPTS):\n+ \
116
+ \ for _attempt in range(settings.retry_max_attempts):\n try:\n return fn(*args)\n\
117
+ \ except Exception as exc:\n last = exc\n- time.sleep(0.25)\n+ \
118
+ \ time.sleep(settings.retry_backoff_ms / 1000)\n raise last\n--\n2.45.2\n"
119
+ src/ledger/__init__.py: '"""svc-ledger: a double-entry ledger service."""
120
+
121
+
122
+ __version__ = "0.9.0"
123
+
124
+ '
125
+ src/ledger/config.py: "\"\"\"The only module in svc-ledger that reads os.environ.\n\nEvery tunable is\
126
+ \ a key in CONFIG_KEYS with its default beside it. docs/runbook.md\ndeliberately does not repeat these\
127
+ \ values: one place, one default.\n\"\"\"\n\nimport os\n\nCONFIG_KEYS = {\n \"retry_max_attempts\"\
128
+ : 5,\n \"retry_backoff_ms\": 250,\n \"settle_batch_size\": 200,\n \"posting_currency\": \"\
129
+ EUR\",\n \"report_top_n\": 10,\n}\n\n\nclass Settings:\n \"\"\"Attribute access over CONFIG_KEYS,\
130
+ \ environment first.\"\"\"\n\n def __init__(self, values):\n self.values = values\n\n \
131
+ \ def __getattr__(self, name):\n if name not in self.values:\n raise AttributeError(f\"\
132
+ unknown config key: {name}\")\n return self.values[name]\n\n\ndef load_settings():\n values\
133
+ \ = {}\n for key, default in CONFIG_KEYS.items():\n raw = os.environ.get(\"LEDGER_\" + key.upper())\n\
134
+ \ values[key] = type(default)(raw) if raw is not None else default\n return Settings(values)\n"
135
+ src/ledger/errors.py: "\"\"\"Exception hierarchy for svc-ledger. Every module raises from here.\"\"\"\
136
+ \n\n\nclass LedgerError(Exception):\n \"\"\"Base class. Nothing raises this directly.\"\"\"\n\n\
137
+ \nclass ValidationError(LedgerError):\n \"\"\"An entry failed validate_entry.\"\"\"\n\n\nclass\
138
+ \ SettlementError(LedgerError):\n \"\"\"A batch could not be settled.\"\"\"\n"
139
+ src/ledger/posting.py: "\"\"\"Post one entry to the ledger. Validates before writing; raises nothing\
140
+ \ itself.\"\"\"\n\nfrom ledger.config import load_settings\nfrom ledger.validate import validate_entry\n\
141
+ \n\ndef post_entry(entry):\n settings = load_settings()\n validate_entry(entry)\n if entry[\"\
142
+ ccy\"] != settings.posting_currency:\n return {\"status\": \"converted\", \"ccy\": settings.posting_currency}\n\
143
+ \ return {\"status\": \"posted\", \"ccy\": entry[\"ccy\"]}\n"
144
+ src/ledger/registry.py: "\"\"\"Handler registry. Resolved once at import time by the service entry point.\"\
145
+ \"\"\n\nHANDLERS = {\n \"settle\": \"ledger.settle:settle_batch\",\n \"post\": \"ledger.posting:post_entry\"\
146
+ ,\n \"report\": \"ledger.report:build_report\",\n}\n"
147
+ src/ledger/report.py: "\"\"\"Per-handler summary. Reads the registry; imports no handler and no settings.\"\
148
+ \"\"\n\nfrom ledger.registry import HANDLERS\n\n\ndef build_report(rows):\n ranked = sorted(rows,\
149
+ \ key=lambda r: r[\"amount\"], reverse=True)\n # Left over from 6b1a903: this 10 was never moved\
150
+ \ into CONFIG_KEYS.\n return {\"handlers\": sorted(HANDLERS), \"top\": ranked[:10]}\n"
151
+ src/ledger/retry.py: "\"\"\"Retry wrapper. Reads its attempt count from settings, never from a constant.\"\
152
+ \"\"\n\nimport time\n\nfrom ledger.config import load_settings\n\n\ndef with_retry(fn, *args):\n \
153
+ \ settings = load_settings()\n last = None\n for _attempt in range(settings.retry_max_attempts):\n\
154
+ \ try:\n return fn(*args)\n except Exception as exc:\n last =\
155
+ \ exc\n time.sleep(settings.retry_backoff_ms / 1000)\n raise last\n"
156
+ src/ledger/settle.py: "\"\"\"Batch settlement. Drives posting through the retry wrapper.\"\"\"\n\nfrom\
157
+ \ ledger.config import load_settings\nfrom ledger.errors import SettlementError\nfrom ledger.posting\
158
+ \ import post_entry\nfrom ledger.retry import with_retry\n\n\ndef settle_batch(entries):\n settings\
159
+ \ = load_settings()\n if len(entries) > settings.settle_batch_size:\n raise SettlementError(f\"\
160
+ batch of {len(entries)} exceeds settle_batch_size\")\n posted = []\n for entry in entries:\n\
161
+ \ posted.append(with_retry(post_entry, entry))\n return posted\n"
162
+ src/ledger/validate.py: "\"\"\"Entry validation. The only module that decides an entry is malformed.\"\
163
+ \"\"\n\nfrom ledger.errors import ValidationError\n\nREQUIRED_FIELDS = (\"amount\", \"ccy\")\n\n\n\
164
+ def validate_entry(entry):\n for field in REQUIRED_FIELDS:\n if field not in entry:\n \
165
+ \ raise ValidationError(f\"missing field {field!r}: {entry!r}\")\n if entry[\"amount\"\
166
+ ] == 0:\n raise ValidationError(f\"amount must be non-zero: {entry!r}\")\n return entry\n"
167
+ tests/test_posting.py: "import pytest\n\nfrom ledger.errors import ValidationError\nfrom ledger.posting\
168
+ \ import post_entry\n\n\ndef test_post_entry_rejects_zero_amount():\n with pytest.raises(ValidationError):\n\
169
+ \ post_entry({\"amount\": 0, \"ccy\": \"EUR\"})\n\n\ndef test_post_entry_converts_foreign_currency():\n\
170
+ \ assert post_entry({\"amount\": 10, \"ccy\": \"USD\"})[\"status\"] == \"converted\"\n"
171
+ tests/test_settle.py: "import pytest\n\nfrom ledger.errors import SettlementError\nfrom ledger.settle\
172
+ \ import settle_batch\n\n\ndef test_settle_batch_rejects_oversized_batch():\n with pytest.raises(SettlementError):\n\
173
+ \ settle_batch([{\"amount\": 1, \"ccy\": \"EUR\"}] * 201)\n"
174
+ prompt: 'issues/142-settlement-timeout.md contains a captured traceback. Work out which commit listed
175
+ in HISTORY.md introduced the check that raised, and which file that check lives in. Read the issue,
176
+ then the file the deepest frame names, then the history. Do not answer from memory. Answer with ONLY
177
+ this JSON, nothing else: {"commit": "<short sha>", "path": "<repo-relative path>"}'
178
+ scoring:
179
+ kind: json_equal
180
+ expected:
181
+ commit: 5c0de41
182
+ path: src/ledger/validate.py
@@ -0,0 +1,180 @@
1
+ # GENERATED by tools/devteam/build_tasks.py from assets/evals/devteam/manifest.yaml.
2
+ # Do not hand-edit: `build_tasks.py check` fails if you do. The task's selection
3
+ # rationale, reference walk and exclusion record live in the manifest.
4
+ name: dt-unread-key
5
+ family: dev-repo-code
6
+ tools:
7
+ - read_file
8
+ - list_files
9
+ workspace:
10
+ HISTORY.md: '# Release history
11
+
12
+
13
+ `git log --oneline --date=short --pretty=''%h %ad %s''`, newest first. Patch files
14
+
15
+ for selected commits are under `patches/`.
16
+
17
+
18
+ ```
19
+
20
+ 9f2c1ab 2026-07-29 fix(retry): read retry_max_attempts from settings, not the constant
21
+
22
+ 7d4e88c 2026-07-22 feat(settle): batch settlement with per-entry retry
23
+
24
+ 6b1a903 2026-07-15 refactor(config): move every tunable into CONFIG_KEYS
25
+
26
+ 5c0de41 2026-07-08 fix(validate): reject zero-amount entries
27
+
28
+ 4a9bb27 2026-07-01 feat(report): render a per-handler summary
29
+
30
+ 3e88f10 2026-06-24 feat(posting): post_entry writes one entry
31
+
32
+ 2d5a6c9 2026-06-17 feat(errors): ValidationError and SettlementError
33
+
34
+ 1c4f0b8 2026-06-10 chore: initial layout
35
+
36
+ ```
37
+
38
+
39
+ Branch `main` is at `9f2c1ab`. Tags: `v0.9.0` on `9f2c1ab`, `v0.8.0` on
40
+
41
+ `5c0de41`.
42
+
43
+ '
44
+ README.md: '# svc-ledger
45
+
46
+
47
+ Double-entry ledger service. Three handlers, one settings module, no globals.
48
+
49
+
50
+ - Module map and the call order between modules: `docs/architecture.md`
51
+
52
+ - Operating procedure, escalation, and where the tunables live: `docs/runbook.md`
53
+
54
+ - Release history, newest first, with commit ids: `HISTORY.md`
55
+
56
+ - Patch files for selected commits: `patches/`
57
+
58
+ - Open incidents, including captured tracebacks: `issues/`
59
+
60
+
61
+ Handlers are resolved at import time from `src/ledger/registry.py`. Nothing else
62
+
63
+ in the service knows the handler names.
64
+
65
+ '
66
+ docs/architecture.md: "# Architecture\n\nThe three entry points are the handlers listed in `src/ledger/registry.py`.\n\
67
+ \n- `settle.py` drives a batch. It wraps `posting.post_entry` in\n `retry.with_retry`, and raises\
68
+ \ its own error when a batch is too large.\n- `posting.py` writes one entry. It validates first, via\n\
69
+ \ `validate.validate_entry`, and defines no exception of its own.\n- `report.py` renders a summary.\
70
+ \ It reads the registry and imports no handler.\n- `validate.py` is the only module that decides an\
71
+ \ entry is malformed.\n- `errors.py` defines every exception class in the service. No other module\n\
72
+ \ declares one.\n\nEvery tunable is a key in `src/ledger/config.py`; that module is the only one\n\
73
+ that touches `os.environ`. Modules ask `load_settings()` for values, so an\nattribute read on a `Settings`\
74
+ \ object is how you tell that a module consumes a\nconfig key.\n"
75
+ docs/runbook.md: '# Runbook
76
+
77
+
78
+ Attempt counts, backoff intervals and batch sizes are **not written on this
79
+
80
+ page** and never will be. Every one of them is a key with its default in
81
+
82
+ `src/ledger/config.py` — read that file for the current value. Do not answer an
83
+
84
+ operational question about a number from this page; it will be stale.
85
+
86
+
87
+ Escalation: three consecutive settlement failures page the on-call. Foreign
88
+
89
+ currency entries are converted, not rejected. A malformed entry is a caller bug
90
+
91
+ and is never retried past the wrapper''s own attempt budget.
92
+
93
+ '
94
+ issues/142-settlement-timeout.md: "# 142 — settlement aborts on a zero-amount entry\n\nReported by ops\
95
+ \ on 2026-08-02 against `main` (`9f2c1ab`). One entry in a batch\nof 40 carried `amount: 0`; the whole\
96
+ \ batch aborted after the wrapper exhausted\nits attempts. Captured traceback, verbatim from the worker\
97
+ \ log:\n\n```\nTraceback (most recent call last):\n File \"src/ledger/settle.py\", line 15, in settle_batch\n\
98
+ \ posted.append(with_retry(post_entry, entry))\n File \"src/ledger/retry.py\", line 17, in with_retry\n\
99
+ \ raise last\n File \"src/ledger/posting.py\", line 9, in post_entry\n validate_entry(entry)\n\
100
+ \ File \"src/ledger/validate.py\", line 13, in validate_entry\n raise ValidationError(f\"amount\
101
+ \ must be non-zero: {entry!r}\")\nledger.errors.ValidationError: amount must be non-zero: {'amount':\
102
+ \ 0, 'ccy': 'EUR'}\n```\n\nOps question: which change put this check in, and should a caller bug really\n\
103
+ burn the whole attempt budget? Runbook says a malformed entry is never retried\npast the wrapper's\
104
+ \ budget, so behaviour matches the runbook; the argument is\nabout whether the budget should apply\
105
+ \ at all.\n"
106
+ patches/0009-retry-budget.patch: "From 9f2c1ab5e2d84c1f0a7b6935cc41d0e2a8f37b41 Mon Sep 17 00:00:00\
107
+ \ 2001\nDate: Wed, 29 Jul 2026 09:14:02 +0000\nSubject: [PATCH] fix(retry): read retry_max_attempts\
108
+ \ from settings, not the\n constant\n\nThe wrapper had its own hard-coded attempt count, so raising\
109
+ \ the deployed\nbudget changed nothing. It now asks load_settings() like every other module.\n---\n\
110
+ \ src/ledger/retry.py | 8 +++++---\n 1 file changed, 5 insertions(+), 3 deletions(-)\n\ndiff --git\
111
+ \ a/src/ledger/retry.py b/src/ledger/retry.py\n--- a/src/ledger/retry.py\n+++ b/src/ledger/retry.py\n\
112
+ @@ -1,14 +1,15 @@\n-\"\"\"Retry wrapper. Attempts are fixed at the module constant below.\"\"\"\n\
113
+ +\"\"\"Retry wrapper. Reads its attempt count from settings, never from a constant.\"\"\"\n\n import\
114
+ \ time\n\n-_MAX_ATTEMPTS = 3\n+from ledger.config import load_settings\n\n\n def with_retry(fn, *args):\n\
115
+ + settings = load_settings()\n last = None\n- for _attempt in range(_MAX_ATTEMPTS):\n+ \
116
+ \ for _attempt in range(settings.retry_max_attempts):\n try:\n return fn(*args)\n\
117
+ \ except Exception as exc:\n last = exc\n- time.sleep(0.25)\n+ \
118
+ \ time.sleep(settings.retry_backoff_ms / 1000)\n raise last\n--\n2.45.2\n"
119
+ src/ledger/__init__.py: '"""svc-ledger: a double-entry ledger service."""
120
+
121
+
122
+ __version__ = "0.9.0"
123
+
124
+ '
125
+ src/ledger/config.py: "\"\"\"The only module in svc-ledger that reads os.environ.\n\nEvery tunable is\
126
+ \ a key in CONFIG_KEYS with its default beside it. docs/runbook.md\ndeliberately does not repeat these\
127
+ \ values: one place, one default.\n\"\"\"\n\nimport os\n\nCONFIG_KEYS = {\n \"retry_max_attempts\"\
128
+ : 5,\n \"retry_backoff_ms\": 250,\n \"settle_batch_size\": 200,\n \"posting_currency\": \"\
129
+ EUR\",\n \"report_top_n\": 10,\n}\n\n\nclass Settings:\n \"\"\"Attribute access over CONFIG_KEYS,\
130
+ \ environment first.\"\"\"\n\n def __init__(self, values):\n self.values = values\n\n \
131
+ \ def __getattr__(self, name):\n if name not in self.values:\n raise AttributeError(f\"\
132
+ unknown config key: {name}\")\n return self.values[name]\n\n\ndef load_settings():\n values\
133
+ \ = {}\n for key, default in CONFIG_KEYS.items():\n raw = os.environ.get(\"LEDGER_\" + key.upper())\n\
134
+ \ values[key] = type(default)(raw) if raw is not None else default\n return Settings(values)\n"
135
+ src/ledger/errors.py: "\"\"\"Exception hierarchy for svc-ledger. Every module raises from here.\"\"\"\
136
+ \n\n\nclass LedgerError(Exception):\n \"\"\"Base class. Nothing raises this directly.\"\"\"\n\n\
137
+ \nclass ValidationError(LedgerError):\n \"\"\"An entry failed validate_entry.\"\"\"\n\n\nclass\
138
+ \ SettlementError(LedgerError):\n \"\"\"A batch could not be settled.\"\"\"\n"
139
+ src/ledger/posting.py: "\"\"\"Post one entry to the ledger. Validates before writing; raises nothing\
140
+ \ itself.\"\"\"\n\nfrom ledger.config import load_settings\nfrom ledger.validate import validate_entry\n\
141
+ \n\ndef post_entry(entry):\n settings = load_settings()\n validate_entry(entry)\n if entry[\"\
142
+ ccy\"] != settings.posting_currency:\n return {\"status\": \"converted\", \"ccy\": settings.posting_currency}\n\
143
+ \ return {\"status\": \"posted\", \"ccy\": entry[\"ccy\"]}\n"
144
+ src/ledger/registry.py: "\"\"\"Handler registry. Resolved once at import time by the service entry point.\"\
145
+ \"\"\n\nHANDLERS = {\n \"settle\": \"ledger.settle:settle_batch\",\n \"post\": \"ledger.posting:post_entry\"\
146
+ ,\n \"report\": \"ledger.report:build_report\",\n}\n"
147
+ src/ledger/report.py: "\"\"\"Per-handler summary. Reads the registry; imports no handler and no settings.\"\
148
+ \"\"\n\nfrom ledger.registry import HANDLERS\n\n\ndef build_report(rows):\n ranked = sorted(rows,\
149
+ \ key=lambda r: r[\"amount\"], reverse=True)\n # Left over from 6b1a903: this 10 was never moved\
150
+ \ into CONFIG_KEYS.\n return {\"handlers\": sorted(HANDLERS), \"top\": ranked[:10]}\n"
151
+ src/ledger/retry.py: "\"\"\"Retry wrapper. Reads its attempt count from settings, never from a constant.\"\
152
+ \"\"\n\nimport time\n\nfrom ledger.config import load_settings\n\n\ndef with_retry(fn, *args):\n \
153
+ \ settings = load_settings()\n last = None\n for _attempt in range(settings.retry_max_attempts):\n\
154
+ \ try:\n return fn(*args)\n except Exception as exc:\n last =\
155
+ \ exc\n time.sleep(settings.retry_backoff_ms / 1000)\n raise last\n"
156
+ src/ledger/settle.py: "\"\"\"Batch settlement. Drives posting through the retry wrapper.\"\"\"\n\nfrom\
157
+ \ ledger.config import load_settings\nfrom ledger.errors import SettlementError\nfrom ledger.posting\
158
+ \ import post_entry\nfrom ledger.retry import with_retry\n\n\ndef settle_batch(entries):\n settings\
159
+ \ = load_settings()\n if len(entries) > settings.settle_batch_size:\n raise SettlementError(f\"\
160
+ batch of {len(entries)} exceeds settle_batch_size\")\n posted = []\n for entry in entries:\n\
161
+ \ posted.append(with_retry(post_entry, entry))\n return posted\n"
162
+ src/ledger/validate.py: "\"\"\"Entry validation. The only module that decides an entry is malformed.\"\
163
+ \"\"\n\nfrom ledger.errors import ValidationError\n\nREQUIRED_FIELDS = (\"amount\", \"ccy\")\n\n\n\
164
+ def validate_entry(entry):\n for field in REQUIRED_FIELDS:\n if field not in entry:\n \
165
+ \ raise ValidationError(f\"missing field {field!r}: {entry!r}\")\n if entry[\"amount\"\
166
+ ] == 0:\n raise ValidationError(f\"amount must be non-zero: {entry!r}\")\n return entry\n"
167
+ tests/test_posting.py: "import pytest\n\nfrom ledger.errors import ValidationError\nfrom ledger.posting\
168
+ \ import post_entry\n\n\ndef test_post_entry_rejects_zero_amount():\n with pytest.raises(ValidationError):\n\
169
+ \ post_entry({\"amount\": 0, \"ccy\": \"EUR\"})\n\n\ndef test_post_entry_converts_foreign_currency():\n\
170
+ \ assert post_entry({\"amount\": 10, \"ccy\": \"USD\"})[\"status\"] == \"converted\"\n"
171
+ tests/test_settle.py: "import pytest\n\nfrom ledger.errors import SettlementError\nfrom ledger.settle\
172
+ \ import settle_batch\n\n\ndef test_settle_batch_rejects_oversized_batch():\n with pytest.raises(SettlementError):\n\
173
+ \ settle_batch([{\"amount\": 1, \"ccy\": \"EUR\"}] * 201)\n"
174
+ prompt: 'src/ledger/config.py declares five keys in CONFIG_KEYS. Exactly one of them is read by no module
175
+ under src/ledger/. Which key? Read config.py for the key list, then read each module that could consume
176
+ a key. Do not answer from memory. Answer with ONLY this JSON, nothing else: {"key": "<key name>"}'
177
+ scoring:
178
+ kind: json_equal
179
+ expected:
180
+ key: report_top_n
@@ -0,0 +1,38 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell LARGE-IN. Corpus `inventory.xlsx`, 12,000 data rows, 258,129 extracted bytes
6
+ # (~64,532 est. tokens, 1.97x WORKER_NUM_CTX). The answer's data row is <= 400, so it
7
+ # sits INSIDE the declared head truncation (8,621 B = rendered rows 0..400; the cut moved
8
+ # there under Amendment 1, 2026-08-20, bar section 12 — pre-registration read 12,288 B =
9
+ # rows 0..570 and was sized on an estimator measured to be 2.9x wrong).
10
+ # The `paste` arm CAN answer here. This is the cell that can refute the reader.
11
+ name: doc-large-in-137
12
+ family: document-read
13
+ tools: []
14
+ document_setup:
15
+ - path: inventory.xlsx
16
+ seed: 4021
17
+ sheets:
18
+ - name: stock
19
+ rows: 12000
20
+ columns:
21
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
22
+ - {name: region, kind: choice, values: [north, south, east, west]}
23
+ - {name: units, kind: int, low: 1000, high: 9999}
24
+ answers:
25
+ question_sku: stock!A138
26
+ expected_region: stock!B138
27
+ expected_units: stock!C138
28
+ prompt: >-
29
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
30
+ and units, one row per sku. Find the single row whose sku is exactly SKU-000137 and
31
+ report that row's region and units. Do not compute anything and do not summarise the
32
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
33
+ {"region": "<region>", "units": <integer>}
34
+ scoring:
35
+ kind: json_equal
36
+ expected:
37
+ region: east
38
+ units: 7726
@@ -0,0 +1,44 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell LARGE-IN. Corpus `inventory.xlsx`, 12,000 data rows, 258,129 extracted bytes
6
+ # (~64,532 est. tokens, 1.97x WORKER_NUM_CTX). The answer's data row is <= 400, so it
7
+ # sits INSIDE the head truncation (8,621 B = rendered rows 0..400). The `paste` arm CAN
8
+ # answer here. This is the cell that can refute the reader.
9
+ #
10
+ # REPLACES doc-large-in-529 under Amendment 1 (2026-08-20, bar section 12). 529 was picked
11
+ # as the near-the-cut IN row when the cut was at data row 570, i.e. 41 rows inside it.
12
+ # Amendment 1 re-sized PASTE_MAX_BYTES 12,288 -> 8,621 on a MEASURED bytes-per-token ratio
13
+ # and the cut moved to data row 400, which put 529 OUTSIDE. This row is the same design
14
+ # element re-derived at the new boundary: 400 - 359 = 41, the identical offset, so a
15
+ # boundary-effect failure still has somewhere to show. The corpus, the seed, the columns,
16
+ # the prompt shape and the scoring are byte-identical to the row it replaces.
17
+ name: doc-large-in-359
18
+ family: document-read
19
+ tools: []
20
+ document_setup:
21
+ - path: inventory.xlsx
22
+ seed: 4021
23
+ sheets:
24
+ - name: stock
25
+ rows: 12000
26
+ columns:
27
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
28
+ - {name: region, kind: choice, values: [north, south, east, west]}
29
+ - {name: units, kind: int, low: 1000, high: 9999}
30
+ answers:
31
+ question_sku: stock!A360
32
+ expected_region: stock!B360
33
+ expected_units: stock!C360
34
+ prompt: >-
35
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
36
+ and units, one row per sku. Find the single row whose sku is exactly SKU-000359 and
37
+ report that row's region and units. Do not compute anything and do not summarise the
38
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
39
+ {"region": "<region>", "units": <integer>}
40
+ scoring:
41
+ kind: json_equal
42
+ expected:
43
+ region: south
44
+ units: 1187
@@ -0,0 +1,38 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell LARGE-IN. Corpus `inventory.xlsx`, 12,000 data rows, 258,129 extracted bytes
6
+ # (~64,532 est. tokens, 1.97x WORKER_NUM_CTX). The answer's data row is <= 400, so it
7
+ # sits INSIDE the declared head truncation (8,621 B = rendered rows 0..400; the cut moved
8
+ # there under Amendment 1, 2026-08-20, bar section 12 — pre-registration read 12,288 B =
9
+ # rows 0..570 and was sized on an estimator measured to be 2.9x wrong).
10
+ # The `paste` arm CAN answer here. This is the cell that can refute the reader.
11
+ name: doc-large-in-372
12
+ family: document-read
13
+ tools: []
14
+ document_setup:
15
+ - path: inventory.xlsx
16
+ seed: 4021
17
+ sheets:
18
+ - name: stock
19
+ rows: 12000
20
+ columns:
21
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
22
+ - {name: region, kind: choice, values: [north, south, east, west]}
23
+ - {name: units, kind: int, low: 1000, high: 9999}
24
+ answers:
25
+ question_sku: stock!A373
26
+ expected_region: stock!B373
27
+ expected_units: stock!C373
28
+ prompt: >-
29
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
30
+ and units, one row per sku. Find the single row whose sku is exactly SKU-000372 and
31
+ report that row's region and units. Do not compute anything and do not summarise the
32
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
33
+ {"region": "<region>", "units": <integer>}
34
+ scoring:
35
+ kind: json_equal
36
+ expected:
37
+ region: south
38
+ units: 7796
@@ -0,0 +1,37 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell LARGE-OUT. Corpus `inventory.xlsx`, 12,000 data rows, 258,129 extracted bytes
6
+ # (~64,532 est. tokens, 1.97x WORKER_NUM_CTX). The answer's data row is > 400 (it was
7
+ # > 570 as pre-registered; Amendment 1, 2026-08-20, moved the cut in), so it
8
+ # sits OUTSIDE the declared head truncation. The `paste` arm cannot answer here
9
+ # by construction. This is the cell that carries the job's claim.
10
+ name: doc-large-out-11764
11
+ family: document-read
12
+ tools: []
13
+ document_setup:
14
+ - path: inventory.xlsx
15
+ seed: 4021
16
+ sheets:
17
+ - name: stock
18
+ rows: 12000
19
+ columns:
20
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
21
+ - {name: region, kind: choice, values: [north, south, east, west]}
22
+ - {name: units, kind: int, low: 1000, high: 9999}
23
+ answers:
24
+ question_sku: stock!A11765
25
+ expected_region: stock!B11765
26
+ expected_units: stock!C11765
27
+ prompt: >-
28
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
29
+ and units, one row per sku. Find the single row whose sku is exactly SKU-011764 and
30
+ report that row's region and units. Do not compute anything and do not summarise the
31
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
32
+ {"region": "<region>", "units": <integer>}
33
+ scoring:
34
+ kind: json_equal
35
+ expected:
36
+ region: south
37
+ units: 7509
@@ -0,0 +1,37 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell LARGE-OUT. Corpus `inventory.xlsx`, 12,000 data rows, 258,129 extracted bytes
6
+ # (~64,532 est. tokens, 1.97x WORKER_NUM_CTX). The answer's data row is > 400 (it was
7
+ # > 570 as pre-registered; Amendment 1, 2026-08-20, moved the cut in), so it
8
+ # sits OUTSIDE the declared head truncation. The `paste` arm cannot answer here
9
+ # by construction. This is the cell that carries the job's claim.
10
+ name: doc-large-out-4137
11
+ family: document-read
12
+ tools: []
13
+ document_setup:
14
+ - path: inventory.xlsx
15
+ seed: 4021
16
+ sheets:
17
+ - name: stock
18
+ rows: 12000
19
+ columns:
20
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
21
+ - {name: region, kind: choice, values: [north, south, east, west]}
22
+ - {name: units, kind: int, low: 1000, high: 9999}
23
+ answers:
24
+ question_sku: stock!A4138
25
+ expected_region: stock!B4138
26
+ expected_units: stock!C4138
27
+ prompt: >-
28
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
29
+ and units, one row per sku. Find the single row whose sku is exactly SKU-004137 and
30
+ report that row's region and units. Do not compute anything and do not summarise the
31
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
32
+ {"region": "<region>", "units": <integer>}
33
+ scoring:
34
+ kind: json_equal
35
+ expected:
36
+ region: north
37
+ units: 7508
@@ -0,0 +1,37 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell LARGE-OUT. Corpus `inventory.xlsx`, 12,000 data rows, 258,129 extracted bytes
6
+ # (~64,532 est. tokens, 1.97x WORKER_NUM_CTX). The answer's data row is > 400 (it was
7
+ # > 570 as pre-registered; Amendment 1, 2026-08-20, moved the cut in), so it
8
+ # sits OUTSIDE the declared head truncation. The `paste` arm cannot answer here
9
+ # by construction. This is the cell that carries the job's claim.
10
+ name: doc-large-out-8022
11
+ family: document-read
12
+ tools: []
13
+ document_setup:
14
+ - path: inventory.xlsx
15
+ seed: 4021
16
+ sheets:
17
+ - name: stock
18
+ rows: 12000
19
+ columns:
20
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
21
+ - {name: region, kind: choice, values: [north, south, east, west]}
22
+ - {name: units, kind: int, low: 1000, high: 9999}
23
+ answers:
24
+ question_sku: stock!A8023
25
+ expected_region: stock!B8023
26
+ expected_units: stock!C8023
27
+ prompt: >-
28
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
29
+ and units, one row per sku. Find the single row whose sku is exactly SKU-008022 and
30
+ report that row's region and units. Do not compute anything and do not summarise the
31
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
32
+ {"region": "<region>", "units": <integer>}
33
+ scoring:
34
+ kind: json_equal
35
+ expected:
36
+ region: south
37
+ units: 8935
@@ -0,0 +1,37 @@
1
+ # PRE-REGISTERED by docs/eval-data/2026-08-20-document-read-bar.md, before any arm ran.
2
+ # `expected` was READ OUT OF the generated document (the same slice `answers:` names),
3
+ # never hand-written: the bar's gate G-1 rebuilds the fixture and fails the run if the
4
+ # two ever disagree.
5
+ # Cell SMALL-IN. Corpus `inventory-small.xlsx`, 400 data rows, 8,620 extracted bytes
6
+ # (~2,155 est. tokens, 0.07x WORKER_NUM_CTX). At PASTE_MAX_BYTES = 8,621 (Amendment 1,
7
+ # 2026-08-20; pre-registration read 12,288) the
8
+ # WHOLE corpus fits, so the `paste` arm here is a COMPLETE paste. This is the cell that
9
+ # separates "the reader works" from "paging works".
10
+ name: doc-small-137
11
+ family: document-read
12
+ tools: []
13
+ document_setup:
14
+ - path: inventory-small.xlsx
15
+ seed: 4021
16
+ sheets:
17
+ - name: stock
18
+ rows: 400
19
+ columns:
20
+ - {name: sku, kind: key, prefix: 'SKU-', width: 6}
21
+ - {name: region, kind: choice, values: [north, south, east, west]}
22
+ - {name: units, kind: int, low: 1000, high: 9999}
23
+ answers:
24
+ question_sku: stock!A138
25
+ expected_region: stock!B138
26
+ expected_units: stock!C138
27
+ prompt: >-
28
+ A workbook is attached to this task. Its sheet `stock` has the columns sku, region
29
+ and units, one row per sku. Find the single row whose sku is exactly SKU-000137 and
30
+ report that row's region and units. Do not compute anything and do not summarise the
31
+ sheet; read the one row. Answer with ONLY this JSON, nothing else:
32
+ {"region": "<region>", "units": <integer>}
33
+ scoring:
34
+ kind: json_equal
35
+ expected:
36
+ region: east
37
+ units: 7726