bantamkit 0.27.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bantamkit/__init__.py +32 -0
- bantamkit/agent.py +458 -0
- bantamkit/assets/contracts/default.yaml +90 -0
- bantamkit/assets/evals/devteam/manifest.yaml +351 -0
- bantamkit/assets/evals/devteam/repo/HISTORY.md +18 -0
- bantamkit/assets/evals/devteam/repo/README.md +12 -0
- bantamkit/assets/evals/devteam/repo/docs/architecture.md +17 -0
- bantamkit/assets/evals/devteam/repo/docs/runbook.md +10 -0
- bantamkit/assets/evals/devteam/repo/issues/142-settlement-timeout.md +23 -0
- bantamkit/assets/evals/devteam/repo/patches/0009-retry-budget.patch +38 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/__init__.py +3 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/config.py +35 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/errors.py +13 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/posting.py +12 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/registry.py +7 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/report.py +9 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/retry.py +17 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/settle.py +16 -0
- bantamkit/assets/evals/devteam/repo/src/ledger/validate.py +14 -0
- bantamkit/assets/evals/devteam/repo/tests/test_posting.py +13 -0
- bantamkit/assets/evals/devteam/repo/tests/test_settle.py +9 -0
- bantamkit/assets/evals/devteam/tasks/dt-error-contract.yaml +186 -0
- bantamkit/assets/evals/devteam/tasks/dt-handler-map.yaml +183 -0
- bantamkit/assets/evals/devteam/tasks/dt-patch-before-after.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-retry-attempts.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-settlement-config.yaml +185 -0
- bantamkit/assets/evals/devteam/tasks/dt-symbol-home.yaml +181 -0
- bantamkit/assets/evals/devteam/tasks/dt-trace-blame.yaml +182 -0
- bantamkit/assets/evals/devteam/tasks/dt-unread-key.yaml +180 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-137.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-359.yaml +44 -0
- bantamkit/assets/evals/document/tasks/doc-large-in-372.yaml +38 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-11764.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-4137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-large-out-8022.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-137.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-261.yaml +37 -0
- bantamkit/assets/evals/document/tasks/doc-small-388.yaml +37 -0
- bantamkit/assets/evals/fixtures/.gitkeep +0 -0
- bantamkit/assets/evals/fixtures/catalog.json +6 -0
- bantamkit/assets/evals/perturbations/task-completion.yaml +576 -0
- bantamkit/assets/evals/tasks/.gitkeep +0 -0
- bantamkit/assets/evals/tasks/extract-contact.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-invoice.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-order.yaml +15 -0
- bantamkit/assets/evals/tasks/extract-schedule.yaml +14 -0
- bantamkit/assets/evals/tasks/extract-versions.yaml +17 -0
- bantamkit/assets/evals/tasks/nav-prod-port.yaml +84 -0
- bantamkit/assets/evals/tasks/nav-release-bundle.yaml +87 -0
- bantamkit/assets/evals/tasks/recall-audit-retention.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-cache-ttl.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-db-port.yaml +17 -0
- bantamkit/assets/evals/tasks/recall-deploy.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-env-endpoint.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-oncall-rotation.yaml +21 -0
- bantamkit/assets/evals/tasks/recall-oncall.yaml +13 -0
- bantamkit/assets/evals/tasks/recall-org-quota.yaml +18 -0
- bantamkit/assets/evals/tasks/recall-owner.yaml +13 -0
- bantamkit/assets/evals/tasks/shop-basket-total.yaml +10 -0
- bantamkit/assets/evals/tasks/shop-cheapest.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-compare.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-gadget-value.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-stock-total.yaml +9 -0
- bantamkit/assets/evals/tasks/shop-total.yaml +9 -0
- bantamkit/assets/profiles/default.yaml +31 -0
- bantamkit/assets/profiles/patient.yaml +31 -0
- bantamkit/assets/rubrics/.gitkeep +0 -0
- bantamkit/assets/rubrics/code-quality.yaml +20 -0
- bantamkit/assets/rubrics/grounded-completion.yaml +37 -0
- bantamkit/assets/rubrics/task-completion.yaml +28 -0
- bantamkit/assets/schemas/shiftwork-checkpoint.json +188 -0
- bantamkit/assets/skills/.gitkeep +0 -0
- bantamkit/assets/skills/file-graph.md +7 -0
- bantamkit/assets/skills/memory.md +35 -0
- bantamkit/assets/tools/.gitkeep +0 -0
- bantamkit/assets/tools/bantamkit_read.json +48 -0
- bantamkit/assets/tools/bantamkit_status.json +25 -0
- bantamkit/assets/tools/build_identity.json +17 -0
- bantamkit/assets/tools/document_list.json +12 -0
- bantamkit/assets/tools/document_read.json +31 -0
- bantamkit/assets/tools/file_graph.json +12 -0
- bantamkit/assets/tools/memory_compact.json +31 -0
- bantamkit/assets/tools/memory_recall.json +38 -0
- bantamkit/assets/tools/memory_save.json +61 -0
- bantamkit/assets/tools/shiftwork_clock_in.json +25 -0
- bantamkit/assets/tools/shiftwork_clock_out.json +60 -0
- bantamkit/assets/tools/shiftwork_status.json +25 -0
- bantamkit/assets/tools/skill_audit.json +70 -0
- bantamkit/assets/tools/validate_json.json +31 -0
- bantamkit/assets.py +67 -0
- bantamkit/budget.py +114 -0
- bantamkit/client.py +329 -0
- bantamkit/contract.py +522 -0
- bantamkit/criticreplay.py +3241 -0
- bantamkit/critique.py +301 -0
- bantamkit/docread.py +1744 -0
- bantamkit/evalrun.py +2003 -0
- bantamkit/eventlog.py +282 -0
- bantamkit/filegraph.py +218 -0
- bantamkit/loopguard.py +101 -0
- bantamkit/mcpreport.py +763 -0
- bantamkit/mcpserver.py +1334 -0
- bantamkit/memory/__init__.py +28 -0
- bantamkit/memory/__main__.py +291 -0
- bantamkit/memory/component.py +569 -0
- bantamkit/memory/divergence.py +744 -0
- bantamkit/memory/layers.py +257 -0
- bantamkit/memory/store.py +940 -0
- bantamkit/pdfread.py +1402 -0
- bantamkit/profile.py +46 -0
- bantamkit/shiftwork.py +212 -0
- bantamkit/skillaudit.py +853 -0
- bantamkit/statusline.py +313 -0
- bantamkit/structured.py +125 -0
- bantamkit/textutil.py +30 -0
- bantamkit-0.27.0.dist-info/METADATA +207 -0
- bantamkit-0.27.0.dist-info/RECORD +119 -0
- bantamkit-0.27.0.dist-info/WHEEL +4 -0
- bantamkit-0.27.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
# The dev-team workload: source of truth.
|
|
2
|
+
#
|
|
3
|
+
# Dated 2026-08-17. Written by M2 of job `devteam-workload-and-null-control`
|
|
4
|
+
# BEFORE any arm was run, and before any token number for this workload existed.
|
|
5
|
+
#
|
|
6
|
+
# `assets/evals/tasks/` is FROZEN. This is a new asset beside it, run through the
|
|
7
|
+
# existing `--tasks` flag (`evalrun.py:736-738`); nothing in the frozen suite is
|
|
8
|
+
# edited, and the `workspace:` mechanism (`evalrun.py:93-122`, attached at
|
|
9
|
+
# `evalrun.py:435`) is extended in content only, never in code.
|
|
10
|
+
#
|
|
11
|
+
# `tasks/*.yaml` is GENERATED from this file by `tools/devteam/build_tasks.py`.
|
|
12
|
+
# Edit this file, re-run the builder, never hand-edit a generated task.
|
|
13
|
+
|
|
14
|
+
version: 1
|
|
15
|
+
date: 2026-08-17
|
|
16
|
+
|
|
17
|
+
surface:
|
|
18
|
+
name: svc-ledger
|
|
19
|
+
root: repo
|
|
20
|
+
origin: synthesised
|
|
21
|
+
# Recorded per the brief: pick one, record the threat accepted, say what detects it.
|
|
22
|
+
threat_accepted: unrealism
|
|
23
|
+
threat_statement: >
|
|
24
|
+
The surface is authored, not observed. Nothing guarantees its file sizes,
|
|
25
|
+
import fan-in, reference-chain depth or prose/code ratio resemble a repo a
|
|
26
|
+
dev team actually works on, and every one of those shapes the number of
|
|
27
|
+
reads a run makes — which is the quantity under measurement.
|
|
28
|
+
detector: >
|
|
29
|
+
A structural comparison against this repository's own runtime package
|
|
30
|
+
(runtime-py/src/bantamkit), printed by
|
|
31
|
+
docs/eval-data/2026-08-17-devteam-workload-measurements.py. Unrealism is
|
|
32
|
+
DETECTED if the synthesised surface falls outside the real package on
|
|
33
|
+
bytes-per-file range, internal imports per module, or maximum
|
|
34
|
+
internal-import fan-in. The detector is structural only: it cannot detect
|
|
35
|
+
unrealistic *task* choice, which is what the exclusion list below is for.
|
|
36
|
+
rejected_origins:
|
|
37
|
+
- origin: vendored open-source snapshot
|
|
38
|
+
threat: memorisation
|
|
39
|
+
why_rejected: >
|
|
40
|
+
A memorised repo can be answered without reading it. That suppresses
|
|
41
|
+
reads, and read count is the quantity under measurement — so the
|
|
42
|
+
contamination moves the measurand itself, in a direction nothing here can
|
|
43
|
+
bound. Detecting it needs a tools-removed arm, and M2 runs no arm.
|
|
44
|
+
- origin: point at bantamkit itself
|
|
45
|
+
threat: self-reference, and the asset stops being frozen
|
|
46
|
+
why_rejected: >
|
|
47
|
+
The content would change with every commit, so two arms run at different
|
|
48
|
+
shas would not be comparable — an evidence asset has to be immutable. And
|
|
49
|
+
this repo's own docs/eval*.md answer questions about the very mechanisms
|
|
50
|
+
being measured, so a run could read the answer rather than derive it.
|
|
51
|
+
|
|
52
|
+
# Every task gets the WHOLE repo as its workspace, never a curated subset: a
|
|
53
|
+
# workspace containing only the answer path is a pointer chase with no selection
|
|
54
|
+
# to do, and selection is the dev-team surface. The reference walk below is the
|
|
55
|
+
# path a depth-first pointer-follower takes; it is not a restriction on the run.
|
|
56
|
+
workspace_scope: whole-repo
|
|
57
|
+
|
|
58
|
+
# The reference strategy the walks are stated against. Named explicitly because
|
|
59
|
+
# re-read pressure is only defined relative to a strategy.
|
|
60
|
+
reference_strategy: >
|
|
61
|
+
Depth-first pointer-following without memoisation: read the file the prompt or
|
|
62
|
+
the previous observation points at, and resolve each pointer at the moment it
|
|
63
|
+
is encountered. A memoising solver re-reads nothing, so the walk-derived
|
|
64
|
+
pressure below is an UPPER bound on what a run will realise, not a prediction.
|
|
65
|
+
|
|
66
|
+
tasks:
|
|
67
|
+
|
|
68
|
+
- name: dt-retry-attempts
|
|
69
|
+
family: dev-repo-code
|
|
70
|
+
prompt: >-
|
|
71
|
+
How many attempts does the retry wrapper of svc-ledger make? Start with
|
|
72
|
+
list_files, then follow the documentation pointers from README.md. Do not
|
|
73
|
+
answer from memory, and do not answer with a number taken from a page that
|
|
74
|
+
tells you the value lives elsewhere. Answer with ONLY this JSON, nothing
|
|
75
|
+
else: {"attempts": <number>}
|
|
76
|
+
scoring:
|
|
77
|
+
kind: json_equal
|
|
78
|
+
expected: {attempts: 5}
|
|
79
|
+
rationale: >
|
|
80
|
+
The low-pressure control, and the discriminator against answering from
|
|
81
|
+
prose. Two wrong answers are reachable without reading config.py: the
|
|
82
|
+
runbook offers none (it refuses on purpose) and patches/0009 shows the
|
|
83
|
+
pre-patch constant 3. Only src/ledger/config.py carries 5. Included because
|
|
84
|
+
a workload needs tasks the mechanism cannot possibly help on; excluding them
|
|
85
|
+
is how a bar gets rigged.
|
|
86
|
+
walk:
|
|
87
|
+
- {path: README.md, pointer: 'README.md', pointer_in: prompt}
|
|
88
|
+
- {path: docs/runbook.md, pointer: 'docs/runbook.md'}
|
|
89
|
+
- {path: src/ledger/config.py, pointer: 'src/ledger/config.py'}
|
|
90
|
+
answer_evidence: ['"retry_max_attempts": 5']
|
|
91
|
+
|
|
92
|
+
- name: dt-symbol-home
|
|
93
|
+
family: dev-repo-code
|
|
94
|
+
prompt: >-
|
|
95
|
+
Which file of svc-ledger DEFINES the exception class that settle_batch
|
|
96
|
+
raises when a batch is too large? Use list_files first, then read the module
|
|
97
|
+
that raises and follow its import to the module that defines the class. Do
|
|
98
|
+
not answer from memory. Answer with ONLY this JSON, nothing else: {"path":
|
|
99
|
+
"<repo-relative path>"}
|
|
100
|
+
scoring:
|
|
101
|
+
kind: json_equal
|
|
102
|
+
expected: {path: src/ledger/errors.py}
|
|
103
|
+
rationale: >
|
|
104
|
+
Symbol resolution across files — M1 gap 4 — reduced to the one form this
|
|
105
|
+
harness can score today: "which file defines X", answered by following a
|
|
106
|
+
real import statement. Distinguishes the file that RAISES (settle.py) from
|
|
107
|
+
the file that DEFINES (errors.py), which is the distinction a symbol index
|
|
108
|
+
would exist to make and which no mechanism in the repo provides.
|
|
109
|
+
walk:
|
|
110
|
+
- {path: docs/architecture.md, pointer: 'docs/architecture.md', pointer_in: listing}
|
|
111
|
+
- {path: src/ledger/settle.py, pointer: 'settle.py'}
|
|
112
|
+
- {path: src/ledger/errors.py, pointer: 'from ledger.errors import SettlementError'}
|
|
113
|
+
answer_evidence: ['class SettlementError(LedgerError):']
|
|
114
|
+
|
|
115
|
+
- name: dt-handler-map
|
|
116
|
+
family: dev-repo-code
|
|
117
|
+
prompt: >-
|
|
118
|
+
For each of the three handlers registered in src/ledger/registry.py, report
|
|
119
|
+
whether the file that defines its entry function imports ledger.config. Read
|
|
120
|
+
the registry, then read each of the three files it points at. Do not answer
|
|
121
|
+
from memory. Answer with ONLY this JSON, nothing else: {"settle":
|
|
122
|
+
<true|false>, "post": <true|false>, "report": <true|false>}
|
|
123
|
+
scoring:
|
|
124
|
+
kind: json_equal
|
|
125
|
+
expected: {settle: true, post: true, report: false}
|
|
126
|
+
rationale: >
|
|
127
|
+
Cross-file references as a fan-out rather than a chain — M1 gap 5 — and the
|
|
128
|
+
only task here whose answer is not uniform, so a run cannot pass it by
|
|
129
|
+
guessing a constant. report.py deliberately does not import config, which is
|
|
130
|
+
also what makes dt-unread-key have a unique answer. Breadth, zero re-read
|
|
131
|
+
pressure by construction.
|
|
132
|
+
walk:
|
|
133
|
+
- {path: src/ledger/registry.py, pointer: 'src/ledger/registry.py', pointer_in: prompt}
|
|
134
|
+
- {path: src/ledger/settle.py, pointer: 'ledger.settle:settle_batch', pointer_in: src/ledger/registry.py}
|
|
135
|
+
- {path: src/ledger/posting.py, pointer: 'ledger.posting:post_entry', pointer_in: src/ledger/registry.py}
|
|
136
|
+
- {path: src/ledger/report.py, pointer: 'ledger.report:build_report', pointer_in: src/ledger/registry.py}
|
|
137
|
+
answer_evidence: ['from ledger.registry import HANDLERS']
|
|
138
|
+
|
|
139
|
+
- name: dt-trace-blame
|
|
140
|
+
family: dev-repo-history
|
|
141
|
+
prompt: >-
|
|
142
|
+
issues/142-settlement-timeout.md contains a captured traceback. Work out
|
|
143
|
+
which commit listed in HISTORY.md introduced the check that raised, and
|
|
144
|
+
which file that check lives in. Read the issue, then the file the deepest
|
|
145
|
+
frame names, then the history. Do not answer from memory. Answer with ONLY
|
|
146
|
+
this JSON, nothing else: {"commit": "<short sha>", "path": "<repo-relative
|
|
147
|
+
path>"}
|
|
148
|
+
scoring:
|
|
149
|
+
kind: json_equal
|
|
150
|
+
expected: {commit: 5c0de41, path: src/ledger/validate.py}
|
|
151
|
+
rationale: >
|
|
152
|
+
This is the task that puts HISTORY on the surface — M1 gap 6, and the
|
|
153
|
+
brief's "at least one of diff or history". It joins three different kinds of
|
|
154
|
+
artifact (a stack trace, source, a commit log) which is the join a dev
|
|
155
|
+
actually performs and which 0 of the 22 frozen tasks contain. The deepest
|
|
156
|
+
frame names validate.py:13; only one commit subject in HISTORY.md is about
|
|
157
|
+
rejecting zero-amount entries.
|
|
158
|
+
walk:
|
|
159
|
+
- {path: issues/142-settlement-timeout.md, pointer: 'issues/142-settlement-timeout.md', pointer_in: prompt}
|
|
160
|
+
- {path: src/ledger/validate.py, pointer: '"src/ledger/validate.py", line 13'}
|
|
161
|
+
- {path: HISTORY.md, pointer: 'HISTORY.md', pointer_in: prompt}
|
|
162
|
+
answer_evidence: ['5c0de41 2026-07-08 fix(validate): reject zero-amount entries']
|
|
163
|
+
|
|
164
|
+
- name: dt-patch-before-after
|
|
165
|
+
family: dev-repo-history
|
|
166
|
+
prompt: >-
|
|
167
|
+
patches/0009-retry-budget.patch changed how src/ledger/retry.py obtains its
|
|
168
|
+
attempt count. Report the attempt count the code used BEFORE that patch, and
|
|
169
|
+
the attempt count it uses at HEAD. Read the patch, then the file at HEAD,
|
|
170
|
+
then whatever that file reads its value from. Do not answer from memory.
|
|
171
|
+
Answer with ONLY this JSON, nothing else: {"before": <number>, "head":
|
|
172
|
+
<number>}
|
|
173
|
+
scoring:
|
|
174
|
+
kind: json_equal
|
|
175
|
+
expected: {before: 3, head: 5}
|
|
176
|
+
rationale: >
|
|
177
|
+
A DIFF as first-class input — the readable half of M1 gap 3. Scoring an
|
|
178
|
+
edit is impossible today (no write tool, and score_output has no diff kind,
|
|
179
|
+
evalrun.py:233-247), so this task asks the run to READ a patch, which is
|
|
180
|
+
what the harness can score. It is also the one task where two files disagree
|
|
181
|
+
on purpose: the patch's removed line says 3, config.py says 5, and only a
|
|
182
|
+
run that reads both gets both fields right.
|
|
183
|
+
walk:
|
|
184
|
+
- {path: patches/0009-retry-budget.patch, pointer: 'patches/0009-retry-budget.patch', pointer_in: prompt}
|
|
185
|
+
- {path: src/ledger/retry.py, pointer: 'src/ledger/retry.py'}
|
|
186
|
+
- {path: src/ledger/config.py, pointer: 'from ledger.config import load_settings'}
|
|
187
|
+
answer_evidence: ['- for _attempt in range(_MAX_ATTEMPTS):', '"retry_max_attempts": 5']
|
|
188
|
+
|
|
189
|
+
- name: dt-error-contract
|
|
190
|
+
family: dev-repo-code
|
|
191
|
+
prompt: >-
|
|
192
|
+
Two questions about svc-ledger's exceptions. (a) An entry with amount 0
|
|
193
|
+
reaches posting.post_entry — name the exception class that leaves
|
|
194
|
+
post_entry, and the file that DEFINES it. (b) Name the exception class
|
|
195
|
+
settle.settle_batch raises when a batch is too large, and the file that
|
|
196
|
+
DEFINES it. For each, read the module that raises and then the module that
|
|
197
|
+
defines the class; do not answer from memory. Answer with ONLY this JSON,
|
|
198
|
+
nothing else: {"a_class": "<name>", "a_path": "<path>", "b_class": "<name>",
|
|
199
|
+
"b_path": "<path>"}
|
|
200
|
+
scoring:
|
|
201
|
+
kind: json_equal
|
|
202
|
+
expected:
|
|
203
|
+
a_class: ValidationError
|
|
204
|
+
a_path: src/ledger/errors.py
|
|
205
|
+
b_class: SettlementError
|
|
206
|
+
b_path: src/ledger/errors.py
|
|
207
|
+
rationale: >
|
|
208
|
+
The first of the two shared-hub tasks: two independent chains
|
|
209
|
+
(posting -> validate -> errors, and settle -> errors) both terminate at
|
|
210
|
+
errors.py, so a depth-first solver reads errors.py twice. The re-read is a
|
|
211
|
+
property of the repo's import graph, not of the wording — which is the only
|
|
212
|
+
honest way this workload generates re-read pressure at all. Also covers the
|
|
213
|
+
raises-vs-defines distinction on a second, deeper chain than dt-symbol-home.
|
|
214
|
+
walk:
|
|
215
|
+
- {path: src/ledger/posting.py, pointer: 'posting.post_entry', pointer_in: prompt}
|
|
216
|
+
- {path: src/ledger/validate.py, pointer: 'from ledger.validate import validate_entry'}
|
|
217
|
+
- {path: src/ledger/errors.py, pointer: 'from ledger.errors import ValidationError'}
|
|
218
|
+
- {path: src/ledger/settle.py, pointer: 'settle.settle_batch', pointer_in: prompt}
|
|
219
|
+
- {path: src/ledger/errors.py, pointer: 'from ledger.errors import SettlementError'}
|
|
220
|
+
answer_evidence: ['class ValidationError(LedgerError):', 'class SettlementError(LedgerError):']
|
|
221
|
+
|
|
222
|
+
- name: dt-settlement-config
|
|
223
|
+
family: dev-repo-code
|
|
224
|
+
prompt: >-
|
|
225
|
+
The settlement path of svc-ledger is settle.settle_batch ->
|
|
226
|
+
retry.with_retry -> posting.post_entry. For each of those three modules,
|
|
227
|
+
find every config key it reads and report that key's default. Resolve each
|
|
228
|
+
key's default in src/ledger/config.py at the moment you encounter the key;
|
|
229
|
+
do not answer from memory. Answer with ONLY this JSON, nothing else:
|
|
230
|
+
{"settle_batch_size": <number>, "retry_max_attempts": <number>,
|
|
231
|
+
"retry_backoff_ms": <number>, "posting_currency": "<string>"}
|
|
232
|
+
scoring:
|
|
233
|
+
kind: json_equal
|
|
234
|
+
expected:
|
|
235
|
+
settle_batch_size: 200
|
|
236
|
+
retry_max_attempts: 5
|
|
237
|
+
retry_backoff_ms: 250
|
|
238
|
+
posting_currency: EUR
|
|
239
|
+
rationale: >
|
|
240
|
+
The highest-pressure task in the workload, and the second shared-hub one:
|
|
241
|
+
three modules on one call path each read config.py, so a depth-first solver
|
|
242
|
+
reads config.py three times. This is the structure a real settings module
|
|
243
|
+
has — one hub, many consumers — so the pressure comes from the repo, not
|
|
244
|
+
from an instruction to re-read. It is the task the collapse mechanism has
|
|
245
|
+
the most to act on, which is exactly why the bar is committed before it runs.
|
|
246
|
+
walk:
|
|
247
|
+
- {path: src/ledger/settle.py, pointer: 'settle.settle_batch', pointer_in: prompt}
|
|
248
|
+
- {path: src/ledger/config.py, pointer: 'from ledger.config import load_settings'}
|
|
249
|
+
- {path: src/ledger/retry.py, pointer: 'retry.with_retry', pointer_in: prompt}
|
|
250
|
+
- {path: src/ledger/config.py, pointer: 'from ledger.config import load_settings'}
|
|
251
|
+
- {path: src/ledger/posting.py, pointer: 'posting.post_entry', pointer_in: prompt}
|
|
252
|
+
- {path: src/ledger/config.py, pointer: 'from ledger.config import load_settings'}
|
|
253
|
+
answer_evidence: ['"settle_batch_size": 200', '"retry_backoff_ms": 250', '"posting_currency": "EUR"']
|
|
254
|
+
|
|
255
|
+
- name: dt-unread-key
|
|
256
|
+
family: dev-repo-code
|
|
257
|
+
prompt: >-
|
|
258
|
+
src/ledger/config.py declares five keys in CONFIG_KEYS. Exactly one of them
|
|
259
|
+
is read by no module under src/ledger/. Which key? Read config.py for the
|
|
260
|
+
key list, then read each module that could consume a key. Do not answer from
|
|
261
|
+
memory. Answer with ONLY this JSON, nothing else: {"key": "<key name>"}
|
|
262
|
+
scoring:
|
|
263
|
+
kind: json_equal
|
|
264
|
+
expected: {key: report_top_n}
|
|
265
|
+
rationale: >
|
|
266
|
+
A negative question — "which of these is used nowhere" — which cannot be
|
|
267
|
+
answered from any single file and has no pointer chain to follow. It is the
|
|
268
|
+
breadth end of the workload: correctness requires covering the surface, not
|
|
269
|
+
following it. Included as a counterweight to the two chain-shaped
|
|
270
|
+
high-pressure tasks, so the bar cannot be read as depending on chains.
|
|
271
|
+
walk:
|
|
272
|
+
- {path: src/ledger/config.py, pointer: 'src/ledger/config.py', pointer_in: prompt}
|
|
273
|
+
- {path: src/ledger/settle.py, pointer: 'src/ledger/settle.py', pointer_in: listing}
|
|
274
|
+
- {path: src/ledger/posting.py, pointer: 'src/ledger/posting.py', pointer_in: listing}
|
|
275
|
+
- {path: src/ledger/retry.py, pointer: 'src/ledger/retry.py', pointer_in: listing}
|
|
276
|
+
- {path: src/ledger/validate.py, pointer: 'src/ledger/validate.py', pointer_in: listing}
|
|
277
|
+
- {path: src/ledger/report.py, pointer: 'src/ledger/report.py', pointer_in: listing}
|
|
278
|
+
answer_evidence: ['"report_top_n": 10', 'ranked[:10]']
|
|
279
|
+
|
|
280
|
+
# What was considered and rejected. Per the brief, this is as much the record as
|
|
281
|
+
# what was kept: a workload chosen after seeing which tasks a mechanism helps is
|
|
282
|
+
# a rigged bar, so the rejections are written down with their reasons.
|
|
283
|
+
excluded:
|
|
284
|
+
|
|
285
|
+
- candidate: dt-verify-runbook — read the runbook, check the code, then re-read
|
|
286
|
+
the runbook to confirm the exact wording.
|
|
287
|
+
reason: >
|
|
288
|
+
The second read is only NECESSARY if the agent has forgotten the first, and
|
|
289
|
+
nothing in the task derives that. The pressure would be asserted rather than
|
|
290
|
+
derived, which is precisely the rig M5 is briefed to find.
|
|
291
|
+
|
|
292
|
+
- candidate: any task instructing the run to read a file twice, or to re-read
|
|
293
|
+
everything before answering.
|
|
294
|
+
reason: >
|
|
295
|
+
Manufactures re-read pressure by fiat. The mechanism under test pays only on
|
|
296
|
+
repeat reads, so an instruction to repeat reads is an instruction to produce
|
|
297
|
+
the result.
|
|
298
|
+
|
|
299
|
+
- candidate: authoring repo files larger than the observation budget so that
|
|
300
|
+
each read is truncated and must be taken in slices.
|
|
301
|
+
reason: >
|
|
302
|
+
Raises pressure by breaking the reader rather than by describing real work.
|
|
303
|
+
It would also corrupt the accounting: `truncate` runs AFTER FileAccessGraph
|
|
304
|
+
returns (agent.py:198, budget 4096 B in assets/profiles/default.yaml), so
|
|
305
|
+
filegraph.py:84's `size` — the full observation's bytes — would overstate
|
|
306
|
+
the bytes actually removed from context. Every file in this surface is kept
|
|
307
|
+
under 4096 B and the measurement script verifies it.
|
|
308
|
+
|
|
309
|
+
- candidate: dt-apply-patch — have the run produce the edit rather than read it.
|
|
310
|
+
reason: >
|
|
311
|
+
Not scoreable. There is no write tool anywhere (`WORKSPACE_TOOLS`,
|
|
312
|
+
evalrun.py:84) and `score_output` supports exactly three kinds, none of them
|
|
313
|
+
a diff (evalrun.py:233-247). M1 gap 3. Scoring an edit needs a fourth kind
|
|
314
|
+
and a mutable workspace; both are out of this unit's scope.
|
|
315
|
+
|
|
316
|
+
- candidate: dt-changed-file — a task where a workspace file changes mid-run so
|
|
317
|
+
that FileRead.changed fires.
|
|
318
|
+
reason: >
|
|
319
|
+
Unreachable for the same missing write tool. `FileRead.changed`
|
|
320
|
+
(filegraph.py:21) can only ever be False and the "CHANGED since your last
|
|
321
|
+
read" branch (filegraph.py:90) is dead in any suite this harness can run.
|
|
322
|
+
M1 gap 2. Recorded here so no one reads its absence as an oversight.
|
|
323
|
+
|
|
324
|
+
- candidate: dt-symbol-search — "where is X defined", answered with a search or
|
|
325
|
+
grep tool.
|
|
326
|
+
reason: >
|
|
327
|
+
Would measure a tool that does not exist. The only file tools are read_file
|
|
328
|
+
and list_files (evalrun.py:84). dt-symbol-home asks the same question in the
|
|
329
|
+
form the harness can actually serve: follow a real import.
|
|
330
|
+
|
|
331
|
+
- candidate: dt-config-fanin — read all nine modules under src/ledger/ and
|
|
332
|
+
report the most-imported one.
|
|
333
|
+
reason: >
|
|
334
|
+
Infeasible, not uninteresting. Nine reads plus an answer turn is exactly
|
|
335
|
+
`max_turns` 10 (assets/profiles/default.yaml), leaving no slack for a schema
|
|
336
|
+
retry or a critique round — so a turns-exhausted run would be scored as a
|
|
337
|
+
wrong answer and the task would measure the turn budget. dt-unread-key keeps
|
|
338
|
+
the breadth shape at six reads.
|
|
339
|
+
|
|
340
|
+
- candidate: any task whose answer is an arithmetic aggregate over many files
|
|
341
|
+
(counts, sums, totals).
|
|
342
|
+
reason: >
|
|
343
|
+
A failure would be an arithmetic failure, not a navigation failure. It adds
|
|
344
|
+
variance without adding surface, and the frozen `shop-` family already
|
|
345
|
+
measures arithmetic-under-tool-use.
|
|
346
|
+
|
|
347
|
+
- candidate: selecting tasks by which ones filegraph helps.
|
|
348
|
+
reason: >
|
|
349
|
+
Refused by construction, and the construction is the evidence: no arm has
|
|
350
|
+
run, the bar is committed before any run, and the pressure figures are
|
|
351
|
+
derived from the repo's import graph rather than from any trace.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Release history
|
|
2
|
+
|
|
3
|
+
`git log --oneline --date=short --pretty='%h %ad %s'`, newest first. Patch files
|
|
4
|
+
for selected commits are under `patches/`.
|
|
5
|
+
|
|
6
|
+
```
|
|
7
|
+
9f2c1ab 2026-07-29 fix(retry): read retry_max_attempts from settings, not the constant
|
|
8
|
+
7d4e88c 2026-07-22 feat(settle): batch settlement with per-entry retry
|
|
9
|
+
6b1a903 2026-07-15 refactor(config): move every tunable into CONFIG_KEYS
|
|
10
|
+
5c0de41 2026-07-08 fix(validate): reject zero-amount entries
|
|
11
|
+
4a9bb27 2026-07-01 feat(report): render a per-handler summary
|
|
12
|
+
3e88f10 2026-06-24 feat(posting): post_entry writes one entry
|
|
13
|
+
2d5a6c9 2026-06-17 feat(errors): ValidationError and SettlementError
|
|
14
|
+
1c4f0b8 2026-06-10 chore: initial layout
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Branch `main` is at `9f2c1ab`. Tags: `v0.9.0` on `9f2c1ab`, `v0.8.0` on
|
|
18
|
+
`5c0de41`.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# svc-ledger
|
|
2
|
+
|
|
3
|
+
Double-entry ledger service. Three handlers, one settings module, no globals.
|
|
4
|
+
|
|
5
|
+
- Module map and the call order between modules: `docs/architecture.md`
|
|
6
|
+
- Operating procedure, escalation, and where the tunables live: `docs/runbook.md`
|
|
7
|
+
- Release history, newest first, with commit ids: `HISTORY.md`
|
|
8
|
+
- Patch files for selected commits: `patches/`
|
|
9
|
+
- Open incidents, including captured tracebacks: `issues/`
|
|
10
|
+
|
|
11
|
+
Handlers are resolved at import time from `src/ledger/registry.py`. Nothing else
|
|
12
|
+
in the service knows the handler names.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# Architecture
|
|
2
|
+
|
|
3
|
+
The three entry points are the handlers listed in `src/ledger/registry.py`.
|
|
4
|
+
|
|
5
|
+
- `settle.py` drives a batch. It wraps `posting.post_entry` in
|
|
6
|
+
`retry.with_retry`, and raises its own error when a batch is too large.
|
|
7
|
+
- `posting.py` writes one entry. It validates first, via
|
|
8
|
+
`validate.validate_entry`, and defines no exception of its own.
|
|
9
|
+
- `report.py` renders a summary. It reads the registry and imports no handler.
|
|
10
|
+
- `validate.py` is the only module that decides an entry is malformed.
|
|
11
|
+
- `errors.py` defines every exception class in the service. No other module
|
|
12
|
+
declares one.
|
|
13
|
+
|
|
14
|
+
Every tunable is a key in `src/ledger/config.py`; that module is the only one
|
|
15
|
+
that touches `os.environ`. Modules ask `load_settings()` for values, so an
|
|
16
|
+
attribute read on a `Settings` object is how you tell that a module consumes a
|
|
17
|
+
config key.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Runbook
|
|
2
|
+
|
|
3
|
+
Attempt counts, backoff intervals and batch sizes are **not written on this
|
|
4
|
+
page** and never will be. Every one of them is a key with its default in
|
|
5
|
+
`src/ledger/config.py` — read that file for the current value. Do not answer an
|
|
6
|
+
operational question about a number from this page; it will be stale.
|
|
7
|
+
|
|
8
|
+
Escalation: three consecutive settlement failures page the on-call. Foreign
|
|
9
|
+
currency entries are converted, not rejected. A malformed entry is a caller bug
|
|
10
|
+
and is never retried past the wrapper's own attempt budget.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# 142 — settlement aborts on a zero-amount entry
|
|
2
|
+
|
|
3
|
+
Reported by ops on 2026-08-02 against `main` (`9f2c1ab`). One entry in a batch
|
|
4
|
+
of 40 carried `amount: 0`; the whole batch aborted after the wrapper exhausted
|
|
5
|
+
its attempts. Captured traceback, verbatim from the worker log:
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
Traceback (most recent call last):
|
|
9
|
+
File "src/ledger/settle.py", line 15, in settle_batch
|
|
10
|
+
posted.append(with_retry(post_entry, entry))
|
|
11
|
+
File "src/ledger/retry.py", line 17, in with_retry
|
|
12
|
+
raise last
|
|
13
|
+
File "src/ledger/posting.py", line 9, in post_entry
|
|
14
|
+
validate_entry(entry)
|
|
15
|
+
File "src/ledger/validate.py", line 13, in validate_entry
|
|
16
|
+
raise ValidationError(f"amount must be non-zero: {entry!r}")
|
|
17
|
+
ledger.errors.ValidationError: amount must be non-zero: {'amount': 0, 'ccy': 'EUR'}
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
Ops question: which change put this check in, and should a caller bug really
|
|
21
|
+
burn the whole attempt budget? Runbook says a malformed entry is never retried
|
|
22
|
+
past the wrapper's budget, so behaviour matches the runbook; the argument is
|
|
23
|
+
about whether the budget should apply at all.
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
From 9f2c1ab5e2d84c1f0a7b6935cc41d0e2a8f37b41 Mon Sep 17 00:00:00 2001
|
|
2
|
+
Date: Wed, 29 Jul 2026 09:14:02 +0000
|
|
3
|
+
Subject: [PATCH] fix(retry): read retry_max_attempts from settings, not the
|
|
4
|
+
constant
|
|
5
|
+
|
|
6
|
+
The wrapper had its own hard-coded attempt count, so raising the deployed
|
|
7
|
+
budget changed nothing. It now asks load_settings() like every other module.
|
|
8
|
+
---
|
|
9
|
+
src/ledger/retry.py | 8 +++++---
|
|
10
|
+
1 file changed, 5 insertions(+), 3 deletions(-)
|
|
11
|
+
|
|
12
|
+
diff --git a/src/ledger/retry.py b/src/ledger/retry.py
|
|
13
|
+
--- a/src/ledger/retry.py
|
|
14
|
+
+++ b/src/ledger/retry.py
|
|
15
|
+
@@ -1,14 +1,15 @@
|
|
16
|
+
-"""Retry wrapper. Attempts are fixed at the module constant below."""
|
|
17
|
+
+"""Retry wrapper. Reads its attempt count from settings, never from a constant."""
|
|
18
|
+
|
|
19
|
+
import time
|
|
20
|
+
|
|
21
|
+
-_MAX_ATTEMPTS = 3
|
|
22
|
+
+from ledger.config import load_settings
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def with_retry(fn, *args):
|
|
26
|
+
+ settings = load_settings()
|
|
27
|
+
last = None
|
|
28
|
+
- for _attempt in range(_MAX_ATTEMPTS):
|
|
29
|
+
+ for _attempt in range(settings.retry_max_attempts):
|
|
30
|
+
try:
|
|
31
|
+
return fn(*args)
|
|
32
|
+
except Exception as exc:
|
|
33
|
+
last = exc
|
|
34
|
+
- time.sleep(0.25)
|
|
35
|
+
+ time.sleep(settings.retry_backoff_ms / 1000)
|
|
36
|
+
raise last
|
|
37
|
+
--
|
|
38
|
+
2.45.2
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""The only module in svc-ledger that reads os.environ.
|
|
2
|
+
|
|
3
|
+
Every tunable is a key in CONFIG_KEYS with its default beside it. docs/runbook.md
|
|
4
|
+
deliberately does not repeat these values: one place, one default.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
|
|
9
|
+
CONFIG_KEYS = {
|
|
10
|
+
"retry_max_attempts": 5,
|
|
11
|
+
"retry_backoff_ms": 250,
|
|
12
|
+
"settle_batch_size": 200,
|
|
13
|
+
"posting_currency": "EUR",
|
|
14
|
+
"report_top_n": 10,
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Settings:
|
|
19
|
+
"""Attribute access over CONFIG_KEYS, environment first."""
|
|
20
|
+
|
|
21
|
+
def __init__(self, values):
|
|
22
|
+
self.values = values
|
|
23
|
+
|
|
24
|
+
def __getattr__(self, name):
|
|
25
|
+
if name not in self.values:
|
|
26
|
+
raise AttributeError(f"unknown config key: {name}")
|
|
27
|
+
return self.values[name]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def load_settings():
|
|
31
|
+
values = {}
|
|
32
|
+
for key, default in CONFIG_KEYS.items():
|
|
33
|
+
raw = os.environ.get("LEDGER_" + key.upper())
|
|
34
|
+
values[key] = type(default)(raw) if raw is not None else default
|
|
35
|
+
return Settings(values)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Exception hierarchy for svc-ledger. Every module raises from here."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class LedgerError(Exception):
|
|
5
|
+
"""Base class. Nothing raises this directly."""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ValidationError(LedgerError):
|
|
9
|
+
"""An entry failed validate_entry."""
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class SettlementError(LedgerError):
|
|
13
|
+
"""A batch could not be settled."""
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Post one entry to the ledger. Validates before writing; raises nothing itself."""
|
|
2
|
+
|
|
3
|
+
from ledger.config import load_settings
|
|
4
|
+
from ledger.validate import validate_entry
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def post_entry(entry):
|
|
8
|
+
settings = load_settings()
|
|
9
|
+
validate_entry(entry)
|
|
10
|
+
if entry["ccy"] != settings.posting_currency:
|
|
11
|
+
return {"status": "converted", "ccy": settings.posting_currency}
|
|
12
|
+
return {"status": "posted", "ccy": entry["ccy"]}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Per-handler summary. Reads the registry; imports no handler and no settings."""
|
|
2
|
+
|
|
3
|
+
from ledger.registry import HANDLERS
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def build_report(rows):
|
|
7
|
+
ranked = sorted(rows, key=lambda r: r["amount"], reverse=True)
|
|
8
|
+
# Left over from 6b1a903: this 10 was never moved into CONFIG_KEYS.
|
|
9
|
+
return {"handlers": sorted(HANDLERS), "top": ranked[:10]}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Retry wrapper. Reads its attempt count from settings, never from a constant."""
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
|
|
5
|
+
from ledger.config import load_settings
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def with_retry(fn, *args):
|
|
9
|
+
settings = load_settings()
|
|
10
|
+
last = None
|
|
11
|
+
for _attempt in range(settings.retry_max_attempts):
|
|
12
|
+
try:
|
|
13
|
+
return fn(*args)
|
|
14
|
+
except Exception as exc:
|
|
15
|
+
last = exc
|
|
16
|
+
time.sleep(settings.retry_backoff_ms / 1000)
|
|
17
|
+
raise last
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Batch settlement. Drives posting through the retry wrapper."""
|
|
2
|
+
|
|
3
|
+
from ledger.config import load_settings
|
|
4
|
+
from ledger.errors import SettlementError
|
|
5
|
+
from ledger.posting import post_entry
|
|
6
|
+
from ledger.retry import with_retry
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def settle_batch(entries):
|
|
10
|
+
settings = load_settings()
|
|
11
|
+
if len(entries) > settings.settle_batch_size:
|
|
12
|
+
raise SettlementError(f"batch of {len(entries)} exceeds settle_batch_size")
|
|
13
|
+
posted = []
|
|
14
|
+
for entry in entries:
|
|
15
|
+
posted.append(with_retry(post_entry, entry))
|
|
16
|
+
return posted
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Entry validation. The only module that decides an entry is malformed."""
|
|
2
|
+
|
|
3
|
+
from ledger.errors import ValidationError
|
|
4
|
+
|
|
5
|
+
REQUIRED_FIELDS = ("amount", "ccy")
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def validate_entry(entry):
|
|
9
|
+
for field in REQUIRED_FIELDS:
|
|
10
|
+
if field not in entry:
|
|
11
|
+
raise ValidationError(f"missing field {field!r}: {entry!r}")
|
|
12
|
+
if entry["amount"] == 0:
|
|
13
|
+
raise ValidationError(f"amount must be non-zero: {entry!r}")
|
|
14
|
+
return entry
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import pytest
|
|
2
|
+
|
|
3
|
+
from ledger.errors import ValidationError
|
|
4
|
+
from ledger.posting import post_entry
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def test_post_entry_rejects_zero_amount():
|
|
8
|
+
with pytest.raises(ValidationError):
|
|
9
|
+
post_entry({"amount": 0, "ccy": "EUR"})
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_post_entry_converts_foreign_currency():
|
|
13
|
+
assert post_entry({"amount": 10, "ccy": "USD"})["status"] == "converted"
|