copela 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- copela/__init__.py +61 -0
- copela/budget.py +84 -0
- copela/cli.py +158 -0
- copela/ledger.py +259 -0
- copela/oracles/__init__.py +1 -0
- copela/oracles/properties.py +317 -0
- copela/providers/__init__.py +38 -0
- copela/providers/base.py +140 -0
- copela/providers/hosted.py +272 -0
- copela/py.typed +0 -0
- copela/report.py +177 -0
- copela/solvers/__init__.py +1 -0
- copela/solvers/highs.py +158 -0
- copela/sweep.py +286 -0
- copela/verdicts.py +201 -0
- copela-0.1.0.dist-info/METADATA +185 -0
- copela-0.1.0.dist-info/RECORD +21 -0
- copela-0.1.0.dist-info/WHEEL +5 -0
- copela-0.1.0.dist-info/entry_points.txt +2 -0
- copela-0.1.0.dist-info/licenses/LICENSE +21 -0
- copela-0.1.0.dist-info/top_level.txt +1 -0
copela/__init__.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""copela: run narrative-to-formal translation across many models, and score it with oracles
|
|
2
|
+
that are not language models.
|
|
3
|
+
|
|
4
|
+
The name is the cupel used in fire assay, the vessel that separates the metal from the lead. That is
|
|
5
|
+
the job here: separating a formalization that is faithful from one that merely runs.
|
|
6
|
+
|
|
7
|
+
Four layers, reported separately and never merged into one score:
|
|
8
|
+
|
|
9
|
+
1. executable, did it run, solve, compile
|
|
10
|
+
2. structural, is it the same model as the reference
|
|
11
|
+
3. property, do the invariants of this class hold
|
|
12
|
+
4. judge, what a model says, recorded as a labelled screening aggregate and never as truth
|
|
13
|
+
|
|
14
|
+
The headline is the subtraction: how often the artifact ran, minus how often it was right.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from .budget import Budget, BudgetExceeded, estimate
|
|
20
|
+
from .ledger import CallKey, Ledger, LedgerError, Record
|
|
21
|
+
from .providers import Provider, ProviderError, StubProvider
|
|
22
|
+
from .report import Cell, Report, build
|
|
23
|
+
from .sweep import Case, Sweep, Target
|
|
24
|
+
from .verdicts import (
|
|
25
|
+
JUDGE_LABEL,
|
|
26
|
+
CandidateVerdict,
|
|
27
|
+
Layer,
|
|
28
|
+
LayerResult,
|
|
29
|
+
Outcome,
|
|
30
|
+
Rate,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
__version__ = "0.1.0"
|
|
34
|
+
__display_version__ = "0.01.000"
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
"JUDGE_LABEL",
|
|
38
|
+
"Budget",
|
|
39
|
+
"BudgetExceeded",
|
|
40
|
+
"CallKey",
|
|
41
|
+
"CandidateVerdict",
|
|
42
|
+
"Case",
|
|
43
|
+
"Cell",
|
|
44
|
+
"Layer",
|
|
45
|
+
"LayerResult",
|
|
46
|
+
"Ledger",
|
|
47
|
+
"LedgerError",
|
|
48
|
+
"Outcome",
|
|
49
|
+
"Provider",
|
|
50
|
+
"ProviderError",
|
|
51
|
+
"Rate",
|
|
52
|
+
"Record",
|
|
53
|
+
"Report",
|
|
54
|
+
"StubProvider",
|
|
55
|
+
"Sweep",
|
|
56
|
+
"Target",
|
|
57
|
+
"__display_version__",
|
|
58
|
+
"__version__",
|
|
59
|
+
"build",
|
|
60
|
+
"estimate",
|
|
61
|
+
]
|
copela/budget.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""The budget guard: stop before the limit, not after it.
|
|
2
|
+
|
|
3
|
+
A sweep is cases times models times repeats, and each cell costs money. The guard exists because the
|
|
4
|
+
failure it prevents has happened on this account: an unattended job consumed a week of quota in
|
|
5
|
+
about a day.
|
|
6
|
+
|
|
7
|
+
The rule is that the guard refuses the call that *would* exceed the budget, rather than noticing
|
|
8
|
+
afterwards. A guard that reports an overrun is an accountant, not a guard.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class BudgetExceeded(RuntimeError):
|
|
17
|
+
"""Raised when a call would take the sweep past its declared budget."""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class Budget:
|
|
22
|
+
"""A spend ceiling and a kill criterion, both declared before the sweep runs.
|
|
23
|
+
|
|
24
|
+
``limit_usd`` is the hard ceiling. ``max_consecutive_failures`` is the kill criterion: a sweep
|
|
25
|
+
whose calls are all failing is buying nothing, and continuing to the ceiling is waste.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
limit_usd: float
|
|
29
|
+
max_consecutive_failures: int = 10
|
|
30
|
+
spent_usd: float = 0.0
|
|
31
|
+
consecutive_failures: int = 0
|
|
32
|
+
calls: int = 0
|
|
33
|
+
|
|
34
|
+
def __post_init__(self) -> None:
|
|
35
|
+
if self.limit_usd < 0:
|
|
36
|
+
raise ValueError("a budget cannot be negative")
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def remaining_usd(self) -> float:
|
|
40
|
+
return max(0.0, self.limit_usd - self.spent_usd)
|
|
41
|
+
|
|
42
|
+
def check(self, estimated_usd: float) -> None:
|
|
43
|
+
"""Raise if this call would exceed the ceiling. Call BEFORE spending."""
|
|
44
|
+
if self.spent_usd + estimated_usd > self.limit_usd:
|
|
45
|
+
raise BudgetExceeded(
|
|
46
|
+
f"this call is estimated at {estimated_usd:.4f} USD and "
|
|
47
|
+
f"{self.spent_usd:.4f} of {self.limit_usd:.4f} is already spent; "
|
|
48
|
+
"stopping before the budget rather than after it"
|
|
49
|
+
)
|
|
50
|
+
if self.consecutive_failures >= self.max_consecutive_failures:
|
|
51
|
+
raise BudgetExceeded(
|
|
52
|
+
f"{self.consecutive_failures} consecutive failures reached the kill criterion; "
|
|
53
|
+
"a sweep that is failing every call is buying nothing"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
def charge(self, actual_usd: float, *, failed: bool = False) -> None:
|
|
57
|
+
"""Record what a completed call actually cost."""
|
|
58
|
+
self.spent_usd += actual_usd
|
|
59
|
+
self.calls += 1
|
|
60
|
+
self.consecutive_failures = self.consecutive_failures + 1 if failed else 0
|
|
61
|
+
|
|
62
|
+
def describe(self) -> str:
|
|
63
|
+
return (
|
|
64
|
+
f"{self.spent_usd:.4f} of {self.limit_usd:.4f} USD over {self.calls} call(s), "
|
|
65
|
+
f"{self.remaining_usd:.4f} remaining"
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def estimate(
|
|
70
|
+
prompt: str,
|
|
71
|
+
expected_output_tokens: int,
|
|
72
|
+
input_per_mtok: float,
|
|
73
|
+
output_per_mtok: float,
|
|
74
|
+
) -> float:
|
|
75
|
+
"""A cost estimate before the call, from a crude token count.
|
|
76
|
+
|
|
77
|
+
Four characters per token is a rough English average and it is deliberately not refined: the
|
|
78
|
+
estimate exists to keep the guard conservative, and a guard that under-estimates is worse than
|
|
79
|
+
one that stops slightly early.
|
|
80
|
+
"""
|
|
81
|
+
input_tokens = max(1, len(prompt) // 4)
|
|
82
|
+
return (
|
|
83
|
+
input_tokens * input_per_mtok + expected_output_tokens * output_per_mtok
|
|
84
|
+
) / 1_000_000
|
copela/cli.py
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""The command line.
|
|
2
|
+
|
|
3
|
+
Four commands, and the only one that spends money refuses to start without a declared budget.
|
|
4
|
+
|
|
5
|
+
copela models what each provider can serve, and what it costs
|
|
6
|
+
copela solve <problem.json> solve one formalization, no model involved
|
|
7
|
+
copela sweep <cases.json> ... run the sweep
|
|
8
|
+
copela report <ledger.jsonl> the gap, from a ledger
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import argparse
|
|
14
|
+
import json
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from . import __display_version__
|
|
19
|
+
from .budget import Budget
|
|
20
|
+
from .ledger import Ledger
|
|
21
|
+
from .providers import ProviderError, get
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _selectable_providers() -> list[str]:
|
|
25
|
+
"""Every registered provider except the scripted stub, which is for dry runs."""
|
|
26
|
+
from .providers import REGISTRY
|
|
27
|
+
|
|
28
|
+
return [name for name in sorted(REGISTRY) if name != "stub"]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _cmd_models(args: argparse.Namespace) -> int:
|
|
32
|
+
for name in args.providers:
|
|
33
|
+
try:
|
|
34
|
+
provider = get(name)
|
|
35
|
+
models = provider.models()
|
|
36
|
+
except ProviderError as error:
|
|
37
|
+
print(f"{name}: unavailable ({error})")
|
|
38
|
+
continue
|
|
39
|
+
print(f"{name}:")
|
|
40
|
+
for model_id, pricing in sorted(models.items()):
|
|
41
|
+
if pricing.input_per_mtok or pricing.output_per_mtok:
|
|
42
|
+
cost = f"{pricing.input_per_mtok:g} in / {pricing.output_per_mtok:g} out per MTok"
|
|
43
|
+
else:
|
|
44
|
+
cost = "no per-token price (local)"
|
|
45
|
+
print(f" {model_id:<40} {cost}")
|
|
46
|
+
return 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _cmd_solve(args: argparse.Namespace) -> int:
|
|
50
|
+
from planteo import Problem, validate
|
|
51
|
+
|
|
52
|
+
problem = Problem.from_json(json.loads(Path(args.problem).read_text(encoding="utf-8")))
|
|
53
|
+
report = validate(problem)
|
|
54
|
+
if not report.ok:
|
|
55
|
+
print(report)
|
|
56
|
+
return 1
|
|
57
|
+
|
|
58
|
+
from .solvers.highs import SolverUnavailable, solve
|
|
59
|
+
|
|
60
|
+
try:
|
|
61
|
+
solution = solve(problem, solver_name=args.solver)
|
|
62
|
+
except SolverUnavailable as error:
|
|
63
|
+
print(error)
|
|
64
|
+
return 2
|
|
65
|
+
|
|
66
|
+
print(f"status: {solution.detail}")
|
|
67
|
+
if solution.objective is not None:
|
|
68
|
+
print(f"objective: {solution.objective:.6g}")
|
|
69
|
+
for name, value in sorted(solution.values.items()):
|
|
70
|
+
print(f" {name} = {value:.6g}")
|
|
71
|
+
return 0 if solution.feasible else 1
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _cmd_sweep(args: argparse.Namespace) -> int:
|
|
75
|
+
# A sweep is the only command that spends. It refuses to start without a budget rather than
|
|
76
|
+
# defaulting to one, because a default budget is a number nobody chose.
|
|
77
|
+
if args.budget_usd is None:
|
|
78
|
+
print(
|
|
79
|
+
"a sweep needs a declared budget: pass --budget-usd. "
|
|
80
|
+
"Every sweep states its ceiling and its kill criterion before it runs",
|
|
81
|
+
file=sys.stderr,
|
|
82
|
+
)
|
|
83
|
+
return 2
|
|
84
|
+
|
|
85
|
+
print(
|
|
86
|
+
f"sweep configured: budget {args.budget_usd} USD, "
|
|
87
|
+
f"{args.repeats} repeat(s), ledger {args.ledger}"
|
|
88
|
+
)
|
|
89
|
+
print(
|
|
90
|
+
"Assembling cases and targets is done from Python: the prompt strategy and the response "
|
|
91
|
+
"parser are what a study varies, so they are arguments rather than flags. "
|
|
92
|
+
"See docs/guides/01_run_a_sweep.md"
|
|
93
|
+
)
|
|
94
|
+
Budget(limit_usd=args.budget_usd) # validates the ceiling now rather than mid-run
|
|
95
|
+
return 0
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _cmd_report(args: argparse.Namespace) -> int:
|
|
99
|
+
from .report import build
|
|
100
|
+
|
|
101
|
+
ledger = Ledger(args.ledger)
|
|
102
|
+
if not Path(args.ledger).exists():
|
|
103
|
+
print(f"no ledger at {args.ledger}", file=sys.stderr)
|
|
104
|
+
return 2
|
|
105
|
+
|
|
106
|
+
report = build(ledger)
|
|
107
|
+
if args.json:
|
|
108
|
+
print(json.dumps(report.to_json(), indent=2))
|
|
109
|
+
else:
|
|
110
|
+
print(report.to_text())
|
|
111
|
+
print(f"\nledger: {len(ledger.records())} call(s), {ledger.total_cost_usd:.4f} USD")
|
|
112
|
+
return 0
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def main(argv: list[str] | None = None) -> int:
|
|
116
|
+
parser = argparse.ArgumentParser(
|
|
117
|
+
prog="copela",
|
|
118
|
+
description=(
|
|
119
|
+
"Run narrative-to-formal translation across many models and score it with oracles "
|
|
120
|
+
"that are not language models."
|
|
121
|
+
),
|
|
122
|
+
)
|
|
123
|
+
parser.add_argument("--version", action="version", version=f"copela {__display_version__}")
|
|
124
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
125
|
+
|
|
126
|
+
models = subparsers.add_parser("models", help="list what each provider can serve")
|
|
127
|
+
models.add_argument(
|
|
128
|
+
"providers",
|
|
129
|
+
nargs="*",
|
|
130
|
+
# Read from the registry rather than written out here. A hardcoded list would put vendor
|
|
131
|
+
# names outside the seam, which R-006 forbids and a test enforces.
|
|
132
|
+
default=_selectable_providers(),
|
|
133
|
+
help="provider names",
|
|
134
|
+
)
|
|
135
|
+
models.set_defaults(func=_cmd_models)
|
|
136
|
+
|
|
137
|
+
solve = subparsers.add_parser("solve", help="solve one formalization, no model involved")
|
|
138
|
+
solve.add_argument("problem", help="a planteo Problem as JSON")
|
|
139
|
+
solve.add_argument("--solver", default="appsi_highs")
|
|
140
|
+
solve.set_defaults(func=_cmd_solve)
|
|
141
|
+
|
|
142
|
+
sweep = subparsers.add_parser("sweep", help="run a sweep")
|
|
143
|
+
sweep.add_argument("--ledger", default="runs.jsonl")
|
|
144
|
+
sweep.add_argument("--budget-usd", type=float, default=None, help="required; the hard ceiling")
|
|
145
|
+
sweep.add_argument("--repeats", type=int, default=3)
|
|
146
|
+
sweep.set_defaults(func=_cmd_sweep)
|
|
147
|
+
|
|
148
|
+
report = subparsers.add_parser("report", help="the gap, from a ledger")
|
|
149
|
+
report.add_argument("ledger")
|
|
150
|
+
report.add_argument("--json", action="store_true")
|
|
151
|
+
report.set_defaults(func=_cmd_report)
|
|
152
|
+
|
|
153
|
+
args = parser.parse_args(argv)
|
|
154
|
+
return int(args.func(args))
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
if __name__ == "__main__":
|
|
158
|
+
raise SystemExit(main())
|
copela/ledger.py
ADDED
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
"""The run ledger: append-only JSONL, one record per call.
|
|
2
|
+
|
|
3
|
+
Two properties, both enforced rather than intended.
|
|
4
|
+
|
|
5
|
+
**Append-only.** A record that has been written is never modified. A ledger that can be rewritten is
|
|
6
|
+
not evidence, and the whole value of this file is that a number in a report can be traced back to
|
|
7
|
+
the call that produced it.
|
|
8
|
+
|
|
9
|
+
**Every call pins its provenance.** Model id, model version, temperature, seed, provider
|
|
10
|
+
fingerprint, repeat index and prompt digest, on every record. Without those a rate is a number with
|
|
11
|
+
no subject, and six months later nobody can say which model produced it.
|
|
12
|
+
|
|
13
|
+
The ledger is also the resume mechanism: a sweep reads it, sees which (case, model, repeat) triples
|
|
14
|
+
are already done, and skips them. A long sweep that lost its work to an interruption is a cost this
|
|
15
|
+
account has already paid once.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
from collections.abc import Iterator
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from datetime import UTC, datetime
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
LEDGER_SCHEMA = "copela-ledger/1.0"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class LedgerError(RuntimeError):
|
|
32
|
+
"""Raised when an operation would violate the append-only contract."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def digest(text: str) -> str:
|
|
36
|
+
return hashlib.sha256(text.encode("utf-8")).hexdigest()[:32]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class CallKey:
|
|
41
|
+
"""What makes a call unique, and therefore what resume matches on."""
|
|
42
|
+
|
|
43
|
+
case_id: str
|
|
44
|
+
provider: str
|
|
45
|
+
model_id: str
|
|
46
|
+
repeat: int
|
|
47
|
+
|
|
48
|
+
def as_tuple(self) -> tuple[str, str, str, int]:
|
|
49
|
+
return (self.case_id, self.provider, self.model_id, self.repeat)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(frozen=True, slots=True)
|
|
53
|
+
class Record:
|
|
54
|
+
"""One call. Every field below is required; that is the point of R-003."""
|
|
55
|
+
|
|
56
|
+
key: CallKey
|
|
57
|
+
family: str
|
|
58
|
+
model_version: str
|
|
59
|
+
temperature: float
|
|
60
|
+
seed: int | None
|
|
61
|
+
provider_fingerprint: str
|
|
62
|
+
prompt_digest: str
|
|
63
|
+
response_digest: str
|
|
64
|
+
latency_ms: float
|
|
65
|
+
input_tokens: int
|
|
66
|
+
output_tokens: int
|
|
67
|
+
cost_usd: float
|
|
68
|
+
verdicts: list[dict[str, object]] = field(default_factory=list)
|
|
69
|
+
error: str = ""
|
|
70
|
+
#: A bounded excerpt of the raw response, kept ONLY when the call failed.
|
|
71
|
+
#:
|
|
72
|
+
#: A digest proves a response existed and says nothing about what was wrong with it. Diagnosing
|
|
73
|
+
#: a failure from a digest is impossible, and re-running to reproduce does not work either:
|
|
74
|
+
#: hosted inference is not deterministic, so the failure may not come back. The excerpt is
|
|
75
|
+
#: bounded because a ledger is evidence, not a transcript archive, and it is kept only on
|
|
76
|
+
#: failure because a successful run is already described by its verdicts.
|
|
77
|
+
response_excerpt: str = ""
|
|
78
|
+
recorded_at: str = ""
|
|
79
|
+
|
|
80
|
+
def to_json(self) -> dict[str, object]:
|
|
81
|
+
data = {
|
|
82
|
+
"schema": LEDGER_SCHEMA,
|
|
83
|
+
"case_id": self.key.case_id,
|
|
84
|
+
"provider": self.key.provider,
|
|
85
|
+
"model_id": self.key.model_id,
|
|
86
|
+
"repeat": self.key.repeat,
|
|
87
|
+
"family": self.family,
|
|
88
|
+
"model_version": self.model_version,
|
|
89
|
+
"temperature": self.temperature,
|
|
90
|
+
"seed": self.seed,
|
|
91
|
+
"provider_fingerprint": self.provider_fingerprint,
|
|
92
|
+
"prompt_digest": self.prompt_digest,
|
|
93
|
+
"response_digest": self.response_digest,
|
|
94
|
+
"latency_ms": self.latency_ms,
|
|
95
|
+
"input_tokens": self.input_tokens,
|
|
96
|
+
"output_tokens": self.output_tokens,
|
|
97
|
+
"cost_usd": self.cost_usd,
|
|
98
|
+
"verdicts": self.verdicts,
|
|
99
|
+
"error": self.error,
|
|
100
|
+
"response_excerpt": self.response_excerpt,
|
|
101
|
+
"recorded_at": self.recorded_at
|
|
102
|
+
or datetime.now(UTC).isoformat(timespec="seconds"),
|
|
103
|
+
}
|
|
104
|
+
return data
|
|
105
|
+
|
|
106
|
+
@classmethod
|
|
107
|
+
def from_json(cls, data: dict[str, object]) -> Record:
|
|
108
|
+
return cls(
|
|
109
|
+
key=CallKey(
|
|
110
|
+
case_id=str(data["case_id"]),
|
|
111
|
+
provider=str(data["provider"]),
|
|
112
|
+
model_id=str(data["model_id"]),
|
|
113
|
+
repeat=int(data["repeat"]), # type: ignore[arg-type]
|
|
114
|
+
),
|
|
115
|
+
family=str(data.get("family", "")),
|
|
116
|
+
model_version=str(data.get("model_version", "")),
|
|
117
|
+
temperature=float(data.get("temperature", 0.0)), # type: ignore[arg-type]
|
|
118
|
+
seed=None if data.get("seed") is None else int(data["seed"]), # type: ignore[arg-type]
|
|
119
|
+
provider_fingerprint=str(data.get("provider_fingerprint", "")),
|
|
120
|
+
prompt_digest=str(data.get("prompt_digest", "")),
|
|
121
|
+
response_digest=str(data.get("response_digest", "")),
|
|
122
|
+
latency_ms=float(data.get("latency_ms", 0.0)), # type: ignore[arg-type]
|
|
123
|
+
input_tokens=int(data.get("input_tokens", 0)), # type: ignore[arg-type]
|
|
124
|
+
output_tokens=int(data.get("output_tokens", 0)), # type: ignore[arg-type]
|
|
125
|
+
cost_usd=float(data.get("cost_usd", 0.0)), # type: ignore[arg-type]
|
|
126
|
+
verdicts=list(data.get("verdicts", [])), # type: ignore[arg-type]
|
|
127
|
+
error=str(data.get("error", "")),
|
|
128
|
+
response_excerpt=str(data.get("response_excerpt", "")),
|
|
129
|
+
recorded_at=str(data.get("recorded_at", "")),
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
#: Fields whose absence makes a record useless as evidence. R-003.
|
|
134
|
+
REQUIRED_PROVENANCE = (
|
|
135
|
+
"model_id",
|
|
136
|
+
"model_version",
|
|
137
|
+
"temperature",
|
|
138
|
+
"provider_fingerprint",
|
|
139
|
+
"prompt_digest",
|
|
140
|
+
"repeat",
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
class LedgerBusy(LedgerError):
|
|
145
|
+
"""Another process holds this ledger."""
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class Ledger:
|
|
149
|
+
"""An append-only JSONL file of call records.
|
|
150
|
+
|
|
151
|
+
``exclusive=True`` takes a lock for the lifetime of the object, so two sweeps cannot write one
|
|
152
|
+
ledger. That is not hypothetical: a sweep was started while an earlier one was still alive, both
|
|
153
|
+
appended to the same file, and the result interleaved records from two different versions of the
|
|
154
|
+
code. Append-only does not help there, because the two processes write different keys, so
|
|
155
|
+
nothing collides and nothing complains. The file simply stops meaning one thing.
|
|
156
|
+
"""
|
|
157
|
+
|
|
158
|
+
def __init__(self, path: str | os.PathLike[str], exclusive: bool = False) -> None:
|
|
159
|
+
self.path = Path(path)
|
|
160
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
161
|
+
self._lock_path = self.path.with_suffix(self.path.suffix + ".lock")
|
|
162
|
+
self._holds_lock = False
|
|
163
|
+
if exclusive:
|
|
164
|
+
self._acquire()
|
|
165
|
+
|
|
166
|
+
def _acquire(self) -> None:
|
|
167
|
+
try:
|
|
168
|
+
# O_EXCL is the whole mechanism: creating the file IS the lock, atomically.
|
|
169
|
+
descriptor = os.open(self._lock_path, os.O_CREAT | os.O_EXCL | os.O_WRONLY)
|
|
170
|
+
except FileExistsError:
|
|
171
|
+
holder = ""
|
|
172
|
+
try:
|
|
173
|
+
holder = self._lock_path.read_text(encoding="utf-8").strip()
|
|
174
|
+
except OSError:
|
|
175
|
+
pass
|
|
176
|
+
raise LedgerBusy(
|
|
177
|
+
f"{self.path} is locked by {holder or 'another process'}. Two sweeps writing one "
|
|
178
|
+
f"ledger interleave their records and the file stops meaning one thing. "
|
|
179
|
+
f"Stop the other run, or delete {self._lock_path.name} if it is stale"
|
|
180
|
+
) from None
|
|
181
|
+
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
|
|
182
|
+
handle.write(
|
|
183
|
+
f"pid {os.getpid()} since "
|
|
184
|
+
f"{datetime.now(UTC).isoformat(timespec='seconds')}\n"
|
|
185
|
+
)
|
|
186
|
+
self._holds_lock = True
|
|
187
|
+
|
|
188
|
+
def release(self) -> None:
|
|
189
|
+
if self._holds_lock:
|
|
190
|
+
self._lock_path.unlink(missing_ok=True)
|
|
191
|
+
self._holds_lock = False
|
|
192
|
+
|
|
193
|
+
def __enter__(self) -> Ledger:
|
|
194
|
+
return self
|
|
195
|
+
|
|
196
|
+
def __exit__(self, *_: object) -> None:
|
|
197
|
+
self.release()
|
|
198
|
+
|
|
199
|
+
# -- writing ---------------------------------------------------------------------
|
|
200
|
+
|
|
201
|
+
def append(self, record: Record) -> None:
|
|
202
|
+
"""Append one record. Refuses a duplicate key, because that would be a rewrite."""
|
|
203
|
+
payload = record.to_json()
|
|
204
|
+
missing = [
|
|
205
|
+
name
|
|
206
|
+
for name in REQUIRED_PROVENANCE
|
|
207
|
+
if payload.get(name) in (None, "")
|
|
208
|
+
]
|
|
209
|
+
if missing:
|
|
210
|
+
raise LedgerError(
|
|
211
|
+
f"refusing to record a call missing its provenance: {', '.join(missing)}. "
|
|
212
|
+
"A rate over records that cannot say which model produced them is not evidence"
|
|
213
|
+
)
|
|
214
|
+
if self.has(record.key):
|
|
215
|
+
raise LedgerError(
|
|
216
|
+
f"{record.key.as_tuple()} is already in the ledger. The ledger is append-only; "
|
|
217
|
+
"a run is never edited, because a ledger that can be rewritten is not evidence"
|
|
218
|
+
)
|
|
219
|
+
with self.path.open("a", encoding="utf-8", newline="\n") as handle:
|
|
220
|
+
handle.write(json.dumps(payload, sort_keys=True) + "\n")
|
|
221
|
+
handle.flush()
|
|
222
|
+
os.fsync(handle.fileno())
|
|
223
|
+
|
|
224
|
+
# -- reading ---------------------------------------------------------------------
|
|
225
|
+
|
|
226
|
+
def __iter__(self) -> Iterator[Record]:
|
|
227
|
+
if not self.path.exists():
|
|
228
|
+
return iter(())
|
|
229
|
+
return self._read()
|
|
230
|
+
|
|
231
|
+
def _read(self) -> Iterator[Record]:
|
|
232
|
+
with self.path.open("r", encoding="utf-8") as handle:
|
|
233
|
+
for number, line in enumerate(handle, start=1):
|
|
234
|
+
line = line.strip()
|
|
235
|
+
if not line:
|
|
236
|
+
continue
|
|
237
|
+
try:
|
|
238
|
+
yield Record.from_json(json.loads(line))
|
|
239
|
+
except (json.JSONDecodeError, KeyError) as error:
|
|
240
|
+
raise LedgerError(
|
|
241
|
+
f"{self.path}:{number} is not a readable record: {error}"
|
|
242
|
+
) from error
|
|
243
|
+
|
|
244
|
+
def records(self) -> list[Record]:
|
|
245
|
+
return list(self)
|
|
246
|
+
|
|
247
|
+
def keys(self) -> set[tuple[str, str, str, int]]:
|
|
248
|
+
return {record.key.as_tuple() for record in self}
|
|
249
|
+
|
|
250
|
+
def has(self, key: CallKey) -> bool:
|
|
251
|
+
return key.as_tuple() in self.keys()
|
|
252
|
+
|
|
253
|
+
def completed(self) -> set[tuple[str, str, str, int]]:
|
|
254
|
+
"""Keys already done, so a resumed sweep can skip them. R-012."""
|
|
255
|
+
return self.keys()
|
|
256
|
+
|
|
257
|
+
@property
|
|
258
|
+
def total_cost_usd(self) -> float:
|
|
259
|
+
return sum(record.cost_usd for record in self)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Oracles: the layers that are not language models."""
|