copela 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
copela/__init__.py ADDED
@@ -0,0 +1,61 @@
1
+ """copela: run narrative-to-formal translation across many models, and score it with oracles
2
+ that are not language models.
3
+
4
+ The name is the cupel used in fire assay, the vessel that separates the metal from the lead. That is
5
+ the job here: separating a formalization that is faithful from one that merely runs.
6
+
7
+ Four layers, reported separately and never merged into one score:
8
+
9
+ 1. executable, did it run, solve, compile
10
+ 2. structural, is it the same model as the reference
11
+ 3. property, do the invariants of this class hold
12
+ 4. judge, what a model says, recorded as a labelled screening aggregate and never as truth
13
+
14
+ The headline is the subtraction: how often the artifact ran, minus how often it was right.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from .budget import Budget, BudgetExceeded, estimate
20
+ from .ledger import CallKey, Ledger, LedgerError, Record
21
+ from .providers import Provider, ProviderError, StubProvider
22
+ from .report import Cell, Report, build
23
+ from .sweep import Case, Sweep, Target
24
+ from .verdicts import (
25
+ JUDGE_LABEL,
26
+ CandidateVerdict,
27
+ Layer,
28
+ LayerResult,
29
+ Outcome,
30
+ Rate,
31
+ )
32
+
33
+ __version__ = "0.1.0"
34
+ __display_version__ = "0.01.000"
35
+
36
+ __all__ = [
37
+ "JUDGE_LABEL",
38
+ "Budget",
39
+ "BudgetExceeded",
40
+ "CallKey",
41
+ "CandidateVerdict",
42
+ "Case",
43
+ "Cell",
44
+ "Layer",
45
+ "LayerResult",
46
+ "Ledger",
47
+ "LedgerError",
48
+ "Outcome",
49
+ "Provider",
50
+ "ProviderError",
51
+ "Rate",
52
+ "Record",
53
+ "Report",
54
+ "StubProvider",
55
+ "Sweep",
56
+ "Target",
57
+ "__display_version__",
58
+ "__version__",
59
+ "build",
60
+ "estimate",
61
+ ]
copela/budget.py ADDED
@@ -0,0 +1,84 @@
1
+ """The budget guard: stop before the limit, not after it.
2
+
3
+ A sweep is cases times models times repeats, and each cell costs money. The guard exists because the
4
+ failure it prevents has happened on this account: an unattended job consumed a week of quota in
5
+ about a day.
6
+
7
+ The rule is that the guard refuses the call that *would* exceed the budget, rather than noticing
8
+ afterwards. A guard that reports an overrun is an accountant, not a guard.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass
14
+
15
+
16
+ class BudgetExceeded(RuntimeError):
17
+ """Raised when a call would take the sweep past its declared budget."""
18
+
19
+
20
+ @dataclass
21
+ class Budget:
22
+ """A spend ceiling and a kill criterion, both declared before the sweep runs.
23
+
24
+ ``limit_usd`` is the hard ceiling. ``max_consecutive_failures`` is the kill criterion: a sweep
25
+ whose calls are all failing is buying nothing, and continuing to the ceiling is waste.
26
+ """
27
+
28
+ limit_usd: float
29
+ max_consecutive_failures: int = 10
30
+ spent_usd: float = 0.0
31
+ consecutive_failures: int = 0
32
+ calls: int = 0
33
+
34
+ def __post_init__(self) -> None:
35
+ if self.limit_usd < 0:
36
+ raise ValueError("a budget cannot be negative")
37
+
38
+ @property
39
+ def remaining_usd(self) -> float:
40
+ return max(0.0, self.limit_usd - self.spent_usd)
41
+
42
+ def check(self, estimated_usd: float) -> None:
43
+ """Raise if this call would exceed the ceiling. Call BEFORE spending."""
44
+ if self.spent_usd + estimated_usd > self.limit_usd:
45
+ raise BudgetExceeded(
46
+ f"this call is estimated at {estimated_usd:.4f} USD and "
47
+ f"{self.spent_usd:.4f} of {self.limit_usd:.4f} is already spent; "
48
+ "stopping before the budget rather than after it"
49
+ )
50
+ if self.consecutive_failures >= self.max_consecutive_failures:
51
+ raise BudgetExceeded(
52
+ f"{self.consecutive_failures} consecutive failures reached the kill criterion; "
53
+ "a sweep that is failing every call is buying nothing"
54
+ )
55
+
56
+ def charge(self, actual_usd: float, *, failed: bool = False) -> None:
57
+ """Record what a completed call actually cost."""
58
+ self.spent_usd += actual_usd
59
+ self.calls += 1
60
+ self.consecutive_failures = self.consecutive_failures + 1 if failed else 0
61
+
62
+ def describe(self) -> str:
63
+ return (
64
+ f"{self.spent_usd:.4f} of {self.limit_usd:.4f} USD over {self.calls} call(s), "
65
+ f"{self.remaining_usd:.4f} remaining"
66
+ )
67
+
68
+
69
+ def estimate(
70
+ prompt: str,
71
+ expected_output_tokens: int,
72
+ input_per_mtok: float,
73
+ output_per_mtok: float,
74
+ ) -> float:
75
+ """A cost estimate before the call, from a crude token count.
76
+
77
+ Four characters per token is a rough English average and it is deliberately not refined: the
78
+ estimate exists to keep the guard conservative, and a guard that under-estimates is worse than
79
+ one that stops slightly early.
80
+ """
81
+ input_tokens = max(1, len(prompt) // 4)
82
+ return (
83
+ input_tokens * input_per_mtok + expected_output_tokens * output_per_mtok
84
+ ) / 1_000_000
copela/cli.py ADDED
@@ -0,0 +1,158 @@
1
+ """The command line.
2
+
3
+ Four commands, and the only one that spends money refuses to start without a declared budget.
4
+
5
+ copela models what each provider can serve, and what it costs
6
+ copela solve <problem.json> solve one formalization, no model involved
7
+ copela sweep <cases.json> ... run the sweep
8
+ copela report <ledger.jsonl> the gap, from a ledger
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import argparse
14
+ import json
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ from . import __display_version__
19
+ from .budget import Budget
20
+ from .ledger import Ledger
21
+ from .providers import ProviderError, get
22
+
23
+
24
+ def _selectable_providers() -> list[str]:
25
+ """Every registered provider except the scripted stub, which is for dry runs."""
26
+ from .providers import REGISTRY
27
+
28
+ return [name for name in sorted(REGISTRY) if name != "stub"]
29
+
30
+
31
+ def _cmd_models(args: argparse.Namespace) -> int:
32
+ for name in args.providers:
33
+ try:
34
+ provider = get(name)
35
+ models = provider.models()
36
+ except ProviderError as error:
37
+ print(f"{name}: unavailable ({error})")
38
+ continue
39
+ print(f"{name}:")
40
+ for model_id, pricing in sorted(models.items()):
41
+ if pricing.input_per_mtok or pricing.output_per_mtok:
42
+ cost = f"{pricing.input_per_mtok:g} in / {pricing.output_per_mtok:g} out per MTok"
43
+ else:
44
+ cost = "no per-token price (local)"
45
+ print(f" {model_id:<40} {cost}")
46
+ return 0
47
+
48
+
49
+ def _cmd_solve(args: argparse.Namespace) -> int:
50
+ from planteo import Problem, validate
51
+
52
+ problem = Problem.from_json(json.loads(Path(args.problem).read_text(encoding="utf-8")))
53
+ report = validate(problem)
54
+ if not report.ok:
55
+ print(report)
56
+ return 1
57
+
58
+ from .solvers.highs import SolverUnavailable, solve
59
+
60
+ try:
61
+ solution = solve(problem, solver_name=args.solver)
62
+ except SolverUnavailable as error:
63
+ print(error)
64
+ return 2
65
+
66
+ print(f"status: {solution.detail}")
67
+ if solution.objective is not None:
68
+ print(f"objective: {solution.objective:.6g}")
69
+ for name, value in sorted(solution.values.items()):
70
+ print(f" {name} = {value:.6g}")
71
+ return 0 if solution.feasible else 1
72
+
73
+
74
+ def _cmd_sweep(args: argparse.Namespace) -> int:
75
+ # A sweep is the only command that spends. It refuses to start without a budget rather than
76
+ # defaulting to one, because a default budget is a number nobody chose.
77
+ if args.budget_usd is None:
78
+ print(
79
+ "a sweep needs a declared budget: pass --budget-usd. "
80
+ "Every sweep states its ceiling and its kill criterion before it runs",
81
+ file=sys.stderr,
82
+ )
83
+ return 2
84
+
85
+ print(
86
+ f"sweep configured: budget {args.budget_usd} USD, "
87
+ f"{args.repeats} repeat(s), ledger {args.ledger}"
88
+ )
89
+ print(
90
+ "Assembling cases and targets is done from Python: the prompt strategy and the response "
91
+ "parser are what a study varies, so they are arguments rather than flags. "
92
+ "See docs/guides/01_run_a_sweep.md"
93
+ )
94
+ Budget(limit_usd=args.budget_usd) # validates the ceiling now rather than mid-run
95
+ return 0
96
+
97
+
98
+ def _cmd_report(args: argparse.Namespace) -> int:
99
+ from .report import build
100
+
101
+ ledger = Ledger(args.ledger)
102
+ if not Path(args.ledger).exists():
103
+ print(f"no ledger at {args.ledger}", file=sys.stderr)
104
+ return 2
105
+
106
+ report = build(ledger)
107
+ if args.json:
108
+ print(json.dumps(report.to_json(), indent=2))
109
+ else:
110
+ print(report.to_text())
111
+ print(f"\nledger: {len(ledger.records())} call(s), {ledger.total_cost_usd:.4f} USD")
112
+ return 0
113
+
114
+
115
+ def main(argv: list[str] | None = None) -> int:
116
+ parser = argparse.ArgumentParser(
117
+ prog="copela",
118
+ description=(
119
+ "Run narrative-to-formal translation across many models and score it with oracles "
120
+ "that are not language models."
121
+ ),
122
+ )
123
+ parser.add_argument("--version", action="version", version=f"copela {__display_version__}")
124
+ subparsers = parser.add_subparsers(dest="command", required=True)
125
+
126
+ models = subparsers.add_parser("models", help="list what each provider can serve")
127
+ models.add_argument(
128
+ "providers",
129
+ nargs="*",
130
+ # Read from the registry rather than written out here. A hardcoded list would put vendor
131
+ # names outside the seam, which R-006 forbids and a test enforces.
132
+ default=_selectable_providers(),
133
+ help="provider names",
134
+ )
135
+ models.set_defaults(func=_cmd_models)
136
+
137
+ solve = subparsers.add_parser("solve", help="solve one formalization, no model involved")
138
+ solve.add_argument("problem", help="a planteo Problem as JSON")
139
+ solve.add_argument("--solver", default="appsi_highs")
140
+ solve.set_defaults(func=_cmd_solve)
141
+
142
+ sweep = subparsers.add_parser("sweep", help="run a sweep")
143
+ sweep.add_argument("--ledger", default="runs.jsonl")
144
+ sweep.add_argument("--budget-usd", type=float, default=None, help="required; the hard ceiling")
145
+ sweep.add_argument("--repeats", type=int, default=3)
146
+ sweep.set_defaults(func=_cmd_sweep)
147
+
148
+ report = subparsers.add_parser("report", help="the gap, from a ledger")
149
+ report.add_argument("ledger")
150
+ report.add_argument("--json", action="store_true")
151
+ report.set_defaults(func=_cmd_report)
152
+
153
+ args = parser.parse_args(argv)
154
+ return int(args.func(args))
155
+
156
+
157
+ if __name__ == "__main__":
158
+ raise SystemExit(main())
copela/ledger.py ADDED
@@ -0,0 +1,259 @@
1
+ """The run ledger: append-only JSONL, one record per call.
2
+
3
+ Two properties, both enforced rather than intended.
4
+
5
+ **Append-only.** A record that has been written is never modified. A ledger that can be rewritten is
6
+ not evidence, and the whole value of this file is that a number in a report can be traced back to
7
+ the call that produced it.
8
+
9
+ **Every call pins its provenance.** Model id, model version, temperature, seed, provider
10
+ fingerprint, repeat index and prompt digest, on every record. Without those a rate is a number with
11
+ no subject, and six months later nobody can say which model produced it.
12
+
13
+ The ledger is also the resume mechanism: a sweep reads it, sees which (case, model, repeat) triples
14
+ are already done, and skips them. A long sweep that lost its work to an interruption is a cost this
15
+ account has already paid once.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import json
22
+ import os
23
+ from collections.abc import Iterator
24
+ from dataclasses import dataclass, field
25
+ from datetime import UTC, datetime
26
+ from pathlib import Path
27
+
28
+ LEDGER_SCHEMA = "copela-ledger/1.0"
29
+
30
+
31
+ class LedgerError(RuntimeError):
32
+ """Raised when an operation would violate the append-only contract."""
33
+
34
+
35
+ def digest(text: str) -> str:
36
+ return hashlib.sha256(text.encode("utf-8")).hexdigest()[:32]
37
+
38
+
39
+ @dataclass(frozen=True, slots=True)
40
+ class CallKey:
41
+ """What makes a call unique, and therefore what resume matches on."""
42
+
43
+ case_id: str
44
+ provider: str
45
+ model_id: str
46
+ repeat: int
47
+
48
+ def as_tuple(self) -> tuple[str, str, str, int]:
49
+ return (self.case_id, self.provider, self.model_id, self.repeat)
50
+
51
+
52
+ @dataclass(frozen=True, slots=True)
53
+ class Record:
54
+ """One call. Every field below is required; that is the point of R-003."""
55
+
56
+ key: CallKey
57
+ family: str
58
+ model_version: str
59
+ temperature: float
60
+ seed: int | None
61
+ provider_fingerprint: str
62
+ prompt_digest: str
63
+ response_digest: str
64
+ latency_ms: float
65
+ input_tokens: int
66
+ output_tokens: int
67
+ cost_usd: float
68
+ verdicts: list[dict[str, object]] = field(default_factory=list)
69
+ error: str = ""
70
+ #: A bounded excerpt of the raw response, kept ONLY when the call failed.
71
+ #:
72
+ #: A digest proves a response existed and says nothing about what was wrong with it. Diagnosing
73
+ #: a failure from a digest is impossible, and re-running to reproduce does not work either:
74
+ #: hosted inference is not deterministic, so the failure may not come back. The excerpt is
75
+ #: bounded because a ledger is evidence, not a transcript archive, and it is kept only on
76
+ #: failure because a successful run is already described by its verdicts.
77
+ response_excerpt: str = ""
78
+ recorded_at: str = ""
79
+
80
+ def to_json(self) -> dict[str, object]:
81
+ data = {
82
+ "schema": LEDGER_SCHEMA,
83
+ "case_id": self.key.case_id,
84
+ "provider": self.key.provider,
85
+ "model_id": self.key.model_id,
86
+ "repeat": self.key.repeat,
87
+ "family": self.family,
88
+ "model_version": self.model_version,
89
+ "temperature": self.temperature,
90
+ "seed": self.seed,
91
+ "provider_fingerprint": self.provider_fingerprint,
92
+ "prompt_digest": self.prompt_digest,
93
+ "response_digest": self.response_digest,
94
+ "latency_ms": self.latency_ms,
95
+ "input_tokens": self.input_tokens,
96
+ "output_tokens": self.output_tokens,
97
+ "cost_usd": self.cost_usd,
98
+ "verdicts": self.verdicts,
99
+ "error": self.error,
100
+ "response_excerpt": self.response_excerpt,
101
+ "recorded_at": self.recorded_at
102
+ or datetime.now(UTC).isoformat(timespec="seconds"),
103
+ }
104
+ return data
105
+
106
+ @classmethod
107
+ def from_json(cls, data: dict[str, object]) -> Record:
108
+ return cls(
109
+ key=CallKey(
110
+ case_id=str(data["case_id"]),
111
+ provider=str(data["provider"]),
112
+ model_id=str(data["model_id"]),
113
+ repeat=int(data["repeat"]), # type: ignore[arg-type]
114
+ ),
115
+ family=str(data.get("family", "")),
116
+ model_version=str(data.get("model_version", "")),
117
+ temperature=float(data.get("temperature", 0.0)), # type: ignore[arg-type]
118
+ seed=None if data.get("seed") is None else int(data["seed"]), # type: ignore[arg-type]
119
+ provider_fingerprint=str(data.get("provider_fingerprint", "")),
120
+ prompt_digest=str(data.get("prompt_digest", "")),
121
+ response_digest=str(data.get("response_digest", "")),
122
+ latency_ms=float(data.get("latency_ms", 0.0)), # type: ignore[arg-type]
123
+ input_tokens=int(data.get("input_tokens", 0)), # type: ignore[arg-type]
124
+ output_tokens=int(data.get("output_tokens", 0)), # type: ignore[arg-type]
125
+ cost_usd=float(data.get("cost_usd", 0.0)), # type: ignore[arg-type]
126
+ verdicts=list(data.get("verdicts", [])), # type: ignore[arg-type]
127
+ error=str(data.get("error", "")),
128
+ response_excerpt=str(data.get("response_excerpt", "")),
129
+ recorded_at=str(data.get("recorded_at", "")),
130
+ )
131
+
132
+
133
+ #: Fields whose absence makes a record useless as evidence. R-003.
134
+ REQUIRED_PROVENANCE = (
135
+ "model_id",
136
+ "model_version",
137
+ "temperature",
138
+ "provider_fingerprint",
139
+ "prompt_digest",
140
+ "repeat",
141
+ )
142
+
143
+
144
+ class LedgerBusy(LedgerError):
145
+ """Another process holds this ledger."""
146
+
147
+
148
+ class Ledger:
149
+ """An append-only JSONL file of call records.
150
+
151
+ ``exclusive=True`` takes a lock for the lifetime of the object, so two sweeps cannot write one
152
+ ledger. That is not hypothetical: a sweep was started while an earlier one was still alive, both
153
+ appended to the same file, and the result interleaved records from two different versions of the
154
+ code. Append-only does not help there, because the two processes write different keys, so
155
+ nothing collides and nothing complains. The file simply stops meaning one thing.
156
+ """
157
+
158
+ def __init__(self, path: str | os.PathLike[str], exclusive: bool = False) -> None:
159
+ self.path = Path(path)
160
+ self.path.parent.mkdir(parents=True, exist_ok=True)
161
+ self._lock_path = self.path.with_suffix(self.path.suffix + ".lock")
162
+ self._holds_lock = False
163
+ if exclusive:
164
+ self._acquire()
165
+
166
+ def _acquire(self) -> None:
167
+ try:
168
+ # O_EXCL is the whole mechanism: creating the file IS the lock, atomically.
169
+ descriptor = os.open(self._lock_path, os.O_CREAT | os.O_EXCL | os.O_WRONLY)
170
+ except FileExistsError:
171
+ holder = ""
172
+ try:
173
+ holder = self._lock_path.read_text(encoding="utf-8").strip()
174
+ except OSError:
175
+ pass
176
+ raise LedgerBusy(
177
+ f"{self.path} is locked by {holder or 'another process'}. Two sweeps writing one "
178
+ f"ledger interleave their records and the file stops meaning one thing. "
179
+ f"Stop the other run, or delete {self._lock_path.name} if it is stale"
180
+ ) from None
181
+ with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
182
+ handle.write(
183
+ f"pid {os.getpid()} since "
184
+ f"{datetime.now(UTC).isoformat(timespec='seconds')}\n"
185
+ )
186
+ self._holds_lock = True
187
+
188
+ def release(self) -> None:
189
+ if self._holds_lock:
190
+ self._lock_path.unlink(missing_ok=True)
191
+ self._holds_lock = False
192
+
193
+ def __enter__(self) -> Ledger:
194
+ return self
195
+
196
+ def __exit__(self, *_: object) -> None:
197
+ self.release()
198
+
199
+ # -- writing ---------------------------------------------------------------------
200
+
201
+ def append(self, record: Record) -> None:
202
+ """Append one record. Refuses a duplicate key, because that would be a rewrite."""
203
+ payload = record.to_json()
204
+ missing = [
205
+ name
206
+ for name in REQUIRED_PROVENANCE
207
+ if payload.get(name) in (None, "")
208
+ ]
209
+ if missing:
210
+ raise LedgerError(
211
+ f"refusing to record a call missing its provenance: {', '.join(missing)}. "
212
+ "A rate over records that cannot say which model produced them is not evidence"
213
+ )
214
+ if self.has(record.key):
215
+ raise LedgerError(
216
+ f"{record.key.as_tuple()} is already in the ledger. The ledger is append-only; "
217
+ "a run is never edited, because a ledger that can be rewritten is not evidence"
218
+ )
219
+ with self.path.open("a", encoding="utf-8", newline="\n") as handle:
220
+ handle.write(json.dumps(payload, sort_keys=True) + "\n")
221
+ handle.flush()
222
+ os.fsync(handle.fileno())
223
+
224
+ # -- reading ---------------------------------------------------------------------
225
+
226
+ def __iter__(self) -> Iterator[Record]:
227
+ if not self.path.exists():
228
+ return iter(())
229
+ return self._read()
230
+
231
+ def _read(self) -> Iterator[Record]:
232
+ with self.path.open("r", encoding="utf-8") as handle:
233
+ for number, line in enumerate(handle, start=1):
234
+ line = line.strip()
235
+ if not line:
236
+ continue
237
+ try:
238
+ yield Record.from_json(json.loads(line))
239
+ except (json.JSONDecodeError, KeyError) as error:
240
+ raise LedgerError(
241
+ f"{self.path}:{number} is not a readable record: {error}"
242
+ ) from error
243
+
244
+ def records(self) -> list[Record]:
245
+ return list(self)
246
+
247
+ def keys(self) -> set[tuple[str, str, str, int]]:
248
+ return {record.key.as_tuple() for record in self}
249
+
250
+ def has(self, key: CallKey) -> bool:
251
+ return key.as_tuple() in self.keys()
252
+
253
+ def completed(self) -> set[tuple[str, str, str, int]]:
254
+ """Keys already done, so a resumed sweep can skip them. R-012."""
255
+ return self.keys()
256
+
257
+ @property
258
+ def total_cost_usd(self) -> float:
259
+ return sum(record.cost_usd for record in self)
@@ -0,0 +1 @@
1
+ """Oracles: the layers that are not language models."""