fleetproof 0.2.1__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fleetproof-0.2.1/src/fleetproof.egg-info → fleetproof-0.3.0}/PKG-INFO +63 -1
- {fleetproof-0.2.1 → fleetproof-0.3.0}/README.md +62 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/pyproject.toml +1 -1
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/cli.py +160 -2
- fleetproof-0.3.0/src/fleetproof/config.py +131 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/hookgate.py +42 -2
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/ledger.py +123 -9
- fleetproof-0.3.0/src/fleetproof/telemetry.py +783 -0
- fleetproof-0.3.0/src/fleetproof/telemetry_export.py +690 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0/src/fleetproof.egg-info}/PKG-INFO +63 -1
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/SOURCES.txt +6 -1
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_cli.py +18 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_ledger.py +129 -0
- fleetproof-0.3.0/tests/test_telemetry.py +585 -0
- fleetproof-0.3.0/tests/test_telemetry_export.py +384 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/LICENSE +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/setup.cfg +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/__init__.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/__main__.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/checker.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/checks.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/report.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/runlog.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/dependency_links.txt +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/entry_points.txt +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/requires.txt +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/top_level.txt +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_adversarial_regressions.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_checker.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_checks.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_hookgate.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_report.py +0 -0
- {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_runlog.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fleetproof
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Independent, out-of-band verification for agent fleets.
|
|
5
5
|
Author: Gathrazio
|
|
6
6
|
License: MIT
|
|
@@ -74,6 +74,7 @@ No language model sits in the grading path. Grading is comparison.
|
|
|
74
74
|
| SubagentStop hook | The per-subagent gate: records what the agent claimed, grades it at that agent's tier, blocks a false "done". |
|
|
75
75
|
| `fleetproof fleet` | The dispatch board: every dispatch, its state, its tier, and whether anything graded it. |
|
|
76
76
|
| `fleetproof report` | One self-contained HTML file: per run, claimed-done vs. independently-verified. |
|
|
77
|
+
| `fleetproof telemetry` | v0.3: per-dispatch outcome records, a local reliability summary, and a strict-allowlist export. |
|
|
77
78
|
|
|
78
79
|
Runtime dependencies: none (Python standard library only). A tool whose job is
|
|
79
80
|
being trustworthy should add as little dependency and supply-chain surface as it can.
|
|
@@ -207,6 +208,67 @@ transition trail, and which process wrote each transition.
|
|
|
207
208
|
is also why the lookup is scoped to non-terminal dispatches: an agent id can be
|
|
208
209
|
reused once a dispatch is closed out.
|
|
209
210
|
|
|
211
|
+
## Verification telemetry (v0.3)
|
|
212
|
+
|
|
213
|
+
The ledger records what happened to each dispatch. v0.3 turns those records into
|
|
214
|
+
reliability data an operator can actually read — and, when they choose to, share.
|
|
215
|
+
|
|
216
|
+
Enable it per repo with `.fleetproof/config.json`:
|
|
217
|
+
|
|
218
|
+
```json
|
|
219
|
+
{ "telemetry_era": "2026-08-21" }
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
From that date, every dispatch gains a `telemetry.json`, written on the checker/hook
|
|
223
|
+
side at verdict or close time — never by the agent being measured. Runs from before
|
|
224
|
+
the date read back as pre-telemetry and are never back-filled or guessed.
|
|
225
|
+
|
|
226
|
+
Each finished dispatch derives one of nine **outcome classes** from its transition
|
|
227
|
+
history — bookkeeping, not judgment:
|
|
228
|
+
|
|
229
|
+
- `verified` — the claim survived the checks.
|
|
230
|
+
- `near_miss` — a gate-blocked retry whose work product *changed* before passing:
|
|
231
|
+
a real failure, caught before acceptance.
|
|
232
|
+
- `verifier_flake` — the retry passed with the work product *unchanged*: the
|
|
233
|
+
contradiction was wrong, not the work. Counted separately so flaky checks can't
|
|
234
|
+
inflate the near-miss number.
|
|
235
|
+
- `contradicted` — the claim did not survive.
|
|
236
|
+
- `ungraded` — checks existed but no verdict ever landed. This is a control
|
|
237
|
+
failure and every summary says so; it is never folded into a benign class.
|
|
238
|
+
- `unverifiable` — reported, but nothing in the claim was checkable. Never counts
|
|
239
|
+
as success.
|
|
240
|
+
- `silent_idle` / `terminated_unreported` / `terminated_unclassified` — died
|
|
241
|
+
without reporting, split by the recorded terminate reason.
|
|
242
|
+
|
|
243
|
+
The commands:
|
|
244
|
+
|
|
245
|
+
```
|
|
246
|
+
fleetproof telemetry summary # local-only reliability read
|
|
247
|
+
fleetproof telemetry loss <run-id> # one question per failure: hours or dollars
|
|
248
|
+
fleetproof telemetry export --recipient X # allowlist extract for sharing
|
|
249
|
+
fleetproof telemetry anchor # record the current chain head
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
`summary` prints outcome and severity distributions with their numerators and
|
|
253
|
+
denominators spelled out, override rates, and the corpus's own integrity rates
|
|
254
|
+
(stop-only captures, missing telemetry files) — a dataset that can't measure its
|
|
255
|
+
own holes isn't worth reading.
|
|
256
|
+
|
|
257
|
+
`export` is built the opposite way from most exports: a strict per-field
|
|
258
|
+
**allowlist** — closed enums, counts, bands, versions, and salted identifiers.
|
|
259
|
+
Free text, prompts, paths, and every name you chose yourself never leave the
|
|
260
|
+
machine; timestamps coarsen to dates. Each export carries a completeness manifest
|
|
261
|
+
(runs per day per class, gaps visible) and a methodology note that says plainly
|
|
262
|
+
what `verified` means here: conformance to *your own declared checks* at a
|
|
263
|
+
measured coverage — not a judgment of work quality.
|
|
264
|
+
|
|
265
|
+
The honest trust model, stated rather than implied: local records are
|
|
266
|
+
operator-attested. The checker writes verdicts from a separate process, and v0.3
|
|
267
|
+
chains a rolling hash over the records as they're written (`anchor` gives you a
|
|
268
|
+
head you can sign or store elsewhere) — but a local ledger on a shared filesystem
|
|
269
|
+
is not tamper-proof against everything that can write to it, and FleetProof will
|
|
270
|
+
not pretend otherwise.
|
|
271
|
+
|
|
210
272
|
## Install (each line is one command in Claude Code)
|
|
211
273
|
|
|
212
274
|
```
|
|
@@ -53,6 +53,7 @@ No language model sits in the grading path. Grading is comparison.
|
|
|
53
53
|
| SubagentStop hook | The per-subagent gate: records what the agent claimed, grades it at that agent's tier, blocks a false "done". |
|
|
54
54
|
| `fleetproof fleet` | The dispatch board: every dispatch, its state, its tier, and whether anything graded it. |
|
|
55
55
|
| `fleetproof report` | One self-contained HTML file: per run, claimed-done vs. independently-verified. |
|
|
56
|
+
| `fleetproof telemetry` | v0.3: per-dispatch outcome records, a local reliability summary, and a strict-allowlist export. |
|
|
56
57
|
|
|
57
58
|
Runtime dependencies: none (Python standard library only). A tool whose job is
|
|
58
59
|
being trustworthy should add as little dependency and supply-chain surface as it can.
|
|
@@ -186,6 +187,67 @@ transition trail, and which process wrote each transition.
|
|
|
186
187
|
is also why the lookup is scoped to non-terminal dispatches: an agent id can be
|
|
187
188
|
reused once a dispatch is closed out.
|
|
188
189
|
|
|
190
|
+
## Verification telemetry (v0.3)
|
|
191
|
+
|
|
192
|
+
The ledger records what happened to each dispatch. v0.3 turns those records into
|
|
193
|
+
reliability data an operator can actually read — and, when they choose to, share.
|
|
194
|
+
|
|
195
|
+
Enable it per repo with `.fleetproof/config.json`:
|
|
196
|
+
|
|
197
|
+
```json
|
|
198
|
+
{ "telemetry_era": "2026-08-21" }
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
From that date, every dispatch gains a `telemetry.json`, written on the checker/hook
|
|
202
|
+
side at verdict or close time — never by the agent being measured. Runs from before
|
|
203
|
+
the date read back as pre-telemetry and are never back-filled or guessed.
|
|
204
|
+
|
|
205
|
+
Each finished dispatch derives one of nine **outcome classes** from its transition
|
|
206
|
+
history — bookkeeping, not judgment:
|
|
207
|
+
|
|
208
|
+
- `verified` — the claim survived the checks.
|
|
209
|
+
- `near_miss` — a gate-blocked retry whose work product *changed* before passing:
|
|
210
|
+
a real failure, caught before acceptance.
|
|
211
|
+
- `verifier_flake` — the retry passed with the work product *unchanged*: the
|
|
212
|
+
contradiction was wrong, not the work. Counted separately so flaky checks can't
|
|
213
|
+
inflate the near-miss number.
|
|
214
|
+
- `contradicted` — the claim did not survive.
|
|
215
|
+
- `ungraded` — checks existed but no verdict ever landed. This is a control
|
|
216
|
+
failure and every summary says so; it is never folded into a benign class.
|
|
217
|
+
- `unverifiable` — reported, but nothing in the claim was checkable. Never counts
|
|
218
|
+
as success.
|
|
219
|
+
- `silent_idle` / `terminated_unreported` / `terminated_unclassified` — died
|
|
220
|
+
without reporting, split by the recorded terminate reason.
|
|
221
|
+
|
|
222
|
+
The commands:
|
|
223
|
+
|
|
224
|
+
```
|
|
225
|
+
fleetproof telemetry summary # local-only reliability read
|
|
226
|
+
fleetproof telemetry loss <run-id> # one question per failure: hours or dollars
|
|
227
|
+
fleetproof telemetry export --recipient X # allowlist extract for sharing
|
|
228
|
+
fleetproof telemetry anchor # record the current chain head
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
`summary` prints outcome and severity distributions with their numerators and
|
|
232
|
+
denominators spelled out, override rates, and the corpus's own integrity rates
|
|
233
|
+
(stop-only captures, missing telemetry files) — a dataset that can't measure its
|
|
234
|
+
own holes isn't worth reading.
|
|
235
|
+
|
|
236
|
+
`export` is built the opposite way from most exports: a strict per-field
|
|
237
|
+
**allowlist** — closed enums, counts, bands, versions, and salted identifiers.
|
|
238
|
+
Free text, prompts, paths, and every name you chose yourself never leave the
|
|
239
|
+
machine; timestamps coarsen to dates. Each export carries a completeness manifest
|
|
240
|
+
(runs per day per class, gaps visible) and a methodology note that says plainly
|
|
241
|
+
what `verified` means here: conformance to *your own declared checks* at a
|
|
242
|
+
measured coverage — not a judgment of work quality.
|
|
243
|
+
|
|
244
|
+
The honest trust model, stated rather than implied: local records are
|
|
245
|
+
operator-attested. The checker writes verdicts from a separate process, and v0.3
|
|
246
|
+
chains a rolling hash over the records as they're written (`anchor` gives you a
|
|
247
|
+
head you can sign or store elsewhere) — but a local ledger on a shared filesystem
|
|
248
|
+
is not tamper-proof against everything that can write to it, and FleetProof will
|
|
249
|
+
not pretend otherwise.
|
|
250
|
+
|
|
189
251
|
## Install (each line is one command in Claude Code)
|
|
190
252
|
|
|
191
253
|
```
|
|
@@ -15,6 +15,8 @@ surface as it can. Subcommands:
|
|
|
15
15
|
subagent-stop SubagentStop-hook entry: the per-subagent report-and-verify gate
|
|
16
16
|
dispatch ledger verbs: new / report / close a dispatch
|
|
17
17
|
fleet the dispatch board — every dispatch, its state, and whether it reported
|
|
18
|
+
telemetry summary (local-only aggregates) / export (allowlisted bundle) /
|
|
19
|
+
anchor (chain head) / loss (the one raw-loss question per failure)
|
|
18
20
|
|
|
19
21
|
Output is plain ASCII on purpose: these commands get read in cp1252 consoles on
|
|
20
22
|
Windows, where a stray unicode glyph is a UnicodeEncodeError, not a nicer table.
|
|
@@ -27,7 +29,7 @@ import json
|
|
|
27
29
|
import os
|
|
28
30
|
import shutil
|
|
29
31
|
import sys
|
|
30
|
-
from datetime import datetime, timedelta, timezone
|
|
32
|
+
from datetime import date, datetime, timedelta, timezone
|
|
31
33
|
from pathlib import Path
|
|
32
34
|
|
|
33
35
|
# The CLI must not record its own invocations.
|
|
@@ -51,6 +53,8 @@ from .hookgate import (
|
|
|
51
53
|
subagent_stop_main,
|
|
52
54
|
)
|
|
53
55
|
from .ledger import (
|
|
56
|
+
REASON_OPERATOR_CLOSE,
|
|
57
|
+
VALID_TERMINATE_REASONS,
|
|
54
58
|
LedgerError,
|
|
55
59
|
close_dispatch,
|
|
56
60
|
create_dispatch,
|
|
@@ -76,6 +80,7 @@ from .runlog import (
|
|
|
76
80
|
project_root,
|
|
77
81
|
runs_dir,
|
|
78
82
|
)
|
|
83
|
+
from .telemetry import TelemetryError, build_telemetry, record_failure_loss
|
|
79
84
|
|
|
80
85
|
|
|
81
86
|
def _cmd_init(args: argparse.Namespace) -> int:
|
|
@@ -348,10 +353,19 @@ def _cmd_dispatch_report(args: argparse.Namespace) -> int:
|
|
|
348
353
|
|
|
349
354
|
def _cmd_dispatch_close(args: argparse.Namespace) -> int:
|
|
350
355
|
try:
|
|
351
|
-
record = close_dispatch(args.run_id)
|
|
356
|
+
record = close_dispatch(args.run_id, reason=args.reason)
|
|
352
357
|
except LedgerError as e:
|
|
353
358
|
_emit_error("ledger_error", str(e), args.format)
|
|
354
359
|
return 1
|
|
360
|
+
# The close finalizes the lifecycle, so the telemetry record derives here.
|
|
361
|
+
# Best-effort: a telemetry failure must not turn a successful close into a
|
|
362
|
+
# failed command — the summary surfaces the missing record as an integrity
|
|
363
|
+
# defect instead.
|
|
364
|
+
try:
|
|
365
|
+
build_telemetry(record.run_id)
|
|
366
|
+
except Exception as e:
|
|
367
|
+
print(f"[fleetproof] telemetry build failed for {record.run_id}: {e}",
|
|
368
|
+
file=sys.stderr)
|
|
355
369
|
if args.format == "json":
|
|
356
370
|
print(json.dumps(record.to_dict(), indent=2))
|
|
357
371
|
else:
|
|
@@ -416,6 +430,105 @@ def _cmd_fleet(args: argparse.Namespace) -> int:
|
|
|
416
430
|
return 0
|
|
417
431
|
|
|
418
432
|
|
|
433
|
+
def _parse_date(value: str | None, label: str) -> "date | None":
|
|
434
|
+
if value is None:
|
|
435
|
+
return None
|
|
436
|
+
try:
|
|
437
|
+
return date.fromisoformat(value)
|
|
438
|
+
except ValueError as e:
|
|
439
|
+
raise ValueError(f"--{label} must be YYYY-MM-DD: {e}") from e
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _cmd_telemetry_summary(args: argparse.Namespace) -> int:
|
|
443
|
+
from .telemetry_export import summarize
|
|
444
|
+
summary = summarize()
|
|
445
|
+
if args.format == "json":
|
|
446
|
+
print(json.dumps(summary, indent=2))
|
|
447
|
+
return 0
|
|
448
|
+
print("Telemetry summary (LOCAL-ONLY: clear-text names and exact times;")
|
|
449
|
+
print("anything shareable goes through `fleetproof telemetry export`).")
|
|
450
|
+
for name, block in summary["windows"].items():
|
|
451
|
+
counts = block["counts"]
|
|
452
|
+
shown = {k: v for k, v in counts.items() if v}
|
|
453
|
+
print(f"\n[{name}] {block['total_dispatches']} dispatch(es), "
|
|
454
|
+
f"{block['classifiable_dispatches']} classifiable")
|
|
455
|
+
print(f" outcomes: {shown if shown else 'none'}")
|
|
456
|
+
for metric in ("delivery_failure_rate", "false_claim_rate",
|
|
457
|
+
"near_miss_rate", "verifier_flake_rate",
|
|
458
|
+
"ungraded_rate", "unverifiable_rate",
|
|
459
|
+
"telemetry_missing_rate", "stop_only_fraction"):
|
|
460
|
+
m = block[metric]
|
|
461
|
+
rate = f"{m['rate']:.3f}" if m["rate"] is not None else "n/a"
|
|
462
|
+
print(f" {metric}: {m['numerator']}/{m['denominator']} = {rate}")
|
|
463
|
+
sev = block["severity_distribution"]
|
|
464
|
+
print(f" severity: {({k: v for k, v in sev.items() if v}) or 'no failures'}")
|
|
465
|
+
override = block["override_rate_by_source"]
|
|
466
|
+
for source in ("hook", "operator"):
|
|
467
|
+
m = override[source]
|
|
468
|
+
rate = f"{m['rate']:.3f}" if m["rate"] is not None else "n/a"
|
|
469
|
+
print(f" override_rate[{source}]: {m['numerator']}/{m['denominator']} = {rate}")
|
|
470
|
+
timeline = summary["spec_hash_timeline"]
|
|
471
|
+
if timeline:
|
|
472
|
+
print("\nspec-hash timeline:")
|
|
473
|
+
for entry in timeline:
|
|
474
|
+
print(f" {entry['first_seen']} {short_spec_hash(entry['spec_sha256'])}")
|
|
475
|
+
return 0
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def _cmd_telemetry_export(args: argparse.Namespace) -> int:
|
|
479
|
+
from .telemetry_export import export_telemetry
|
|
480
|
+
try:
|
|
481
|
+
since = _parse_date(args.since, "since")
|
|
482
|
+
until = _parse_date(args.until, "until")
|
|
483
|
+
except ValueError as e:
|
|
484
|
+
_emit_error("bad_date", str(e), args.format)
|
|
485
|
+
return 2
|
|
486
|
+
try:
|
|
487
|
+
out = export_telemetry(
|
|
488
|
+
args.recipient,
|
|
489
|
+
out_dir=Path(args.output) if args.output else None,
|
|
490
|
+
since=since, until=until)
|
|
491
|
+
except TelemetryError as e:
|
|
492
|
+
_emit_error("telemetry_error", str(e), args.format)
|
|
493
|
+
return 1
|
|
494
|
+
if args.format == "json":
|
|
495
|
+
print(json.dumps({"ok": True, "export_dir": str(out)}))
|
|
496
|
+
else:
|
|
497
|
+
print(f"Wrote export bundle to {out}")
|
|
498
|
+
print("Label: identified, minimized (single-party exports are "
|
|
499
|
+
"identified by construction).")
|
|
500
|
+
return 0
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def _cmd_telemetry_anchor(args: argparse.Namespace) -> int:
|
|
504
|
+
from .telemetry_export import anchor_chain
|
|
505
|
+
anchor = anchor_chain()
|
|
506
|
+
if args.format == "json":
|
|
507
|
+
print(json.dumps(anchor, indent=2))
|
|
508
|
+
else:
|
|
509
|
+
print(f"chain head: {anchor['chain_head']}")
|
|
510
|
+
print(f"entries: {anchor['chain_entries']}")
|
|
511
|
+
print(anchor["note"])
|
|
512
|
+
return 0
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def _cmd_telemetry_loss(args: argparse.Namespace) -> int:
|
|
516
|
+
try:
|
|
517
|
+
telemetry = record_failure_loss(
|
|
518
|
+
args.run_id, args.value, args.unit,
|
|
519
|
+
default_accepted=args.default_accepted or None)
|
|
520
|
+
except TelemetryError as e:
|
|
521
|
+
_emit_error("telemetry_error", str(e), args.format)
|
|
522
|
+
return 1
|
|
523
|
+
if args.format == "json":
|
|
524
|
+
print(json.dumps(telemetry, indent=2))
|
|
525
|
+
else:
|
|
526
|
+
print(f"{args.run_id}: severity {telemetry['failure.severity_band']} "
|
|
527
|
+
f"(floor {telemetry['failure.severity_floor']}), "
|
|
528
|
+
f"loss {telemetry['failure.estimated_loss']}")
|
|
529
|
+
return 0
|
|
530
|
+
|
|
531
|
+
|
|
419
532
|
def _emit_error(code: str, message: str, fmt: str) -> None:
|
|
420
533
|
if fmt == "json":
|
|
421
534
|
print(json.dumps({"ok": False, "error_code": code, "message": message}), file=sys.stderr)
|
|
@@ -508,9 +621,54 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
508
621
|
|
|
509
622
|
p_dclose = dsub.add_parser("close", help="Terminate a dispatch.")
|
|
510
623
|
p_dclose.add_argument("run_id")
|
|
624
|
+
# Defaulted, not required: the CLI is the operator's close path, so the
|
|
625
|
+
# honest machine-set reason for it is operator-close. Scripted sweepers and
|
|
626
|
+
# session-teardown callers override it to say what they actually are.
|
|
627
|
+
p_dclose.add_argument(
|
|
628
|
+
"--reason", choices=sorted(VALID_TERMINATE_REASONS),
|
|
629
|
+
default=REASON_OPERATOR_CLOSE,
|
|
630
|
+
help="Why this dispatch is being terminated (default: operator-close).")
|
|
511
631
|
_add_format(p_dclose)
|
|
512
632
|
p_dclose.set_defaults(func=_cmd_dispatch_close)
|
|
513
633
|
|
|
634
|
+
p_tel = sub.add_parser("telemetry", help="Verification-telemetry surfaces.")
|
|
635
|
+
tsub = p_tel.add_subparsers(dest="telemetry_command", required=True)
|
|
636
|
+
|
|
637
|
+
p_tsum = tsub.add_parser(
|
|
638
|
+
"summary",
|
|
639
|
+
help="Local-only outcome/severity/integrity summary with trailing windows.")
|
|
640
|
+
_add_format(p_tsum)
|
|
641
|
+
p_tsum.set_defaults(func=_cmd_telemetry_summary)
|
|
642
|
+
|
|
643
|
+
p_texp = tsub.add_parser(
|
|
644
|
+
"export",
|
|
645
|
+
help="Write the allowlisted, date-bucketed export bundle for one recipient.")
|
|
646
|
+
p_texp.add_argument("--recipient", required=True,
|
|
647
|
+
help="Counterparty label; selects (or mints) the export salt.")
|
|
648
|
+
p_texp.add_argument("-o", "--output", default=None, help="Output directory.")
|
|
649
|
+
p_texp.add_argument("--since", default=None, help="Window start, YYYY-MM-DD.")
|
|
650
|
+
p_texp.add_argument("--until", default=None, help="Window end, YYYY-MM-DD.")
|
|
651
|
+
_add_format(p_texp)
|
|
652
|
+
p_texp.set_defaults(func=_cmd_telemetry_export)
|
|
653
|
+
|
|
654
|
+
p_tanc = tsub.add_parser(
|
|
655
|
+
"anchor",
|
|
656
|
+
help="Record the telemetry chain head as an anchorable value.")
|
|
657
|
+
_add_format(p_tanc)
|
|
658
|
+
p_tanc.set_defaults(func=_cmd_telemetry_anchor)
|
|
659
|
+
|
|
660
|
+
p_tloss = tsub.add_parser(
|
|
661
|
+
"loss",
|
|
662
|
+
help="Record the one raw loss quantity for a failed run (hours or usd).")
|
|
663
|
+
p_tloss.add_argument("run_id")
|
|
664
|
+
p_tloss.add_argument("--value", type=float, required=True,
|
|
665
|
+
help="The raw quantity; the band is derived, never chosen.")
|
|
666
|
+
p_tloss.add_argument("--unit", choices=["hours", "usd"], required=True)
|
|
667
|
+
p_tloss.add_argument("--default-accepted", action="store_true",
|
|
668
|
+
help="The suggested default was accepted unchanged.")
|
|
669
|
+
_add_format(p_tloss)
|
|
670
|
+
p_tloss.set_defaults(func=_cmd_telemetry_loss)
|
|
671
|
+
|
|
514
672
|
p_fleet = sub.add_parser("fleet", help="The dispatch board, newest first.")
|
|
515
673
|
p_fleet.add_argument("--open", action="store_true",
|
|
516
674
|
help="Only dispatches that have not terminated.")
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Deployment-level telemetry configuration.
|
|
2
|
+
|
|
3
|
+
Verification telemetry needs a handful of facts that are properties of the
|
|
4
|
+
*deployment*, not of any single run: when telemetry capture was turned on, what
|
|
5
|
+
kind of work this deployment does, and how the operator wants it identified in
|
|
6
|
+
aggregates. Asking per run would violate the "automatic or it won't exist"
|
|
7
|
+
posture, so they live in one JSON file the operator edits once:
|
|
8
|
+
|
|
9
|
+
.fleetproof/config.json
|
|
10
|
+
{
|
|
11
|
+
"telemetry_era": "2026-08-20",
|
|
12
|
+
"run_context": "production",
|
|
13
|
+
"deployment_id": "my-fleet",
|
|
14
|
+
"operator_id": "me",
|
|
15
|
+
"eval_suite_id": null
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
Everything is optional and the file itself is optional: an absent or unreadable
|
|
19
|
+
config reads back as empty, and every consumer treats a missing value as "not
|
|
20
|
+
configured" rather than inventing one.
|
|
21
|
+
|
|
22
|
+
``telemetry_era`` is the dated cutover that marks when this deployment started
|
|
23
|
+
capturing telemetry. It is stamped into each dispatch record *at creation* so
|
|
24
|
+
era membership is a property of the dispatch itself, never inferred from
|
|
25
|
+
whether a telemetry file happens to exist later — file absence is deletable,
|
|
26
|
+
a stamp written at dispatch time is on the record the aggregate reads.
|
|
27
|
+
|
|
28
|
+
The config file is resolved next to the runs directory (the ``.fleetproof/``
|
|
29
|
+
project marker), so tests that override the runs dir get an isolated config
|
|
30
|
+
for free, the same way the drift marker does.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import json
|
|
36
|
+
from datetime import date, datetime, timezone
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Any
|
|
39
|
+
|
|
40
|
+
from .runlog import runs_dir
|
|
41
|
+
|
|
42
|
+
CONFIG_FILENAME = "config.json"
|
|
43
|
+
|
|
44
|
+
# The kinds of work a deployment can declare itself to be doing. An aggregate
|
|
45
|
+
# that cannot tell real work from drills or synthetic load is worth less than
|
|
46
|
+
# one that can, so the value is per-deployment config — never per-run prompting,
|
|
47
|
+
# and never guessed from the work itself.
|
|
48
|
+
RUN_CONTEXT_PRODUCTION = "production"
|
|
49
|
+
RUN_CONTEXT_DEVELOPMENT = "development"
|
|
50
|
+
RUN_CONTEXT_DRILL = "drill"
|
|
51
|
+
RUN_CONTEXT_SYNTHETIC = "synthetic"
|
|
52
|
+
VALID_RUN_CONTEXTS = frozenset({
|
|
53
|
+
RUN_CONTEXT_PRODUCTION,
|
|
54
|
+
RUN_CONTEXT_DEVELOPMENT,
|
|
55
|
+
RUN_CONTEXT_DRILL,
|
|
56
|
+
RUN_CONTEXT_SYNTHETIC,
|
|
57
|
+
})
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def config_path() -> Path:
|
|
61
|
+
return runs_dir().parent / CONFIG_FILENAME
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def load_config() -> dict[str, Any]:
|
|
65
|
+
"""The deployment config, or an empty dict when absent/unreadable.
|
|
66
|
+
|
|
67
|
+
A malformed config must not take down a hook, so every failure mode reads
|
|
68
|
+
as "not configured" — the consequence is a pre-telemetry-shaped record,
|
|
69
|
+
which is the honest description of a deployment whose config is broken.
|
|
70
|
+
"""
|
|
71
|
+
try:
|
|
72
|
+
data = json.loads(config_path().read_text(encoding="utf-8-sig"))
|
|
73
|
+
except (OSError, json.JSONDecodeError):
|
|
74
|
+
return {}
|
|
75
|
+
return data if isinstance(data, dict) else {}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def telemetry_era_stamp(config: dict[str, Any] | None = None) -> str | None:
|
|
79
|
+
"""The era value to stamp onto a dispatch created now, or None.
|
|
80
|
+
|
|
81
|
+
Returns the configured cutover date string once the cutover date has
|
|
82
|
+
arrived; a future-dated cutover (or no cutover at all) yields None, and the
|
|
83
|
+
dispatch is created pre-telemetry. The comparison is by date, matching the
|
|
84
|
+
granularity the config declares.
|
|
85
|
+
"""
|
|
86
|
+
cfg = config if config is not None else load_config()
|
|
87
|
+
raw = cfg.get("telemetry_era")
|
|
88
|
+
if not isinstance(raw, str) or not raw.strip():
|
|
89
|
+
return None
|
|
90
|
+
try:
|
|
91
|
+
cutover = date.fromisoformat(raw.strip())
|
|
92
|
+
except ValueError:
|
|
93
|
+
return None
|
|
94
|
+
if datetime.now(timezone.utc).date() < cutover:
|
|
95
|
+
return None
|
|
96
|
+
return raw.strip()
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def run_context(config: dict[str, Any] | None = None) -> str:
|
|
100
|
+
"""The deployment's declared run context; ``production`` when unconfigured.
|
|
101
|
+
|
|
102
|
+
Defaulting to ``production`` is deliberate: a deployment that never touched
|
|
103
|
+
the config is doing its real work, and letting real work default into a
|
|
104
|
+
discounted bucket would under-count the corpus that matters. Drills and
|
|
105
|
+
synthetic load are the exceptional cases, so they are the ones that require
|
|
106
|
+
an explicit declaration. An unknown value degrades to the default rather
|
|
107
|
+
than crashing a hook on a typo.
|
|
108
|
+
"""
|
|
109
|
+
cfg = config if config is not None else load_config()
|
|
110
|
+
value = cfg.get("run_context")
|
|
111
|
+
if isinstance(value, str) and value in VALID_RUN_CONTEXTS:
|
|
112
|
+
return value
|
|
113
|
+
return RUN_CONTEXT_PRODUCTION
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _config_str(key: str, config: dict[str, Any] | None = None) -> str | None:
|
|
117
|
+
cfg = config if config is not None else load_config()
|
|
118
|
+
value = cfg.get(key)
|
|
119
|
+
return value if isinstance(value, str) and value.strip() else None
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def deployment_id(config: dict[str, Any] | None = None) -> str | None:
|
|
123
|
+
return _config_str("deployment_id", config)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def operator_id(config: dict[str, Any] | None = None) -> str | None:
|
|
127
|
+
return _config_str("operator_id", config)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def eval_suite_id(config: dict[str, Any] | None = None) -> str | None:
|
|
131
|
+
return _config_str("eval_suite_id", config)
|
|
@@ -41,7 +41,13 @@ from typing import Any
|
|
|
41
41
|
|
|
42
42
|
from pathlib import Path
|
|
43
43
|
|
|
44
|
-
from .checker import
|
|
44
|
+
from .checker import (
|
|
45
|
+
CheckReport,
|
|
46
|
+
run_checks,
|
|
47
|
+
select_checks,
|
|
48
|
+
session_spec_baseline,
|
|
49
|
+
spec_drifted,
|
|
50
|
+
)
|
|
45
51
|
from .checks import (
|
|
46
52
|
SPEC_DRIFT_NOTE,
|
|
47
53
|
CheckSpecError,
|
|
@@ -66,6 +72,7 @@ from .ledger import (
|
|
|
66
72
|
record_verdict,
|
|
67
73
|
)
|
|
68
74
|
from .runlog import SESSION_ID_ENV, record, runs_dir
|
|
75
|
+
from .telemetry import build_telemetry
|
|
69
76
|
|
|
70
77
|
|
|
71
78
|
def _read_hook_input() -> dict[str, Any]:
|
|
@@ -433,13 +440,34 @@ def _pin_drift_reason(pinned: str | None, current: str | None) -> str:
|
|
|
433
440
|
|
|
434
441
|
|
|
435
442
|
def _try_close(run_id: str) -> None:
|
|
436
|
-
"""Terminate a dispatch, tolerating an already-closed one.
|
|
443
|
+
"""Terminate a dispatch, tolerating an already-closed one.
|
|
444
|
+
|
|
445
|
+
No terminate reason is passed: the reason vocabulary exists to split apart
|
|
446
|
+
the ways a dispatch dies *from ``dispatched``*, and this close only ever
|
|
447
|
+
runs after a report is on record (post-verdict, or post-grading with an
|
|
448
|
+
empty selection) — a state where the reason plays no part in how the
|
|
449
|
+
outcome reads back.
|
|
450
|
+
"""
|
|
437
451
|
try:
|
|
438
452
|
close_dispatch(run_id, by="hook")
|
|
439
453
|
except LedgerError as e:
|
|
440
454
|
sys.stderr.write(f"[fleetproof] could not close dispatch {run_id}: {e}\n")
|
|
441
455
|
|
|
442
456
|
|
|
457
|
+
def _try_build_telemetry(run_id: str, check_report=None, checks=None) -> None:
|
|
458
|
+
"""Best-effort telemetry build, on the checker's side of the boundary.
|
|
459
|
+
|
|
460
|
+
A telemetry failure must not wedge the gate. The cost of swallowing one is
|
|
461
|
+
an era-stamped run without telemetry.json — which the summary surfaces as
|
|
462
|
+
a ``telemetry_missing`` integrity defect, the visible form a capture
|
|
463
|
+
failure is supposed to take.
|
|
464
|
+
"""
|
|
465
|
+
try:
|
|
466
|
+
build_telemetry(run_id, check_report=check_report, checks=checks)
|
|
467
|
+
except Exception as e:
|
|
468
|
+
sys.stderr.write(f"[fleetproof] telemetry build failed for {run_id}: {e}\n")
|
|
469
|
+
|
|
470
|
+
|
|
443
471
|
def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
|
|
444
472
|
"""The per-subagent gate. Returns ``(decision_or_None, exit_code)``.
|
|
445
473
|
|
|
@@ -526,6 +554,14 @@ def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
|
|
|
526
554
|
selected = select_checks(checks, dispatch.tier) if checks else []
|
|
527
555
|
if not selected:
|
|
528
556
|
_try_close(dispatch.run_id)
|
|
557
|
+
# The grading *happened* and selected nothing — recorded as an empty
|
|
558
|
+
# CheckReport so the telemetry layer can tell "checked nothing on
|
|
559
|
+
# purpose" (unverifiable) apart from "grading never ran" (ungraded).
|
|
560
|
+
_try_build_telemetry(
|
|
561
|
+
dispatch.run_id,
|
|
562
|
+
check_report=CheckReport(spec_sha256=spec_hash(), tier=dispatch.tier),
|
|
563
|
+
checks=[],
|
|
564
|
+
)
|
|
529
565
|
return None, 0
|
|
530
566
|
|
|
531
567
|
report = run_checks(selected, record_to_log=True, tier=dispatch.tier)
|
|
@@ -536,6 +572,7 @@ def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
|
|
|
536
572
|
detail="; ".join(r.id for r in report.blocking_failures),
|
|
537
573
|
by=VERDICT_BY,
|
|
538
574
|
)
|
|
575
|
+
_try_build_telemetry(dispatch.run_id, check_report=report, checks=selected)
|
|
539
576
|
return _subagent_block(_failure_reason(report), _evidence_context(report)), 0
|
|
540
577
|
|
|
541
578
|
record_verdict(
|
|
@@ -545,6 +582,9 @@ def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
|
|
|
545
582
|
by=VERDICT_BY,
|
|
546
583
|
)
|
|
547
584
|
_try_close(dispatch.run_id)
|
|
585
|
+
# Built after the close, so the outcome class derives from a finished
|
|
586
|
+
# lifecycle rather than a snapshot mid-transition.
|
|
587
|
+
_try_build_telemetry(dispatch.run_id, check_report=report, checks=selected)
|
|
548
588
|
return None, 0
|
|
549
589
|
|
|
550
590
|
|