fleetproof 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {fleetproof-0.2.1/src/fleetproof.egg-info → fleetproof-0.3.0}/PKG-INFO +63 -1
  2. {fleetproof-0.2.1 → fleetproof-0.3.0}/README.md +62 -0
  3. {fleetproof-0.2.1 → fleetproof-0.3.0}/pyproject.toml +1 -1
  4. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/cli.py +160 -2
  5. fleetproof-0.3.0/src/fleetproof/config.py +131 -0
  6. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/hookgate.py +42 -2
  7. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/ledger.py +123 -9
  8. fleetproof-0.3.0/src/fleetproof/telemetry.py +783 -0
  9. fleetproof-0.3.0/src/fleetproof/telemetry_export.py +690 -0
  10. {fleetproof-0.2.1 → fleetproof-0.3.0/src/fleetproof.egg-info}/PKG-INFO +63 -1
  11. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/SOURCES.txt +6 -1
  12. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_cli.py +18 -0
  13. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_ledger.py +129 -0
  14. fleetproof-0.3.0/tests/test_telemetry.py +585 -0
  15. fleetproof-0.3.0/tests/test_telemetry_export.py +384 -0
  16. {fleetproof-0.2.1 → fleetproof-0.3.0}/LICENSE +0 -0
  17. {fleetproof-0.2.1 → fleetproof-0.3.0}/setup.cfg +0 -0
  18. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/__init__.py +0 -0
  19. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/__main__.py +0 -0
  20. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/checker.py +0 -0
  21. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/checks.py +0 -0
  22. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/report.py +0 -0
  23. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof/runlog.py +0 -0
  24. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/dependency_links.txt +0 -0
  25. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/entry_points.txt +0 -0
  26. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/requires.txt +0 -0
  27. {fleetproof-0.2.1 → fleetproof-0.3.0}/src/fleetproof.egg-info/top_level.txt +0 -0
  28. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_adversarial_regressions.py +0 -0
  29. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_checker.py +0 -0
  30. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_checks.py +0 -0
  31. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_hookgate.py +0 -0
  32. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_report.py +0 -0
  33. {fleetproof-0.2.1 → fleetproof-0.3.0}/tests/test_runlog.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fleetproof
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: Independent, out-of-band verification for agent fleets.
5
5
  Author: Gathrazio
6
6
  License: MIT
@@ -74,6 +74,7 @@ No language model sits in the grading path. Grading is comparison.
74
74
  | SubagentStop hook | The per-subagent gate: records what the agent claimed, grades it at that agent's tier, blocks a false "done". |
75
75
  | `fleetproof fleet` | The dispatch board: every dispatch, its state, its tier, and whether anything graded it. |
76
76
  | `fleetproof report` | One self-contained HTML file: per run, claimed-done vs. independently-verified. |
77
+ | `fleetproof telemetry` | v0.3: per-dispatch outcome records, a local reliability summary, and a strict-allowlist export. |
77
78
 
78
79
  Runtime dependencies: none (Python standard library only). A tool whose job is
79
80
  being trustworthy should add as little dependency and supply-chain surface as it can.
@@ -207,6 +208,67 @@ transition trail, and which process wrote each transition.
207
208
  is also why the lookup is scoped to non-terminal dispatches: an agent id can be
208
209
  reused once a dispatch is closed out.
209
210
 
211
+ ## Verification telemetry (v0.3)
212
+
213
+ The ledger records what happened to each dispatch. v0.3 turns those records into
214
+ reliability data an operator can actually read — and, when they choose to, share.
215
+
216
+ Enable it per repo with `.fleetproof/config.json`:
217
+
218
+ ```json
219
+ { "telemetry_era": "2026-08-21" }
220
+ ```
221
+
222
+ From that date, every dispatch gains a `telemetry.json`, written on the checker/hook
223
+ side at verdict or close time — never by the agent being measured. Runs from before
224
+ the date read back as pre-telemetry and are never back-filled or guessed.
225
+
226
+ Each finished dispatch derives one of nine **outcome classes** from its transition
227
+ history — bookkeeping, not judgment:
228
+
229
+ - `verified` — the claim survived the checks.
230
+ - `near_miss` — a gate-blocked retry whose work product *changed* before passing:
231
+ a real failure, caught before acceptance.
232
+ - `verifier_flake` — the retry passed with the work product *unchanged*: the
233
+ contradiction was wrong, not the work. Counted separately so flaky checks can't
234
+ inflate the near-miss number.
235
+ - `contradicted` — the claim did not survive.
236
+ - `ungraded` — checks existed but no verdict ever landed. This is a control
237
+ failure and every summary says so; it is never folded into a benign class.
238
+ - `unverifiable` — reported, but nothing in the claim was checkable. Never counts
239
+ as success.
240
+ - `silent_idle` / `terminated_unreported` / `terminated_unclassified` — died
241
+ without reporting, split by the recorded terminate reason.
242
+
243
+ The commands:
244
+
245
+ ```
246
+ fleetproof telemetry summary # local-only reliability read
247
+ fleetproof telemetry loss <run-id> # one question per failure: hours or dollars
248
+ fleetproof telemetry export --recipient X # allowlist extract for sharing
249
+ fleetproof telemetry anchor # record the current chain head
250
+ ```
251
+
252
+ `summary` prints outcome and severity distributions with their numerators and
253
+ denominators spelled out, override rates, and the corpus's own integrity rates
254
+ (stop-only captures, missing telemetry files) — a dataset that can't measure its
255
+ own holes isn't worth reading.
256
+
257
+ `export` is built the opposite way from most exports: a strict per-field
258
+ **allowlist** — closed enums, counts, bands, versions, and salted identifiers.
259
+ Free text, prompts, paths, and every name you chose yourself never leave the
260
+ machine; timestamps coarsen to dates. Each export carries a completeness manifest
261
+ (runs per day per class, gaps visible) and a methodology note that says plainly
262
+ what `verified` means here: conformance to *your own declared checks* at a
263
+ measured coverage — not a judgment of work quality.
264
+
265
+ The honest trust model, stated rather than implied: local records are
266
+ operator-attested. The checker writes verdicts from a separate process, and v0.3
267
+ chains a rolling hash over the records as they're written (`anchor` gives you a
268
+ head you can sign or store elsewhere) — but a local ledger on a shared filesystem
269
+ is not tamper-proof against everything that can write to it, and FleetProof will
270
+ not pretend otherwise.
271
+
210
272
  ## Install (each line is one command in Claude Code)
211
273
 
212
274
  ```
@@ -53,6 +53,7 @@ No language model sits in the grading path. Grading is comparison.
53
53
  | SubagentStop hook | The per-subagent gate: records what the agent claimed, grades it at that agent's tier, blocks a false "done". |
54
54
  | `fleetproof fleet` | The dispatch board: every dispatch, its state, its tier, and whether anything graded it. |
55
55
  | `fleetproof report` | One self-contained HTML file: per run, claimed-done vs. independently-verified. |
56
+ | `fleetproof telemetry` | v0.3: per-dispatch outcome records, a local reliability summary, and a strict-allowlist export. |
56
57
 
57
58
  Runtime dependencies: none (Python standard library only). A tool whose job is
58
59
  being trustworthy should add as little dependency and supply-chain surface as it can.
@@ -186,6 +187,67 @@ transition trail, and which process wrote each transition.
186
187
  is also why the lookup is scoped to non-terminal dispatches: an agent id can be
187
188
  reused once a dispatch is closed out.
188
189
 
190
+ ## Verification telemetry (v0.3)
191
+
192
+ The ledger records what happened to each dispatch. v0.3 turns those records into
193
+ reliability data an operator can actually read — and, when they choose to, share.
194
+
195
+ Enable it per repo with `.fleetproof/config.json`:
196
+
197
+ ```json
198
+ { "telemetry_era": "2026-08-21" }
199
+ ```
200
+
201
+ From that date, every dispatch gains a `telemetry.json`, written on the checker/hook
202
+ side at verdict or close time — never by the agent being measured. Runs from before
203
+ the date read back as pre-telemetry and are never back-filled or guessed.
204
+
205
+ Each finished dispatch derives one of nine **outcome classes** from its transition
206
+ history — bookkeeping, not judgment:
207
+
208
+ - `verified` — the claim survived the checks.
209
+ - `near_miss` — a gate-blocked retry whose work product *changed* before passing:
210
+ a real failure, caught before acceptance.
211
+ - `verifier_flake` — the retry passed with the work product *unchanged*: the
212
+ contradiction was wrong, not the work. Counted separately so flaky checks can't
213
+ inflate the near-miss number.
214
+ - `contradicted` — the claim did not survive.
215
+ - `ungraded` — checks existed but no verdict ever landed. This is a control
216
+ failure and every summary says so; it is never folded into a benign class.
217
+ - `unverifiable` — reported, but nothing in the claim was checkable. Never counts
218
+ as success.
219
+ - `silent_idle` / `terminated_unreported` / `terminated_unclassified` — died
220
+ without reporting, split by the recorded terminate reason.
221
+
222
+ The commands:
223
+
224
+ ```
225
+ fleetproof telemetry summary # local-only reliability read
226
+ fleetproof telemetry loss <run-id> # one question per failure: hours or dollars
227
+ fleetproof telemetry export --recipient X # allowlist extract for sharing
228
+ fleetproof telemetry anchor # record the current chain head
229
+ ```
230
+
231
+ `summary` prints outcome and severity distributions with their numerators and
232
+ denominators spelled out, override rates, and the corpus's own integrity rates
233
+ (stop-only captures, missing telemetry files) — a dataset that can't measure its
234
+ own holes isn't worth reading.
235
+
236
+ `export` is built the opposite way from most exports: a strict per-field
237
+ **allowlist** — closed enums, counts, bands, versions, and salted identifiers.
238
+ Free text, prompts, paths, and every name you chose yourself never leave the
239
+ machine; timestamps coarsen to dates. Each export carries a completeness manifest
240
+ (runs per day per class, gaps visible) and a methodology note that says plainly
241
+ what `verified` means here: conformance to *your own declared checks* at a
242
+ measured coverage — not a judgment of work quality.
243
+
244
+ The honest trust model, stated rather than implied: local records are
245
+ operator-attested. The checker writes verdicts from a separate process, and v0.3
246
+ chains a rolling hash over the records as they're written (`anchor` gives you a
247
+ head you can sign or store elsewhere) — but a local ledger on a shared filesystem
248
+ is not tamper-proof against everything that can write to it, and FleetProof will
249
+ not pretend otherwise.
250
+
189
251
  ## Install (each line is one command in Claude Code)
190
252
 
191
253
  ```
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "fleetproof"
7
- version = "0.2.1"
7
+ version = "0.3.0"
8
8
  description = "Independent, out-of-band verification for agent fleets."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -15,6 +15,8 @@ surface as it can. Subcommands:
15
15
  subagent-stop SubagentStop-hook entry: the per-subagent report-and-verify gate
16
16
  dispatch ledger verbs: new / report / close a dispatch
17
17
  fleet the dispatch board — every dispatch, its state, and whether it reported
18
+ telemetry summary (local-only aggregates) / export (allowlisted bundle) /
19
+ anchor (chain head) / loss (the one raw-loss question per failure)
18
20
 
19
21
  Output is plain ASCII on purpose: these commands get read in cp1252 consoles on
20
22
  Windows, where a stray unicode glyph is a UnicodeEncodeError, not a nicer table.
@@ -27,7 +29,7 @@ import json
27
29
  import os
28
30
  import shutil
29
31
  import sys
30
- from datetime import datetime, timedelta, timezone
32
+ from datetime import date, datetime, timedelta, timezone
31
33
  from pathlib import Path
32
34
 
33
35
  # The CLI must not record its own invocations.
@@ -51,6 +53,8 @@ from .hookgate import (
51
53
  subagent_stop_main,
52
54
  )
53
55
  from .ledger import (
56
+ REASON_OPERATOR_CLOSE,
57
+ VALID_TERMINATE_REASONS,
54
58
  LedgerError,
55
59
  close_dispatch,
56
60
  create_dispatch,
@@ -76,6 +80,7 @@ from .runlog import (
76
80
  project_root,
77
81
  runs_dir,
78
82
  )
83
+ from .telemetry import TelemetryError, build_telemetry, record_failure_loss
79
84
 
80
85
 
81
86
  def _cmd_init(args: argparse.Namespace) -> int:
@@ -348,10 +353,19 @@ def _cmd_dispatch_report(args: argparse.Namespace) -> int:
348
353
 
349
354
  def _cmd_dispatch_close(args: argparse.Namespace) -> int:
350
355
  try:
351
- record = close_dispatch(args.run_id)
356
+ record = close_dispatch(args.run_id, reason=args.reason)
352
357
  except LedgerError as e:
353
358
  _emit_error("ledger_error", str(e), args.format)
354
359
  return 1
360
+ # The close finalizes the lifecycle, so the telemetry record derives here.
361
+ # Best-effort: a telemetry failure must not turn a successful close into a
362
+ # failed command — the summary surfaces the missing record as an integrity
363
+ # defect instead.
364
+ try:
365
+ build_telemetry(record.run_id)
366
+ except Exception as e:
367
+ print(f"[fleetproof] telemetry build failed for {record.run_id}: {e}",
368
+ file=sys.stderr)
355
369
  if args.format == "json":
356
370
  print(json.dumps(record.to_dict(), indent=2))
357
371
  else:
@@ -416,6 +430,105 @@ def _cmd_fleet(args: argparse.Namespace) -> int:
416
430
  return 0
417
431
 
418
432
 
433
+ def _parse_date(value: str | None, label: str) -> "date | None":
434
+ if value is None:
435
+ return None
436
+ try:
437
+ return date.fromisoformat(value)
438
+ except ValueError as e:
439
+ raise ValueError(f"--{label} must be YYYY-MM-DD: {e}") from e
440
+
441
+
442
+ def _cmd_telemetry_summary(args: argparse.Namespace) -> int:
443
+ from .telemetry_export import summarize
444
+ summary = summarize()
445
+ if args.format == "json":
446
+ print(json.dumps(summary, indent=2))
447
+ return 0
448
+ print("Telemetry summary (LOCAL-ONLY: clear-text names and exact times;")
449
+ print("anything shareable goes through `fleetproof telemetry export`).")
450
+ for name, block in summary["windows"].items():
451
+ counts = block["counts"]
452
+ shown = {k: v for k, v in counts.items() if v}
453
+ print(f"\n[{name}] {block['total_dispatches']} dispatch(es), "
454
+ f"{block['classifiable_dispatches']} classifiable")
455
+ print(f" outcomes: {shown if shown else 'none'}")
456
+ for metric in ("delivery_failure_rate", "false_claim_rate",
457
+ "near_miss_rate", "verifier_flake_rate",
458
+ "ungraded_rate", "unverifiable_rate",
459
+ "telemetry_missing_rate", "stop_only_fraction"):
460
+ m = block[metric]
461
+ rate = f"{m['rate']:.3f}" if m["rate"] is not None else "n/a"
462
+ print(f" {metric}: {m['numerator']}/{m['denominator']} = {rate}")
463
+ sev = block["severity_distribution"]
464
+ print(f" severity: {({k: v for k, v in sev.items() if v}) or 'no failures'}")
465
+ override = block["override_rate_by_source"]
466
+ for source in ("hook", "operator"):
467
+ m = override[source]
468
+ rate = f"{m['rate']:.3f}" if m["rate"] is not None else "n/a"
469
+ print(f" override_rate[{source}]: {m['numerator']}/{m['denominator']} = {rate}")
470
+ timeline = summary["spec_hash_timeline"]
471
+ if timeline:
472
+ print("\nspec-hash timeline:")
473
+ for entry in timeline:
474
+ print(f" {entry['first_seen']} {short_spec_hash(entry['spec_sha256'])}")
475
+ return 0
476
+
477
+
478
+ def _cmd_telemetry_export(args: argparse.Namespace) -> int:
479
+ from .telemetry_export import export_telemetry
480
+ try:
481
+ since = _parse_date(args.since, "since")
482
+ until = _parse_date(args.until, "until")
483
+ except ValueError as e:
484
+ _emit_error("bad_date", str(e), args.format)
485
+ return 2
486
+ try:
487
+ out = export_telemetry(
488
+ args.recipient,
489
+ out_dir=Path(args.output) if args.output else None,
490
+ since=since, until=until)
491
+ except TelemetryError as e:
492
+ _emit_error("telemetry_error", str(e), args.format)
493
+ return 1
494
+ if args.format == "json":
495
+ print(json.dumps({"ok": True, "export_dir": str(out)}))
496
+ else:
497
+ print(f"Wrote export bundle to {out}")
498
+ print("Label: identified, minimized (single-party exports are "
499
+ "identified by construction).")
500
+ return 0
501
+
502
+
503
+ def _cmd_telemetry_anchor(args: argparse.Namespace) -> int:
504
+ from .telemetry_export import anchor_chain
505
+ anchor = anchor_chain()
506
+ if args.format == "json":
507
+ print(json.dumps(anchor, indent=2))
508
+ else:
509
+ print(f"chain head: {anchor['chain_head']}")
510
+ print(f"entries: {anchor['chain_entries']}")
511
+ print(anchor["note"])
512
+ return 0
513
+
514
+
515
+ def _cmd_telemetry_loss(args: argparse.Namespace) -> int:
516
+ try:
517
+ telemetry = record_failure_loss(
518
+ args.run_id, args.value, args.unit,
519
+ default_accepted=args.default_accepted or None)
520
+ except TelemetryError as e:
521
+ _emit_error("telemetry_error", str(e), args.format)
522
+ return 1
523
+ if args.format == "json":
524
+ print(json.dumps(telemetry, indent=2))
525
+ else:
526
+ print(f"{args.run_id}: severity {telemetry['failure.severity_band']} "
527
+ f"(floor {telemetry['failure.severity_floor']}), "
528
+ f"loss {telemetry['failure.estimated_loss']}")
529
+ return 0
530
+
531
+
419
532
  def _emit_error(code: str, message: str, fmt: str) -> None:
420
533
  if fmt == "json":
421
534
  print(json.dumps({"ok": False, "error_code": code, "message": message}), file=sys.stderr)
@@ -508,9 +621,54 @@ def build_parser() -> argparse.ArgumentParser:
508
621
 
509
622
  p_dclose = dsub.add_parser("close", help="Terminate a dispatch.")
510
623
  p_dclose.add_argument("run_id")
624
+ # Defaulted, not required: the CLI is the operator's close path, so the
625
+ # honest machine-set reason for it is operator-close. Scripted sweepers and
626
+ # session-teardown callers override it to say what they actually are.
627
+ p_dclose.add_argument(
628
+ "--reason", choices=sorted(VALID_TERMINATE_REASONS),
629
+ default=REASON_OPERATOR_CLOSE,
630
+ help="Why this dispatch is being terminated (default: operator-close).")
511
631
  _add_format(p_dclose)
512
632
  p_dclose.set_defaults(func=_cmd_dispatch_close)
513
633
 
634
+ p_tel = sub.add_parser("telemetry", help="Verification-telemetry surfaces.")
635
+ tsub = p_tel.add_subparsers(dest="telemetry_command", required=True)
636
+
637
+ p_tsum = tsub.add_parser(
638
+ "summary",
639
+ help="Local-only outcome/severity/integrity summary with trailing windows.")
640
+ _add_format(p_tsum)
641
+ p_tsum.set_defaults(func=_cmd_telemetry_summary)
642
+
643
+ p_texp = tsub.add_parser(
644
+ "export",
645
+ help="Write the allowlisted, date-bucketed export bundle for one recipient.")
646
+ p_texp.add_argument("--recipient", required=True,
647
+ help="Counterparty label; selects (or mints) the export salt.")
648
+ p_texp.add_argument("-o", "--output", default=None, help="Output directory.")
649
+ p_texp.add_argument("--since", default=None, help="Window start, YYYY-MM-DD.")
650
+ p_texp.add_argument("--until", default=None, help="Window end, YYYY-MM-DD.")
651
+ _add_format(p_texp)
652
+ p_texp.set_defaults(func=_cmd_telemetry_export)
653
+
654
+ p_tanc = tsub.add_parser(
655
+ "anchor",
656
+ help="Record the telemetry chain head as an anchorable value.")
657
+ _add_format(p_tanc)
658
+ p_tanc.set_defaults(func=_cmd_telemetry_anchor)
659
+
660
+ p_tloss = tsub.add_parser(
661
+ "loss",
662
+ help="Record the one raw loss quantity for a failed run (hours or usd).")
663
+ p_tloss.add_argument("run_id")
664
+ p_tloss.add_argument("--value", type=float, required=True,
665
+ help="The raw quantity; the band is derived, never chosen.")
666
+ p_tloss.add_argument("--unit", choices=["hours", "usd"], required=True)
667
+ p_tloss.add_argument("--default-accepted", action="store_true",
668
+ help="The suggested default was accepted unchanged.")
669
+ _add_format(p_tloss)
670
+ p_tloss.set_defaults(func=_cmd_telemetry_loss)
671
+
514
672
  p_fleet = sub.add_parser("fleet", help="The dispatch board, newest first.")
515
673
  p_fleet.add_argument("--open", action="store_true",
516
674
  help="Only dispatches that have not terminated.")
@@ -0,0 +1,131 @@
1
+ """Deployment-level telemetry configuration.
2
+
3
+ Verification telemetry needs a handful of facts that are properties of the
4
+ *deployment*, not of any single run: when telemetry capture was turned on, what
5
+ kind of work this deployment does, and how the operator wants it identified in
6
+ aggregates. Asking per run would violate the "automatic or it won't exist"
7
+ posture, so they live in one JSON file the operator edits once:
8
+
9
+ .fleetproof/config.json
10
+ {
11
+ "telemetry_era": "2026-08-20",
12
+ "run_context": "production",
13
+ "deployment_id": "my-fleet",
14
+ "operator_id": "me",
15
+ "eval_suite_id": null
16
+ }
17
+
18
+ Everything is optional and the file itself is optional: an absent or unreadable
19
+ config reads back as empty, and every consumer treats a missing value as "not
20
+ configured" rather than inventing one.
21
+
22
+ ``telemetry_era`` is the dated cutover that marks when this deployment started
23
+ capturing telemetry. It is stamped into each dispatch record *at creation* so
24
+ era membership is a property of the dispatch itself, never inferred from
25
+ whether a telemetry file happens to exist later — file absence is deletable,
26
+ a stamp written at dispatch time is on the record the aggregate reads.
27
+
28
+ The config file is resolved next to the runs directory (the ``.fleetproof/``
29
+ project marker), so tests that override the runs dir get an isolated config
30
+ for free, the same way the drift marker does.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import json
36
+ from datetime import date, datetime, timezone
37
+ from pathlib import Path
38
+ from typing import Any
39
+
40
+ from .runlog import runs_dir
41
+
42
+ CONFIG_FILENAME = "config.json"
43
+
44
+ # The kinds of work a deployment can declare itself to be doing. An aggregate
45
+ # that cannot tell real work from drills or synthetic load is worth less than
46
+ # one that can, so the value is per-deployment config — never per-run prompting,
47
+ # and never guessed from the work itself.
48
+ RUN_CONTEXT_PRODUCTION = "production"
49
+ RUN_CONTEXT_DEVELOPMENT = "development"
50
+ RUN_CONTEXT_DRILL = "drill"
51
+ RUN_CONTEXT_SYNTHETIC = "synthetic"
52
+ VALID_RUN_CONTEXTS = frozenset({
53
+ RUN_CONTEXT_PRODUCTION,
54
+ RUN_CONTEXT_DEVELOPMENT,
55
+ RUN_CONTEXT_DRILL,
56
+ RUN_CONTEXT_SYNTHETIC,
57
+ })
58
+
59
+
60
+ def config_path() -> Path:
61
+ return runs_dir().parent / CONFIG_FILENAME
62
+
63
+
64
+ def load_config() -> dict[str, Any]:
65
+ """The deployment config, or an empty dict when absent/unreadable.
66
+
67
+ A malformed config must not take down a hook, so every failure mode reads
68
+ as "not configured" — the consequence is a pre-telemetry-shaped record,
69
+ which is the honest description of a deployment whose config is broken.
70
+ """
71
+ try:
72
+ data = json.loads(config_path().read_text(encoding="utf-8-sig"))
73
+ except (OSError, json.JSONDecodeError):
74
+ return {}
75
+ return data if isinstance(data, dict) else {}
76
+
77
+
78
+ def telemetry_era_stamp(config: dict[str, Any] | None = None) -> str | None:
79
+ """The era value to stamp onto a dispatch created now, or None.
80
+
81
+ Returns the configured cutover date string once the cutover date has
82
+ arrived; a future-dated cutover (or no cutover at all) yields None, and the
83
+ dispatch is created pre-telemetry. The comparison is by date, matching the
84
+ granularity the config declares.
85
+ """
86
+ cfg = config if config is not None else load_config()
87
+ raw = cfg.get("telemetry_era")
88
+ if not isinstance(raw, str) or not raw.strip():
89
+ return None
90
+ try:
91
+ cutover = date.fromisoformat(raw.strip())
92
+ except ValueError:
93
+ return None
94
+ if datetime.now(timezone.utc).date() < cutover:
95
+ return None
96
+ return raw.strip()
97
+
98
+
99
+ def run_context(config: dict[str, Any] | None = None) -> str:
100
+ """The deployment's declared run context; ``production`` when unconfigured.
101
+
102
+ Defaulting to ``production`` is deliberate: a deployment that never touched
103
+ the config is doing its real work, and letting real work default into a
104
+ discounted bucket would under-count the corpus that matters. Drills and
105
+ synthetic load are the exceptional cases, so they are the ones that require
106
+ an explicit declaration. An unknown value degrades to the default rather
107
+ than crashing a hook on a typo.
108
+ """
109
+ cfg = config if config is not None else load_config()
110
+ value = cfg.get("run_context")
111
+ if isinstance(value, str) and value in VALID_RUN_CONTEXTS:
112
+ return value
113
+ return RUN_CONTEXT_PRODUCTION
114
+
115
+
116
+ def _config_str(key: str, config: dict[str, Any] | None = None) -> str | None:
117
+ cfg = config if config is not None else load_config()
118
+ value = cfg.get(key)
119
+ return value if isinstance(value, str) and value.strip() else None
120
+
121
+
122
+ def deployment_id(config: dict[str, Any] | None = None) -> str | None:
123
+ return _config_str("deployment_id", config)
124
+
125
+
126
+ def operator_id(config: dict[str, Any] | None = None) -> str | None:
127
+ return _config_str("operator_id", config)
128
+
129
+
130
+ def eval_suite_id(config: dict[str, Any] | None = None) -> str | None:
131
+ return _config_str("eval_suite_id", config)
@@ -41,7 +41,13 @@ from typing import Any
41
41
 
42
42
  from pathlib import Path
43
43
 
44
- from .checker import run_checks, select_checks, session_spec_baseline, spec_drifted
44
+ from .checker import (
45
+ CheckReport,
46
+ run_checks,
47
+ select_checks,
48
+ session_spec_baseline,
49
+ spec_drifted,
50
+ )
45
51
  from .checks import (
46
52
  SPEC_DRIFT_NOTE,
47
53
  CheckSpecError,
@@ -66,6 +72,7 @@ from .ledger import (
66
72
  record_verdict,
67
73
  )
68
74
  from .runlog import SESSION_ID_ENV, record, runs_dir
75
+ from .telemetry import build_telemetry
69
76
 
70
77
 
71
78
  def _read_hook_input() -> dict[str, Any]:
@@ -433,13 +440,34 @@ def _pin_drift_reason(pinned: str | None, current: str | None) -> str:
433
440
 
434
441
 
435
442
  def _try_close(run_id: str) -> None:
436
- """Terminate a dispatch, tolerating an already-closed one."""
443
+ """Terminate a dispatch, tolerating an already-closed one.
444
+
445
+ No terminate reason is passed: the reason vocabulary exists to split apart
446
+ the ways a dispatch dies *from ``dispatched``*, and this close only ever
447
+ runs after a report is on record (post-verdict, or post-grading with an
448
+ empty selection) — a state where the reason plays no part in how the
449
+ outcome reads back.
450
+ """
437
451
  try:
438
452
  close_dispatch(run_id, by="hook")
439
453
  except LedgerError as e:
440
454
  sys.stderr.write(f"[fleetproof] could not close dispatch {run_id}: {e}\n")
441
455
 
442
456
 
457
+ def _try_build_telemetry(run_id: str, check_report=None, checks=None) -> None:
458
+ """Best-effort telemetry build, on the checker's side of the boundary.
459
+
460
+ A telemetry failure must not wedge the gate. The cost of swallowing one is
461
+ an era-stamped run without telemetry.json — which the summary surfaces as
462
+ a ``telemetry_missing`` integrity defect, the visible form a capture
463
+ failure is supposed to take.
464
+ """
465
+ try:
466
+ build_telemetry(run_id, check_report=check_report, checks=checks)
467
+ except Exception as e:
468
+ sys.stderr.write(f"[fleetproof] telemetry build failed for {run_id}: {e}\n")
469
+
470
+
443
471
  def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
444
472
  """The per-subagent gate. Returns ``(decision_or_None, exit_code)``.
445
473
 
@@ -526,6 +554,14 @@ def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
526
554
  selected = select_checks(checks, dispatch.tier) if checks else []
527
555
  if not selected:
528
556
  _try_close(dispatch.run_id)
557
+ # The grading *happened* and selected nothing — recorded as an empty
558
+ # CheckReport so the telemetry layer can tell "checked nothing on
559
+ # purpose" (unverifiable) apart from "grading never ran" (ungraded).
560
+ _try_build_telemetry(
561
+ dispatch.run_id,
562
+ check_report=CheckReport(spec_sha256=spec_hash(), tier=dispatch.tier),
563
+ checks=[],
564
+ )
529
565
  return None, 0
530
566
 
531
567
  report = run_checks(selected, record_to_log=True, tier=dispatch.tier)
@@ -536,6 +572,7 @@ def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
536
572
  detail="; ".join(r.id for r in report.blocking_failures),
537
573
  by=VERDICT_BY,
538
574
  )
575
+ _try_build_telemetry(dispatch.run_id, check_report=report, checks=selected)
539
576
  return _subagent_block(_failure_reason(report), _evidence_context(report)), 0
540
577
 
541
578
  record_verdict(
@@ -545,6 +582,9 @@ def subagent_stop(payload: dict[str, Any]) -> tuple[dict[str, Any] | None, int]:
545
582
  by=VERDICT_BY,
546
583
  )
547
584
  _try_close(dispatch.run_id)
585
+ # Built after the close, so the outcome class derives from a finished
586
+ # lifecycle rather than a snapshot mid-transition.
587
+ _try_build_telemetry(dispatch.run_id, check_report=report, checks=selected)
548
588
  return None, 0
549
589
 
550
590