robotruth 0.1.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- robotruth/__init__.py +12 -0
- robotruth/_env.py +35 -0
- robotruth/audit.py +84 -0
- robotruth/cli.py +272 -0
- robotruth/compare.py +105 -0
- robotruth/contract/__init__.py +39 -0
- robotruth/contract/check.py +165 -0
- robotruth/contract/extract.py +462 -0
- robotruth/contract/spec.py +149 -0
- robotruth/fingerprint/__init__.py +36 -0
- robotruth/fingerprint/cell.py +194 -0
- robotruth/fingerprint/common.py +64 -0
- robotruth/fingerprint/drift.py +163 -0
- robotruth/fingerprint/unit.py +183 -0
- robotruth/guard/__init__.py +93 -0
- robotruth/guard/adapters.py +179 -0
- robotruth/guard/cli.py +100 -0
- robotruth/guard/conformal.py +261 -0
- robotruth/guard/metrics.py +173 -0
- robotruth/guard/monitor.py +540 -0
- robotruth/guard/scores.py +612 -0
- robotruth/judge/__init__.py +69 -0
- robotruth/judge/calibrate.py +367 -0
- robotruth/judge/cli.py +158 -0
- robotruth/judge/failbench.py +91 -0
- robotruth/judge/features.py +255 -0
- robotruth/judge/judge.py +371 -0
- robotruth/judge/open_vlm.py +107 -0
- robotruth/judge/vlm.py +329 -0
- robotruth/report.py +207 -0
- robotruth/results.py +86 -0
- robotruth/schema/__init__.py +41 -0
- robotruth/schema/episode.py +177 -0
- robotruth/schema/lerobot_ingest.py +240 -0
- robotruth/schema/log.py +109 -0
- robotruth/schema/metrics.py +95 -0
- robotruth/stats/__init__.py +44 -0
- robotruth/stats/intervals.py +94 -0
- robotruth/stats/pairwise.py +84 -0
- robotruth/stats/planning.py +56 -0
- robotruth/stats/schedule.py +59 -0
- robotruth/stats/sequential.py +136 -0
- robotruth/stats/timing.py +108 -0
- robotruth-0.1.6.dist-info/METADATA +531 -0
- robotruth-0.1.6.dist-info/RECORD +48 -0
- robotruth-0.1.6.dist-info/WHEEL +4 -0
- robotruth-0.1.6.dist-info/entry_points.txt +2 -0
- robotruth-0.1.6.dist-info/licenses/LICENSE +201 -0
robotruth/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""robotruth: Robot CI for learned robot policies.
|
|
2
|
+
|
|
3
|
+
Three truths, one report:
|
|
4
|
+
|
|
5
|
+
- configuration truth: is the policy you evaluated the policy you deployed?
|
|
6
|
+
(`robotruth.contract`)
|
|
7
|
+
- statistical truth: is checkpoint B really better than A? (`robotruth.stats`)
|
|
8
|
+
- outcome truth: did the episode succeed, when did it fail, and why?
|
|
9
|
+
(`robotruth.schema`, `robotruth.judge`, later modules)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.6"
|
robotruth/_env.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Tiny .env loader that tolerates PowerShell's UTF-16 output. Never prints values."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def load_env(path: Path | str = ".env") -> dict[str, str]:
|
|
10
|
+
p = Path(path)
|
|
11
|
+
if not p.exists():
|
|
12
|
+
return {}
|
|
13
|
+
raw = p.read_bytes()
|
|
14
|
+
for enc in ("utf-8-sig", "utf-16", "utf-16-le", "latin-1"):
|
|
15
|
+
try:
|
|
16
|
+
text = raw.decode(enc)
|
|
17
|
+
if "=" in text:
|
|
18
|
+
break
|
|
19
|
+
except UnicodeDecodeError:
|
|
20
|
+
continue
|
|
21
|
+
out: dict[str, str] = {}
|
|
22
|
+
for line in text.splitlines():
|
|
23
|
+
line = line.strip().lstrip("")
|
|
24
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
25
|
+
continue
|
|
26
|
+
k, v = line.split("=", 1)
|
|
27
|
+
out[k.strip()] = v.strip().strip('"').strip("'")
|
|
28
|
+
for k, v in out.items():
|
|
29
|
+
os.environ.setdefault(k, v)
|
|
30
|
+
return out
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def hf_token() -> str | None:
|
|
34
|
+
env = load_env(Path(__file__).resolve().parents[2] / ".env")
|
|
35
|
+
return env.get("HF_TOKEN") or os.environ.get("HF_TOKEN")
|
robotruth/audit.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Claims audit: which published or internal comparisons survive an interval?
|
|
2
|
+
|
|
3
|
+
Input: a CSV with columns `policy, task, successes, trials` (one row per policy and task, or
|
|
4
|
+
task blank for a pooled number). Output: every pairwise comparison within a task with the
|
|
5
|
+
difference, its interval, and whether it is resolved at the requested confidence. This is
|
|
6
|
+
the tool behind the first public evidence artifact: run the numbers papers actually report
|
|
7
|
+
through it and count how many claims hold up.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import csv
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from itertools import combinations
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
from scipy import stats as sps
|
|
19
|
+
|
|
20
|
+
from robotruth.report import Report, Section
|
|
21
|
+
from robotruth.stats.intervals import wilson
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class Claim:
|
|
26
|
+
policy: str
|
|
27
|
+
task: str
|
|
28
|
+
successes: int
|
|
29
|
+
trials: int
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def read_claims(path: Path | str) -> list[Claim]:
|
|
33
|
+
with Path(path).open(newline="", encoding="utf-8") as f:
|
|
34
|
+
rows = list(csv.DictReader(f))
|
|
35
|
+
out = []
|
|
36
|
+
for r in rows:
|
|
37
|
+
out.append(Claim(r["policy"].strip(), (r.get("task") or "ALL").strip() or "ALL", int(float(r["successes"])), int(float(r["trials"]))))
|
|
38
|
+
return out
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def newcombe_diff_ci(k1: int, n1: int, k2: int, n2: int, alpha: float = 0.05) -> tuple[float, float, float]:
|
|
42
|
+
"""Newcombe hybrid score interval for p2 - p1 (two independent proportions)."""
|
|
43
|
+
w1, w2 = wilson(k1, n1, alpha), wilson(k2, n2, alpha)
|
|
44
|
+
d = k2 / n2 - k1 / n1
|
|
45
|
+
lo = d - np.sqrt((w1.upper - w1.estimate) ** 2 + (w2.estimate - w2.lower) ** 2)
|
|
46
|
+
hi = d + np.sqrt((w1.estimate - w1.lower) ** 2 + (w2.upper - w2.estimate) ** 2)
|
|
47
|
+
return float(d), float(lo), float(hi)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def fisher_p(k1: int, n1: int, k2: int, n2: int) -> float:
|
|
51
|
+
table = [[k1, n1 - k1], [k2, n2 - k2]]
|
|
52
|
+
return float(sps.fisher_exact(table)[1])
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def audit_claims(claims: list[Claim], alpha: float = 0.05, title: str = "robotruth claims audit") -> Report:
|
|
56
|
+
rep = Report(title)
|
|
57
|
+
by_task: dict[str, list[Claim]] = {}
|
|
58
|
+
for c in claims:
|
|
59
|
+
by_task.setdefault(c.task, []).append(c)
|
|
60
|
+
|
|
61
|
+
rate_rows = []
|
|
62
|
+
for c in claims:
|
|
63
|
+
w = wilson(c.successes, c.trials, alpha)
|
|
64
|
+
rate_rows.append({"task": c.task, "policy": c.policy, "trials": c.trials,
|
|
65
|
+
"rate": f"{w.estimate:.2f} [{w.lower:.2f}, {w.upper:.2f}]", "interval width": round(w.width, 3)})
|
|
66
|
+
rep.add(Section("Reported rates with intervals", "What each number actually says once the interval is attached.", rate_rows))
|
|
67
|
+
|
|
68
|
+
comp_rows = []
|
|
69
|
+
n_resolved = 0
|
|
70
|
+
for task, cs in by_task.items():
|
|
71
|
+
for a, b in combinations(cs, 2):
|
|
72
|
+
d, lo, hi = newcombe_diff_ci(a.successes, a.trials, b.successes, b.trials, alpha)
|
|
73
|
+
p = fisher_p(a.successes, a.trials, b.successes, b.trials)
|
|
74
|
+
resolved = (lo > 0) or (hi < 0)
|
|
75
|
+
n_resolved += resolved
|
|
76
|
+
comp_rows.append({"task": task, "A": a.policy, "B": b.policy, "diff (B-A)": round(d, 3),
|
|
77
|
+
"interval": f"[{lo:+.2f}, {hi:+.2f}]", "Fisher p": round(p, 4),
|
|
78
|
+
"resolved": "yes" if resolved else "no"})
|
|
79
|
+
n_comp = len(comp_rows)
|
|
80
|
+
body = (f"{n_comp} pairwise comparisons within tasks; {n_resolved} resolved at {100*(1-alpha):.0f}% confidence, "
|
|
81
|
+
f"{n_comp - n_resolved} are inside the noise.")
|
|
82
|
+
rep.add(Section("Pairwise comparisons", body, comp_rows, verdict="INFO"))
|
|
83
|
+
rep.verdict = "WARN" if n_comp and n_resolved < n_comp else "PASS"
|
|
84
|
+
return rep
|
robotruth/cli.py
ADDED
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""robotruth command line."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import json
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Optional
|
|
9
|
+
|
|
10
|
+
import typer
|
|
11
|
+
from rich.console import Console
|
|
12
|
+
from rich.table import Table
|
|
13
|
+
|
|
14
|
+
from robotruth import __version__
|
|
15
|
+
from robotruth.contract import ExecSpec, Severity, check_contract, extract_spec
|
|
16
|
+
from robotruth.results import Results
|
|
17
|
+
|
|
18
|
+
app = typer.Typer(help="Robot CI: contract checks, honest statistics, episode records, drift fingerprints, outcome judging and runtime monitoring for learned robot policies.", no_args_is_help=True)
|
|
19
|
+
contract_app = typer.Typer(help="Module 1: executable-policy contract manifests.", no_args_is_help=True)
|
|
20
|
+
stats_app = typer.Typer(help="Module 2: honest evaluation statistics.", no_args_is_help=True)
|
|
21
|
+
episodes_app = typer.Typer(help="Module 3: episode records, fleet metrics, MCAP bridge.", no_args_is_help=True)
|
|
22
|
+
fingerprint_app = typer.Typer(help="Module 4: cell and robot-unit fingerprints and drift.", no_args_is_help=True)
|
|
23
|
+
app.add_typer(contract_app, name="contract")
|
|
24
|
+
app.add_typer(stats_app, name="stats")
|
|
25
|
+
app.add_typer(episodes_app, name="episodes")
|
|
26
|
+
app.add_typer(fingerprint_app, name="fingerprint")
|
|
27
|
+
from robotruth.judge.cli import judge_app # noqa: E402
|
|
28
|
+
from robotruth.guard.cli import guard_app # noqa: E402
|
|
29
|
+
app.add_typer(judge_app, name="judge")
|
|
30
|
+
app.add_typer(guard_app, name="guard")
|
|
31
|
+
console = Console()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _version_cb(value: bool):
|
|
35
|
+
if value:
|
|
36
|
+
console.print(f"robotruth {__version__}")
|
|
37
|
+
raise typer.Exit()
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@app.callback()
|
|
41
|
+
def _main(version: bool = typer.Option(False, "--version", help="Show version and exit.", callback=_version_cb, is_eager=True)):
|
|
42
|
+
pass
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@contract_app.command("extract")
|
|
46
|
+
def contract_extract(path: Path = typer.Argument(..., help="Checkpoint directory."),
|
|
47
|
+
out: Path = typer.Option(Path("execspec.json"), "--out", "-o"),
|
|
48
|
+
family: Optional[str] = typer.Option(None, help="lerobot | openpi | gr00t | custom (auto-detected if omitted)"),
|
|
49
|
+
role: str = typer.Option("unspecified", help="evaluated | deployed"),
|
|
50
|
+
overlay: Optional[Path] = typer.Option(None, help="JSON with fields the extractor cannot read (control_hz, gripper_convention, cameras...).")):
|
|
51
|
+
"""Build an ExecSpec manifest from a checkpoint directory."""
|
|
52
|
+
ov = json.loads(overlay.read_text(encoding="utf-8")) if overlay else None
|
|
53
|
+
spec = extract_spec(path, family=family, overlay=ov, role=role)
|
|
54
|
+
spec.to_json(out)
|
|
55
|
+
console.print(f"[green]wrote[/] {out} family={spec.policy.family} fingerprint={spec.fingerprint()[:16]}")
|
|
56
|
+
if spec.unverified:
|
|
57
|
+
console.print(f"[yellow]unverified fields ({len(spec.unverified)}):[/] " + ", ".join(spec.unverified))
|
|
58
|
+
console.print("Supply them with --overlay or a robotruth.spec.json next to the checkpoint, or the check will fail closed.")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@contract_app.command("check")
|
|
62
|
+
def contract_check(evaluated: Path, deployed: Path,
|
|
63
|
+
allow_unknown: bool = typer.Option(False, help="Downgrade unknown fatal fields to warnings (not recommended)."),
|
|
64
|
+
out: Optional[Path] = typer.Option(None, "--out", "-o", help="Write a JSON report here.")):
|
|
65
|
+
"""Compare the evaluated manifest against the deployed one. Exit code 1 on FAIL."""
|
|
66
|
+
res = check_contract(ExecSpec.from_json(evaluated), ExecSpec.from_json(deployed), allow_unknown=allow_unknown)
|
|
67
|
+
table = Table(title=res.summary())
|
|
68
|
+
for col in ("severity", "field", "evaluated", "deployed", "why"):
|
|
69
|
+
table.add_column(col, overflow="fold")
|
|
70
|
+
colours = {Severity.FATAL: "red", Severity.WARN: "yellow", Severity.INFO: "blue", Severity.OK: "green"}
|
|
71
|
+
for f in sorted(res.findings, key=lambda f: list(Severity).index(f.severity)):
|
|
72
|
+
table.add_row(f"[{colours[f.severity]}]{f.severity.value}[/]", f.field, _short(f.evaluated), _short(f.deployed), f.message)
|
|
73
|
+
console.print(table)
|
|
74
|
+
if out:
|
|
75
|
+
out.write_text(json.dumps({"passed": res.passed, "summary": res.summary(),
|
|
76
|
+
"findings": [f.__dict__ | {"severity": f.severity.value} for f in res.findings]},
|
|
77
|
+
indent=2, default=str), encoding="utf-8")
|
|
78
|
+
raise typer.Exit(code=0 if res.passed else 1)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _short(v) -> str:
|
|
82
|
+
s = str(v)
|
|
83
|
+
return s if len(s) <= 24 else s[:10] + "..." + s[-10:]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@stats_app.command("plan")
|
|
87
|
+
def stats_plan(p_a: float = typer.Option(..., help="Baseline success rate, e.g. 0.80"),
|
|
88
|
+
p_b: float = typer.Option(..., help="Rate you hope to detect, e.g. 0.90"),
|
|
89
|
+
alpha: float = 0.05, power: float = 0.8,
|
|
90
|
+
rho: float = typer.Option(0.4, help="Within-pair outcome correlation for the paired design.")):
|
|
91
|
+
"""How many trials you need before you start."""
|
|
92
|
+
from robotruth.stats.planning import required_trials_two_proportions
|
|
93
|
+
console.print(str(required_trials_two_proportions(p_a, p_b, alpha, power, paired=False)))
|
|
94
|
+
console.print(str(required_trials_two_proportions(p_a, p_b, alpha, power, paired=True, rho=rho)))
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@stats_app.command("schedule")
|
|
98
|
+
def stats_schedule(policies: str = typer.Option(..., help="Comma-separated policy names."),
|
|
99
|
+
conditions: str = typer.Option(..., help="Comma-separated condition labels (object pose, lighting...)."),
|
|
100
|
+
repeats: int = 1, seed: int = 0,
|
|
101
|
+
out: Path = typer.Option(Path("schedule.csv")),
|
|
102
|
+
key_out: Path = typer.Option(Path("schedule_key.json"), help="Blinding key; keep it away from the operator.")):
|
|
103
|
+
"""Write a blinded, interleaved A/B/n schedule."""
|
|
104
|
+
from robotruth.stats.schedule import interleaved_schedule
|
|
105
|
+
s = interleaved_schedule([p.strip() for p in policies.split(",")], [c.strip() for c in conditions.split(",")], repeats, seed)
|
|
106
|
+
rows = s.to_rows()
|
|
107
|
+
with out.open("w", newline="", encoding="utf-8") as f:
|
|
108
|
+
w = csv.DictWriter(f, fieldnames=list(rows[0].keys()))
|
|
109
|
+
w.writeheader(); w.writerows(rows)
|
|
110
|
+
key_out.write_text(json.dumps(s.key, indent=2), encoding="utf-8")
|
|
111
|
+
console.print(f"[green]wrote[/] {out} ({len(rows)} slots) and blinding key {key_out}")
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@stats_app.command("compare")
|
|
115
|
+
def stats_compare(results: Path = typer.Argument(..., help="Results CSV (see robotruth.results)."),
|
|
116
|
+
a: str = typer.Option(..., help="Baseline policy name."),
|
|
117
|
+
b: str = typer.Option(..., help="Candidate policy name."),
|
|
118
|
+
alpha: float = 0.05,
|
|
119
|
+
margin: float = typer.Option(0.0, help="Minimum improvement that counts, e.g. 0.02."),
|
|
120
|
+
out_dir: Path = typer.Option(Path("reports"))):
|
|
121
|
+
"""Is B better than A? Writes a Markdown and HTML report. Exit 1 if B is worse."""
|
|
122
|
+
from robotruth.compare import compare_policies
|
|
123
|
+
res = Results.read_csv(results)
|
|
124
|
+
rep = compare_policies(res, a, b, alpha=alpha, margin=margin)
|
|
125
|
+
md, html = rep.write(out_dir, stem=f"compare_{a}_vs_{b}")
|
|
126
|
+
console.print(rep.to_markdown())
|
|
127
|
+
console.print(f"[green]wrote[/] {md} and {html}")
|
|
128
|
+
raise typer.Exit(code=1 if rep.verdict == "FAIL" else 0)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@stats_app.command("audit")
|
|
132
|
+
def stats_audit(claims_csv: Path = typer.Argument(..., help="CSV: policy, task, successes, trials"),
|
|
133
|
+
alpha: float = 0.05, out_dir: Path = typer.Option(Path("reports"))):
|
|
134
|
+
"""Which reported comparisons survive an interval? Pairwise Newcombe intervals and Fisher tests within each task."""
|
|
135
|
+
from robotruth.audit import audit_claims, read_claims
|
|
136
|
+
rep = audit_claims(read_claims(claims_csv), alpha)
|
|
137
|
+
md, html = rep.write(out_dir, stem="claims_audit")
|
|
138
|
+
console.print(rep.to_markdown())
|
|
139
|
+
console.print(f"[green]wrote[/] {md} and {html}")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
@stats_app.command("interval")
|
|
143
|
+
def stats_interval(k: int, n: int, alpha: float = 0.05):
|
|
144
|
+
"""Interval for k successes out of n. Because a rate without an interval is not a result."""
|
|
145
|
+
from robotruth.stats.intervals import clopper_pearson, wilson
|
|
146
|
+
console.print(str(wilson(k, n, alpha)))
|
|
147
|
+
console.print(str(clopper_pearson(k, n, alpha)))
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
@episodes_app.command("validate")
|
|
151
|
+
def episodes_validate(log: Path):
|
|
152
|
+
"""Validate a JSONL episode log against the schema."""
|
|
153
|
+
from robotruth.schema import EpisodeLog
|
|
154
|
+
ok, errors = EpisodeLog(log).validate()
|
|
155
|
+
console.print(f"{ok} valid records, {len(errors)} errors")
|
|
156
|
+
for line, err in errors[:20]:
|
|
157
|
+
console.print(f" [red]line {line}[/]: {err}")
|
|
158
|
+
raise typer.Exit(code=1 if errors else 0)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@episodes_app.command("to-results")
|
|
162
|
+
def episodes_to_results(log: Path, out: Path = typer.Option(Path("results.csv"), "--out", "-o")):
|
|
163
|
+
"""Export episodes with an outcome to the results CSV used by `robotruth stats compare`."""
|
|
164
|
+
from robotruth.schema import EpisodeLog
|
|
165
|
+
p = EpisodeLog(log).to_results_csv(out)
|
|
166
|
+
console.print(f"[green]wrote[/] {p}")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@episodes_app.command("metrics")
|
|
170
|
+
def episodes_metrics(log: Path, alpha: float = 0.05, out: Optional[Path] = typer.Option(None, "--out", "-o"),
|
|
171
|
+
out_dir: Optional[Path] = typer.Option(None, help="Write fleet_report.md and fleet_report.html here.")):
|
|
172
|
+
"""Fleet metrics: success, autonomous fraction, interventions per hour, MTBI, failure Pareto. Markdown to stdout; --out-dir adds a styled HTML report."""
|
|
173
|
+
from robotruth.report import Report, Section
|
|
174
|
+
from robotruth.schema import EpisodeLog, fleet_metrics
|
|
175
|
+
fm = fleet_metrics(EpisodeLog(log), alpha)
|
|
176
|
+
md = fm.to_markdown()
|
|
177
|
+
console.print(md)
|
|
178
|
+
if out:
|
|
179
|
+
out.write_text(md, encoding="utf-8")
|
|
180
|
+
if out_dir:
|
|
181
|
+
rep = Report(f"robotruth fleet report: {log.name}")
|
|
182
|
+
rep.add(Section("Fleet metrics", md))
|
|
183
|
+
md_path, html_path = rep.write(out_dir, stem="fleet_report")
|
|
184
|
+
console.print(f"[green]wrote[/] {md_path} and {html_path}")
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
@episodes_app.command("to-mcap")
|
|
188
|
+
def episodes_to_mcap(log: Path, out: Path = typer.Option(Path("episodes.mcap"), "--out", "-o")):
|
|
189
|
+
"""Write the log as JSON messages on /robotruth/episode for Foxglove or Rerun."""
|
|
190
|
+
from robotruth.schema import EpisodeLog
|
|
191
|
+
console.print(f"[green]wrote[/] {EpisodeLog(log).to_mcap(out)}")
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
@episodes_app.command("from-mcap")
|
|
195
|
+
def episodes_from_mcap(mcap_file: Path, out: Path = typer.Option(Path("episodes.jsonl"), "--out", "-o")):
|
|
196
|
+
"""Extract /robotruth/episode messages from an MCAP file into a JSONL log."""
|
|
197
|
+
from robotruth.schema import EpisodeLog
|
|
198
|
+
log = EpisodeLog.from_mcap(mcap_file, out)
|
|
199
|
+
console.print(f"[green]wrote[/] {log.path} ({len(log.records())} records)")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
@episodes_app.command("from-lerobot")
|
|
203
|
+
def episodes_from_lerobot(dataset_dir: Path, out: Path = typer.Option(Path("episodes.jsonl"), "--out", "-o"),
|
|
204
|
+
policy: Optional[str] = typer.Option(None, help="Policy label; defaults to dataset:<repo_id>."),
|
|
205
|
+
max_episodes: Optional[int] = None):
|
|
206
|
+
"""Convert a LeRobot dataset directory into an episode log (success from next.success or next.reward, interventions from flag columns)."""
|
|
207
|
+
from robotruth.schema import EpisodeLog
|
|
208
|
+
from robotruth.schema.lerobot_ingest import LeRobotDataset
|
|
209
|
+
ds = LeRobotDataset(dataset_dir)
|
|
210
|
+
recs = ds.to_records(policy, max_episodes)
|
|
211
|
+
if out.exists():
|
|
212
|
+
out.unlink()
|
|
213
|
+
log = EpisodeLog(out)
|
|
214
|
+
log.extend(recs)
|
|
215
|
+
labelled = sum(r.outcome.success is not None for r in recs)
|
|
216
|
+
console.print(f"[green]wrote[/] {out}: {len(recs)} episodes from {ds.repo_id} ({ds.robot}, {ds.fps:.0f} fps, {len(ds.camera_keys())} cameras); {labelled} with success labels")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
@episodes_app.command("taxonomy")
|
|
220
|
+
def episodes_taxonomy():
|
|
221
|
+
"""Print the failure taxonomy."""
|
|
222
|
+
from robotruth.schema import FAILURE_SUBCLASSES
|
|
223
|
+
for cls, subs in FAILURE_SUBCLASSES.items():
|
|
224
|
+
console.print(f"[bold]{cls}[/]: " + ", ".join(subs))
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
@fingerprint_app.command("unit")
|
|
228
|
+
def fingerprint_unit(excitation_csv: Path, unit_id: str = typer.Option("unknown"), out: Path = typer.Option(Path("unit_fingerprint.json"), "--out", "-o")):
|
|
229
|
+
"""Fingerprint a robot unit from a fixed excitation trajectory log (t, cmd_<j>, meas_<j> columns)."""
|
|
230
|
+
from robotruth.fingerprint.unit import read_excitation_csv, unit_fingerprint
|
|
231
|
+
t, cmd, meas = read_excitation_csv(excitation_csv)
|
|
232
|
+
fp = unit_fingerprint(t, cmd, meas, unit_id)
|
|
233
|
+
fp.to_json(out)
|
|
234
|
+
for j, st in fp.joints.items():
|
|
235
|
+
console.print(f"{j}: rmse={st.rmse:.4f} lag={1000*st.lag_s:.0f}ms backlash={st.backlash:.4f} offset={st.steady_error:+.4f} gain={st.gain:.3f}")
|
|
236
|
+
console.print(f"[green]wrote[/] {out} fingerprint={fp.fingerprint[:16]}")
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
@fingerprint_app.command("cell")
|
|
240
|
+
def fingerprint_cell(image: Path, camera: str = typer.Option("cam"), aruco_dict: str = typer.Option("4x4_50"),
|
|
241
|
+
out: Path = typer.Option(Path("cell_fingerprint.json"), "--out", "-o")):
|
|
242
|
+
"""Fingerprint a workspace camera view: ArUco marker positions, exposure, sharpness, colour balance."""
|
|
243
|
+
from robotruth.fingerprint import cell_fingerprint
|
|
244
|
+
fp = cell_fingerprint(image, camera, aruco_dict)
|
|
245
|
+
fp.to_json(out)
|
|
246
|
+
console.print(f"{len(fp.markers)} markers, luma={fp.photometrics.mean_luma:.0f}, sharpness={fp.photometrics.sharpness:.0f}")
|
|
247
|
+
console.print(f"[green]wrote[/] {out} fingerprint={fp.fingerprint[:16]}")
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
@fingerprint_app.command("diff")
|
|
251
|
+
def fingerprint_diff(reference: Path, current: Path, out: Optional[Path] = typer.Option(None, "--out", "-o")):
|
|
252
|
+
"""Compare two fingerprints (unit or cell) against drift tolerances. Exit 1 on FAIL."""
|
|
253
|
+
from robotruth.fingerprint import CellFingerprint, UnitFingerprint, diff_cell, diff_unit
|
|
254
|
+
ref_d = json.loads(reference.read_text(encoding="utf-8"))
|
|
255
|
+
if "joints" in ref_d:
|
|
256
|
+
rep = diff_unit(UnitFingerprint.from_json(reference), UnitFingerprint.from_json(current))
|
|
257
|
+
else:
|
|
258
|
+
rep = diff_cell(CellFingerprint.from_json(reference), CellFingerprint.from_json(current))
|
|
259
|
+
table = Table(title=rep.summary())
|
|
260
|
+
for col in ("severity", "metric", "reference", "current", "delta", "why"):
|
|
261
|
+
table.add_column(col, overflow="fold")
|
|
262
|
+
colours = {"fail": "red", "warn": "yellow", "ok": "green"}
|
|
263
|
+
for f in sorted(rep.findings, key=lambda f: ["fail", "warn", "ok"].index(f.severity)):
|
|
264
|
+
table.add_row(f"[{colours[f.severity]}]{f.severity}[/]", f.metric, _short(f.reference), _short(f.current), f"{f.delta:.4g}", f.message)
|
|
265
|
+
console.print(table)
|
|
266
|
+
if out:
|
|
267
|
+
out.write_text(json.dumps(rep.to_dict(), indent=2, default=str), encoding="utf-8")
|
|
268
|
+
raise typer.Exit(code=0 if rep.passed else 1)
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
if __name__ == "__main__":
|
|
272
|
+
app()
|
robotruth/compare.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Compare two policies from a results table and build the report.
|
|
2
|
+
|
|
3
|
+
This is the function behind `robotruth stats compare`. It applies every rule in module 2:
|
|
4
|
+
intervals on every rate, paired analysis when pair ids exist, an anytime-valid sequential
|
|
5
|
+
verdict, time-to-success with censoring, per-task breakdown, and an explicit statement of
|
|
6
|
+
how many trials would have been needed if the comparison is inconclusive.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Optional
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
from robotruth.report import Report, Section
|
|
16
|
+
from robotruth.results import Results
|
|
17
|
+
from robotruth.stats.intervals import bootstrap_diff_ci, paired_diff_ci, wilson
|
|
18
|
+
from robotruth.stats.planning import required_trials_two_proportions
|
|
19
|
+
from robotruth.stats.sequential import sequential_paired_test
|
|
20
|
+
from robotruth.stats.timing import logrank_test
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def compare_policies(res: Results, a: str, b: str, alpha: float = 0.05, margin: float = 0.0,
|
|
24
|
+
title: Optional[str] = None) -> Report:
|
|
25
|
+
rep = Report(title or f"robotruth: is {b} better than {a}?")
|
|
26
|
+
ra, rb = res.subset(a), res.subset(b)
|
|
27
|
+
if not ra.rows or not rb.rows:
|
|
28
|
+
raise ValueError(f"no rows for one of the policies: {a}={len(ra.rows)}, {b}={len(rb.rows)}")
|
|
29
|
+
|
|
30
|
+
# 1. Rates with intervals, overall and per task.
|
|
31
|
+
rows = []
|
|
32
|
+
for task in [None] + res.tasks():
|
|
33
|
+
sa, sb = ra.subset(task=task), rb.subset(task=task)
|
|
34
|
+
if not sa.rows or not sb.rows:
|
|
35
|
+
continue
|
|
36
|
+
wa = wilson(int(sa.successes().sum()), len(sa.rows), alpha)
|
|
37
|
+
wb = wilson(int(sb.successes().sum()), len(sb.rows), alpha)
|
|
38
|
+
rows.append({
|
|
39
|
+
"task": task or "ALL", f"{a} n": wa.n, f"{a} rate": f"{wa.estimate:.2f} [{wa.lower:.2f}, {wa.upper:.2f}]",
|
|
40
|
+
f"{b} n": wb.n, f"{b} rate": f"{wb.estimate:.2f} [{wb.lower:.2f}, {wb.upper:.2f}]",
|
|
41
|
+
"intervals overlap": "yes" if (wa.lower <= wb.upper and wb.lower <= wa.upper) else "no",
|
|
42
|
+
})
|
|
43
|
+
rep.add(Section("Success rates with Wilson intervals",
|
|
44
|
+
f"Every rate carries a {100*(1-alpha):.0f}% interval. Overlapping intervals do not prove equality; "
|
|
45
|
+
"non-overlap is strong evidence of a difference.", rows))
|
|
46
|
+
|
|
47
|
+
# 2. Difference estimate: paired if possible.
|
|
48
|
+
paired = res.has_pairs()
|
|
49
|
+
if paired:
|
|
50
|
+
xa, xb, pairs = res.paired(a, b)
|
|
51
|
+
if xa.size >= 2:
|
|
52
|
+
ci = paired_diff_ci(xa, xb, alpha)
|
|
53
|
+
boot = bootstrap_diff_ci(xa, xb, alpha, paired=True)
|
|
54
|
+
n_pairs = xa.size
|
|
55
|
+
body = (f"{n_pairs} matched pairs (same pair_id and task).\n"
|
|
56
|
+
f"Difference in success (B - A): {ci}\n"
|
|
57
|
+
f"Bootstrap check: {boot}\n"
|
|
58
|
+
f"Discordant pairs: A only {int(np.sum((xa==1)&(xb==0)))}, B only {int(np.sum((xa==0)&(xb==1)))}.")
|
|
59
|
+
verdict = "PASS" if ci.lower > margin else ("FAIL" if ci.upper < -margin else "WARN")
|
|
60
|
+
rep.add(Section("Paired difference", body, verdict=verdict))
|
|
61
|
+
# 3. Anytime-valid sequential verdict, in pair order.
|
|
62
|
+
seq = sequential_paired_test(xa, xb, alpha=alpha, margin=margin, stop_early=False)
|
|
63
|
+
first = next((t for t in seq.trace if (t[2] > margin or t[3] < -margin)), None)
|
|
64
|
+
stop_note = (f"An anytime-valid stopping rule would have decided at n={first[0]}."
|
|
65
|
+
if first else "The sequence never excluded zero; more paired trials are needed.")
|
|
66
|
+
rep.add(Section("Anytime-valid sequential verdict",
|
|
67
|
+
f"{seq}\n{stop_note}\nThis interval is valid at every sample size simultaneously "
|
|
68
|
+
"(empirical Bernstein confidence sequence), so peeking after each rollout is allowed.",
|
|
69
|
+
table=[{"n": t[0], "diff": t[1], "lower": t[2], "upper": t[3]} for t in seq.trace[:: max(1, len(seq.trace)//20)]],
|
|
70
|
+
verdict={"B_better": "PASS", "A_better": "FAIL"}.get(seq.decision, "WARN")))
|
|
71
|
+
else:
|
|
72
|
+
paired = False
|
|
73
|
+
if not paired:
|
|
74
|
+
boot = bootstrap_diff_ci(ra.scores(), rb.scores(), alpha, paired=False)
|
|
75
|
+
verdict = "PASS" if boot.lower > margin else ("FAIL" if boot.upper < -margin else "WARN")
|
|
76
|
+
rep.add(Section("Unpaired difference",
|
|
77
|
+
f"No pair_id column found, so trials are treated as independent.\n"
|
|
78
|
+
f"Difference in mean score (B - A): {boot}\n"
|
|
79
|
+
"Interleave and pair your trials next time: it needs fewer rollouts for the same power.",
|
|
80
|
+
verdict=verdict))
|
|
81
|
+
|
|
82
|
+
# 4. Time to success with censoring.
|
|
83
|
+
ta, ea = ra.times()
|
|
84
|
+
tb, eb = rb.times()
|
|
85
|
+
if np.isfinite(ta).all() and np.isfinite(tb).all() and (ea.sum() + eb.sum()) > 0:
|
|
86
|
+
lr = logrank_test(ta, ea, tb, eb)
|
|
87
|
+
rep.add(Section("Time to success (censored)",
|
|
88
|
+
f"{lr}\nFailures are treated as censored at their timeout. A faster policy at equal success "
|
|
89
|
+
"rate is a better policy; binary success cannot see that.",
|
|
90
|
+
verdict="INFO"))
|
|
91
|
+
|
|
92
|
+
# 5. If inconclusive, say what it would take.
|
|
93
|
+
pa = ra.successes().mean()
|
|
94
|
+
pb = rb.successes().mean()
|
|
95
|
+
if abs(pa - pb) > 1e-9 and 0 < pa < 1 and 0 < pb < 1:
|
|
96
|
+
plan_ind = required_trials_two_proportions(pa, pb, alpha, 0.8, paired=False)
|
|
97
|
+
plan_pair = required_trials_two_proportions(pa, pb, alpha, 0.8, paired=True, rho=0.4)
|
|
98
|
+
rep.add(Section("What it would take to resolve this",
|
|
99
|
+
f"Observed rates {pa:.2f} vs {pb:.2f}. To detect that gap with 80% power:\n"
|
|
100
|
+
f"- independent trials: {plan_ind}\n- paired interleaved trials (rho 0.4): {plan_pair}\n"
|
|
101
|
+
f"You have {len(ra.rows)} and {len(rb.rows)} trials.", verdict="INFO"))
|
|
102
|
+
|
|
103
|
+
verdicts = [s.verdict for s in rep.sections if s.verdict in ("PASS", "FAIL", "WARN")]
|
|
104
|
+
rep.verdict = "FAIL" if "FAIL" in verdicts else ("PASS" if verdicts and all(v == "PASS" for v in verdicts) else "WARN")
|
|
105
|
+
return rep
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Module 1: executable-policy contract checker.
|
|
2
|
+
|
|
3
|
+
A checkpoint is not a policy. The thing that runs on a robot is the tuple of weights,
|
|
4
|
+
normalization statistics, action semantics, control rate, chunking, preprocessing, camera
|
|
5
|
+
calibration and embodiment. Swap any one and the same weights become a different policy
|
|
6
|
+
("Same Weights, Different Robot", arXiv 2606.03724: 28/28 to 2/28).
|
|
7
|
+
|
|
8
|
+
This module extracts that tuple into an `ExecSpec` manifest and compares two manifests,
|
|
9
|
+
failing closed on anything it cannot verify.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from robotruth.contract.spec import (
|
|
13
|
+
ActionSemantics,
|
|
14
|
+
CameraSpec,
|
|
15
|
+
Embodiment,
|
|
16
|
+
ExecSpec,
|
|
17
|
+
Normalization,
|
|
18
|
+
ObservationSpec,
|
|
19
|
+
PolicyIdentity,
|
|
20
|
+
RuntimeSpec,
|
|
21
|
+
)
|
|
22
|
+
from robotruth.contract.check import CheckResult, Finding, Severity, check_contract
|
|
23
|
+
from robotruth.contract.extract import extract_spec
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"ActionSemantics",
|
|
27
|
+
"CameraSpec",
|
|
28
|
+
"CheckResult",
|
|
29
|
+
"Embodiment",
|
|
30
|
+
"ExecSpec",
|
|
31
|
+
"Finding",
|
|
32
|
+
"Normalization",
|
|
33
|
+
"ObservationSpec",
|
|
34
|
+
"PolicyIdentity",
|
|
35
|
+
"RuntimeSpec",
|
|
36
|
+
"Severity",
|
|
37
|
+
"check_contract",
|
|
38
|
+
"extract_spec",
|
|
39
|
+
]
|