sediment-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sediment_cli/cli.py ADDED
@@ -0,0 +1,1922 @@
1
+ # SPDX-License-Identifier: AGPL-3.0-or-later
2
+ """
3
+ Sediment CLI: login/logout, remote facts + commit, quarantine, exports, reports, mirror GC.
4
+
5
+ Remote verbs (login/logout/facts/commit/demo) speak HTTP through
6
+ ``sediment_cli.client`` and never open the fact store; only explicit
7
+ local mode (``facts --database-url`` or ``SEDIMENT_DATABASE_URL``) reads the store
8
+ directly, so ``docker compose exec api sediment facts`` is unchanged.
9
+
10
+ Store verbs are presentation-only (ADR 0001): each is a ``FactStore`` method,
11
+ an export pipeline call, or a forwarded report module — the CLI adds no
12
+ semantics of its own. Store-writing verbs read the same settings as the API
13
+ (database URL, mirror path, org id), so there is exactly one configuration
14
+ surface; ``report`` and ``mirror-gc`` never construct ``Settings`` — they
15
+ take ``--org``/``--database-url``/``--mirror-path`` with the ``SEDIMENT_*`` env
16
+ vars as defaults, so a report can cover any org a deployment holds without
17
+ API auth config.
18
+
19
+ Safety posture: the bulk ``quarantine-inference-calls`` form dry-runs by
20
+ default and writes only with ``--apply``, and refuses a filterless
21
+ invocation without ``--all`` (no filters means every inference call in the org).
22
+ Per-fact ``quarantine``/``release`` act immediately — one id, reversible,
23
+ ``--reason`` always required. Output vocabulary is the CONTEXT.md glossary:
24
+ facts are quarantined and released, never deleted.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import argparse
30
+ import hashlib
31
+ import importlib
32
+ import os
33
+ import sys
34
+ from contextlib import ExitStack
35
+ from datetime import datetime
36
+ from itertools import islice
37
+ from pathlib import Path
38
+ from typing import Any
39
+ from urllib.parse import urlencode, urlsplit
40
+
41
+ from pydantic import TypeAdapter
42
+
43
+ from . import __version__, ui
44
+ from .client import (
45
+ ClientError,
46
+ current_url,
47
+ get_json,
48
+ is_loopback_host,
49
+ maybe_warn_version_skew,
50
+ norm_url,
51
+ post_json,
52
+ probe_me,
53
+ read_config,
54
+ validate_server_url,
55
+ write_config,
56
+ )
57
+
58
+ from sediment_core import (
59
+ FactStore,
60
+ FactTable,
61
+ GatewayProvider,
62
+ ForgeProvider,
63
+ NonEmptyId,
64
+ QuarantineRecord,
65
+ normalize_org_id,
66
+ )
67
+ from sediment_core.postgres_engine import DatabaseOperationError
68
+ from sediment_derive import (
69
+ MirrorManager,
70
+ derive_recovery_result,
71
+ inference_fact_id,
72
+ )
73
+ from sediment_export import (
74
+ DPOPolicy,
75
+ DerivationPolicy,
76
+ DerivationScope,
77
+ build_derived_bundle_context,
78
+ open_derived_bundle,
79
+ SFTPolicy,
80
+ export_rlvr,
81
+ export_rlvr_from_bundle,
82
+ project_recovery,
83
+ load_derivation_policy,
84
+ recovery_to_export_rows,
85
+ write_jsonl,
86
+ write_derived_bundle,
87
+ )
88
+
89
+ # The one-line answer to "what is this?", coder-style, shown beside the
90
+ # version on the root --help title. With the title prefix it fills exactly
91
+ # the pinned 80 columns. Keep in step with the root pyproject description.
92
+ _TAGLINE = "Turn AI developer workflow traces into RL-ready training data."
93
+
94
+ _TABLES = [t.value for t in FactTable]
95
+
96
+ # Read-only reports: name -> (module, one-line help). Dispatched
97
+ # before argparse and before Settings exists — a report never needs API auth
98
+ # config (SEDIMENT_ORG_ID + secrets) to read a store, so argv is forwarded
99
+ # verbatim to the module's own parser (`--org` falls back to
100
+ # $SEDIMENT_ORG_ID, `--database-url` to $SEDIMENT_DATABASE_URL). Modules import lazily:
101
+ # `sediment --help` stays fast and settings-free.
102
+ _REPORTS = {
103
+ "model": (
104
+ "sediment_api.reports.model_report",
105
+ "per-model attribution/CI/acceptance; --compare significance",
106
+ ),
107
+ "label-confidence-inspection": (
108
+ "sediment_api.reports.label_confidence_inspection",
109
+ "sample resolved confidence for human validation",
110
+ ),
111
+ "dataset-diagnostics": (
112
+ "sediment_api.reports.dataset_diagnostics",
113
+ "export-side dataset health checks",
114
+ ),
115
+ "recovery-yield": (
116
+ "sediment_api.reports.recovery_yield_report",
117
+ "recovery-pair yield by skip reason",
118
+ ),
119
+ "precision": (
120
+ "sediment_api.reports.precision_report",
121
+ "attribution precision/recall against a ground-truth manifest",
122
+ ),
123
+ "abandonment": (
124
+ "sediment_api.reports.abandonment_report",
125
+ "sessions whose accepted edits never reached a commit",
126
+ ),
127
+ "attribution-share": (
128
+ "sediment_api.reports.attribution_share_report",
129
+ "notes-attribution share and decline alert",
130
+ ),
131
+ "merge-retention": (
132
+ "sediment_api.reports.merge_retention_report",
133
+ "attributed change retention through pull request merge",
134
+ ),
135
+ "lifecycle": (
136
+ "sediment_api.reports.lifecycle_report",
137
+ "accepted-work progression, retention, attrition, and rework evidence",
138
+ ),
139
+ }
140
+
141
+
142
+ def _fail(msg: str) -> int:
143
+ print(ui.error_line(msg), file=sys.stderr)
144
+ return 1
145
+
146
+
147
+ def _database_url(args: argparse.Namespace) -> str:
148
+ database_url = args.database_url or os.environ.get("SEDIMENT_DATABASE_URL")
149
+ if not database_url:
150
+ raise ValueError("set SEDIMENT_DATABASE_URL or pass --database-url")
151
+ return database_url
152
+
153
+
154
+ def cmd_db_upgrade(args: argparse.Namespace) -> int:
155
+ """Upgrade the PostgreSQL physical schema to the supported head."""
156
+ from sediment_core.postgres_migrations import HEAD_REVISION, upgrade_database
157
+
158
+ upgrade_database(_database_url(args))
159
+ print(f"database schema upgraded to {HEAD_REVISION}")
160
+ return 0
161
+
162
+
163
+ def cmd_db_provision(args: argparse.Namespace) -> int:
164
+ """Provision fixed deployment roles using secrets from the environment."""
165
+ from sediment_core.postgres_roles import provision_database
166
+
167
+ values = {}
168
+ for name in (
169
+ "BOOTSTRAP_DATABASE_URL",
170
+ "MIGRATOR_PASSWORD",
171
+ "RUNTIME_PASSWORD",
172
+ "OPERATOR_PASSWORD",
173
+ ):
174
+ value = os.environ.get(f"SEDIMENT_{name}")
175
+ if not value:
176
+ raise ValueError(f"set SEDIMENT_{name} for database provisioning")
177
+ values[name.lower()] = value
178
+ provision_database(**values)
179
+ print("database roles provisioned and schema upgraded")
180
+ return 0
181
+
182
+
183
+ def cmd_db_status(args: argparse.Namespace) -> int:
184
+ """Inspect the PostgreSQL physical schema without changing it."""
185
+ from sediment_core.postgres_migrations import RevisionState, inspect_revision
186
+
187
+ inspection = inspect_revision(_database_url(args))
188
+ if inspection.state is RevisionState.AT_HEAD:
189
+ print(f"database schema: at_head ({inspection.head_revision})")
190
+ else:
191
+ print(
192
+ f"database schema: {inspection.state.value} "
193
+ f"(supported head {inspection.head_revision})"
194
+ )
195
+ return 0
196
+
197
+
198
+ def _public_status(status: int) -> int:
199
+ """Map a dispatched command's semantic failure onto the CLI contract."""
200
+ return 0 if status == 0 else 1
201
+
202
+
203
+ def _run_report(argv: list[str]) -> int:
204
+ is_help = bool(argv) and argv[0] in ("-h", "--help")
205
+ if not argv or is_help:
206
+ # Usage on stdout only when asked for (--help); the bare-invocation
207
+ # error goes to stderr like argparse's own, keeping stdout pipeable.
208
+ stream = sys.stdout if is_help else sys.stderr
209
+ lines = [
210
+ "USAGE:",
211
+ " sediment report <name> [options]",
212
+ "",
213
+ " read-only reports; each takes --help, an empty result exits 0",
214
+ "",
215
+ "REPORTS:",
216
+ ]
217
+ lines += [
218
+ f" {name:<22} {help_text}" for name, (_, help_text) in _REPORTS.items()
219
+ ]
220
+ print(ui.style_help("\n".join(lines), stream=stream), file=stream)
221
+ return 0 if is_help else 2
222
+ name, *rest = argv
223
+ if name not in _REPORTS:
224
+ return _fail(f"unknown report {name!r} (choices: {', '.join(_REPORTS)})")
225
+ module = importlib.import_module(_REPORTS[name][0])
226
+ if rest in (["-h"], ["--help"]):
227
+ return _print_dispatched_help(
228
+ module.build_parser(), prog=f"sediment report {name}"
229
+ )
230
+ try:
231
+ return _public_status(module.main(rest))
232
+ except (DatabaseOperationError, OSError, ValueError) as exc:
233
+ return _fail(str(exc))
234
+
235
+
236
+ def _styled_revision(quarantine_revision: object) -> str:
237
+ """Render revision 0 quietly and later quarantine revisions prominently."""
238
+ revision = str(quarantine_revision)
239
+ if revision == "0":
240
+ return ui.style(revision, "dim")
241
+ return ui.style(revision, "sandstone", "bold")
242
+
243
+
244
+ def _print_facts(
245
+ sessions: int, tables: dict[str, dict[str, int]], quarantine_revision: object
246
+ ) -> None:
247
+ """The §6.2 counts table — shared by the local and remote ``facts`` paths
248
+ so both render identically. ``quarantine_revision`` is an int on both
249
+ paths; the parameter stays ``object`` so either caller formats."""
250
+ print(ui.style(f"{'table':<20} {'total':>7} {'visible':>8}", "dim"))
251
+ print(f"{'sessions':<20} {sessions:>7} {'-':>8}")
252
+ for table in FactTable:
253
+ if table.value not in tables:
254
+ print(f"{table.value:<20} {'unavailable':>7} {'unavailable':>8}")
255
+ continue
256
+ counts = tables[table.value]
257
+ head = f"{table.value:<20} {counts['total']:>7}"
258
+ visible = f"{counts['visible']:>8}"
259
+ if counts["total"] == 0:
260
+ print(ui.style(f"{head} {visible}", "dim"))
261
+ elif counts["visible"] < counts["total"]:
262
+ # Facts hidden from derivations — the number an operator scans for.
263
+ print(f"{head} {ui.style(visible, 'sandstone')}")
264
+ else:
265
+ print(f"{head} {visible}")
266
+ print(f"quarantine_revision: {_styled_revision(quarantine_revision)}")
267
+
268
+
269
+ def _facts_local(store: FactStore, org: str) -> None:
270
+ """Read today's counts straight from the store (local mode)."""
271
+ tables = {
272
+ table.value: {
273
+ "total": store.count_facts(org, table, include_quarantined=True),
274
+ "visible": store.count_facts(org, table),
275
+ }
276
+ for table in store.available_fact_tables()
277
+ }
278
+ _print_facts(store.count_sessions(org), tables, store.quarantine_revision(org))
279
+
280
+
281
+ def cmd_facts(args: argparse.Namespace) -> int:
282
+ """Fact counts per table, total and derivation-facing ("visible" = not
283
+ quarantined). Remote by default (GET /v1/facts); ``--database-url`` or a
284
+ present ``SEDIMENT_DATABASE_URL`` opens the store directly, so
285
+ ``docker compose exec api sediment facts`` is unchanged."""
286
+ if args.database_url or os.environ.get("SEDIMENT_DATABASE_URL"):
287
+ from sediment_api.database import one_shot_fact_store
288
+
289
+ database_url = args.database_url or os.environ["SEDIMENT_DATABASE_URL"]
290
+ org = os.environ.get("SEDIMENT_ORG_ID")
291
+ if not org:
292
+ raise ValueError("set SEDIMENT_ORG_ID for direct fact counts")
293
+ with one_shot_fact_store(database_url, operation="count facts") as store:
294
+ _facts_local(store, normalize_org_id(org))
295
+ return 0
296
+ maybe_warn_version_skew()
297
+ data = get_json("/v1/facts")
298
+ _print_facts(data["sessions"], data["tables"], data["quarantine_revision"])
299
+ return 0
300
+
301
+
302
+ # The demo session's identity. Everything the verb writes carries these, so
303
+ # one glance at a fact row says "synthetic" and one quarantine call by
304
+ # session id removes the lot.
305
+ DEMO_SESSION_ID = "sediment-demo"
306
+ DEMO_CALL_ID = "sediment-demo-call-1"
307
+ # Fixed, not "now": ``occurred_at`` is part of both decision dedup indexes,
308
+ # so a wall-clock event time would store a second decision on every run
309
+ # while the completion (keyed on call_id) collapsed — the reader who runs
310
+ # the verb twice would watch one count move and the other stay put. A
311
+ # constant makes both facts collapse, and a synthetic fact claiming a
312
+ # synthetic time is the honest version anyway. 2026-01-01T00:00:00Z.
313
+ DEMO_EVENT_NANOS = 1767225600000000000
314
+
315
+
316
+ def _demo_completion() -> dict[str, Any]:
317
+ """The gateway envelope a real LiteLLM callback POSTs, with synthetic
318
+ content. ``litellm_call_id`` matches the decision's ``tool_use_id``
319
+ below so the two facts join exactly as a real session's would."""
320
+ return {
321
+ "provider": "litellm",
322
+ "session_id": DEMO_SESSION_ID,
323
+ "user_id": DEMO_SESSION_ID,
324
+ "payload": {
325
+ "model": "sediment-demo-model",
326
+ "litellm_call_id": DEMO_CALL_ID,
327
+ "messages": [{"role": "user", "content": "Add a docstring to greet()."}],
328
+ "response": {
329
+ "choices": [
330
+ {
331
+ "message": {
332
+ "role": "assistant",
333
+ "content": (
334
+ "def greet(name):\n"
335
+ ' """Return a greeting for name."""\n'
336
+ " return f'hello {name}'\n"
337
+ ),
338
+ }
339
+ }
340
+ ]
341
+ },
342
+ "usage": {"prompt_tokens": 12, "completion_tokens": 24},
343
+ "response_time_ms": 350,
344
+ },
345
+ }
346
+
347
+
348
+ def _demo_decision() -> dict[str, Any]:
349
+ """The OTLP/JSON logs batch a harness shim POSTs on an edit-tool call,
350
+ shaped like packages/capture/tests/fixtures/otlp/sediment/tool_decision.json."""
351
+ return {
352
+ "resourceLogs": [
353
+ {
354
+ "resource": {
355
+ "attributes": [
356
+ {
357
+ "key": "user.id",
358
+ "value": {"stringValue": DEMO_SESSION_ID},
359
+ }
360
+ ]
361
+ },
362
+ "scopeLogs": [
363
+ {
364
+ "logRecords": [
365
+ {
366
+ "body": {"stringValue": "sediment.tool_decision"},
367
+ "timeUnixNano": str(DEMO_EVENT_NANOS),
368
+ "attributes": [
369
+ {
370
+ "key": "agent",
371
+ "value": {"stringValue": "claude-code"},
372
+ },
373
+ {
374
+ "key": "session.id",
375
+ "value": {"stringValue": DEMO_SESSION_ID},
376
+ },
377
+ {
378
+ "key": "tool_use_id",
379
+ "value": {"stringValue": DEMO_CALL_ID},
380
+ },
381
+ {
382
+ "key": "decision",
383
+ "value": {"stringValue": "accept"},
384
+ },
385
+ {"key": "explicit", "value": {"boolValue": True}},
386
+ {
387
+ "key": "tool_name",
388
+ "value": {"stringValue": "Edit"},
389
+ },
390
+ {
391
+ "key": "file_path",
392
+ "value": {
393
+ "stringValue": "/sediment-demo/greet.py"
394
+ },
395
+ },
396
+ ],
397
+ }
398
+ ]
399
+ }
400
+ ],
401
+ }
402
+ ]
403
+ }
404
+
405
+
406
+ def _is_loopback(url: str) -> bool:
407
+ # urlsplit() strips brackets from an IPv6 authority before the helper.
408
+ host = urlsplit(url).hostname
409
+ return host is not None and is_loopback_host(host)
410
+
411
+
412
+ def cmd_demo(args: argparse.Namespace) -> int:
413
+ """Plant one synthetic session through the real ingest doors, then print
414
+ the fact counts. This exists so the quickstart ends on a captured fact
415
+ instead of a table of zeros: it proves the ingest path works end to end,
416
+ and proves nothing about the reader's own agent, which is still unwired.
417
+
418
+ Refuses a non-loopback server without ``--force``. Seeding a shared
419
+ deployment with synthetic facts is the failure this guard exists to
420
+ prevent — the facts are real once stored, and removing them is a
421
+ quarantine call, not an undo."""
422
+ url = current_url()
423
+ if not _is_loopback(url) and not args.force:
424
+ print(
425
+ ui.error_line(
426
+ f"{url} is not a local server. `demo` writes synthetic facts, "
427
+ "so it refuses a shared deployment. Pass --force if you meant it."
428
+ ),
429
+ file=sys.stderr,
430
+ )
431
+ return 1
432
+
433
+ maybe_warn_version_skew()
434
+ print(f"posting demo session (synthetic) to {url}")
435
+ inference_call = post_json("/ingest/gateway", _demo_completion())
436
+ if inference_call.get("skipped"):
437
+ raise ClientError(
438
+ "the server skipped the demo inference call: "
439
+ f"{inference_call.get('reason')}"
440
+ )
441
+ # The OTLP door answers {} whether it stored the record or skipped it —
442
+ # per-record results are invisible to an exporter by design. So this
443
+ # says what was *posted*; the counts below are the only authority on
444
+ # what landed, and the check after them is what makes the difference
445
+ # actionable instead of a table the reader has to audit.
446
+ post_json("/v1/logs", _demo_decision())
447
+ print(f" posted 1 inference call and 1 decision as session {DEMO_SESSION_ID}")
448
+ print()
449
+
450
+ data = get_json("/v1/facts")
451
+ _print_facts(data["sessions"], data["tables"], data["quarantine_revision"])
452
+ print()
453
+
454
+ session_data = get_json(f"/v1/facts/session/{DEMO_SESSION_ID}")
455
+ missing = [
456
+ table
457
+ for table in ("inference_calls", "developer_decisions")
458
+ if session_data["tables"].get(table, {}).get("total", 0) == 0
459
+ ]
460
+ if missing:
461
+ print(
462
+ ui.error_line(
463
+ f"posted, but {' and '.join(missing)} stayed empty — the server "
464
+ "took the payload and stored nothing. Check the api logs."
465
+ ),
466
+ file=sys.stderr,
467
+ )
468
+ return 1
469
+
470
+ print(
471
+ ui.style(
472
+ "These are synthetic facts, not your agent's. They prove the "
473
+ "ingest path works end to end.",
474
+ "dim",
475
+ )
476
+ )
477
+ return 0
478
+
479
+
480
+ def _load_server_env(env_file: Path) -> dict[str, str]:
481
+ """KEY=VALUE lines from a previous run's server.env; {} when absent."""
482
+ try:
483
+ lines = env_file.read_text(encoding="utf-8").splitlines()
484
+ except OSError:
485
+ return {}
486
+ pairs = {}
487
+ for line in lines:
488
+ key, sep, value = line.partition("=")
489
+ if sep and key and not key.startswith("#"):
490
+ pairs[key] = value
491
+ return pairs
492
+
493
+
494
+ def cmd_server(args: argparse.Namespace) -> int:
495
+ """Run the API and its optional managed PostgreSQL until interrupted."""
496
+ try:
497
+ with ExitStack() as stack:
498
+ return _run_server(args, stack)
499
+ except KeyboardInterrupt:
500
+ return 0
501
+
502
+
503
+ def _run_server(args: argparse.Namespace, stack: ExitStack) -> int:
504
+ """Provision an evaluation database, then serve with runtime database authority.
505
+
506
+ An explicit bootstrap URL selects an external database.
507
+ Private server.env stores separate capture/operator API tokens and role
508
+ passwords. An exported secret takes precedence over its saved value.
509
+ """
510
+ import secrets as secrets_module
511
+
512
+ from sqlalchemy.engine import make_url
513
+
514
+ from sediment_core.postgres_roles import RUNTIME_ROLE, provision_database
515
+
516
+ from .local_postgres import managed_postgres, server_root
517
+
518
+ bootstrap = os.environ.get("SEDIMENT_BOOTSTRAP_DATABASE_URL")
519
+ try:
520
+ target = make_url(bootstrap) if bootstrap else None
521
+ if target is not None and (
522
+ target.get_backend_name() != "postgresql"
523
+ or not target.host
524
+ or not target.database
525
+ ):
526
+ raise ValueError
527
+ except Exception:
528
+ raise ValueError(
529
+ "bootstrap URL must name an explicit PostgreSQL host and database"
530
+ ) from None
531
+
532
+ root = Path(args.root).expanduser().absolute()
533
+ stack.enter_context(server_root(root))
534
+ (root / "mirror").mkdir(exist_ok=True, mode=0o700)
535
+ env_file = root / "server.env"
536
+ if env_file.is_symlink() or (env_file.exists() and env_file.stat().st_nlink != 1):
537
+ raise ValueError("server.env must be a regular private file")
538
+ stored = _load_server_env(env_file)
539
+ additions = {}
540
+ api_keys = ["SEDIMENT_OPERATOR_TOKEN", "SEDIMENT_GITHUB_WEBHOOK_SECRET"]
541
+ if not (
542
+ os.environ.get("SEDIMENT_INGEST_TOKENS") or stored.get("SEDIMENT_INGEST_TOKENS")
543
+ ):
544
+ api_keys.append("SEDIMENT_API_BEARER_TOKEN")
545
+ database_keys = [
546
+ "SEDIMENT_MIGRATOR_PASSWORD",
547
+ "SEDIMENT_RUNTIME_PASSWORD",
548
+ "SEDIMENT_OPERATOR_PASSWORD",
549
+ ]
550
+ if not bootstrap:
551
+ database_keys.append("SEDIMENT_BOOTSTRAP_PASSWORD")
552
+ for key in (*api_keys, *database_keys):
553
+ if not (os.environ.get(key) or stored.get(key)):
554
+ stored[key] = secrets_module.token_hex(32)
555
+ additions[key] = stored[key]
556
+ existing_content = env_file.read_bytes() if env_file.exists() else b""
557
+ separator = (
558
+ "\n"
559
+ if existing_content and not existing_content.endswith(b"\n") and additions
560
+ else ""
561
+ )
562
+ content = separator + "".join(
563
+ f"{key}={value}\n" for key, value in additions.items()
564
+ )
565
+ fd = os.open(
566
+ env_file, os.O_WRONLY | os.O_CREAT | os.O_APPEND | os.O_NOFOLLOW, 0o600
567
+ )
568
+ with os.fdopen(fd, "w", encoding="utf-8") as stream:
569
+ if os.fstat(stream.fileno()).st_nlink != 1:
570
+ raise ValueError("server.env must be a regular private file")
571
+ os.fchmod(stream.fileno(), 0o600)
572
+ stream.write(content)
573
+
574
+ effective = {
575
+ key: os.environ.get(key) or stored.get(key)
576
+ for key in (
577
+ *api_keys,
578
+ *database_keys,
579
+ "SEDIMENT_API_BEARER_TOKEN",
580
+ "SEDIMENT_INGEST_TOKENS",
581
+ )
582
+ }
583
+ if not bootstrap:
584
+ bootstrap = stack.enter_context(
585
+ managed_postgres(root, effective["SEDIMENT_BOOTSTRAP_PASSWORD"])
586
+ )
587
+ target = make_url(bootstrap)
588
+ provision_database(
589
+ bootstrap,
590
+ migrator_password=effective["SEDIMENT_MIGRATOR_PASSWORD"],
591
+ runtime_password=effective["SEDIMENT_RUNTIME_PASSWORD"],
592
+ operator_password=effective["SEDIMENT_OPERATOR_PASSWORD"],
593
+ )
594
+ os.environ["SEDIMENT_DATABASE_URL"] = target.set(
595
+ drivername="postgresql+psycopg",
596
+ username=RUNTIME_ROLE,
597
+ password=effective["SEDIMENT_RUNTIME_PASSWORD"],
598
+ ).render_as_string(hide_password=False)
599
+ # Only runtime connection authority reaches the API and its worker children.
600
+ for key in (
601
+ *database_keys,
602
+ "SEDIMENT_BOOTSTRAP_PASSWORD",
603
+ "SEDIMENT_BOOTSTRAP_DATABASE_URL",
604
+ "SEDIMENT_MIGRATOR_DATABASE_URL",
605
+ "SEDIMENT_OPERATOR_DATABASE_URL",
606
+ ):
607
+ os.environ.pop(key, None)
608
+ os.environ.setdefault("SEDIMENT_ORG_ID", stored.get("SEDIMENT_ORG_ID", "default"))
609
+ os.environ.setdefault("SEDIMENT_MIRROR_PATH", str(root / "mirror"))
610
+ for key in (*api_keys, "SEDIMENT_API_BEARER_TOKEN", "SEDIMENT_INGEST_TOKENS"):
611
+ if effective.get(key):
612
+ os.environ[key] = effective[key]
613
+
614
+ url = f"http://{args.host}:{args.port}"
615
+ ui.banner("sediment", f"v{__version__}")
616
+ if additions:
617
+ print(f"Generated server credentials in {env_file}")
618
+ else:
619
+ print(f"Using credentials from {env_file} and explicit environment overrides")
620
+ print(f"Serving on {ui.style(url, 'sandstone', 'bold')}")
621
+ print(f"Next: {ui.style(f'sediment login {url}', 'sandstone')}")
622
+
623
+ del bootstrap, target, effective, stored, additions
624
+ import uvicorn
625
+
626
+ uvicorn.run("sediment_api.main:app", host=args.host, port=args.port)
627
+ return 0
628
+
629
+
630
+ def _prompt_token() -> str:
631
+ import getpass
632
+
633
+ return getpass.getpass("Bearer token: ")
634
+
635
+
636
+ def _eval_server_token(url: str, *, capture: bool = False) -> str | None:
637
+ """The token ``sediment server`` generated for *this machine's own* eval
638
+ server, so ``sediment login http://127.0.0.1:8000`` needs nothing pasted
639
+ or piped — the quickstart's whole login step is that one line.
640
+
641
+ Loopback only. The token is a local secret; offering it to whatever
642
+ host the operator typed would hand it to that host. A non-default
643
+ ``server --root`` is not searched either.
644
+ # ponytail: default root only — `--with-token` covers every other
645
+ # posture, and a --root flag on login would be config for a value that
646
+ # does not vary in the quickstart.
647
+ """
648
+ host = urlsplit(url).hostname
649
+ if host is None or not is_loopback_host(host):
650
+ return None
651
+ env = _load_server_env(Path.home() / ".sediment" / "server" / "server.env")
652
+ return (
653
+ env.get("SEDIMENT_API_BEARER_TOKEN" if capture else "SEDIMENT_OPERATOR_TOKEN")
654
+ or None
655
+ )
656
+
657
+
658
+ def _store_login(
659
+ url: str,
660
+ token: str,
661
+ me: dict,
662
+ *,
663
+ capture: bool = False,
664
+ capture_identity: tuple[str, dict] | None = None,
665
+ ) -> int:
666
+ """Persist validated credentials and say which org they resolved to."""
667
+ cfg = read_config()
668
+ expected = "ingest" if capture else "operator"
669
+ if me.get("authority") != expected:
670
+ return _fail(
671
+ f"{expected} authority required for this login; use --capture for an ingest credential"
672
+ )
673
+ cfg["current"] = url
674
+ servers = dict(cfg.get("servers") or {})
675
+ existing = servers.get(url)
676
+ entry = dict(existing) if isinstance(existing, dict) else {}
677
+ prefix = "capture_" if capture else ""
678
+ entry.update(
679
+ {
680
+ f"{prefix}token": token,
681
+ f"{prefix}authority": expected,
682
+ f"{prefix}client_id": me["client_id"],
683
+ "org_id": me["org_id"],
684
+ }
685
+ )
686
+ if capture_identity is not None:
687
+ capture_token, capture_me = capture_identity
688
+ if (
689
+ capture_me.get("authority") != "ingest"
690
+ or capture_me.get("org_id") != me["org_id"]
691
+ ):
692
+ return _fail(
693
+ "ingest authority for the same org is required for capture enrollment"
694
+ )
695
+ entry.update(
696
+ capture_token=capture_token,
697
+ capture_authority="ingest",
698
+ capture_client_id=capture_me["client_id"],
699
+ )
700
+ servers[url] = entry
701
+ cfg["servers"] = servers
702
+ write_config(cfg)
703
+ action = "capture credential enrolled for" if capture else "logged in to"
704
+ print(f"{ui.glyph('✓', 'phosphor')}{action} {url} (org {me['org_id']})")
705
+ return 0
706
+
707
+
708
+ def cmd_login(args: argparse.Namespace) -> int:
709
+ """Store credentials for a server after live-validating the token via
710
+ GET /v1/me. A 401 re-prompts; a connection failure stops. Refuses an
711
+ environment override for the selected authority because it would silently
712
+ win over the stored credential.
713
+
714
+ Nothing needs answering in the quickstart: a loopback URL reuses
715
+ the eval server's own generated token. Every other server — remote, or
716
+ local under a custom ``--root`` — takes ``--with-token``: one line on
717
+ stdin, no prompt and no re-prompt. getpass is not that path, since it
718
+ reads /dev/tty when one exists, so piping into the prompt hangs."""
719
+ capture = bool(getattr(args, "capture", False))
720
+ override = "SEDIMENT_INGEST_TOKEN" if capture else "SEDIMENT_SESSION_TOKEN"
721
+ if os.environ.get(override):
722
+ return _fail(
723
+ f"{override} is set; unset it first — it would "
724
+ "silently override the stored token"
725
+ )
726
+ url = validate_server_url(args.url)
727
+ generated = None if args.with_token else _eval_server_token(url, capture=capture)
728
+ if generated:
729
+ try:
730
+ me = probe_me(url, generated)
731
+ except ClientError as exc:
732
+ # A stale server.env (the server was restarted under a different
733
+ # token) is the one case worth falling through on: nobody typed
734
+ # this token, so asking for one is the fix. Anything else — an
735
+ # unreachable server above all — is the operator's real error and
736
+ # must not be buried under a prompt.
737
+ if str(exc) != "that's not a valid token":
738
+ return _fail(str(exc))
739
+ else:
740
+ print(ui.style("using the token from ~/.sediment/server/server.env", "dim"))
741
+ capture_identity = None
742
+ if not capture and (capture_token := _eval_server_token(url, capture=True)):
743
+ try:
744
+ capture_identity = (capture_token, probe_me(url, capture_token))
745
+ except ClientError as exc:
746
+ return _fail(f"capture enrollment failed: {exc}")
747
+ return _store_login(
748
+ url, generated, me, capture=capture, capture_identity=capture_identity
749
+ )
750
+ while True:
751
+ if args.with_token:
752
+ token = sys.stdin.readline().strip()
753
+ if not token:
754
+ return _fail("no token on stdin")
755
+ else:
756
+ try:
757
+ token = _prompt_token()
758
+ except EOFError:
759
+ # Piped/closed stdin (a 401 re-prompt with no second line to
760
+ # read): clean error, never a traceback — the module contract.
761
+ return _fail("no token provided")
762
+ try:
763
+ me = probe_me(url, token)
764
+ except ClientError as exc:
765
+ msg = str(exc)
766
+ if msg == "that's not a valid token" and not args.with_token:
767
+ print(ui.error_line(msg), file=sys.stderr)
768
+ continue
769
+ return _fail(msg)
770
+ return _store_login(url, token, me, capture=capture)
771
+
772
+
773
+ def cmd_logout(args: argparse.Namespace) -> int:
774
+ """Remove one server's stored credentials (``--server``, default the
775
+ current entry), then say what was removed."""
776
+ cfg = read_config()
777
+ servers = dict(cfg.get("servers") or {})
778
+ if args.server:
779
+ url = norm_url(args.server)
780
+ if url not in servers:
781
+ return _fail(f"no stored credentials for {url}")
782
+ else:
783
+ current = cfg.get("current")
784
+ if not isinstance(current, str) or current not in servers:
785
+ return _fail("not logged in")
786
+ url = current
787
+ removed = servers.pop(url)
788
+ cfg["servers"] = servers
789
+ if cfg.get("current") == url:
790
+ cfg["current"] = None
791
+ write_config(cfg)
792
+ print(
793
+ f"{ui.glyph('✓', 'phosphor')}logged out of {url} (org {removed.get('org_id')})"
794
+ )
795
+ return 0
796
+
797
+
798
+ def cmd_commit(args: argparse.Namespace) -> int:
799
+ """GET /query/commit/{sha} and pretty-print the per-repo attributions,
800
+ decisions, and CI outcomes."""
801
+ selectors = {
802
+ name: getattr(args, name, None)
803
+ for name in (
804
+ "repo",
805
+ "repository_provider",
806
+ "repository_host",
807
+ "repository_id",
808
+ "as_of",
809
+ )
810
+ }
811
+ identity = [
812
+ selectors[name]
813
+ for name in ("repository_provider", "repository_host", "repository_id")
814
+ ]
815
+ if any(value is not None for value in identity) and not all(
816
+ value is not None for value in identity
817
+ ):
818
+ raise ClientError("repository identity requires all three components")
819
+ if selectors["as_of"] is not None:
820
+ selectors["as_of"] = _parse_scope_time(selectors["as_of"], "as_of").isoformat()
821
+ query = urlencode(
822
+ {name: value for name, value in selectors.items() if value is not None}
823
+ )
824
+ path = f"/query/commit/{args.sha}"
825
+ maybe_warn_version_skew()
826
+ data = get_json(f"{path}?{query}" if query else path)
827
+ sha = data.get("commit_sha", args.sha)
828
+ for reason, count in sorted(data.get("repository_skipped", {}).items()):
829
+ print(f"{reason}: {count}")
830
+ for outcome in data.get("unresolved_ci_outcomes", []):
831
+ print(
832
+ f"unresolved repository: ci {outcome['result']} {outcome['workflow_name']}"
833
+ )
834
+ if not data.get("repos"):
835
+ print(f"{sha}: no attributions")
836
+ return 0
837
+ print(ui.style(sha, "bleached", "bold"))
838
+ for repo in data.get("repos", []):
839
+ identity = repo.get("repository_identity")
840
+ label = repo["repo"]
841
+ if identity is not None:
842
+ label += f" ({identity['provider']} {identity['host']} repository {identity['repository_id']})"
843
+ print(f" {ui.style(label, 'sandstone')}")
844
+ observations = repo.get("observed_sessions", [])
845
+ for session in observations:
846
+ print(f" observed Session: {session['session_id']}")
847
+ if not observations:
848
+ print(" Session observations: unavailable")
849
+ if repo.get("session_commit_unobserved"):
850
+ print(f" session_commit_unobserved: {repo['session_commit_unobserved']}")
851
+ for inference_call in repo.get("inference_calls", []):
852
+ print(
853
+ f" {inference_call['inference_call_id']} "
854
+ + ui.style(
855
+ f"{inference_call['gateway_provider']}/"
856
+ f"{inference_call['model_provider'] or '-'}/"
857
+ f"{inference_call['model']}",
858
+ "dim",
859
+ )
860
+ + f" session={inference_call['session_id']} "
861
+ f"attribution={inference_call['attribution_source']} (inferred call/file)"
862
+ )
863
+ print(f" decisions: {repo['decisions']}")
864
+ for outcome in repo.get("ci_outcomes", []):
865
+ color = {"passed": "phosphor", "failed": "iron-oxide"}.get(
866
+ outcome["result"], "dim"
867
+ )
868
+ print(
869
+ f" ci {ui.style(outcome['result'], color)} {outcome['workflow_name']}"
870
+ )
871
+ return 0
872
+
873
+
874
+ def cmd_quarantine_or_release(
875
+ store: FactStore, org: str, args: argparse.Namespace
876
+ ) -> int:
877
+ if args.command == "quarantine":
878
+ store.quarantine_fact(org, args.table, args.fact_id, reason=args.reason)
879
+ print(
880
+ f"{ui.glyph('✓', 'phosphor')}quarantined {args.table}/{args.fact_id} "
881
+ "(reversible: sediment release)"
882
+ )
883
+ else:
884
+ store.release_fact(org, args.table, args.fact_id, reason=args.reason)
885
+ print(f"{ui.glyph('✓', 'phosphor')}released {args.table}/{args.fact_id}")
886
+ return 0
887
+
888
+
889
+ def cmd_quarantine_log(store: FactStore, org: str, args: argparse.Namespace) -> int:
890
+ # ponytail: full-history read, tail applied in the CLI — log rows are
891
+ # small; a store-side LIMIT variant if an org's log ever gets huge.
892
+ log = store.read_quarantine_log(org)
893
+ shown = log if args.all else log[-args.tail :]
894
+ for rec in shown:
895
+ color = "sandstone" if rec.action == "quarantine" else "phosphor"
896
+ print(
897
+ ui.style(rec.recorded_at.isoformat(), "dim")
898
+ + f" {ui.style(f'{rec.action:<10}', color)} "
899
+ f"{rec.fact_table}/{rec.fact_id} {rec.reason}"
900
+ )
901
+ if len(shown) < len(log):
902
+ print(
903
+ ui.style(
904
+ f"... showing last {len(shown)} of {len(log)} rows (--all for all)",
905
+ "dim",
906
+ )
907
+ )
908
+ revision = _styled_revision(store.quarantine_revision(org))
909
+ print(f"{len(log)} rows; quarantine_revision: {revision}")
910
+ return 0
911
+
912
+
913
+ def _resolve_quarantine_inference_filters(
914
+ args: argparse.Namespace,
915
+ ) -> tuple[str | None, tuple[datetime, datetime] | None]:
916
+ provider = None
917
+ if args.provider is not None:
918
+ try:
919
+ provider = GatewayProvider(args.provider).value
920
+ except ValueError:
921
+ choices = ", ".join(item.value for item in GatewayProvider)
922
+ raise ValueError(f"--provider must be one of: {choices}") from None
923
+ between = None
924
+ if args.between:
925
+ try:
926
+ lo, hi = (datetime.fromisoformat(t) for t in args.between)
927
+ except ValueError as exc:
928
+ raise ValueError(f"--between wants two ISO-8601 datetimes: {exc}") from None
929
+ if lo.tzinfo is None or hi.tzinfo is None:
930
+ raise ValueError("--between bounds must be timezone-aware")
931
+ if lo > hi:
932
+ raise ValueError(f"--between bounds are reversed ({lo} > {hi})")
933
+ between = (lo, hi)
934
+ return provider, between
935
+
936
+
937
+ def cmd_quarantine_inference_calls(
938
+ store: FactStore, org: str, args: argparse.Namespace
939
+ ) -> int:
940
+ provider, between = args._quarantine_inference_filters
941
+ n = store.quarantine_inference_calls_where(
942
+ org,
943
+ captured_between=between,
944
+ session_id=args.session_id,
945
+ provider=provider,
946
+ reason=args.reason,
947
+ dry_run=not args.apply,
948
+ )
949
+ if args.apply:
950
+ print(f"{ui.glyph('✓', 'phosphor')}quarantined {n} model-call facts")
951
+ else:
952
+ print(
953
+ f"{n} model-call facts would be quarantined "
954
+ + ui.style("(dry run; add --apply)", "sandstone")
955
+ )
956
+ return 0
957
+
958
+
959
+ def _mirrors(mirror_path: str | None = None) -> MirrorManager | None:
960
+ """The mirror store every export reads, or None once a missing
961
+ SEDIMENT_MIRROR_PATH has been reported."""
962
+ if not mirror_path:
963
+ _fail("SEDIMENT_MIRROR_PATH is not set; the export needs mirrors")
964
+ return None
965
+ return MirrorManager(mirror_path)
966
+
967
+
968
+ def _print_written(written: dict) -> None:
969
+ """The export payoff lines, shared by every export verb so ✓/dim reads
970
+ the same everywhere."""
971
+ if written:
972
+ for path, count in written.items():
973
+ print(f"{ui.glyph('✓', 'phosphor')}wrote {path} ({count} rows)")
974
+ else:
975
+ print(
976
+ ui.style(
977
+ "nothing written (empty projections leave existing files untouched)",
978
+ "dim",
979
+ )
980
+ )
981
+
982
+
983
+ def _default_derivation_policy() -> DerivationPolicy:
984
+ return DerivationPolicy()
985
+
986
+
987
+ def _parse_scope_time(value: str | None, name: str) -> datetime | None:
988
+ if value is None:
989
+ return None
990
+ try:
991
+ parsed = datetime.fromisoformat(value)
992
+ except ValueError as exc:
993
+ raise ValueError(f"{name} must be an RFC 3339 timestamp") from exc
994
+ if parsed.tzinfo is None or parsed.utcoffset() is None:
995
+ raise ValueError(f"{name} must be timezone-aware")
996
+ return parsed
997
+
998
+
999
+ def _prepare_store_command(args: argparse.Namespace) -> None:
1000
+ """Validate store-command semantics before settings or connectivity."""
1001
+ if args.command in {"quarantine", "release", "quarantine-inference-calls"}:
1002
+ # Use the fact model's canonical audit-reason validator at the CLI
1003
+ # trust boundary without constructing or persisting a placeholder fact.
1004
+ QuarantineRecord._reason_required(args.reason)
1005
+ if args.command in {"quarantine", "release"}:
1006
+ TypeAdapter(NonEmptyId).validate_python(args.fact_id)
1007
+ if args.command == "quarantine-log" and not args.all and args.tail < 1:
1008
+ raise ValueError("--tail must be >= 1")
1009
+ if args.command == "quarantine-inference-calls":
1010
+ if args.session_id is not None:
1011
+ TypeAdapter(NonEmptyId).validate_python(args.session_id)
1012
+ filters = _resolve_quarantine_inference_filters(args)
1013
+ args._quarantine_inference_filters = filters
1014
+ provider, between = filters
1015
+ if not (between or args.session_id or provider) and not args.all:
1016
+ raise ValueError(
1017
+ "no filters given — this would quarantine the org's every "
1018
+ "model-call fact; pass --all if that is what you mean"
1019
+ )
1020
+ if args.command == "derive":
1021
+ if args.sample < 0:
1022
+ raise ValueError("--sample must be zero or greater")
1023
+ policy = (
1024
+ load_derivation_policy(Path(args.policy))
1025
+ if args.policy
1026
+ else _default_derivation_policy()
1027
+ )
1028
+ scope = DerivationScope(
1029
+ since=_parse_scope_time(args.since, "--since"),
1030
+ until=_parse_scope_time(args.until, "--until"),
1031
+ users=tuple(args.users) if args.users is not None else None,
1032
+ )
1033
+ args._derivation = (policy, scope)
1034
+
1035
+
1036
+ def cmd_derive(store: FactStore, org: str, args: argparse.Namespace) -> int:
1037
+ mirrors = _mirrors(getattr(args, "_mirror_path", None))
1038
+ if mirrors is None:
1039
+ return 1
1040
+ policy, scope = args._derivation
1041
+ destination = Path(args.out)
1042
+ with build_derived_bundle_context(
1043
+ store, mirrors, org, policy=policy, scope=scope
1044
+ ) as bundle:
1045
+ try:
1046
+ write_derived_bundle(bundle, destination)
1047
+ except OSError as exc:
1048
+ return _fail(str(exc))
1049
+ print(f"{ui.glyph('✓', 'phosphor')}wrote {destination}")
1050
+ print(f"attributed completions: {len(bundle.attributed_completions)}")
1051
+ print(f"rollouts: {len(bundle.rollouts)}")
1052
+ print(f"referenced inference calls: {len(bundle.inference_calls)}")
1053
+ print(f"skipped: {dict(bundle.skipped)}")
1054
+ print(f"excluded: {dict(bundle.excluded)}")
1055
+ print(f"fragmented: {dict(bundle.fragmented)}")
1056
+ print(f"policy digest: {bundle.policy.digest}")
1057
+ print(f"as of: {bundle.as_of.isoformat() if bundle.as_of else 'none'}")
1058
+ for name in (
1059
+ "manifest.json",
1060
+ "attributed_completions.jsonl",
1061
+ "rollouts.jsonl",
1062
+ "inference_calls.jsonl",
1063
+ "inference_call_identities.jsonl",
1064
+ "repository_identities.jsonl",
1065
+ "repository_renames.jsonl",
1066
+ ):
1067
+ with (destination / name).open("rb") as handle:
1068
+ digest = hashlib.file_digest(handle, "sha256").hexdigest()
1069
+ print(f"sha256: {name} {digest}")
1070
+ if args.sample:
1071
+ for name in ("attributed_completions.jsonl", "rollouts.jsonl"):
1072
+ with (destination / name).open(encoding="utf-8") as handle:
1073
+ for line in islice(handle, args.sample):
1074
+ print(f"sample {name}: {line}", end="")
1075
+ return 0
1076
+
1077
+
1078
+ def _attributed_completion_pipeline(
1079
+ store: FactStore | None,
1080
+ mirrors: MirrorManager | None,
1081
+ org: str | None,
1082
+ out_dir: str,
1083
+ base_name: str,
1084
+ *,
1085
+ label: str,
1086
+ extra: dict | None = None,
1087
+ from_bundle: str | None = None,
1088
+ profile: str | None = None,
1089
+ ) -> int:
1090
+ """Project complete evidence groups from a validated, live bundle context."""
1091
+ from sediment_export.bounded_training import project_training_bundle
1092
+
1093
+ with ExitStack() as contexts:
1094
+ if from_bundle is not None:
1095
+ bundle = contexts.enter_context(open_derived_bundle(Path(from_bundle)))
1096
+ else:
1097
+ if store is None or mirrors is None or org is None:
1098
+ raise ValueError("direct export requires the fact store and mirrors")
1099
+ bundle = contexts.enter_context(
1100
+ build_derived_bundle_context(
1101
+ store, mirrors, org, policy=_default_derivation_policy()
1102
+ )
1103
+ )
1104
+ projection = contexts.enter_context(
1105
+ project_training_bundle(
1106
+ bundle, objective=base_name.replace("_", "-"), **(extra or {})
1107
+ )
1108
+ )
1109
+ split_enabled = bundle.policy.eval_fraction > 0
1110
+ if profile is not None:
1111
+ from sediment_export.compatibility import write_compatible_export
1112
+
1113
+ # The consumer adapter reads the staged rows within the open context;
1114
+ # its own published view is the profile's declared envelope.
1115
+ summary = write_compatible_export(
1116
+ projection.rows,
1117
+ out_dir,
1118
+ profile,
1119
+ split_enabled=split_enabled,
1120
+ canonical_skipped=projection.skipped,
1121
+ )
1122
+ written = summary["written"]
1123
+ print(f"compatibility profile: {profile}")
1124
+ print(f"dataset diagnostics: {summary['diagnostics']}")
1125
+ else:
1126
+ result = write_jsonl(
1127
+ projection.rows,
1128
+ Path(out_dir) / f"{base_name}.jsonl",
1129
+ split_enabled=split_enabled,
1130
+ max_bytes=projection.remaining_bytes,
1131
+ )
1132
+ written = result.written
1133
+ count_str = ui.style(str(len(projection.rows)), "bleached", "bold")
1134
+ print(f"{label} projected: {count_str} skipped: {dict(projection.skipped)}")
1135
+ _print_written(written)
1136
+ return 0
1137
+
1138
+
1139
+ def cmd_export_rlvr(
1140
+ store: FactStore | None, org: str | None, args: argparse.Namespace
1141
+ ) -> int:
1142
+ profile_name = getattr(args, "profile", None)
1143
+ if profile_name is not None:
1144
+ from sediment_export.compatibility import get_profile, require_dependencies
1145
+
1146
+ profile = get_profile(profile_name, objective="rlvr")
1147
+ if profile.consumer != args.target:
1148
+ raise ValueError("profile and --target must name the same consumer")
1149
+ require_dependencies(profile)
1150
+ elif getattr(args, "consumer_config", None):
1151
+ raise ValueError("--consumer-config requires --profile")
1152
+ with ExitStack() as contexts:
1153
+ bundle = (
1154
+ contexts.enter_context(open_derived_bundle(Path(args.from_bundle)))
1155
+ if args.from_bundle
1156
+ else None
1157
+ )
1158
+ needs_mirror = bundle is None or args.target != "nemo-gym"
1159
+ mirrors = (
1160
+ _mirrors(getattr(args, "_mirror_path", None)) if needs_mirror else None
1161
+ )
1162
+ if needs_mirror and mirrors is None:
1163
+ return 1
1164
+ if profile_name is not None:
1165
+ from sediment_export.consumer_rlvr import (
1166
+ export_rlvr_profile,
1167
+ load_consumer_settings,
1168
+ )
1169
+
1170
+ if bundle is None:
1171
+ bundle = contexts.enter_context(
1172
+ build_derived_bundle_context(
1173
+ store, mirrors, org, policy=_default_derivation_policy()
1174
+ )
1175
+ )
1176
+ # The consumer adapter reads the validated file-backed bundle within
1177
+ # this context; its own hydrated view is the profile's envelope.
1178
+ summary = export_rlvr_profile(
1179
+ bundle,
1180
+ mirrors,
1181
+ args.out,
1182
+ profile_name,
1183
+ load_consumer_settings(getattr(args, "consumer_config", None)),
1184
+ )
1185
+ print(f"compatibility profile: {profile_name}")
1186
+ print(f"rows: {summary['rows']} skipped: {summary['skipped']}")
1187
+ print(f"canonical skipped: {summary['canonical_skipped']}")
1188
+ print(f"fragmented: {summary['fragmented']}")
1189
+ print(f"dataset diagnostics: {summary['diagnostics']}")
1190
+ _print_written(summary["written"])
1191
+ return 0
1192
+ if bundle is not None:
1193
+ summary = export_rlvr_from_bundle(
1194
+ bundle,
1195
+ mirrors,
1196
+ args.out,
1197
+ target=args.target,
1198
+ )
1199
+ else:
1200
+ summary = export_rlvr(store, mirrors, org, args.out, target=args.target)
1201
+ print(f"rollouts derived: {ui.style(str(summary['rollouts']), 'bleached', 'bold')}")
1202
+ print(f"task rows: {summary['task_rows']} skipped: {summary['task_skipped']}")
1203
+ print(
1204
+ f"rollout rows: {summary['rollout_rows']} "
1205
+ f"skipped: {summary['rollout_skipped']}"
1206
+ )
1207
+ print(f"fragmented: {summary['fragmented']}")
1208
+ _print_written(summary["written"])
1209
+ if summary["environment_manifest"]:
1210
+ print(
1211
+ f"{ui.glyph('✓', 'phosphor')}wrote {summary['environment_manifest']} "
1212
+ "(experimental taskset manifest)"
1213
+ )
1214
+ return 0
1215
+
1216
+
1217
+ def cmd_export_dpo(
1218
+ store: FactStore | None, org: str | None, args: argparse.Namespace
1219
+ ) -> int:
1220
+ profile_name = getattr(args, "profile", None)
1221
+ if profile_name is not None:
1222
+ from sediment_export.compatibility import get_profile, require_dependencies
1223
+
1224
+ require_dependencies(get_profile(profile_name, objective="dpo"))
1225
+ mirrors = (
1226
+ None if args.from_bundle else _mirrors(getattr(args, "_mirror_path", None))
1227
+ )
1228
+ if not args.from_bundle and mirrors is None:
1229
+ return 1
1230
+ return _attributed_completion_pipeline(
1231
+ store,
1232
+ mirrors,
1233
+ org,
1234
+ args.out,
1235
+ "dpo",
1236
+ label="pairs",
1237
+ extra={"policy": DPOPolicy(recipe_id=args.recipe)},
1238
+ from_bundle=args.from_bundle,
1239
+ profile=profile_name,
1240
+ )
1241
+
1242
+
1243
+ def cmd_export_sft(
1244
+ store: FactStore | None, org: str | None, args: argparse.Namespace
1245
+ ) -> int:
1246
+ profile_name = getattr(args, "profile", None)
1247
+ if profile_name is not None:
1248
+ from sediment_export.compatibility import get_profile, require_dependencies
1249
+
1250
+ require_dependencies(get_profile(profile_name, objective="sft"))
1251
+ mirrors = (
1252
+ None if args.from_bundle else _mirrors(getattr(args, "_mirror_path", None))
1253
+ )
1254
+ if not args.from_bundle and mirrors is None:
1255
+ return 1
1256
+ return _attributed_completion_pipeline(
1257
+ store,
1258
+ mirrors,
1259
+ org,
1260
+ args.out,
1261
+ "sft",
1262
+ label="samples",
1263
+ extra={"policy": SFTPolicy(recipe_id=args.recipe)},
1264
+ from_bundle=args.from_bundle,
1265
+ profile=profile_name,
1266
+ )
1267
+
1268
+
1269
+ def cmd_export_diff_sft(
1270
+ store: FactStore | None, org: str | None, args: argparse.Namespace
1271
+ ) -> int:
1272
+ mirrors = _mirrors(getattr(args, "_mirror_path", None))
1273
+ if mirrors is None:
1274
+ return 1
1275
+ return _attributed_completion_pipeline(
1276
+ store,
1277
+ mirrors,
1278
+ org,
1279
+ args.out,
1280
+ "diff_sft",
1281
+ label="samples",
1282
+ extra={"mirrors": mirrors, "policy": SFTPolicy(recipe_id=args.recipe)},
1283
+ from_bundle=args.from_bundle,
1284
+ )
1285
+
1286
+
1287
+ def cmd_export_recovery(store: FactStore, org: str, args: argparse.Namespace) -> int:
1288
+ mirrors = _mirrors(getattr(args, "_mirror_path", None))
1289
+ if mirrors is None:
1290
+ return 1
1291
+ fraction = DerivationPolicy().eval_fraction
1292
+ split_enabled = fraction > 0
1293
+ with store.read_snapshot() as snapshot:
1294
+ derivation = derive_recovery_result(snapshot, mirrors, org)
1295
+ inference_calls = {
1296
+ inference_fact_id(call): call for call in snapshot.read_inference_calls(org)
1297
+ }
1298
+ projection = project_recovery(derivation.pairs, inference_calls, fraction)
1299
+ rows = recovery_to_export_rows(projection.rows)
1300
+ result = write_jsonl(
1301
+ rows, Path(args.out) / "recovery.jsonl", split_enabled=split_enabled
1302
+ )
1303
+ pairs = ui.style(str(len(derivation.pairs)), "bleached", "bold")
1304
+ print(f"pairs derived: {pairs} skipped: {dict(derivation.skipped)}")
1305
+ print(f"recovery rows: {len(projection.rows)} skipped: {dict(projection.skipped)}")
1306
+ _print_written(result.written)
1307
+ return 0
1308
+
1309
+
1310
+ class _CoderFormatter(argparse.HelpFormatter):
1311
+ """Coder-style help, pinned to an 80-column wrap so golden --help files
1312
+ stay deterministic across terminals and CI: a ``USAGE:``
1313
+ block, UPPERCASE section headings, subcommand rows without the
1314
+ ``{a,b,c}`` metavar header, descriptions indented under usage. The
1315
+ layout is identical piped or interactive — color is a separate
1316
+ TTY-gated pass (``ui.style_help``) in ``_ArgumentParser.format_help``."""
1317
+
1318
+ _SECTIONS = {"positional arguments": "arguments"}
1319
+
1320
+ def __init__(self, *args, **kwargs):
1321
+ kwargs.setdefault("width", 80)
1322
+ super().__init__(*args, **kwargs)
1323
+
1324
+ def _format_usage(self, usage, actions, groups, prefix):
1325
+ bare = super()._format_usage(usage, actions, groups, "").strip("\n")
1326
+ indented = "\n".join(f" {line}" for line in bare.splitlines())
1327
+ return f"USAGE:\n{indented}\n\n"
1328
+
1329
+ def start_section(self, heading):
1330
+ if heading:
1331
+ heading = self._SECTIONS.get(heading, heading).upper()
1332
+ super().start_section(heading)
1333
+
1334
+ def _format_text(self, text):
1335
+ # Descriptions sit indented under the USAGE block, coder-style.
1336
+ if "%(prog)" in text:
1337
+ text = text % {"prog": self._prog}
1338
+ return self._fill_text(text, self._width - 2, " ") + "\n\n"
1339
+
1340
+ def _format_action(self, action):
1341
+ if isinstance(action, argparse._SubParsersAction):
1342
+ # Subcommand rows render directly at section indent — no
1343
+ # "{report,mirror-gc,...}" header row above them.
1344
+ return "".join(self._format_action(sub) for sub in action._get_subactions())
1345
+ return super()._format_action(action)
1346
+
1347
+
1348
+ def _propagate_descriptions(parser: argparse.ArgumentParser) -> None:
1349
+ """Coder-style per-command help: a subcommand without an explicit
1350
+ description reuses its one-line listing help under its USAGE block."""
1351
+ for action in parser._actions:
1352
+ if not isinstance(action, argparse._SubParsersAction):
1353
+ continue
1354
+ help_by_name = {p.dest: p.help for p in action._choices_actions}
1355
+ for name, child in action.choices.items():
1356
+ if child.description is None:
1357
+ child.description = help_by_name.get(name)
1358
+ _propagate_descriptions(child)
1359
+
1360
+
1361
+ def _subparsers_action(
1362
+ parser: argparse.ArgumentParser,
1363
+ ) -> argparse._SubParsersAction | None:
1364
+ """The subparsers action on *parser*, or None for a leaf command."""
1365
+ for action in parser._actions:
1366
+ if isinstance(action, argparse._SubParsersAction):
1367
+ return action
1368
+ return None
1369
+
1370
+
1371
+ def _subcommand_display_name(action: argparse._SubParsersAction) -> str:
1372
+ """The name argparse prints for a missing/invalid subcommand — the
1373
+ subparsers ``metavar`` (``<command>``/``<format>``), falling back to
1374
+ ``dest`` the way ``argparse._get_action_name`` does."""
1375
+ if action.metavar not in (None, argparse.SUPPRESS):
1376
+ return action.metavar
1377
+ return action.dest
1378
+
1379
+
1380
+ class _ArgumentParser(argparse.ArgumentParser):
1381
+ """ArgumentParser whose subparsers inherit the coder-style formatter
1382
+ (``add_subparsers`` defaults ``parser_class`` to ``type(self)``) and
1383
+ whose help gets the TTY-gated color pass — plain text everywhere else,
1384
+ so the goldens capture exactly what a pipe sees."""
1385
+
1386
+ def __init__(self, *args, **kwargs):
1387
+ kwargs.setdefault("formatter_class", _CoderFormatter)
1388
+ super().__init__(*args, **kwargs)
1389
+
1390
+ def error(self, message):
1391
+ # A missing or unknown subcommand is the one case where the full
1392
+ # listing IS the answer to "what do I type next?" — print it to
1393
+ # stderr before the one-line diagnostic (stdout stays empty, exit 2).
1394
+ # A bad flag on a leaf command keeps argparse's terse usage line.
1395
+ # This distinguishes the two by matching argparse's own message
1396
+ # text, which is not a stable contract; test_cli_help.py pins the
1397
+ # split on both sides.
1398
+ if (
1399
+ self.prog == "sediment export rlvr"
1400
+ and message.startswith("the following arguments are required:")
1401
+ and "--target" in message
1402
+ ):
1403
+ self.print_usage(sys.stderr)
1404
+ self.exit(
1405
+ 2,
1406
+ f"{self.prog}: error: {message}; --target choices: sediment, "
1407
+ "swe-bench, or nemo-gym; see docs/exports/rlvr-export.md\n",
1408
+ )
1409
+
1410
+ sub = _subparsers_action(self)
1411
+ if sub is not None:
1412
+ name = _subcommand_display_name(sub)
1413
+ if message == f"the following arguments are required: {name}":
1414
+ self._print_message(self.format_help(sys.stderr), sys.stderr)
1415
+ self.exit(2, f"{self.prog}: error: {message}\n")
1416
+ elif message.startswith(f"argument {name}: invalid choice: "):
1417
+ # Drop the ``(choose from 'a', 'b', …)`` tail — the listing
1418
+ # now sits directly above — and name the thing by what this
1419
+ # level calls it: a command up top, a format under `export`.
1420
+ value = message.removeprefix(
1421
+ f"argument {name}: invalid choice: "
1422
+ ).rsplit(" (choose from ", 1)[0]
1423
+ self._print_message(self.format_help(sys.stderr), sys.stderr)
1424
+ noun = name.strip("<>")
1425
+ self.exit(2, f"{self.prog}: error: invalid {noun} {value}\n")
1426
+ self.print_usage(sys.stderr)
1427
+ self.exit(2, f"{self.prog}: error: {message}\n")
1428
+
1429
+ def format_help(self, stream=None):
1430
+ # Coder's section order: COMMANDS/FORMATS listings above OPTIONS.
1431
+ # Stable sort, idempotent — only the default options group moves.
1432
+ # *stream* is where the text is headed: the color pass is TTY-gated
1433
+ # on it, so help written to a redirected stderr stays plain. argparse
1434
+ # calls this with no argument, which gates on stdout as before.
1435
+ self._action_groups.sort(key=lambda g: g.title == "options")
1436
+ text = super().format_help()
1437
+ if self.prog == "sediment":
1438
+ title = ui.style(
1439
+ f"{self.prog} v{__version__}", "bleached", "bold", stream=stream
1440
+ )
1441
+ tagline = ui.style(_TAGLINE, "dim", stream=stream)
1442
+ text = f"{title} — {tagline}\n\n{text}"
1443
+ return ui.style_help(text, stream=stream)
1444
+
1445
+
1446
+ def _print_dispatched_help(parser: argparse.ArgumentParser, *, prog: str) -> int:
1447
+ """Render a forwarded command through the public CLI help contract.
1448
+
1449
+ Report, mirror-GC, and attribution modules keep standalone parsers because
1450
+ their execution seams have different dependency constraints. The installed
1451
+ command still owns their public path and presentation.
1452
+ """
1453
+ parser.prog = prog
1454
+ parser.formatter_class = _CoderFormatter
1455
+ parser._action_groups.sort(key=lambda group: group.title == "options")
1456
+ print(ui.style_help(parser.format_help(), stream=sys.stdout), end="")
1457
+ return 0
1458
+
1459
+
1460
+ def _dpo_profile_name(name: str) -> str:
1461
+ """Keep retired profile migration guidance at the argument boundary."""
1462
+ from sediment_export.compatibility import CompatibilityError, get_profile
1463
+
1464
+ try:
1465
+ return get_profile(name, objective="dpo").id
1466
+ except CompatibilityError as exc:
1467
+ raise argparse.ArgumentTypeError(str(exc)) from exc
1468
+
1469
+
1470
+ def build_parser() -> argparse.ArgumentParser:
1471
+ """The argparse tree, extracted so the golden --help test can walk it."""
1472
+ parser = _ArgumentParser(
1473
+ prog="sediment",
1474
+ usage="sediment <command> [options]",
1475
+ description=__doc__.splitlines()[1],
1476
+ )
1477
+ parser.add_argument(
1478
+ "--version", action="version", version=f"%(prog)s {__version__}"
1479
+ )
1480
+ # prog passed explicitly: argparse otherwise derives it by formatting the
1481
+ # parent usage through the formatter, which would drag the USAGE: heading
1482
+ # into every subcommand's prog.
1483
+ sub = parser.add_subparsers(
1484
+ dest="command",
1485
+ required=True,
1486
+ title="commands",
1487
+ metavar="<command>",
1488
+ prog=parser.prog,
1489
+ )
1490
+
1491
+ # Stubs for --help only — the dispatch in main() runs first.
1492
+ sub.add_parser("report", help="read-only reports (sediment report --help)")
1493
+ sub.add_parser(
1494
+ "delivery", help="prepared capture delivery (sediment delivery --help)"
1495
+ )
1496
+ sub.add_parser(
1497
+ "mirror-gc",
1498
+ help="remove mirrors past push retention (dry-run; --apply deletes)",
1499
+ )
1500
+
1501
+ p_facts = sub.add_parser(
1502
+ "facts", help="fact counts per table (remote; --database-url for direct)"
1503
+ )
1504
+ p_facts.add_argument(
1505
+ "--database-url", help="read PostgreSQL directly instead of the API"
1506
+ )
1507
+ p_facts.set_defaults(func=cmd_facts)
1508
+
1509
+ p_demo = sub.add_parser(
1510
+ "demo", help="plant one synthetic session so facts is non-zero"
1511
+ )
1512
+ p_demo.add_argument(
1513
+ "--force",
1514
+ action="store_true",
1515
+ help="allow a non-loopback server (writes synthetic facts to it)",
1516
+ )
1517
+ p_demo.set_defaults(func=cmd_demo)
1518
+
1519
+ p_db = sub.add_parser(
1520
+ "db", help="provision database roles or inspect and upgrade the schema"
1521
+ )
1522
+ db_sub = p_db.add_subparsers(
1523
+ dest="db_operation",
1524
+ required=True,
1525
+ title="operations",
1526
+ metavar="<operation>",
1527
+ prog=p_db.prog,
1528
+ )
1529
+ for operation, help_text, func in (
1530
+ ("status", "inspect the schema revision without changing it", cmd_db_status),
1531
+ ("upgrade", "upgrade the schema under an advisory lock", cmd_db_upgrade),
1532
+ ):
1533
+ command = db_sub.add_parser(operation, help=help_text)
1534
+ command.add_argument(
1535
+ "--database-url",
1536
+ help="PostgreSQL URL (default: SEDIMENT_DATABASE_URL)",
1537
+ )
1538
+ command.set_defaults(func=func)
1539
+
1540
+ provision = db_sub.add_parser(
1541
+ "provision",
1542
+ help="provision roles and migrate using isolated bootstrap credentials",
1543
+ description=(
1544
+ "Read SEDIMENT_BOOTSTRAP_DATABASE_URL, SEDIMENT_MIGRATOR_PASSWORD, "
1545
+ "SEDIMENT_RUNTIME_PASSWORD, and SEDIMENT_OPERATOR_PASSWORD from "
1546
+ "the environment. Stop services before provisioning."
1547
+ ),
1548
+ )
1549
+ provision.set_defaults(func=cmd_db_provision)
1550
+
1551
+ for verb, help_text in (
1552
+ ("quarantine", "quarantine one fact by id"),
1553
+ ("release", "release one quarantined fact by id"),
1554
+ ):
1555
+ p = sub.add_parser(verb, help=help_text)
1556
+ p.add_argument("table", choices=_TABLES)
1557
+ p.add_argument("fact_id")
1558
+ p.add_argument("--reason", required=True)
1559
+ p.set_defaults(func=cmd_quarantine_or_release)
1560
+
1561
+ p_log = sub.add_parser("quarantine-log", help="the org's quarantine history")
1562
+ p_log.add_argument("--tail", type=int, default=20, help="rows shown (default 20)")
1563
+ p_log.add_argument("--all", action="store_true", help="show the full history")
1564
+ p_log.set_defaults(func=cmd_quarantine_log)
1565
+
1566
+ p_qc = sub.add_parser(
1567
+ "quarantine-inference-calls",
1568
+ help="bulk-quarantine inference calls by filter (dry-run by default)",
1569
+ )
1570
+ p_qc.add_argument("--session-id")
1571
+ p_qc.add_argument(
1572
+ "--provider",
1573
+ metavar="{" + ",".join(provider.value for provider in GatewayProvider) + "}",
1574
+ help="gateway provider filter",
1575
+ )
1576
+ p_qc.add_argument(
1577
+ "--between",
1578
+ nargs=2,
1579
+ metavar=("FROM", "TO"),
1580
+ help="ISO-8601 capture-time bounds, timezone-aware, inclusive",
1581
+ )
1582
+ p_qc.add_argument(
1583
+ "--all", action="store_true", help="allow a filterless (org-wide) run"
1584
+ )
1585
+ p_qc.add_argument(
1586
+ "--apply", action="store_true", help="write; without it, dry-run only"
1587
+ )
1588
+ p_qc.add_argument("--reason", required=True)
1589
+ p_qc.set_defaults(func=cmd_quarantine_inference_calls)
1590
+
1591
+ # Stubs for --help only — the attribution dispatch in main() runs first
1592
+ # (the report/mirror-gc pattern).
1593
+ sub.add_parser(
1594
+ "install",
1595
+ help="wire a repo + this machine: git hooks, agent hooks, telemetry "
1596
+ "env (sediment install --help)",
1597
+ )
1598
+ sub.add_parser(
1599
+ "uninstall", help="remove the per-repo git hooks (--agents: user-level too)"
1600
+ )
1601
+ sub.add_parser("doctor", help="check attribution + server health on this machine")
1602
+
1603
+ p_server = sub.add_parser(
1604
+ "server", help="run a local API server with managed PostgreSQL"
1605
+ )
1606
+ p_server.add_argument("--host", default="127.0.0.1")
1607
+ p_server.add_argument("--port", type=int, default=8000)
1608
+ p_server.add_argument(
1609
+ "--root",
1610
+ default=str(Path.home() / ".sediment" / "server"),
1611
+ help="server data, PostgreSQL binaries, credentials, and logs "
1612
+ "(default: ~/.sediment/server)",
1613
+ )
1614
+ p_server.set_defaults(func=cmd_server)
1615
+
1616
+ p_login = sub.add_parser(
1617
+ "login",
1618
+ help="store credentials for a server",
1619
+ # Explicit description (it beats _propagate_descriptions): the token
1620
+ # resolution order is the thing a reader most needs, and stating it
1621
+ # here feeds both --help and the generated CLI reference.
1622
+ description="store verified operator credentials, or separate ingest "
1623
+ "credentials with --capture. Remote URLs must use "
1624
+ "HTTPS; HTTP is accepted only for localhost or a literal loopback "
1625
+ "IP address. A loopback URL enrolls both credentials sediment server "
1626
+ "generated in ~/.sediment/server/server.env; any other server "
1627
+ "prompts, or takes --with-token on stdin.",
1628
+ )
1629
+ p_login.add_argument(
1630
+ "url",
1631
+ help="HTTPS server base URL or loopback HTTP URL, e.g. http://127.0.0.1:8000",
1632
+ )
1633
+ p_login.add_argument(
1634
+ "--capture",
1635
+ action="store_true",
1636
+ help="enroll an ingest credential for capture without replacing operator login",
1637
+ )
1638
+ p_login.add_argument(
1639
+ "--with-token",
1640
+ action="store_true",
1641
+ help="read the token from stdin instead of prompting (unattended)",
1642
+ )
1643
+ p_login.set_defaults(func=cmd_login)
1644
+
1645
+ p_logout = sub.add_parser("logout", help="remove stored credentials")
1646
+ p_logout.add_argument(
1647
+ "--server", help="server URL to log out of (default: current)"
1648
+ )
1649
+ p_logout.set_defaults(func=cmd_logout)
1650
+
1651
+ p_commit = sub.add_parser("commit", help="pretty-print attributions for a commit")
1652
+ p_commit.add_argument("sha", help="full 40- or 64-char commit SHA")
1653
+ p_commit.add_argument(
1654
+ "--repo", help="observed repository name; ambiguous names require identity"
1655
+ )
1656
+ p_commit.add_argument(
1657
+ "--repository-provider",
1658
+ choices=[provider.value for provider in ForgeProvider],
1659
+ help="forge provider (requires host and ID)",
1660
+ )
1661
+ p_commit.add_argument(
1662
+ "--repository-host", help="forge hostname (requires provider and ID)"
1663
+ )
1664
+ p_commit.add_argument(
1665
+ "--repository-id", help="provider repository ID (requires provider and host)"
1666
+ )
1667
+ p_commit.add_argument(
1668
+ "--as-of", help="inclusive evidence boundary as an aware RFC 3339 timestamp"
1669
+ )
1670
+ p_commit.set_defaults(func=cmd_commit)
1671
+
1672
+ p_derive = sub.add_parser(
1673
+ "derive", help="materialize a reviewed canonical derivation bundle"
1674
+ )
1675
+ p_derive.add_argument("--out", required=True, help="new bundle directory")
1676
+ p_derive.add_argument("--policy", help="strict derivation policy TOML file")
1677
+ p_derive.add_argument("--since", help="inclusive RFC 3339 completion time")
1678
+ p_derive.add_argument("--until", help="exclusive RFC 3339 completion time")
1679
+ p_derive.add_argument(
1680
+ "--users",
1681
+ nargs="+",
1682
+ metavar="USER",
1683
+ help="space-separated user IDs (default: every user in the organization)",
1684
+ )
1685
+ p_derive.add_argument(
1686
+ "--sample",
1687
+ type=int,
1688
+ default=0,
1689
+ metavar="N",
1690
+ help="print the first N canonical rows after writing",
1691
+ )
1692
+ p_derive.set_defaults(func=cmd_derive)
1693
+
1694
+ p_e = sub.add_parser("export", help="run an export projection")
1695
+ e_sub = p_e.add_subparsers(
1696
+ dest="format",
1697
+ required=True,
1698
+ title="formats",
1699
+ metavar="<format>",
1700
+ prog=p_e.prog,
1701
+ )
1702
+
1703
+ p_rlvr = e_sub.add_parser("rlvr", help="export one explicit RLVR target")
1704
+ p_rlvr.add_argument("--out", required=True, help="output directory")
1705
+ p_rlvr.add_argument(
1706
+ "--target",
1707
+ required=True,
1708
+ choices=("sediment", "swe-bench", "nemo-gym"),
1709
+ help="explicit consumer contract",
1710
+ )
1711
+ p_rlvr.add_argument(
1712
+ "--from", dest="from_bundle", help="validated derived bundle directory"
1713
+ )
1714
+ from sediment_export.compatibility import PROFILES
1715
+
1716
+ p_rlvr.add_argument(
1717
+ "--profile",
1718
+ choices=tuple(p.id for p in PROFILES if p.objective == "rlvr"),
1719
+ help="exact optional consumer compatibility profile",
1720
+ )
1721
+ p_rlvr.add_argument(
1722
+ "--consumer-config",
1723
+ help="JSON task and response configuration for the selected profile",
1724
+ )
1725
+ p_rlvr.set_defaults(func=cmd_export_rlvr)
1726
+
1727
+ p_dpo = e_sub.add_parser("dpo", help="dpo.jsonl — chosen/rejected pairs")
1728
+ p_dpo.add_argument("--out", required=True, help="output directory")
1729
+ p_dpo.add_argument(
1730
+ "--recipe",
1731
+ choices=("dpo_human", "dpo_outcome"),
1732
+ default="dpo_human",
1733
+ help="evidence recipe (default: dpo_human; dpo_outcome requires opt-in)",
1734
+ )
1735
+ p_dpo.add_argument(
1736
+ "--from", dest="from_bundle", help="validated derived bundle directory"
1737
+ )
1738
+ p_dpo.add_argument(
1739
+ "--profile",
1740
+ choices=tuple(p.id for p in PROFILES if p.objective == "dpo"),
1741
+ type=_dpo_profile_name,
1742
+ help="exact optional consumer compatibility profile",
1743
+ )
1744
+ p_dpo.set_defaults(func=cmd_export_dpo)
1745
+
1746
+ p_sft = e_sub.add_parser("sft", help="sft.jsonl — supervised training targets")
1747
+ p_sft.add_argument("--out", required=True, help="output directory")
1748
+ p_sft.add_argument(
1749
+ "--recipe",
1750
+ choices=("sft_curated", "sft_verified"),
1751
+ default="sft_curated",
1752
+ help="evidence recipe (default: sft_curated; sft_verified requires opt-in)",
1753
+ )
1754
+ p_sft.add_argument(
1755
+ "--from", dest="from_bundle", help="validated derived bundle directory"
1756
+ )
1757
+ p_sft.add_argument(
1758
+ "--profile",
1759
+ choices=tuple(p.id for p in PROFILES if p.objective == "sft"),
1760
+ help="exact optional consumer compatibility profile",
1761
+ )
1762
+ p_sft.set_defaults(func=cmd_export_sft)
1763
+
1764
+ p_diff = e_sub.add_parser(
1765
+ "diff-sft", help="diff_sft.jsonl — per-commit, per-file training rows"
1766
+ )
1767
+ p_diff.add_argument("--out", required=True, help="output directory")
1768
+ p_diff.add_argument(
1769
+ "--recipe",
1770
+ choices=("sft_curated", "sft_verified"),
1771
+ default="sft_curated",
1772
+ help="SFT evidence recipe (default: sft_curated; sft_verified requires opt-in)",
1773
+ )
1774
+ p_diff.add_argument(
1775
+ "--from", dest="from_bundle", help="validated derived bundle directory"
1776
+ )
1777
+ p_diff.set_defaults(func=cmd_export_diff_sft)
1778
+
1779
+ p_rec = e_sub.add_parser(
1780
+ "recovery",
1781
+ help="recovery.jsonl — red-to-green CI-transition recovery pairs",
1782
+ )
1783
+ p_rec.add_argument("--out", required=True, help="output directory")
1784
+ p_rec.add_argument(
1785
+ "--recipe",
1786
+ choices=("recovery_ci",),
1787
+ default="recovery_ci",
1788
+ help="evidence recipe (default: recovery_ci)",
1789
+ )
1790
+ p_rec.set_defaults(func=cmd_export_recovery)
1791
+
1792
+ _propagate_descriptions(parser)
1793
+ return parser
1794
+
1795
+
1796
+ # Remote verbs speak HTTP through the client seam; they never open the fact
1797
+ # store and never construct Settings.
1798
+ _REMOTE_VERBS = {"login", "logout", "commit", "facts", "demo"}
1799
+
1800
+ # Dispatched pre-argparse to the stdlib-only attribution module.
1801
+ # install/uninstall/doctor get help stubs; the hook-plumbing verbs are
1802
+ # execution-only (hidden from --help).
1803
+ _ATTRIBUTION_VERBS = {
1804
+ "cursor-hook",
1805
+ "install",
1806
+ "uninstall",
1807
+ "doctor",
1808
+ "mark",
1809
+ "stamp",
1810
+ "union-squash-notes",
1811
+ "push-notes",
1812
+ "repair-notes",
1813
+ }
1814
+
1815
+
1816
+ def main(argv: list[str] | None = None) -> int:
1817
+ argv = list(sys.argv[1:] if argv is None else argv)
1818
+ if argv[:1] == ["delivery"]:
1819
+ module = importlib.import_module("sediment_cli.delivery")
1820
+ if argv[1:] in (["-h"], ["--help"]):
1821
+ return _print_dispatched_help(
1822
+ module.build_parser(), prog="sediment delivery"
1823
+ )
1824
+ if len(argv) == 3 and argv[2] in {"-h", "--help"}:
1825
+ parser = module.build_parser(prog="sediment delivery")
1826
+ commands = next(
1827
+ action.choices
1828
+ for action in parser._actions
1829
+ if isinstance(action, argparse._SubParsersAction)
1830
+ )
1831
+ if argv[1] in commands:
1832
+ return _print_dispatched_help(
1833
+ commands[argv[1]], prog=f"sediment delivery {argv[1]}"
1834
+ )
1835
+ return _public_status(module.main(argv[1:], prog="sediment delivery"))
1836
+ # Dispatched before argparse: report/mirror-gc forward argv verbatim to
1837
+ # the module's own parser and never construct Settings (see _REPORTS).
1838
+ if argv[:1] == ["report"]:
1839
+ return _run_report(argv[1:])
1840
+ if argv[:1] == ["mirror-gc"]:
1841
+ module = importlib.import_module("sediment_api.mirror_gc")
1842
+ if argv[1:] in (["-h"], ["--help"]):
1843
+ return _print_dispatched_help(
1844
+ module.build_parser(), prog="sediment mirror-gc"
1845
+ )
1846
+ try:
1847
+ return _public_status(module.main(argv[1:]))
1848
+ except (DatabaseOperationError, OSError, ValueError) as exc:
1849
+ return _fail(str(exc))
1850
+ if argv[:1] == ["transcript"]:
1851
+ module = importlib.import_module("sediment_cli.transcript")
1852
+ return _public_status(module.main(argv[1:]))
1853
+ if argv[:1] and argv[0] in _ATTRIBUTION_VERBS:
1854
+ # Forwarded verbatim to the attribution module's own parser —
1855
+ # stdlib-only, never constructs Settings. The plumbing verbs (mark,
1856
+ # stamp, union-squash-notes, push-notes, repair-notes) get no help
1857
+ # stub: hooks invoke them, humans don't.
1858
+ module = importlib.import_module("sediment_cli.attribution")
1859
+ if argv[0] in {"install", "uninstall", "doctor"} and argv[1:] in (
1860
+ ["-h"],
1861
+ ["--help"],
1862
+ ):
1863
+ action = _subparsers_action(module.build_parser())
1864
+ return _print_dispatched_help(
1865
+ action.choices[argv[0]], prog=f"sediment {argv[0]}"
1866
+ )
1867
+ return _public_status(module.main(argv))
1868
+
1869
+ args = build_parser().parse_args(argv)
1870
+
1871
+ if args.command == "export" and getattr(args, "from_bundle", None):
1872
+ try:
1873
+ args._mirror_path = os.environ.get("SEDIMENT_MIRROR_PATH")
1874
+ return args.func(None, None, args)
1875
+ except (OSError, ValueError) as exc:
1876
+ return _fail(str(exc))
1877
+
1878
+ if args.command == "server":
1879
+ # Pre-Settings dispatch: provision before the API imports its settings.
1880
+ try:
1881
+ return args.func(args)
1882
+ except (DatabaseOperationError, OSError, ValueError) as exc:
1883
+ return _fail(str(exc))
1884
+
1885
+ if args.command == "db":
1886
+ from sediment_core.postgres_migrations import MigrationError
1887
+
1888
+ try:
1889
+ return args.func(args)
1890
+ except (DatabaseOperationError, MigrationError, ValueError) as exc:
1891
+ return _fail(str(exc))
1892
+
1893
+ if args.command in _REMOTE_VERBS:
1894
+ try:
1895
+ return args.func(args)
1896
+ except (ClientError, DatabaseOperationError, ValueError) as exc:
1897
+ return _fail(str(exc))
1898
+
1899
+ try:
1900
+ _prepare_store_command(args)
1901
+
1902
+ # Deferred so --help needs no SEDIMENT_ORG_ID. Keep construction
1903
+ # inside this boundary because Pydantic settings validation is an
1904
+ # expected operator-input failure.
1905
+ from sediment_api.config import settings
1906
+ from sediment_api.database import one_shot_fact_store
1907
+
1908
+ args._mirror_path = settings.mirror_path
1909
+ with one_shot_fact_store(
1910
+ settings.database_url.get_secret_value(), operation=f"run {args.command}"
1911
+ ) as store:
1912
+ return args.func(store, settings.org_id, args)
1913
+ except (DatabaseOperationError, OSError, ValueError) as exc:
1914
+ # One net for every validation error below (empty reason via the
1915
+ # QuarantineRecord validator, naive --between bounds, pydantic
1916
+ # settings): clean `error:` line, not a
1917
+ # traceback.
1918
+ return _fail(str(exc))
1919
+
1920
+
1921
+ if __name__ == "__main__": # pragma: no cover
1922
+ sys.exit(main())