sediment-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sediment_cli/__init__.py +22 -0
- sediment_cli/attribution.py +3881 -0
- sediment_cli/cli.py +1922 -0
- sediment_cli/client.py +323 -0
- sediment_cli/delivery.py +1334 -0
- sediment_cli/local_postgres.py +337 -0
- sediment_cli/transcript.py +1764 -0
- sediment_cli/ui.py +123 -0
- sediment_cli-0.1.0.dist-info/METADATA +16 -0
- sediment_cli-0.1.0.dist-info/RECORD +13 -0
- sediment_cli-0.1.0.dist-info/WHEEL +4 -0
- sediment_cli-0.1.0.dist-info/entry_points.txt +2 -0
- sediment_cli-0.1.0.dist-info/licenses/LICENSE +661 -0
sediment_cli/cli.py
ADDED
|
@@ -0,0 +1,1922 @@
|
|
|
1
|
+
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
"""
|
|
3
|
+
Sediment CLI: login/logout, remote facts + commit, quarantine, exports, reports, mirror GC.
|
|
4
|
+
|
|
5
|
+
Remote verbs (login/logout/facts/commit/demo) speak HTTP through
|
|
6
|
+
``sediment_cli.client`` and never open the fact store; only explicit
|
|
7
|
+
local mode (``facts --database-url`` or ``SEDIMENT_DATABASE_URL``) reads the store
|
|
8
|
+
directly, so ``docker compose exec api sediment facts`` is unchanged.
|
|
9
|
+
|
|
10
|
+
Store verbs are presentation-only (ADR 0001): each is a ``FactStore`` method,
|
|
11
|
+
an export pipeline call, or a forwarded report module — the CLI adds no
|
|
12
|
+
semantics of its own. Store-writing verbs read the same settings as the API
|
|
13
|
+
(database URL, mirror path, org id), so there is exactly one configuration
|
|
14
|
+
surface; ``report`` and ``mirror-gc`` never construct ``Settings`` — they
|
|
15
|
+
take ``--org``/``--database-url``/``--mirror-path`` with the ``SEDIMENT_*`` env
|
|
16
|
+
vars as defaults, so a report can cover any org a deployment holds without
|
|
17
|
+
API auth config.
|
|
18
|
+
|
|
19
|
+
Safety posture: the bulk ``quarantine-inference-calls`` form dry-runs by
|
|
20
|
+
default and writes only with ``--apply``, and refuses a filterless
|
|
21
|
+
invocation without ``--all`` (no filters means every inference call in the org).
|
|
22
|
+
Per-fact ``quarantine``/``release`` act immediately — one id, reversible,
|
|
23
|
+
``--reason`` always required. Output vocabulary is the CONTEXT.md glossary:
|
|
24
|
+
facts are quarantined and released, never deleted.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import hashlib
|
|
31
|
+
import importlib
|
|
32
|
+
import os
|
|
33
|
+
import sys
|
|
34
|
+
from contextlib import ExitStack
|
|
35
|
+
from datetime import datetime
|
|
36
|
+
from itertools import islice
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Any
|
|
39
|
+
from urllib.parse import urlencode, urlsplit
|
|
40
|
+
|
|
41
|
+
from pydantic import TypeAdapter
|
|
42
|
+
|
|
43
|
+
from . import __version__, ui
|
|
44
|
+
from .client import (
|
|
45
|
+
ClientError,
|
|
46
|
+
current_url,
|
|
47
|
+
get_json,
|
|
48
|
+
is_loopback_host,
|
|
49
|
+
maybe_warn_version_skew,
|
|
50
|
+
norm_url,
|
|
51
|
+
post_json,
|
|
52
|
+
probe_me,
|
|
53
|
+
read_config,
|
|
54
|
+
validate_server_url,
|
|
55
|
+
write_config,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
from sediment_core import (
|
|
59
|
+
FactStore,
|
|
60
|
+
FactTable,
|
|
61
|
+
GatewayProvider,
|
|
62
|
+
ForgeProvider,
|
|
63
|
+
NonEmptyId,
|
|
64
|
+
QuarantineRecord,
|
|
65
|
+
normalize_org_id,
|
|
66
|
+
)
|
|
67
|
+
from sediment_core.postgres_engine import DatabaseOperationError
|
|
68
|
+
from sediment_derive import (
|
|
69
|
+
MirrorManager,
|
|
70
|
+
derive_recovery_result,
|
|
71
|
+
inference_fact_id,
|
|
72
|
+
)
|
|
73
|
+
from sediment_export import (
|
|
74
|
+
DPOPolicy,
|
|
75
|
+
DerivationPolicy,
|
|
76
|
+
DerivationScope,
|
|
77
|
+
build_derived_bundle_context,
|
|
78
|
+
open_derived_bundle,
|
|
79
|
+
SFTPolicy,
|
|
80
|
+
export_rlvr,
|
|
81
|
+
export_rlvr_from_bundle,
|
|
82
|
+
project_recovery,
|
|
83
|
+
load_derivation_policy,
|
|
84
|
+
recovery_to_export_rows,
|
|
85
|
+
write_jsonl,
|
|
86
|
+
write_derived_bundle,
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
# The one-line answer to "what is this?", coder-style, shown beside the
|
|
90
|
+
# version on the root --help title. With the title prefix it fills exactly
|
|
91
|
+
# the pinned 80 columns. Keep in step with the root pyproject description.
|
|
92
|
+
_TAGLINE = "Turn AI developer workflow traces into RL-ready training data."
|
|
93
|
+
|
|
94
|
+
_TABLES = [t.value for t in FactTable]
|
|
95
|
+
|
|
96
|
+
# Read-only reports: name -> (module, one-line help). Dispatched
|
|
97
|
+
# before argparse and before Settings exists — a report never needs API auth
|
|
98
|
+
# config (SEDIMENT_ORG_ID + secrets) to read a store, so argv is forwarded
|
|
99
|
+
# verbatim to the module's own parser (`--org` falls back to
|
|
100
|
+
# $SEDIMENT_ORG_ID, `--database-url` to $SEDIMENT_DATABASE_URL). Modules import lazily:
|
|
101
|
+
# `sediment --help` stays fast and settings-free.
|
|
102
|
+
_REPORTS = {
|
|
103
|
+
"model": (
|
|
104
|
+
"sediment_api.reports.model_report",
|
|
105
|
+
"per-model attribution/CI/acceptance; --compare significance",
|
|
106
|
+
),
|
|
107
|
+
"label-confidence-inspection": (
|
|
108
|
+
"sediment_api.reports.label_confidence_inspection",
|
|
109
|
+
"sample resolved confidence for human validation",
|
|
110
|
+
),
|
|
111
|
+
"dataset-diagnostics": (
|
|
112
|
+
"sediment_api.reports.dataset_diagnostics",
|
|
113
|
+
"export-side dataset health checks",
|
|
114
|
+
),
|
|
115
|
+
"recovery-yield": (
|
|
116
|
+
"sediment_api.reports.recovery_yield_report",
|
|
117
|
+
"recovery-pair yield by skip reason",
|
|
118
|
+
),
|
|
119
|
+
"precision": (
|
|
120
|
+
"sediment_api.reports.precision_report",
|
|
121
|
+
"attribution precision/recall against a ground-truth manifest",
|
|
122
|
+
),
|
|
123
|
+
"abandonment": (
|
|
124
|
+
"sediment_api.reports.abandonment_report",
|
|
125
|
+
"sessions whose accepted edits never reached a commit",
|
|
126
|
+
),
|
|
127
|
+
"attribution-share": (
|
|
128
|
+
"sediment_api.reports.attribution_share_report",
|
|
129
|
+
"notes-attribution share and decline alert",
|
|
130
|
+
),
|
|
131
|
+
"merge-retention": (
|
|
132
|
+
"sediment_api.reports.merge_retention_report",
|
|
133
|
+
"attributed change retention through pull request merge",
|
|
134
|
+
),
|
|
135
|
+
"lifecycle": (
|
|
136
|
+
"sediment_api.reports.lifecycle_report",
|
|
137
|
+
"accepted-work progression, retention, attrition, and rework evidence",
|
|
138
|
+
),
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _fail(msg: str) -> int:
|
|
143
|
+
print(ui.error_line(msg), file=sys.stderr)
|
|
144
|
+
return 1
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _database_url(args: argparse.Namespace) -> str:
|
|
148
|
+
database_url = args.database_url or os.environ.get("SEDIMENT_DATABASE_URL")
|
|
149
|
+
if not database_url:
|
|
150
|
+
raise ValueError("set SEDIMENT_DATABASE_URL or pass --database-url")
|
|
151
|
+
return database_url
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def cmd_db_upgrade(args: argparse.Namespace) -> int:
|
|
155
|
+
"""Upgrade the PostgreSQL physical schema to the supported head."""
|
|
156
|
+
from sediment_core.postgres_migrations import HEAD_REVISION, upgrade_database
|
|
157
|
+
|
|
158
|
+
upgrade_database(_database_url(args))
|
|
159
|
+
print(f"database schema upgraded to {HEAD_REVISION}")
|
|
160
|
+
return 0
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def cmd_db_provision(args: argparse.Namespace) -> int:
|
|
164
|
+
"""Provision fixed deployment roles using secrets from the environment."""
|
|
165
|
+
from sediment_core.postgres_roles import provision_database
|
|
166
|
+
|
|
167
|
+
values = {}
|
|
168
|
+
for name in (
|
|
169
|
+
"BOOTSTRAP_DATABASE_URL",
|
|
170
|
+
"MIGRATOR_PASSWORD",
|
|
171
|
+
"RUNTIME_PASSWORD",
|
|
172
|
+
"OPERATOR_PASSWORD",
|
|
173
|
+
):
|
|
174
|
+
value = os.environ.get(f"SEDIMENT_{name}")
|
|
175
|
+
if not value:
|
|
176
|
+
raise ValueError(f"set SEDIMENT_{name} for database provisioning")
|
|
177
|
+
values[name.lower()] = value
|
|
178
|
+
provision_database(**values)
|
|
179
|
+
print("database roles provisioned and schema upgraded")
|
|
180
|
+
return 0
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def cmd_db_status(args: argparse.Namespace) -> int:
|
|
184
|
+
"""Inspect the PostgreSQL physical schema without changing it."""
|
|
185
|
+
from sediment_core.postgres_migrations import RevisionState, inspect_revision
|
|
186
|
+
|
|
187
|
+
inspection = inspect_revision(_database_url(args))
|
|
188
|
+
if inspection.state is RevisionState.AT_HEAD:
|
|
189
|
+
print(f"database schema: at_head ({inspection.head_revision})")
|
|
190
|
+
else:
|
|
191
|
+
print(
|
|
192
|
+
f"database schema: {inspection.state.value} "
|
|
193
|
+
f"(supported head {inspection.head_revision})"
|
|
194
|
+
)
|
|
195
|
+
return 0
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _public_status(status: int) -> int:
|
|
199
|
+
"""Map a dispatched command's semantic failure onto the CLI contract."""
|
|
200
|
+
return 0 if status == 0 else 1
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _run_report(argv: list[str]) -> int:
|
|
204
|
+
is_help = bool(argv) and argv[0] in ("-h", "--help")
|
|
205
|
+
if not argv or is_help:
|
|
206
|
+
# Usage on stdout only when asked for (--help); the bare-invocation
|
|
207
|
+
# error goes to stderr like argparse's own, keeping stdout pipeable.
|
|
208
|
+
stream = sys.stdout if is_help else sys.stderr
|
|
209
|
+
lines = [
|
|
210
|
+
"USAGE:",
|
|
211
|
+
" sediment report <name> [options]",
|
|
212
|
+
"",
|
|
213
|
+
" read-only reports; each takes --help, an empty result exits 0",
|
|
214
|
+
"",
|
|
215
|
+
"REPORTS:",
|
|
216
|
+
]
|
|
217
|
+
lines += [
|
|
218
|
+
f" {name:<22} {help_text}" for name, (_, help_text) in _REPORTS.items()
|
|
219
|
+
]
|
|
220
|
+
print(ui.style_help("\n".join(lines), stream=stream), file=stream)
|
|
221
|
+
return 0 if is_help else 2
|
|
222
|
+
name, *rest = argv
|
|
223
|
+
if name not in _REPORTS:
|
|
224
|
+
return _fail(f"unknown report {name!r} (choices: {', '.join(_REPORTS)})")
|
|
225
|
+
module = importlib.import_module(_REPORTS[name][0])
|
|
226
|
+
if rest in (["-h"], ["--help"]):
|
|
227
|
+
return _print_dispatched_help(
|
|
228
|
+
module.build_parser(), prog=f"sediment report {name}"
|
|
229
|
+
)
|
|
230
|
+
try:
|
|
231
|
+
return _public_status(module.main(rest))
|
|
232
|
+
except (DatabaseOperationError, OSError, ValueError) as exc:
|
|
233
|
+
return _fail(str(exc))
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _styled_revision(quarantine_revision: object) -> str:
|
|
237
|
+
"""Render revision 0 quietly and later quarantine revisions prominently."""
|
|
238
|
+
revision = str(quarantine_revision)
|
|
239
|
+
if revision == "0":
|
|
240
|
+
return ui.style(revision, "dim")
|
|
241
|
+
return ui.style(revision, "sandstone", "bold")
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _print_facts(
|
|
245
|
+
sessions: int, tables: dict[str, dict[str, int]], quarantine_revision: object
|
|
246
|
+
) -> None:
|
|
247
|
+
"""The §6.2 counts table — shared by the local and remote ``facts`` paths
|
|
248
|
+
so both render identically. ``quarantine_revision`` is an int on both
|
|
249
|
+
paths; the parameter stays ``object`` so either caller formats."""
|
|
250
|
+
print(ui.style(f"{'table':<20} {'total':>7} {'visible':>8}", "dim"))
|
|
251
|
+
print(f"{'sessions':<20} {sessions:>7} {'-':>8}")
|
|
252
|
+
for table in FactTable:
|
|
253
|
+
if table.value not in tables:
|
|
254
|
+
print(f"{table.value:<20} {'unavailable':>7} {'unavailable':>8}")
|
|
255
|
+
continue
|
|
256
|
+
counts = tables[table.value]
|
|
257
|
+
head = f"{table.value:<20} {counts['total']:>7}"
|
|
258
|
+
visible = f"{counts['visible']:>8}"
|
|
259
|
+
if counts["total"] == 0:
|
|
260
|
+
print(ui.style(f"{head} {visible}", "dim"))
|
|
261
|
+
elif counts["visible"] < counts["total"]:
|
|
262
|
+
# Facts hidden from derivations — the number an operator scans for.
|
|
263
|
+
print(f"{head} {ui.style(visible, 'sandstone')}")
|
|
264
|
+
else:
|
|
265
|
+
print(f"{head} {visible}")
|
|
266
|
+
print(f"quarantine_revision: {_styled_revision(quarantine_revision)}")
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _facts_local(store: FactStore, org: str) -> None:
|
|
270
|
+
"""Read today's counts straight from the store (local mode)."""
|
|
271
|
+
tables = {
|
|
272
|
+
table.value: {
|
|
273
|
+
"total": store.count_facts(org, table, include_quarantined=True),
|
|
274
|
+
"visible": store.count_facts(org, table),
|
|
275
|
+
}
|
|
276
|
+
for table in store.available_fact_tables()
|
|
277
|
+
}
|
|
278
|
+
_print_facts(store.count_sessions(org), tables, store.quarantine_revision(org))
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def cmd_facts(args: argparse.Namespace) -> int:
|
|
282
|
+
"""Fact counts per table, total and derivation-facing ("visible" = not
|
|
283
|
+
quarantined). Remote by default (GET /v1/facts); ``--database-url`` or a
|
|
284
|
+
present ``SEDIMENT_DATABASE_URL`` opens the store directly, so
|
|
285
|
+
``docker compose exec api sediment facts`` is unchanged."""
|
|
286
|
+
if args.database_url or os.environ.get("SEDIMENT_DATABASE_URL"):
|
|
287
|
+
from sediment_api.database import one_shot_fact_store
|
|
288
|
+
|
|
289
|
+
database_url = args.database_url or os.environ["SEDIMENT_DATABASE_URL"]
|
|
290
|
+
org = os.environ.get("SEDIMENT_ORG_ID")
|
|
291
|
+
if not org:
|
|
292
|
+
raise ValueError("set SEDIMENT_ORG_ID for direct fact counts")
|
|
293
|
+
with one_shot_fact_store(database_url, operation="count facts") as store:
|
|
294
|
+
_facts_local(store, normalize_org_id(org))
|
|
295
|
+
return 0
|
|
296
|
+
maybe_warn_version_skew()
|
|
297
|
+
data = get_json("/v1/facts")
|
|
298
|
+
_print_facts(data["sessions"], data["tables"], data["quarantine_revision"])
|
|
299
|
+
return 0
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
# The demo session's identity. Everything the verb writes carries these, so
|
|
303
|
+
# one glance at a fact row says "synthetic" and one quarantine call by
|
|
304
|
+
# session id removes the lot.
|
|
305
|
+
DEMO_SESSION_ID = "sediment-demo"
|
|
306
|
+
DEMO_CALL_ID = "sediment-demo-call-1"
|
|
307
|
+
# Fixed, not "now": ``occurred_at`` is part of both decision dedup indexes,
|
|
308
|
+
# so a wall-clock event time would store a second decision on every run
|
|
309
|
+
# while the completion (keyed on call_id) collapsed — the reader who runs
|
|
310
|
+
# the verb twice would watch one count move and the other stay put. A
|
|
311
|
+
# constant makes both facts collapse, and a synthetic fact claiming a
|
|
312
|
+
# synthetic time is the honest version anyway. 2026-01-01T00:00:00Z.
|
|
313
|
+
DEMO_EVENT_NANOS = 1767225600000000000
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _demo_completion() -> dict[str, Any]:
|
|
317
|
+
"""The gateway envelope a real LiteLLM callback POSTs, with synthetic
|
|
318
|
+
content. ``litellm_call_id`` matches the decision's ``tool_use_id``
|
|
319
|
+
below so the two facts join exactly as a real session's would."""
|
|
320
|
+
return {
|
|
321
|
+
"provider": "litellm",
|
|
322
|
+
"session_id": DEMO_SESSION_ID,
|
|
323
|
+
"user_id": DEMO_SESSION_ID,
|
|
324
|
+
"payload": {
|
|
325
|
+
"model": "sediment-demo-model",
|
|
326
|
+
"litellm_call_id": DEMO_CALL_ID,
|
|
327
|
+
"messages": [{"role": "user", "content": "Add a docstring to greet()."}],
|
|
328
|
+
"response": {
|
|
329
|
+
"choices": [
|
|
330
|
+
{
|
|
331
|
+
"message": {
|
|
332
|
+
"role": "assistant",
|
|
333
|
+
"content": (
|
|
334
|
+
"def greet(name):\n"
|
|
335
|
+
' """Return a greeting for name."""\n'
|
|
336
|
+
" return f'hello {name}'\n"
|
|
337
|
+
),
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
]
|
|
341
|
+
},
|
|
342
|
+
"usage": {"prompt_tokens": 12, "completion_tokens": 24},
|
|
343
|
+
"response_time_ms": 350,
|
|
344
|
+
},
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _demo_decision() -> dict[str, Any]:
|
|
349
|
+
"""The OTLP/JSON logs batch a harness shim POSTs on an edit-tool call,
|
|
350
|
+
shaped like packages/capture/tests/fixtures/otlp/sediment/tool_decision.json."""
|
|
351
|
+
return {
|
|
352
|
+
"resourceLogs": [
|
|
353
|
+
{
|
|
354
|
+
"resource": {
|
|
355
|
+
"attributes": [
|
|
356
|
+
{
|
|
357
|
+
"key": "user.id",
|
|
358
|
+
"value": {"stringValue": DEMO_SESSION_ID},
|
|
359
|
+
}
|
|
360
|
+
]
|
|
361
|
+
},
|
|
362
|
+
"scopeLogs": [
|
|
363
|
+
{
|
|
364
|
+
"logRecords": [
|
|
365
|
+
{
|
|
366
|
+
"body": {"stringValue": "sediment.tool_decision"},
|
|
367
|
+
"timeUnixNano": str(DEMO_EVENT_NANOS),
|
|
368
|
+
"attributes": [
|
|
369
|
+
{
|
|
370
|
+
"key": "agent",
|
|
371
|
+
"value": {"stringValue": "claude-code"},
|
|
372
|
+
},
|
|
373
|
+
{
|
|
374
|
+
"key": "session.id",
|
|
375
|
+
"value": {"stringValue": DEMO_SESSION_ID},
|
|
376
|
+
},
|
|
377
|
+
{
|
|
378
|
+
"key": "tool_use_id",
|
|
379
|
+
"value": {"stringValue": DEMO_CALL_ID},
|
|
380
|
+
},
|
|
381
|
+
{
|
|
382
|
+
"key": "decision",
|
|
383
|
+
"value": {"stringValue": "accept"},
|
|
384
|
+
},
|
|
385
|
+
{"key": "explicit", "value": {"boolValue": True}},
|
|
386
|
+
{
|
|
387
|
+
"key": "tool_name",
|
|
388
|
+
"value": {"stringValue": "Edit"},
|
|
389
|
+
},
|
|
390
|
+
{
|
|
391
|
+
"key": "file_path",
|
|
392
|
+
"value": {
|
|
393
|
+
"stringValue": "/sediment-demo/greet.py"
|
|
394
|
+
},
|
|
395
|
+
},
|
|
396
|
+
],
|
|
397
|
+
}
|
|
398
|
+
]
|
|
399
|
+
}
|
|
400
|
+
],
|
|
401
|
+
}
|
|
402
|
+
]
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _is_loopback(url: str) -> bool:
|
|
407
|
+
# urlsplit() strips brackets from an IPv6 authority before the helper.
|
|
408
|
+
host = urlsplit(url).hostname
|
|
409
|
+
return host is not None and is_loopback_host(host)
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def cmd_demo(args: argparse.Namespace) -> int:
|
|
413
|
+
"""Plant one synthetic session through the real ingest doors, then print
|
|
414
|
+
the fact counts. This exists so the quickstart ends on a captured fact
|
|
415
|
+
instead of a table of zeros: it proves the ingest path works end to end,
|
|
416
|
+
and proves nothing about the reader's own agent, which is still unwired.
|
|
417
|
+
|
|
418
|
+
Refuses a non-loopback server without ``--force``. Seeding a shared
|
|
419
|
+
deployment with synthetic facts is the failure this guard exists to
|
|
420
|
+
prevent — the facts are real once stored, and removing them is a
|
|
421
|
+
quarantine call, not an undo."""
|
|
422
|
+
url = current_url()
|
|
423
|
+
if not _is_loopback(url) and not args.force:
|
|
424
|
+
print(
|
|
425
|
+
ui.error_line(
|
|
426
|
+
f"{url} is not a local server. `demo` writes synthetic facts, "
|
|
427
|
+
"so it refuses a shared deployment. Pass --force if you meant it."
|
|
428
|
+
),
|
|
429
|
+
file=sys.stderr,
|
|
430
|
+
)
|
|
431
|
+
return 1
|
|
432
|
+
|
|
433
|
+
maybe_warn_version_skew()
|
|
434
|
+
print(f"posting demo session (synthetic) to {url}")
|
|
435
|
+
inference_call = post_json("/ingest/gateway", _demo_completion())
|
|
436
|
+
if inference_call.get("skipped"):
|
|
437
|
+
raise ClientError(
|
|
438
|
+
"the server skipped the demo inference call: "
|
|
439
|
+
f"{inference_call.get('reason')}"
|
|
440
|
+
)
|
|
441
|
+
# The OTLP door answers {} whether it stored the record or skipped it —
|
|
442
|
+
# per-record results are invisible to an exporter by design. So this
|
|
443
|
+
# says what was *posted*; the counts below are the only authority on
|
|
444
|
+
# what landed, and the check after them is what makes the difference
|
|
445
|
+
# actionable instead of a table the reader has to audit.
|
|
446
|
+
post_json("/v1/logs", _demo_decision())
|
|
447
|
+
print(f" posted 1 inference call and 1 decision as session {DEMO_SESSION_ID}")
|
|
448
|
+
print()
|
|
449
|
+
|
|
450
|
+
data = get_json("/v1/facts")
|
|
451
|
+
_print_facts(data["sessions"], data["tables"], data["quarantine_revision"])
|
|
452
|
+
print()
|
|
453
|
+
|
|
454
|
+
session_data = get_json(f"/v1/facts/session/{DEMO_SESSION_ID}")
|
|
455
|
+
missing = [
|
|
456
|
+
table
|
|
457
|
+
for table in ("inference_calls", "developer_decisions")
|
|
458
|
+
if session_data["tables"].get(table, {}).get("total", 0) == 0
|
|
459
|
+
]
|
|
460
|
+
if missing:
|
|
461
|
+
print(
|
|
462
|
+
ui.error_line(
|
|
463
|
+
f"posted, but {' and '.join(missing)} stayed empty — the server "
|
|
464
|
+
"took the payload and stored nothing. Check the api logs."
|
|
465
|
+
),
|
|
466
|
+
file=sys.stderr,
|
|
467
|
+
)
|
|
468
|
+
return 1
|
|
469
|
+
|
|
470
|
+
print(
|
|
471
|
+
ui.style(
|
|
472
|
+
"These are synthetic facts, not your agent's. They prove the "
|
|
473
|
+
"ingest path works end to end.",
|
|
474
|
+
"dim",
|
|
475
|
+
)
|
|
476
|
+
)
|
|
477
|
+
return 0
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _load_server_env(env_file: Path) -> dict[str, str]:
|
|
481
|
+
"""KEY=VALUE lines from a previous run's server.env; {} when absent."""
|
|
482
|
+
try:
|
|
483
|
+
lines = env_file.read_text(encoding="utf-8").splitlines()
|
|
484
|
+
except OSError:
|
|
485
|
+
return {}
|
|
486
|
+
pairs = {}
|
|
487
|
+
for line in lines:
|
|
488
|
+
key, sep, value = line.partition("=")
|
|
489
|
+
if sep and key and not key.startswith("#"):
|
|
490
|
+
pairs[key] = value
|
|
491
|
+
return pairs
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def cmd_server(args: argparse.Namespace) -> int:
|
|
495
|
+
"""Run the API and its optional managed PostgreSQL until interrupted."""
|
|
496
|
+
try:
|
|
497
|
+
with ExitStack() as stack:
|
|
498
|
+
return _run_server(args, stack)
|
|
499
|
+
except KeyboardInterrupt:
|
|
500
|
+
return 0
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def _run_server(args: argparse.Namespace, stack: ExitStack) -> int:
|
|
504
|
+
"""Provision an evaluation database, then serve with runtime database authority.
|
|
505
|
+
|
|
506
|
+
An explicit bootstrap URL selects an external database.
|
|
507
|
+
Private server.env stores separate capture/operator API tokens and role
|
|
508
|
+
passwords. An exported secret takes precedence over its saved value.
|
|
509
|
+
"""
|
|
510
|
+
import secrets as secrets_module
|
|
511
|
+
|
|
512
|
+
from sqlalchemy.engine import make_url
|
|
513
|
+
|
|
514
|
+
from sediment_core.postgres_roles import RUNTIME_ROLE, provision_database
|
|
515
|
+
|
|
516
|
+
from .local_postgres import managed_postgres, server_root
|
|
517
|
+
|
|
518
|
+
bootstrap = os.environ.get("SEDIMENT_BOOTSTRAP_DATABASE_URL")
|
|
519
|
+
try:
|
|
520
|
+
target = make_url(bootstrap) if bootstrap else None
|
|
521
|
+
if target is not None and (
|
|
522
|
+
target.get_backend_name() != "postgresql"
|
|
523
|
+
or not target.host
|
|
524
|
+
or not target.database
|
|
525
|
+
):
|
|
526
|
+
raise ValueError
|
|
527
|
+
except Exception:
|
|
528
|
+
raise ValueError(
|
|
529
|
+
"bootstrap URL must name an explicit PostgreSQL host and database"
|
|
530
|
+
) from None
|
|
531
|
+
|
|
532
|
+
root = Path(args.root).expanduser().absolute()
|
|
533
|
+
stack.enter_context(server_root(root))
|
|
534
|
+
(root / "mirror").mkdir(exist_ok=True, mode=0o700)
|
|
535
|
+
env_file = root / "server.env"
|
|
536
|
+
if env_file.is_symlink() or (env_file.exists() and env_file.stat().st_nlink != 1):
|
|
537
|
+
raise ValueError("server.env must be a regular private file")
|
|
538
|
+
stored = _load_server_env(env_file)
|
|
539
|
+
additions = {}
|
|
540
|
+
api_keys = ["SEDIMENT_OPERATOR_TOKEN", "SEDIMENT_GITHUB_WEBHOOK_SECRET"]
|
|
541
|
+
if not (
|
|
542
|
+
os.environ.get("SEDIMENT_INGEST_TOKENS") or stored.get("SEDIMENT_INGEST_TOKENS")
|
|
543
|
+
):
|
|
544
|
+
api_keys.append("SEDIMENT_API_BEARER_TOKEN")
|
|
545
|
+
database_keys = [
|
|
546
|
+
"SEDIMENT_MIGRATOR_PASSWORD",
|
|
547
|
+
"SEDIMENT_RUNTIME_PASSWORD",
|
|
548
|
+
"SEDIMENT_OPERATOR_PASSWORD",
|
|
549
|
+
]
|
|
550
|
+
if not bootstrap:
|
|
551
|
+
database_keys.append("SEDIMENT_BOOTSTRAP_PASSWORD")
|
|
552
|
+
for key in (*api_keys, *database_keys):
|
|
553
|
+
if not (os.environ.get(key) or stored.get(key)):
|
|
554
|
+
stored[key] = secrets_module.token_hex(32)
|
|
555
|
+
additions[key] = stored[key]
|
|
556
|
+
existing_content = env_file.read_bytes() if env_file.exists() else b""
|
|
557
|
+
separator = (
|
|
558
|
+
"\n"
|
|
559
|
+
if existing_content and not existing_content.endswith(b"\n") and additions
|
|
560
|
+
else ""
|
|
561
|
+
)
|
|
562
|
+
content = separator + "".join(
|
|
563
|
+
f"{key}={value}\n" for key, value in additions.items()
|
|
564
|
+
)
|
|
565
|
+
fd = os.open(
|
|
566
|
+
env_file, os.O_WRONLY | os.O_CREAT | os.O_APPEND | os.O_NOFOLLOW, 0o600
|
|
567
|
+
)
|
|
568
|
+
with os.fdopen(fd, "w", encoding="utf-8") as stream:
|
|
569
|
+
if os.fstat(stream.fileno()).st_nlink != 1:
|
|
570
|
+
raise ValueError("server.env must be a regular private file")
|
|
571
|
+
os.fchmod(stream.fileno(), 0o600)
|
|
572
|
+
stream.write(content)
|
|
573
|
+
|
|
574
|
+
effective = {
|
|
575
|
+
key: os.environ.get(key) or stored.get(key)
|
|
576
|
+
for key in (
|
|
577
|
+
*api_keys,
|
|
578
|
+
*database_keys,
|
|
579
|
+
"SEDIMENT_API_BEARER_TOKEN",
|
|
580
|
+
"SEDIMENT_INGEST_TOKENS",
|
|
581
|
+
)
|
|
582
|
+
}
|
|
583
|
+
if not bootstrap:
|
|
584
|
+
bootstrap = stack.enter_context(
|
|
585
|
+
managed_postgres(root, effective["SEDIMENT_BOOTSTRAP_PASSWORD"])
|
|
586
|
+
)
|
|
587
|
+
target = make_url(bootstrap)
|
|
588
|
+
provision_database(
|
|
589
|
+
bootstrap,
|
|
590
|
+
migrator_password=effective["SEDIMENT_MIGRATOR_PASSWORD"],
|
|
591
|
+
runtime_password=effective["SEDIMENT_RUNTIME_PASSWORD"],
|
|
592
|
+
operator_password=effective["SEDIMENT_OPERATOR_PASSWORD"],
|
|
593
|
+
)
|
|
594
|
+
os.environ["SEDIMENT_DATABASE_URL"] = target.set(
|
|
595
|
+
drivername="postgresql+psycopg",
|
|
596
|
+
username=RUNTIME_ROLE,
|
|
597
|
+
password=effective["SEDIMENT_RUNTIME_PASSWORD"],
|
|
598
|
+
).render_as_string(hide_password=False)
|
|
599
|
+
# Only runtime connection authority reaches the API and its worker children.
|
|
600
|
+
for key in (
|
|
601
|
+
*database_keys,
|
|
602
|
+
"SEDIMENT_BOOTSTRAP_PASSWORD",
|
|
603
|
+
"SEDIMENT_BOOTSTRAP_DATABASE_URL",
|
|
604
|
+
"SEDIMENT_MIGRATOR_DATABASE_URL",
|
|
605
|
+
"SEDIMENT_OPERATOR_DATABASE_URL",
|
|
606
|
+
):
|
|
607
|
+
os.environ.pop(key, None)
|
|
608
|
+
os.environ.setdefault("SEDIMENT_ORG_ID", stored.get("SEDIMENT_ORG_ID", "default"))
|
|
609
|
+
os.environ.setdefault("SEDIMENT_MIRROR_PATH", str(root / "mirror"))
|
|
610
|
+
for key in (*api_keys, "SEDIMENT_API_BEARER_TOKEN", "SEDIMENT_INGEST_TOKENS"):
|
|
611
|
+
if effective.get(key):
|
|
612
|
+
os.environ[key] = effective[key]
|
|
613
|
+
|
|
614
|
+
url = f"http://{args.host}:{args.port}"
|
|
615
|
+
ui.banner("sediment", f"v{__version__}")
|
|
616
|
+
if additions:
|
|
617
|
+
print(f"Generated server credentials in {env_file}")
|
|
618
|
+
else:
|
|
619
|
+
print(f"Using credentials from {env_file} and explicit environment overrides")
|
|
620
|
+
print(f"Serving on {ui.style(url, 'sandstone', 'bold')}")
|
|
621
|
+
print(f"Next: {ui.style(f'sediment login {url}', 'sandstone')}")
|
|
622
|
+
|
|
623
|
+
del bootstrap, target, effective, stored, additions
|
|
624
|
+
import uvicorn
|
|
625
|
+
|
|
626
|
+
uvicorn.run("sediment_api.main:app", host=args.host, port=args.port)
|
|
627
|
+
return 0
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
def _prompt_token() -> str:
|
|
631
|
+
import getpass
|
|
632
|
+
|
|
633
|
+
return getpass.getpass("Bearer token: ")
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
def _eval_server_token(url: str, *, capture: bool = False) -> str | None:
|
|
637
|
+
"""The token ``sediment server`` generated for *this machine's own* eval
|
|
638
|
+
server, so ``sediment login http://127.0.0.1:8000`` needs nothing pasted
|
|
639
|
+
or piped — the quickstart's whole login step is that one line.
|
|
640
|
+
|
|
641
|
+
Loopback only. The token is a local secret; offering it to whatever
|
|
642
|
+
host the operator typed would hand it to that host. A non-default
|
|
643
|
+
``server --root`` is not searched either.
|
|
644
|
+
# ponytail: default root only — `--with-token` covers every other
|
|
645
|
+
# posture, and a --root flag on login would be config for a value that
|
|
646
|
+
# does not vary in the quickstart.
|
|
647
|
+
"""
|
|
648
|
+
host = urlsplit(url).hostname
|
|
649
|
+
if host is None or not is_loopback_host(host):
|
|
650
|
+
return None
|
|
651
|
+
env = _load_server_env(Path.home() / ".sediment" / "server" / "server.env")
|
|
652
|
+
return (
|
|
653
|
+
env.get("SEDIMENT_API_BEARER_TOKEN" if capture else "SEDIMENT_OPERATOR_TOKEN")
|
|
654
|
+
or None
|
|
655
|
+
)
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
def _store_login(
|
|
659
|
+
url: str,
|
|
660
|
+
token: str,
|
|
661
|
+
me: dict,
|
|
662
|
+
*,
|
|
663
|
+
capture: bool = False,
|
|
664
|
+
capture_identity: tuple[str, dict] | None = None,
|
|
665
|
+
) -> int:
|
|
666
|
+
"""Persist validated credentials and say which org they resolved to."""
|
|
667
|
+
cfg = read_config()
|
|
668
|
+
expected = "ingest" if capture else "operator"
|
|
669
|
+
if me.get("authority") != expected:
|
|
670
|
+
return _fail(
|
|
671
|
+
f"{expected} authority required for this login; use --capture for an ingest credential"
|
|
672
|
+
)
|
|
673
|
+
cfg["current"] = url
|
|
674
|
+
servers = dict(cfg.get("servers") or {})
|
|
675
|
+
existing = servers.get(url)
|
|
676
|
+
entry = dict(existing) if isinstance(existing, dict) else {}
|
|
677
|
+
prefix = "capture_" if capture else ""
|
|
678
|
+
entry.update(
|
|
679
|
+
{
|
|
680
|
+
f"{prefix}token": token,
|
|
681
|
+
f"{prefix}authority": expected,
|
|
682
|
+
f"{prefix}client_id": me["client_id"],
|
|
683
|
+
"org_id": me["org_id"],
|
|
684
|
+
}
|
|
685
|
+
)
|
|
686
|
+
if capture_identity is not None:
|
|
687
|
+
capture_token, capture_me = capture_identity
|
|
688
|
+
if (
|
|
689
|
+
capture_me.get("authority") != "ingest"
|
|
690
|
+
or capture_me.get("org_id") != me["org_id"]
|
|
691
|
+
):
|
|
692
|
+
return _fail(
|
|
693
|
+
"ingest authority for the same org is required for capture enrollment"
|
|
694
|
+
)
|
|
695
|
+
entry.update(
|
|
696
|
+
capture_token=capture_token,
|
|
697
|
+
capture_authority="ingest",
|
|
698
|
+
capture_client_id=capture_me["client_id"],
|
|
699
|
+
)
|
|
700
|
+
servers[url] = entry
|
|
701
|
+
cfg["servers"] = servers
|
|
702
|
+
write_config(cfg)
|
|
703
|
+
action = "capture credential enrolled for" if capture else "logged in to"
|
|
704
|
+
print(f"{ui.glyph('✓', 'phosphor')}{action} {url} (org {me['org_id']})")
|
|
705
|
+
return 0
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
def cmd_login(args: argparse.Namespace) -> int:
|
|
709
|
+
"""Store credentials for a server after live-validating the token via
|
|
710
|
+
GET /v1/me. A 401 re-prompts; a connection failure stops. Refuses an
|
|
711
|
+
environment override for the selected authority because it would silently
|
|
712
|
+
win over the stored credential.
|
|
713
|
+
|
|
714
|
+
Nothing needs answering in the quickstart: a loopback URL reuses
|
|
715
|
+
the eval server's own generated token. Every other server — remote, or
|
|
716
|
+
local under a custom ``--root`` — takes ``--with-token``: one line on
|
|
717
|
+
stdin, no prompt and no re-prompt. getpass is not that path, since it
|
|
718
|
+
reads /dev/tty when one exists, so piping into the prompt hangs."""
|
|
719
|
+
capture = bool(getattr(args, "capture", False))
|
|
720
|
+
override = "SEDIMENT_INGEST_TOKEN" if capture else "SEDIMENT_SESSION_TOKEN"
|
|
721
|
+
if os.environ.get(override):
|
|
722
|
+
return _fail(
|
|
723
|
+
f"{override} is set; unset it first — it would "
|
|
724
|
+
"silently override the stored token"
|
|
725
|
+
)
|
|
726
|
+
url = validate_server_url(args.url)
|
|
727
|
+
generated = None if args.with_token else _eval_server_token(url, capture=capture)
|
|
728
|
+
if generated:
|
|
729
|
+
try:
|
|
730
|
+
me = probe_me(url, generated)
|
|
731
|
+
except ClientError as exc:
|
|
732
|
+
# A stale server.env (the server was restarted under a different
|
|
733
|
+
# token) is the one case worth falling through on: nobody typed
|
|
734
|
+
# this token, so asking for one is the fix. Anything else — an
|
|
735
|
+
# unreachable server above all — is the operator's real error and
|
|
736
|
+
# must not be buried under a prompt.
|
|
737
|
+
if str(exc) != "that's not a valid token":
|
|
738
|
+
return _fail(str(exc))
|
|
739
|
+
else:
|
|
740
|
+
print(ui.style("using the token from ~/.sediment/server/server.env", "dim"))
|
|
741
|
+
capture_identity = None
|
|
742
|
+
if not capture and (capture_token := _eval_server_token(url, capture=True)):
|
|
743
|
+
try:
|
|
744
|
+
capture_identity = (capture_token, probe_me(url, capture_token))
|
|
745
|
+
except ClientError as exc:
|
|
746
|
+
return _fail(f"capture enrollment failed: {exc}")
|
|
747
|
+
return _store_login(
|
|
748
|
+
url, generated, me, capture=capture, capture_identity=capture_identity
|
|
749
|
+
)
|
|
750
|
+
while True:
|
|
751
|
+
if args.with_token:
|
|
752
|
+
token = sys.stdin.readline().strip()
|
|
753
|
+
if not token:
|
|
754
|
+
return _fail("no token on stdin")
|
|
755
|
+
else:
|
|
756
|
+
try:
|
|
757
|
+
token = _prompt_token()
|
|
758
|
+
except EOFError:
|
|
759
|
+
# Piped/closed stdin (a 401 re-prompt with no second line to
|
|
760
|
+
# read): clean error, never a traceback — the module contract.
|
|
761
|
+
return _fail("no token provided")
|
|
762
|
+
try:
|
|
763
|
+
me = probe_me(url, token)
|
|
764
|
+
except ClientError as exc:
|
|
765
|
+
msg = str(exc)
|
|
766
|
+
if msg == "that's not a valid token" and not args.with_token:
|
|
767
|
+
print(ui.error_line(msg), file=sys.stderr)
|
|
768
|
+
continue
|
|
769
|
+
return _fail(msg)
|
|
770
|
+
return _store_login(url, token, me, capture=capture)
|
|
771
|
+
|
|
772
|
+
|
|
773
|
+
def cmd_logout(args: argparse.Namespace) -> int:
|
|
774
|
+
"""Remove one server's stored credentials (``--server``, default the
|
|
775
|
+
current entry), then say what was removed."""
|
|
776
|
+
cfg = read_config()
|
|
777
|
+
servers = dict(cfg.get("servers") or {})
|
|
778
|
+
if args.server:
|
|
779
|
+
url = norm_url(args.server)
|
|
780
|
+
if url not in servers:
|
|
781
|
+
return _fail(f"no stored credentials for {url}")
|
|
782
|
+
else:
|
|
783
|
+
current = cfg.get("current")
|
|
784
|
+
if not isinstance(current, str) or current not in servers:
|
|
785
|
+
return _fail("not logged in")
|
|
786
|
+
url = current
|
|
787
|
+
removed = servers.pop(url)
|
|
788
|
+
cfg["servers"] = servers
|
|
789
|
+
if cfg.get("current") == url:
|
|
790
|
+
cfg["current"] = None
|
|
791
|
+
write_config(cfg)
|
|
792
|
+
print(
|
|
793
|
+
f"{ui.glyph('✓', 'phosphor')}logged out of {url} (org {removed.get('org_id')})"
|
|
794
|
+
)
|
|
795
|
+
return 0
|
|
796
|
+
|
|
797
|
+
|
|
798
|
+
def cmd_commit(args: argparse.Namespace) -> int:
|
|
799
|
+
"""GET /query/commit/{sha} and pretty-print the per-repo attributions,
|
|
800
|
+
decisions, and CI outcomes."""
|
|
801
|
+
selectors = {
|
|
802
|
+
name: getattr(args, name, None)
|
|
803
|
+
for name in (
|
|
804
|
+
"repo",
|
|
805
|
+
"repository_provider",
|
|
806
|
+
"repository_host",
|
|
807
|
+
"repository_id",
|
|
808
|
+
"as_of",
|
|
809
|
+
)
|
|
810
|
+
}
|
|
811
|
+
identity = [
|
|
812
|
+
selectors[name]
|
|
813
|
+
for name in ("repository_provider", "repository_host", "repository_id")
|
|
814
|
+
]
|
|
815
|
+
if any(value is not None for value in identity) and not all(
|
|
816
|
+
value is not None for value in identity
|
|
817
|
+
):
|
|
818
|
+
raise ClientError("repository identity requires all three components")
|
|
819
|
+
if selectors["as_of"] is not None:
|
|
820
|
+
selectors["as_of"] = _parse_scope_time(selectors["as_of"], "as_of").isoformat()
|
|
821
|
+
query = urlencode(
|
|
822
|
+
{name: value for name, value in selectors.items() if value is not None}
|
|
823
|
+
)
|
|
824
|
+
path = f"/query/commit/{args.sha}"
|
|
825
|
+
maybe_warn_version_skew()
|
|
826
|
+
data = get_json(f"{path}?{query}" if query else path)
|
|
827
|
+
sha = data.get("commit_sha", args.sha)
|
|
828
|
+
for reason, count in sorted(data.get("repository_skipped", {}).items()):
|
|
829
|
+
print(f"{reason}: {count}")
|
|
830
|
+
for outcome in data.get("unresolved_ci_outcomes", []):
|
|
831
|
+
print(
|
|
832
|
+
f"unresolved repository: ci {outcome['result']} {outcome['workflow_name']}"
|
|
833
|
+
)
|
|
834
|
+
if not data.get("repos"):
|
|
835
|
+
print(f"{sha}: no attributions")
|
|
836
|
+
return 0
|
|
837
|
+
print(ui.style(sha, "bleached", "bold"))
|
|
838
|
+
for repo in data.get("repos", []):
|
|
839
|
+
identity = repo.get("repository_identity")
|
|
840
|
+
label = repo["repo"]
|
|
841
|
+
if identity is not None:
|
|
842
|
+
label += f" ({identity['provider']} {identity['host']} repository {identity['repository_id']})"
|
|
843
|
+
print(f" {ui.style(label, 'sandstone')}")
|
|
844
|
+
observations = repo.get("observed_sessions", [])
|
|
845
|
+
for session in observations:
|
|
846
|
+
print(f" observed Session: {session['session_id']}")
|
|
847
|
+
if not observations:
|
|
848
|
+
print(" Session observations: unavailable")
|
|
849
|
+
if repo.get("session_commit_unobserved"):
|
|
850
|
+
print(f" session_commit_unobserved: {repo['session_commit_unobserved']}")
|
|
851
|
+
for inference_call in repo.get("inference_calls", []):
|
|
852
|
+
print(
|
|
853
|
+
f" {inference_call['inference_call_id']} "
|
|
854
|
+
+ ui.style(
|
|
855
|
+
f"{inference_call['gateway_provider']}/"
|
|
856
|
+
f"{inference_call['model_provider'] or '-'}/"
|
|
857
|
+
f"{inference_call['model']}",
|
|
858
|
+
"dim",
|
|
859
|
+
)
|
|
860
|
+
+ f" session={inference_call['session_id']} "
|
|
861
|
+
f"attribution={inference_call['attribution_source']} (inferred call/file)"
|
|
862
|
+
)
|
|
863
|
+
print(f" decisions: {repo['decisions']}")
|
|
864
|
+
for outcome in repo.get("ci_outcomes", []):
|
|
865
|
+
color = {"passed": "phosphor", "failed": "iron-oxide"}.get(
|
|
866
|
+
outcome["result"], "dim"
|
|
867
|
+
)
|
|
868
|
+
print(
|
|
869
|
+
f" ci {ui.style(outcome['result'], color)} {outcome['workflow_name']}"
|
|
870
|
+
)
|
|
871
|
+
return 0
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def cmd_quarantine_or_release(
|
|
875
|
+
store: FactStore, org: str, args: argparse.Namespace
|
|
876
|
+
) -> int:
|
|
877
|
+
if args.command == "quarantine":
|
|
878
|
+
store.quarantine_fact(org, args.table, args.fact_id, reason=args.reason)
|
|
879
|
+
print(
|
|
880
|
+
f"{ui.glyph('✓', 'phosphor')}quarantined {args.table}/{args.fact_id} "
|
|
881
|
+
"(reversible: sediment release)"
|
|
882
|
+
)
|
|
883
|
+
else:
|
|
884
|
+
store.release_fact(org, args.table, args.fact_id, reason=args.reason)
|
|
885
|
+
print(f"{ui.glyph('✓', 'phosphor')}released {args.table}/{args.fact_id}")
|
|
886
|
+
return 0
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
def cmd_quarantine_log(store: FactStore, org: str, args: argparse.Namespace) -> int:
|
|
890
|
+
# ponytail: full-history read, tail applied in the CLI — log rows are
|
|
891
|
+
# small; a store-side LIMIT variant if an org's log ever gets huge.
|
|
892
|
+
log = store.read_quarantine_log(org)
|
|
893
|
+
shown = log if args.all else log[-args.tail :]
|
|
894
|
+
for rec in shown:
|
|
895
|
+
color = "sandstone" if rec.action == "quarantine" else "phosphor"
|
|
896
|
+
print(
|
|
897
|
+
ui.style(rec.recorded_at.isoformat(), "dim")
|
|
898
|
+
+ f" {ui.style(f'{rec.action:<10}', color)} "
|
|
899
|
+
f"{rec.fact_table}/{rec.fact_id} {rec.reason}"
|
|
900
|
+
)
|
|
901
|
+
if len(shown) < len(log):
|
|
902
|
+
print(
|
|
903
|
+
ui.style(
|
|
904
|
+
f"... showing last {len(shown)} of {len(log)} rows (--all for all)",
|
|
905
|
+
"dim",
|
|
906
|
+
)
|
|
907
|
+
)
|
|
908
|
+
revision = _styled_revision(store.quarantine_revision(org))
|
|
909
|
+
print(f"{len(log)} rows; quarantine_revision: {revision}")
|
|
910
|
+
return 0
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
def _resolve_quarantine_inference_filters(
|
|
914
|
+
args: argparse.Namespace,
|
|
915
|
+
) -> tuple[str | None, tuple[datetime, datetime] | None]:
|
|
916
|
+
provider = None
|
|
917
|
+
if args.provider is not None:
|
|
918
|
+
try:
|
|
919
|
+
provider = GatewayProvider(args.provider).value
|
|
920
|
+
except ValueError:
|
|
921
|
+
choices = ", ".join(item.value for item in GatewayProvider)
|
|
922
|
+
raise ValueError(f"--provider must be one of: {choices}") from None
|
|
923
|
+
between = None
|
|
924
|
+
if args.between:
|
|
925
|
+
try:
|
|
926
|
+
lo, hi = (datetime.fromisoformat(t) for t in args.between)
|
|
927
|
+
except ValueError as exc:
|
|
928
|
+
raise ValueError(f"--between wants two ISO-8601 datetimes: {exc}") from None
|
|
929
|
+
if lo.tzinfo is None or hi.tzinfo is None:
|
|
930
|
+
raise ValueError("--between bounds must be timezone-aware")
|
|
931
|
+
if lo > hi:
|
|
932
|
+
raise ValueError(f"--between bounds are reversed ({lo} > {hi})")
|
|
933
|
+
between = (lo, hi)
|
|
934
|
+
return provider, between
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
def cmd_quarantine_inference_calls(
|
|
938
|
+
store: FactStore, org: str, args: argparse.Namespace
|
|
939
|
+
) -> int:
|
|
940
|
+
provider, between = args._quarantine_inference_filters
|
|
941
|
+
n = store.quarantine_inference_calls_where(
|
|
942
|
+
org,
|
|
943
|
+
captured_between=between,
|
|
944
|
+
session_id=args.session_id,
|
|
945
|
+
provider=provider,
|
|
946
|
+
reason=args.reason,
|
|
947
|
+
dry_run=not args.apply,
|
|
948
|
+
)
|
|
949
|
+
if args.apply:
|
|
950
|
+
print(f"{ui.glyph('✓', 'phosphor')}quarantined {n} model-call facts")
|
|
951
|
+
else:
|
|
952
|
+
print(
|
|
953
|
+
f"{n} model-call facts would be quarantined "
|
|
954
|
+
+ ui.style("(dry run; add --apply)", "sandstone")
|
|
955
|
+
)
|
|
956
|
+
return 0
|
|
957
|
+
|
|
958
|
+
|
|
959
|
+
def _mirrors(mirror_path: str | None = None) -> MirrorManager | None:
|
|
960
|
+
"""The mirror store every export reads, or None once a missing
|
|
961
|
+
SEDIMENT_MIRROR_PATH has been reported."""
|
|
962
|
+
if not mirror_path:
|
|
963
|
+
_fail("SEDIMENT_MIRROR_PATH is not set; the export needs mirrors")
|
|
964
|
+
return None
|
|
965
|
+
return MirrorManager(mirror_path)
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def _print_written(written: dict) -> None:
|
|
969
|
+
"""The export payoff lines, shared by every export verb so ✓/dim reads
|
|
970
|
+
the same everywhere."""
|
|
971
|
+
if written:
|
|
972
|
+
for path, count in written.items():
|
|
973
|
+
print(f"{ui.glyph('✓', 'phosphor')}wrote {path} ({count} rows)")
|
|
974
|
+
else:
|
|
975
|
+
print(
|
|
976
|
+
ui.style(
|
|
977
|
+
"nothing written (empty projections leave existing files untouched)",
|
|
978
|
+
"dim",
|
|
979
|
+
)
|
|
980
|
+
)
|
|
981
|
+
|
|
982
|
+
|
|
983
|
+
def _default_derivation_policy() -> DerivationPolicy:
|
|
984
|
+
return DerivationPolicy()
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
def _parse_scope_time(value: str | None, name: str) -> datetime | None:
|
|
988
|
+
if value is None:
|
|
989
|
+
return None
|
|
990
|
+
try:
|
|
991
|
+
parsed = datetime.fromisoformat(value)
|
|
992
|
+
except ValueError as exc:
|
|
993
|
+
raise ValueError(f"{name} must be an RFC 3339 timestamp") from exc
|
|
994
|
+
if parsed.tzinfo is None or parsed.utcoffset() is None:
|
|
995
|
+
raise ValueError(f"{name} must be timezone-aware")
|
|
996
|
+
return parsed
|
|
997
|
+
|
|
998
|
+
|
|
999
|
+
def _prepare_store_command(args: argparse.Namespace) -> None:
|
|
1000
|
+
"""Validate store-command semantics before settings or connectivity."""
|
|
1001
|
+
if args.command in {"quarantine", "release", "quarantine-inference-calls"}:
|
|
1002
|
+
# Use the fact model's canonical audit-reason validator at the CLI
|
|
1003
|
+
# trust boundary without constructing or persisting a placeholder fact.
|
|
1004
|
+
QuarantineRecord._reason_required(args.reason)
|
|
1005
|
+
if args.command in {"quarantine", "release"}:
|
|
1006
|
+
TypeAdapter(NonEmptyId).validate_python(args.fact_id)
|
|
1007
|
+
if args.command == "quarantine-log" and not args.all and args.tail < 1:
|
|
1008
|
+
raise ValueError("--tail must be >= 1")
|
|
1009
|
+
if args.command == "quarantine-inference-calls":
|
|
1010
|
+
if args.session_id is not None:
|
|
1011
|
+
TypeAdapter(NonEmptyId).validate_python(args.session_id)
|
|
1012
|
+
filters = _resolve_quarantine_inference_filters(args)
|
|
1013
|
+
args._quarantine_inference_filters = filters
|
|
1014
|
+
provider, between = filters
|
|
1015
|
+
if not (between or args.session_id or provider) and not args.all:
|
|
1016
|
+
raise ValueError(
|
|
1017
|
+
"no filters given — this would quarantine the org's every "
|
|
1018
|
+
"model-call fact; pass --all if that is what you mean"
|
|
1019
|
+
)
|
|
1020
|
+
if args.command == "derive":
|
|
1021
|
+
if args.sample < 0:
|
|
1022
|
+
raise ValueError("--sample must be zero or greater")
|
|
1023
|
+
policy = (
|
|
1024
|
+
load_derivation_policy(Path(args.policy))
|
|
1025
|
+
if args.policy
|
|
1026
|
+
else _default_derivation_policy()
|
|
1027
|
+
)
|
|
1028
|
+
scope = DerivationScope(
|
|
1029
|
+
since=_parse_scope_time(args.since, "--since"),
|
|
1030
|
+
until=_parse_scope_time(args.until, "--until"),
|
|
1031
|
+
users=tuple(args.users) if args.users is not None else None,
|
|
1032
|
+
)
|
|
1033
|
+
args._derivation = (policy, scope)
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
def cmd_derive(store: FactStore, org: str, args: argparse.Namespace) -> int:
|
|
1037
|
+
mirrors = _mirrors(getattr(args, "_mirror_path", None))
|
|
1038
|
+
if mirrors is None:
|
|
1039
|
+
return 1
|
|
1040
|
+
policy, scope = args._derivation
|
|
1041
|
+
destination = Path(args.out)
|
|
1042
|
+
with build_derived_bundle_context(
|
|
1043
|
+
store, mirrors, org, policy=policy, scope=scope
|
|
1044
|
+
) as bundle:
|
|
1045
|
+
try:
|
|
1046
|
+
write_derived_bundle(bundle, destination)
|
|
1047
|
+
except OSError as exc:
|
|
1048
|
+
return _fail(str(exc))
|
|
1049
|
+
print(f"{ui.glyph('✓', 'phosphor')}wrote {destination}")
|
|
1050
|
+
print(f"attributed completions: {len(bundle.attributed_completions)}")
|
|
1051
|
+
print(f"rollouts: {len(bundle.rollouts)}")
|
|
1052
|
+
print(f"referenced inference calls: {len(bundle.inference_calls)}")
|
|
1053
|
+
print(f"skipped: {dict(bundle.skipped)}")
|
|
1054
|
+
print(f"excluded: {dict(bundle.excluded)}")
|
|
1055
|
+
print(f"fragmented: {dict(bundle.fragmented)}")
|
|
1056
|
+
print(f"policy digest: {bundle.policy.digest}")
|
|
1057
|
+
print(f"as of: {bundle.as_of.isoformat() if bundle.as_of else 'none'}")
|
|
1058
|
+
for name in (
|
|
1059
|
+
"manifest.json",
|
|
1060
|
+
"attributed_completions.jsonl",
|
|
1061
|
+
"rollouts.jsonl",
|
|
1062
|
+
"inference_calls.jsonl",
|
|
1063
|
+
"inference_call_identities.jsonl",
|
|
1064
|
+
"repository_identities.jsonl",
|
|
1065
|
+
"repository_renames.jsonl",
|
|
1066
|
+
):
|
|
1067
|
+
with (destination / name).open("rb") as handle:
|
|
1068
|
+
digest = hashlib.file_digest(handle, "sha256").hexdigest()
|
|
1069
|
+
print(f"sha256: {name} {digest}")
|
|
1070
|
+
if args.sample:
|
|
1071
|
+
for name in ("attributed_completions.jsonl", "rollouts.jsonl"):
|
|
1072
|
+
with (destination / name).open(encoding="utf-8") as handle:
|
|
1073
|
+
for line in islice(handle, args.sample):
|
|
1074
|
+
print(f"sample {name}: {line}", end="")
|
|
1075
|
+
return 0
|
|
1076
|
+
|
|
1077
|
+
|
|
1078
|
+
def _attributed_completion_pipeline(
|
|
1079
|
+
store: FactStore | None,
|
|
1080
|
+
mirrors: MirrorManager | None,
|
|
1081
|
+
org: str | None,
|
|
1082
|
+
out_dir: str,
|
|
1083
|
+
base_name: str,
|
|
1084
|
+
*,
|
|
1085
|
+
label: str,
|
|
1086
|
+
extra: dict | None = None,
|
|
1087
|
+
from_bundle: str | None = None,
|
|
1088
|
+
profile: str | None = None,
|
|
1089
|
+
) -> int:
|
|
1090
|
+
"""Project complete evidence groups from a validated, live bundle context."""
|
|
1091
|
+
from sediment_export.bounded_training import project_training_bundle
|
|
1092
|
+
|
|
1093
|
+
with ExitStack() as contexts:
|
|
1094
|
+
if from_bundle is not None:
|
|
1095
|
+
bundle = contexts.enter_context(open_derived_bundle(Path(from_bundle)))
|
|
1096
|
+
else:
|
|
1097
|
+
if store is None or mirrors is None or org is None:
|
|
1098
|
+
raise ValueError("direct export requires the fact store and mirrors")
|
|
1099
|
+
bundle = contexts.enter_context(
|
|
1100
|
+
build_derived_bundle_context(
|
|
1101
|
+
store, mirrors, org, policy=_default_derivation_policy()
|
|
1102
|
+
)
|
|
1103
|
+
)
|
|
1104
|
+
projection = contexts.enter_context(
|
|
1105
|
+
project_training_bundle(
|
|
1106
|
+
bundle, objective=base_name.replace("_", "-"), **(extra or {})
|
|
1107
|
+
)
|
|
1108
|
+
)
|
|
1109
|
+
split_enabled = bundle.policy.eval_fraction > 0
|
|
1110
|
+
if profile is not None:
|
|
1111
|
+
from sediment_export.compatibility import write_compatible_export
|
|
1112
|
+
|
|
1113
|
+
# The consumer adapter reads the staged rows within the open context;
|
|
1114
|
+
# its own published view is the profile's declared envelope.
|
|
1115
|
+
summary = write_compatible_export(
|
|
1116
|
+
projection.rows,
|
|
1117
|
+
out_dir,
|
|
1118
|
+
profile,
|
|
1119
|
+
split_enabled=split_enabled,
|
|
1120
|
+
canonical_skipped=projection.skipped,
|
|
1121
|
+
)
|
|
1122
|
+
written = summary["written"]
|
|
1123
|
+
print(f"compatibility profile: {profile}")
|
|
1124
|
+
print(f"dataset diagnostics: {summary['diagnostics']}")
|
|
1125
|
+
else:
|
|
1126
|
+
result = write_jsonl(
|
|
1127
|
+
projection.rows,
|
|
1128
|
+
Path(out_dir) / f"{base_name}.jsonl",
|
|
1129
|
+
split_enabled=split_enabled,
|
|
1130
|
+
max_bytes=projection.remaining_bytes,
|
|
1131
|
+
)
|
|
1132
|
+
written = result.written
|
|
1133
|
+
count_str = ui.style(str(len(projection.rows)), "bleached", "bold")
|
|
1134
|
+
print(f"{label} projected: {count_str} skipped: {dict(projection.skipped)}")
|
|
1135
|
+
_print_written(written)
|
|
1136
|
+
return 0
|
|
1137
|
+
|
|
1138
|
+
|
|
1139
|
+
def cmd_export_rlvr(
|
|
1140
|
+
store: FactStore | None, org: str | None, args: argparse.Namespace
|
|
1141
|
+
) -> int:
|
|
1142
|
+
profile_name = getattr(args, "profile", None)
|
|
1143
|
+
if profile_name is not None:
|
|
1144
|
+
from sediment_export.compatibility import get_profile, require_dependencies
|
|
1145
|
+
|
|
1146
|
+
profile = get_profile(profile_name, objective="rlvr")
|
|
1147
|
+
if profile.consumer != args.target:
|
|
1148
|
+
raise ValueError("profile and --target must name the same consumer")
|
|
1149
|
+
require_dependencies(profile)
|
|
1150
|
+
elif getattr(args, "consumer_config", None):
|
|
1151
|
+
raise ValueError("--consumer-config requires --profile")
|
|
1152
|
+
with ExitStack() as contexts:
|
|
1153
|
+
bundle = (
|
|
1154
|
+
contexts.enter_context(open_derived_bundle(Path(args.from_bundle)))
|
|
1155
|
+
if args.from_bundle
|
|
1156
|
+
else None
|
|
1157
|
+
)
|
|
1158
|
+
needs_mirror = bundle is None or args.target != "nemo-gym"
|
|
1159
|
+
mirrors = (
|
|
1160
|
+
_mirrors(getattr(args, "_mirror_path", None)) if needs_mirror else None
|
|
1161
|
+
)
|
|
1162
|
+
if needs_mirror and mirrors is None:
|
|
1163
|
+
return 1
|
|
1164
|
+
if profile_name is not None:
|
|
1165
|
+
from sediment_export.consumer_rlvr import (
|
|
1166
|
+
export_rlvr_profile,
|
|
1167
|
+
load_consumer_settings,
|
|
1168
|
+
)
|
|
1169
|
+
|
|
1170
|
+
if bundle is None:
|
|
1171
|
+
bundle = contexts.enter_context(
|
|
1172
|
+
build_derived_bundle_context(
|
|
1173
|
+
store, mirrors, org, policy=_default_derivation_policy()
|
|
1174
|
+
)
|
|
1175
|
+
)
|
|
1176
|
+
# The consumer adapter reads the validated file-backed bundle within
|
|
1177
|
+
# this context; its own hydrated view is the profile's envelope.
|
|
1178
|
+
summary = export_rlvr_profile(
|
|
1179
|
+
bundle,
|
|
1180
|
+
mirrors,
|
|
1181
|
+
args.out,
|
|
1182
|
+
profile_name,
|
|
1183
|
+
load_consumer_settings(getattr(args, "consumer_config", None)),
|
|
1184
|
+
)
|
|
1185
|
+
print(f"compatibility profile: {profile_name}")
|
|
1186
|
+
print(f"rows: {summary['rows']} skipped: {summary['skipped']}")
|
|
1187
|
+
print(f"canonical skipped: {summary['canonical_skipped']}")
|
|
1188
|
+
print(f"fragmented: {summary['fragmented']}")
|
|
1189
|
+
print(f"dataset diagnostics: {summary['diagnostics']}")
|
|
1190
|
+
_print_written(summary["written"])
|
|
1191
|
+
return 0
|
|
1192
|
+
if bundle is not None:
|
|
1193
|
+
summary = export_rlvr_from_bundle(
|
|
1194
|
+
bundle,
|
|
1195
|
+
mirrors,
|
|
1196
|
+
args.out,
|
|
1197
|
+
target=args.target,
|
|
1198
|
+
)
|
|
1199
|
+
else:
|
|
1200
|
+
summary = export_rlvr(store, mirrors, org, args.out, target=args.target)
|
|
1201
|
+
print(f"rollouts derived: {ui.style(str(summary['rollouts']), 'bleached', 'bold')}")
|
|
1202
|
+
print(f"task rows: {summary['task_rows']} skipped: {summary['task_skipped']}")
|
|
1203
|
+
print(
|
|
1204
|
+
f"rollout rows: {summary['rollout_rows']} "
|
|
1205
|
+
f"skipped: {summary['rollout_skipped']}"
|
|
1206
|
+
)
|
|
1207
|
+
print(f"fragmented: {summary['fragmented']}")
|
|
1208
|
+
_print_written(summary["written"])
|
|
1209
|
+
if summary["environment_manifest"]:
|
|
1210
|
+
print(
|
|
1211
|
+
f"{ui.glyph('✓', 'phosphor')}wrote {summary['environment_manifest']} "
|
|
1212
|
+
"(experimental taskset manifest)"
|
|
1213
|
+
)
|
|
1214
|
+
return 0
|
|
1215
|
+
|
|
1216
|
+
|
|
1217
|
+
def cmd_export_dpo(
|
|
1218
|
+
store: FactStore | None, org: str | None, args: argparse.Namespace
|
|
1219
|
+
) -> int:
|
|
1220
|
+
profile_name = getattr(args, "profile", None)
|
|
1221
|
+
if profile_name is not None:
|
|
1222
|
+
from sediment_export.compatibility import get_profile, require_dependencies
|
|
1223
|
+
|
|
1224
|
+
require_dependencies(get_profile(profile_name, objective="dpo"))
|
|
1225
|
+
mirrors = (
|
|
1226
|
+
None if args.from_bundle else _mirrors(getattr(args, "_mirror_path", None))
|
|
1227
|
+
)
|
|
1228
|
+
if not args.from_bundle and mirrors is None:
|
|
1229
|
+
return 1
|
|
1230
|
+
return _attributed_completion_pipeline(
|
|
1231
|
+
store,
|
|
1232
|
+
mirrors,
|
|
1233
|
+
org,
|
|
1234
|
+
args.out,
|
|
1235
|
+
"dpo",
|
|
1236
|
+
label="pairs",
|
|
1237
|
+
extra={"policy": DPOPolicy(recipe_id=args.recipe)},
|
|
1238
|
+
from_bundle=args.from_bundle,
|
|
1239
|
+
profile=profile_name,
|
|
1240
|
+
)
|
|
1241
|
+
|
|
1242
|
+
|
|
1243
|
+
def cmd_export_sft(
|
|
1244
|
+
store: FactStore | None, org: str | None, args: argparse.Namespace
|
|
1245
|
+
) -> int:
|
|
1246
|
+
profile_name = getattr(args, "profile", None)
|
|
1247
|
+
if profile_name is not None:
|
|
1248
|
+
from sediment_export.compatibility import get_profile, require_dependencies
|
|
1249
|
+
|
|
1250
|
+
require_dependencies(get_profile(profile_name, objective="sft"))
|
|
1251
|
+
mirrors = (
|
|
1252
|
+
None if args.from_bundle else _mirrors(getattr(args, "_mirror_path", None))
|
|
1253
|
+
)
|
|
1254
|
+
if not args.from_bundle and mirrors is None:
|
|
1255
|
+
return 1
|
|
1256
|
+
return _attributed_completion_pipeline(
|
|
1257
|
+
store,
|
|
1258
|
+
mirrors,
|
|
1259
|
+
org,
|
|
1260
|
+
args.out,
|
|
1261
|
+
"sft",
|
|
1262
|
+
label="samples",
|
|
1263
|
+
extra={"policy": SFTPolicy(recipe_id=args.recipe)},
|
|
1264
|
+
from_bundle=args.from_bundle,
|
|
1265
|
+
profile=profile_name,
|
|
1266
|
+
)
|
|
1267
|
+
|
|
1268
|
+
|
|
1269
|
+
def cmd_export_diff_sft(
|
|
1270
|
+
store: FactStore | None, org: str | None, args: argparse.Namespace
|
|
1271
|
+
) -> int:
|
|
1272
|
+
mirrors = _mirrors(getattr(args, "_mirror_path", None))
|
|
1273
|
+
if mirrors is None:
|
|
1274
|
+
return 1
|
|
1275
|
+
return _attributed_completion_pipeline(
|
|
1276
|
+
store,
|
|
1277
|
+
mirrors,
|
|
1278
|
+
org,
|
|
1279
|
+
args.out,
|
|
1280
|
+
"diff_sft",
|
|
1281
|
+
label="samples",
|
|
1282
|
+
extra={"mirrors": mirrors, "policy": SFTPolicy(recipe_id=args.recipe)},
|
|
1283
|
+
from_bundle=args.from_bundle,
|
|
1284
|
+
)
|
|
1285
|
+
|
|
1286
|
+
|
|
1287
|
+
def cmd_export_recovery(store: FactStore, org: str, args: argparse.Namespace) -> int:
|
|
1288
|
+
mirrors = _mirrors(getattr(args, "_mirror_path", None))
|
|
1289
|
+
if mirrors is None:
|
|
1290
|
+
return 1
|
|
1291
|
+
fraction = DerivationPolicy().eval_fraction
|
|
1292
|
+
split_enabled = fraction > 0
|
|
1293
|
+
with store.read_snapshot() as snapshot:
|
|
1294
|
+
derivation = derive_recovery_result(snapshot, mirrors, org)
|
|
1295
|
+
inference_calls = {
|
|
1296
|
+
inference_fact_id(call): call for call in snapshot.read_inference_calls(org)
|
|
1297
|
+
}
|
|
1298
|
+
projection = project_recovery(derivation.pairs, inference_calls, fraction)
|
|
1299
|
+
rows = recovery_to_export_rows(projection.rows)
|
|
1300
|
+
result = write_jsonl(
|
|
1301
|
+
rows, Path(args.out) / "recovery.jsonl", split_enabled=split_enabled
|
|
1302
|
+
)
|
|
1303
|
+
pairs = ui.style(str(len(derivation.pairs)), "bleached", "bold")
|
|
1304
|
+
print(f"pairs derived: {pairs} skipped: {dict(derivation.skipped)}")
|
|
1305
|
+
print(f"recovery rows: {len(projection.rows)} skipped: {dict(projection.skipped)}")
|
|
1306
|
+
_print_written(result.written)
|
|
1307
|
+
return 0
|
|
1308
|
+
|
|
1309
|
+
|
|
1310
|
+
class _CoderFormatter(argparse.HelpFormatter):
|
|
1311
|
+
"""Coder-style help, pinned to an 80-column wrap so golden --help files
|
|
1312
|
+
stay deterministic across terminals and CI: a ``USAGE:``
|
|
1313
|
+
block, UPPERCASE section headings, subcommand rows without the
|
|
1314
|
+
``{a,b,c}`` metavar header, descriptions indented under usage. The
|
|
1315
|
+
layout is identical piped or interactive — color is a separate
|
|
1316
|
+
TTY-gated pass (``ui.style_help``) in ``_ArgumentParser.format_help``."""
|
|
1317
|
+
|
|
1318
|
+
_SECTIONS = {"positional arguments": "arguments"}
|
|
1319
|
+
|
|
1320
|
+
def __init__(self, *args, **kwargs):
|
|
1321
|
+
kwargs.setdefault("width", 80)
|
|
1322
|
+
super().__init__(*args, **kwargs)
|
|
1323
|
+
|
|
1324
|
+
def _format_usage(self, usage, actions, groups, prefix):
|
|
1325
|
+
bare = super()._format_usage(usage, actions, groups, "").strip("\n")
|
|
1326
|
+
indented = "\n".join(f" {line}" for line in bare.splitlines())
|
|
1327
|
+
return f"USAGE:\n{indented}\n\n"
|
|
1328
|
+
|
|
1329
|
+
def start_section(self, heading):
|
|
1330
|
+
if heading:
|
|
1331
|
+
heading = self._SECTIONS.get(heading, heading).upper()
|
|
1332
|
+
super().start_section(heading)
|
|
1333
|
+
|
|
1334
|
+
def _format_text(self, text):
|
|
1335
|
+
# Descriptions sit indented under the USAGE block, coder-style.
|
|
1336
|
+
if "%(prog)" in text:
|
|
1337
|
+
text = text % {"prog": self._prog}
|
|
1338
|
+
return self._fill_text(text, self._width - 2, " ") + "\n\n"
|
|
1339
|
+
|
|
1340
|
+
def _format_action(self, action):
|
|
1341
|
+
if isinstance(action, argparse._SubParsersAction):
|
|
1342
|
+
# Subcommand rows render directly at section indent — no
|
|
1343
|
+
# "{report,mirror-gc,...}" header row above them.
|
|
1344
|
+
return "".join(self._format_action(sub) for sub in action._get_subactions())
|
|
1345
|
+
return super()._format_action(action)
|
|
1346
|
+
|
|
1347
|
+
|
|
1348
|
+
def _propagate_descriptions(parser: argparse.ArgumentParser) -> None:
|
|
1349
|
+
"""Coder-style per-command help: a subcommand without an explicit
|
|
1350
|
+
description reuses its one-line listing help under its USAGE block."""
|
|
1351
|
+
for action in parser._actions:
|
|
1352
|
+
if not isinstance(action, argparse._SubParsersAction):
|
|
1353
|
+
continue
|
|
1354
|
+
help_by_name = {p.dest: p.help for p in action._choices_actions}
|
|
1355
|
+
for name, child in action.choices.items():
|
|
1356
|
+
if child.description is None:
|
|
1357
|
+
child.description = help_by_name.get(name)
|
|
1358
|
+
_propagate_descriptions(child)
|
|
1359
|
+
|
|
1360
|
+
|
|
1361
|
+
def _subparsers_action(
|
|
1362
|
+
parser: argparse.ArgumentParser,
|
|
1363
|
+
) -> argparse._SubParsersAction | None:
|
|
1364
|
+
"""The subparsers action on *parser*, or None for a leaf command."""
|
|
1365
|
+
for action in parser._actions:
|
|
1366
|
+
if isinstance(action, argparse._SubParsersAction):
|
|
1367
|
+
return action
|
|
1368
|
+
return None
|
|
1369
|
+
|
|
1370
|
+
|
|
1371
|
+
def _subcommand_display_name(action: argparse._SubParsersAction) -> str:
|
|
1372
|
+
"""The name argparse prints for a missing/invalid subcommand — the
|
|
1373
|
+
subparsers ``metavar`` (``<command>``/``<format>``), falling back to
|
|
1374
|
+
``dest`` the way ``argparse._get_action_name`` does."""
|
|
1375
|
+
if action.metavar not in (None, argparse.SUPPRESS):
|
|
1376
|
+
return action.metavar
|
|
1377
|
+
return action.dest
|
|
1378
|
+
|
|
1379
|
+
|
|
1380
|
+
class _ArgumentParser(argparse.ArgumentParser):
|
|
1381
|
+
"""ArgumentParser whose subparsers inherit the coder-style formatter
|
|
1382
|
+
(``add_subparsers`` defaults ``parser_class`` to ``type(self)``) and
|
|
1383
|
+
whose help gets the TTY-gated color pass — plain text everywhere else,
|
|
1384
|
+
so the goldens capture exactly what a pipe sees."""
|
|
1385
|
+
|
|
1386
|
+
def __init__(self, *args, **kwargs):
|
|
1387
|
+
kwargs.setdefault("formatter_class", _CoderFormatter)
|
|
1388
|
+
super().__init__(*args, **kwargs)
|
|
1389
|
+
|
|
1390
|
+
def error(self, message):
|
|
1391
|
+
# A missing or unknown subcommand is the one case where the full
|
|
1392
|
+
# listing IS the answer to "what do I type next?" — print it to
|
|
1393
|
+
# stderr before the one-line diagnostic (stdout stays empty, exit 2).
|
|
1394
|
+
# A bad flag on a leaf command keeps argparse's terse usage line.
|
|
1395
|
+
# This distinguishes the two by matching argparse's own message
|
|
1396
|
+
# text, which is not a stable contract; test_cli_help.py pins the
|
|
1397
|
+
# split on both sides.
|
|
1398
|
+
if (
|
|
1399
|
+
self.prog == "sediment export rlvr"
|
|
1400
|
+
and message.startswith("the following arguments are required:")
|
|
1401
|
+
and "--target" in message
|
|
1402
|
+
):
|
|
1403
|
+
self.print_usage(sys.stderr)
|
|
1404
|
+
self.exit(
|
|
1405
|
+
2,
|
|
1406
|
+
f"{self.prog}: error: {message}; --target choices: sediment, "
|
|
1407
|
+
"swe-bench, or nemo-gym; see docs/exports/rlvr-export.md\n",
|
|
1408
|
+
)
|
|
1409
|
+
|
|
1410
|
+
sub = _subparsers_action(self)
|
|
1411
|
+
if sub is not None:
|
|
1412
|
+
name = _subcommand_display_name(sub)
|
|
1413
|
+
if message == f"the following arguments are required: {name}":
|
|
1414
|
+
self._print_message(self.format_help(sys.stderr), sys.stderr)
|
|
1415
|
+
self.exit(2, f"{self.prog}: error: {message}\n")
|
|
1416
|
+
elif message.startswith(f"argument {name}: invalid choice: "):
|
|
1417
|
+
# Drop the ``(choose from 'a', 'b', …)`` tail — the listing
|
|
1418
|
+
# now sits directly above — and name the thing by what this
|
|
1419
|
+
# level calls it: a command up top, a format under `export`.
|
|
1420
|
+
value = message.removeprefix(
|
|
1421
|
+
f"argument {name}: invalid choice: "
|
|
1422
|
+
).rsplit(" (choose from ", 1)[0]
|
|
1423
|
+
self._print_message(self.format_help(sys.stderr), sys.stderr)
|
|
1424
|
+
noun = name.strip("<>")
|
|
1425
|
+
self.exit(2, f"{self.prog}: error: invalid {noun} {value}\n")
|
|
1426
|
+
self.print_usage(sys.stderr)
|
|
1427
|
+
self.exit(2, f"{self.prog}: error: {message}\n")
|
|
1428
|
+
|
|
1429
|
+
def format_help(self, stream=None):
|
|
1430
|
+
# Coder's section order: COMMANDS/FORMATS listings above OPTIONS.
|
|
1431
|
+
# Stable sort, idempotent — only the default options group moves.
|
|
1432
|
+
# *stream* is where the text is headed: the color pass is TTY-gated
|
|
1433
|
+
# on it, so help written to a redirected stderr stays plain. argparse
|
|
1434
|
+
# calls this with no argument, which gates on stdout as before.
|
|
1435
|
+
self._action_groups.sort(key=lambda g: g.title == "options")
|
|
1436
|
+
text = super().format_help()
|
|
1437
|
+
if self.prog == "sediment":
|
|
1438
|
+
title = ui.style(
|
|
1439
|
+
f"{self.prog} v{__version__}", "bleached", "bold", stream=stream
|
|
1440
|
+
)
|
|
1441
|
+
tagline = ui.style(_TAGLINE, "dim", stream=stream)
|
|
1442
|
+
text = f"{title} — {tagline}\n\n{text}"
|
|
1443
|
+
return ui.style_help(text, stream=stream)
|
|
1444
|
+
|
|
1445
|
+
|
|
1446
|
+
def _print_dispatched_help(parser: argparse.ArgumentParser, *, prog: str) -> int:
|
|
1447
|
+
"""Render a forwarded command through the public CLI help contract.
|
|
1448
|
+
|
|
1449
|
+
Report, mirror-GC, and attribution modules keep standalone parsers because
|
|
1450
|
+
their execution seams have different dependency constraints. The installed
|
|
1451
|
+
command still owns their public path and presentation.
|
|
1452
|
+
"""
|
|
1453
|
+
parser.prog = prog
|
|
1454
|
+
parser.formatter_class = _CoderFormatter
|
|
1455
|
+
parser._action_groups.sort(key=lambda group: group.title == "options")
|
|
1456
|
+
print(ui.style_help(parser.format_help(), stream=sys.stdout), end="")
|
|
1457
|
+
return 0
|
|
1458
|
+
|
|
1459
|
+
|
|
1460
|
+
def _dpo_profile_name(name: str) -> str:
|
|
1461
|
+
"""Keep retired profile migration guidance at the argument boundary."""
|
|
1462
|
+
from sediment_export.compatibility import CompatibilityError, get_profile
|
|
1463
|
+
|
|
1464
|
+
try:
|
|
1465
|
+
return get_profile(name, objective="dpo").id
|
|
1466
|
+
except CompatibilityError as exc:
|
|
1467
|
+
raise argparse.ArgumentTypeError(str(exc)) from exc
|
|
1468
|
+
|
|
1469
|
+
|
|
1470
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
1471
|
+
"""The argparse tree, extracted so the golden --help test can walk it."""
|
|
1472
|
+
parser = _ArgumentParser(
|
|
1473
|
+
prog="sediment",
|
|
1474
|
+
usage="sediment <command> [options]",
|
|
1475
|
+
description=__doc__.splitlines()[1],
|
|
1476
|
+
)
|
|
1477
|
+
parser.add_argument(
|
|
1478
|
+
"--version", action="version", version=f"%(prog)s {__version__}"
|
|
1479
|
+
)
|
|
1480
|
+
# prog passed explicitly: argparse otherwise derives it by formatting the
|
|
1481
|
+
# parent usage through the formatter, which would drag the USAGE: heading
|
|
1482
|
+
# into every subcommand's prog.
|
|
1483
|
+
sub = parser.add_subparsers(
|
|
1484
|
+
dest="command",
|
|
1485
|
+
required=True,
|
|
1486
|
+
title="commands",
|
|
1487
|
+
metavar="<command>",
|
|
1488
|
+
prog=parser.prog,
|
|
1489
|
+
)
|
|
1490
|
+
|
|
1491
|
+
# Stubs for --help only — the dispatch in main() runs first.
|
|
1492
|
+
sub.add_parser("report", help="read-only reports (sediment report --help)")
|
|
1493
|
+
sub.add_parser(
|
|
1494
|
+
"delivery", help="prepared capture delivery (sediment delivery --help)"
|
|
1495
|
+
)
|
|
1496
|
+
sub.add_parser(
|
|
1497
|
+
"mirror-gc",
|
|
1498
|
+
help="remove mirrors past push retention (dry-run; --apply deletes)",
|
|
1499
|
+
)
|
|
1500
|
+
|
|
1501
|
+
p_facts = sub.add_parser(
|
|
1502
|
+
"facts", help="fact counts per table (remote; --database-url for direct)"
|
|
1503
|
+
)
|
|
1504
|
+
p_facts.add_argument(
|
|
1505
|
+
"--database-url", help="read PostgreSQL directly instead of the API"
|
|
1506
|
+
)
|
|
1507
|
+
p_facts.set_defaults(func=cmd_facts)
|
|
1508
|
+
|
|
1509
|
+
p_demo = sub.add_parser(
|
|
1510
|
+
"demo", help="plant one synthetic session so facts is non-zero"
|
|
1511
|
+
)
|
|
1512
|
+
p_demo.add_argument(
|
|
1513
|
+
"--force",
|
|
1514
|
+
action="store_true",
|
|
1515
|
+
help="allow a non-loopback server (writes synthetic facts to it)",
|
|
1516
|
+
)
|
|
1517
|
+
p_demo.set_defaults(func=cmd_demo)
|
|
1518
|
+
|
|
1519
|
+
p_db = sub.add_parser(
|
|
1520
|
+
"db", help="provision database roles or inspect and upgrade the schema"
|
|
1521
|
+
)
|
|
1522
|
+
db_sub = p_db.add_subparsers(
|
|
1523
|
+
dest="db_operation",
|
|
1524
|
+
required=True,
|
|
1525
|
+
title="operations",
|
|
1526
|
+
metavar="<operation>",
|
|
1527
|
+
prog=p_db.prog,
|
|
1528
|
+
)
|
|
1529
|
+
for operation, help_text, func in (
|
|
1530
|
+
("status", "inspect the schema revision without changing it", cmd_db_status),
|
|
1531
|
+
("upgrade", "upgrade the schema under an advisory lock", cmd_db_upgrade),
|
|
1532
|
+
):
|
|
1533
|
+
command = db_sub.add_parser(operation, help=help_text)
|
|
1534
|
+
command.add_argument(
|
|
1535
|
+
"--database-url",
|
|
1536
|
+
help="PostgreSQL URL (default: SEDIMENT_DATABASE_URL)",
|
|
1537
|
+
)
|
|
1538
|
+
command.set_defaults(func=func)
|
|
1539
|
+
|
|
1540
|
+
provision = db_sub.add_parser(
|
|
1541
|
+
"provision",
|
|
1542
|
+
help="provision roles and migrate using isolated bootstrap credentials",
|
|
1543
|
+
description=(
|
|
1544
|
+
"Read SEDIMENT_BOOTSTRAP_DATABASE_URL, SEDIMENT_MIGRATOR_PASSWORD, "
|
|
1545
|
+
"SEDIMENT_RUNTIME_PASSWORD, and SEDIMENT_OPERATOR_PASSWORD from "
|
|
1546
|
+
"the environment. Stop services before provisioning."
|
|
1547
|
+
),
|
|
1548
|
+
)
|
|
1549
|
+
provision.set_defaults(func=cmd_db_provision)
|
|
1550
|
+
|
|
1551
|
+
for verb, help_text in (
|
|
1552
|
+
("quarantine", "quarantine one fact by id"),
|
|
1553
|
+
("release", "release one quarantined fact by id"),
|
|
1554
|
+
):
|
|
1555
|
+
p = sub.add_parser(verb, help=help_text)
|
|
1556
|
+
p.add_argument("table", choices=_TABLES)
|
|
1557
|
+
p.add_argument("fact_id")
|
|
1558
|
+
p.add_argument("--reason", required=True)
|
|
1559
|
+
p.set_defaults(func=cmd_quarantine_or_release)
|
|
1560
|
+
|
|
1561
|
+
p_log = sub.add_parser("quarantine-log", help="the org's quarantine history")
|
|
1562
|
+
p_log.add_argument("--tail", type=int, default=20, help="rows shown (default 20)")
|
|
1563
|
+
p_log.add_argument("--all", action="store_true", help="show the full history")
|
|
1564
|
+
p_log.set_defaults(func=cmd_quarantine_log)
|
|
1565
|
+
|
|
1566
|
+
p_qc = sub.add_parser(
|
|
1567
|
+
"quarantine-inference-calls",
|
|
1568
|
+
help="bulk-quarantine inference calls by filter (dry-run by default)",
|
|
1569
|
+
)
|
|
1570
|
+
p_qc.add_argument("--session-id")
|
|
1571
|
+
p_qc.add_argument(
|
|
1572
|
+
"--provider",
|
|
1573
|
+
metavar="{" + ",".join(provider.value for provider in GatewayProvider) + "}",
|
|
1574
|
+
help="gateway provider filter",
|
|
1575
|
+
)
|
|
1576
|
+
p_qc.add_argument(
|
|
1577
|
+
"--between",
|
|
1578
|
+
nargs=2,
|
|
1579
|
+
metavar=("FROM", "TO"),
|
|
1580
|
+
help="ISO-8601 capture-time bounds, timezone-aware, inclusive",
|
|
1581
|
+
)
|
|
1582
|
+
p_qc.add_argument(
|
|
1583
|
+
"--all", action="store_true", help="allow a filterless (org-wide) run"
|
|
1584
|
+
)
|
|
1585
|
+
p_qc.add_argument(
|
|
1586
|
+
"--apply", action="store_true", help="write; without it, dry-run only"
|
|
1587
|
+
)
|
|
1588
|
+
p_qc.add_argument("--reason", required=True)
|
|
1589
|
+
p_qc.set_defaults(func=cmd_quarantine_inference_calls)
|
|
1590
|
+
|
|
1591
|
+
# Stubs for --help only — the attribution dispatch in main() runs first
|
|
1592
|
+
# (the report/mirror-gc pattern).
|
|
1593
|
+
sub.add_parser(
|
|
1594
|
+
"install",
|
|
1595
|
+
help="wire a repo + this machine: git hooks, agent hooks, telemetry "
|
|
1596
|
+
"env (sediment install --help)",
|
|
1597
|
+
)
|
|
1598
|
+
sub.add_parser(
|
|
1599
|
+
"uninstall", help="remove the per-repo git hooks (--agents: user-level too)"
|
|
1600
|
+
)
|
|
1601
|
+
sub.add_parser("doctor", help="check attribution + server health on this machine")
|
|
1602
|
+
|
|
1603
|
+
p_server = sub.add_parser(
|
|
1604
|
+
"server", help="run a local API server with managed PostgreSQL"
|
|
1605
|
+
)
|
|
1606
|
+
p_server.add_argument("--host", default="127.0.0.1")
|
|
1607
|
+
p_server.add_argument("--port", type=int, default=8000)
|
|
1608
|
+
p_server.add_argument(
|
|
1609
|
+
"--root",
|
|
1610
|
+
default=str(Path.home() / ".sediment" / "server"),
|
|
1611
|
+
help="server data, PostgreSQL binaries, credentials, and logs "
|
|
1612
|
+
"(default: ~/.sediment/server)",
|
|
1613
|
+
)
|
|
1614
|
+
p_server.set_defaults(func=cmd_server)
|
|
1615
|
+
|
|
1616
|
+
p_login = sub.add_parser(
|
|
1617
|
+
"login",
|
|
1618
|
+
help="store credentials for a server",
|
|
1619
|
+
# Explicit description (it beats _propagate_descriptions): the token
|
|
1620
|
+
# resolution order is the thing a reader most needs, and stating it
|
|
1621
|
+
# here feeds both --help and the generated CLI reference.
|
|
1622
|
+
description="store verified operator credentials, or separate ingest "
|
|
1623
|
+
"credentials with --capture. Remote URLs must use "
|
|
1624
|
+
"HTTPS; HTTP is accepted only for localhost or a literal loopback "
|
|
1625
|
+
"IP address. A loopback URL enrolls both credentials sediment server "
|
|
1626
|
+
"generated in ~/.sediment/server/server.env; any other server "
|
|
1627
|
+
"prompts, or takes --with-token on stdin.",
|
|
1628
|
+
)
|
|
1629
|
+
p_login.add_argument(
|
|
1630
|
+
"url",
|
|
1631
|
+
help="HTTPS server base URL or loopback HTTP URL, e.g. http://127.0.0.1:8000",
|
|
1632
|
+
)
|
|
1633
|
+
p_login.add_argument(
|
|
1634
|
+
"--capture",
|
|
1635
|
+
action="store_true",
|
|
1636
|
+
help="enroll an ingest credential for capture without replacing operator login",
|
|
1637
|
+
)
|
|
1638
|
+
p_login.add_argument(
|
|
1639
|
+
"--with-token",
|
|
1640
|
+
action="store_true",
|
|
1641
|
+
help="read the token from stdin instead of prompting (unattended)",
|
|
1642
|
+
)
|
|
1643
|
+
p_login.set_defaults(func=cmd_login)
|
|
1644
|
+
|
|
1645
|
+
p_logout = sub.add_parser("logout", help="remove stored credentials")
|
|
1646
|
+
p_logout.add_argument(
|
|
1647
|
+
"--server", help="server URL to log out of (default: current)"
|
|
1648
|
+
)
|
|
1649
|
+
p_logout.set_defaults(func=cmd_logout)
|
|
1650
|
+
|
|
1651
|
+
p_commit = sub.add_parser("commit", help="pretty-print attributions for a commit")
|
|
1652
|
+
p_commit.add_argument("sha", help="full 40- or 64-char commit SHA")
|
|
1653
|
+
p_commit.add_argument(
|
|
1654
|
+
"--repo", help="observed repository name; ambiguous names require identity"
|
|
1655
|
+
)
|
|
1656
|
+
p_commit.add_argument(
|
|
1657
|
+
"--repository-provider",
|
|
1658
|
+
choices=[provider.value for provider in ForgeProvider],
|
|
1659
|
+
help="forge provider (requires host and ID)",
|
|
1660
|
+
)
|
|
1661
|
+
p_commit.add_argument(
|
|
1662
|
+
"--repository-host", help="forge hostname (requires provider and ID)"
|
|
1663
|
+
)
|
|
1664
|
+
p_commit.add_argument(
|
|
1665
|
+
"--repository-id", help="provider repository ID (requires provider and host)"
|
|
1666
|
+
)
|
|
1667
|
+
p_commit.add_argument(
|
|
1668
|
+
"--as-of", help="inclusive evidence boundary as an aware RFC 3339 timestamp"
|
|
1669
|
+
)
|
|
1670
|
+
p_commit.set_defaults(func=cmd_commit)
|
|
1671
|
+
|
|
1672
|
+
p_derive = sub.add_parser(
|
|
1673
|
+
"derive", help="materialize a reviewed canonical derivation bundle"
|
|
1674
|
+
)
|
|
1675
|
+
p_derive.add_argument("--out", required=True, help="new bundle directory")
|
|
1676
|
+
p_derive.add_argument("--policy", help="strict derivation policy TOML file")
|
|
1677
|
+
p_derive.add_argument("--since", help="inclusive RFC 3339 completion time")
|
|
1678
|
+
p_derive.add_argument("--until", help="exclusive RFC 3339 completion time")
|
|
1679
|
+
p_derive.add_argument(
|
|
1680
|
+
"--users",
|
|
1681
|
+
nargs="+",
|
|
1682
|
+
metavar="USER",
|
|
1683
|
+
help="space-separated user IDs (default: every user in the organization)",
|
|
1684
|
+
)
|
|
1685
|
+
p_derive.add_argument(
|
|
1686
|
+
"--sample",
|
|
1687
|
+
type=int,
|
|
1688
|
+
default=0,
|
|
1689
|
+
metavar="N",
|
|
1690
|
+
help="print the first N canonical rows after writing",
|
|
1691
|
+
)
|
|
1692
|
+
p_derive.set_defaults(func=cmd_derive)
|
|
1693
|
+
|
|
1694
|
+
p_e = sub.add_parser("export", help="run an export projection")
|
|
1695
|
+
e_sub = p_e.add_subparsers(
|
|
1696
|
+
dest="format",
|
|
1697
|
+
required=True,
|
|
1698
|
+
title="formats",
|
|
1699
|
+
metavar="<format>",
|
|
1700
|
+
prog=p_e.prog,
|
|
1701
|
+
)
|
|
1702
|
+
|
|
1703
|
+
p_rlvr = e_sub.add_parser("rlvr", help="export one explicit RLVR target")
|
|
1704
|
+
p_rlvr.add_argument("--out", required=True, help="output directory")
|
|
1705
|
+
p_rlvr.add_argument(
|
|
1706
|
+
"--target",
|
|
1707
|
+
required=True,
|
|
1708
|
+
choices=("sediment", "swe-bench", "nemo-gym"),
|
|
1709
|
+
help="explicit consumer contract",
|
|
1710
|
+
)
|
|
1711
|
+
p_rlvr.add_argument(
|
|
1712
|
+
"--from", dest="from_bundle", help="validated derived bundle directory"
|
|
1713
|
+
)
|
|
1714
|
+
from sediment_export.compatibility import PROFILES
|
|
1715
|
+
|
|
1716
|
+
p_rlvr.add_argument(
|
|
1717
|
+
"--profile",
|
|
1718
|
+
choices=tuple(p.id for p in PROFILES if p.objective == "rlvr"),
|
|
1719
|
+
help="exact optional consumer compatibility profile",
|
|
1720
|
+
)
|
|
1721
|
+
p_rlvr.add_argument(
|
|
1722
|
+
"--consumer-config",
|
|
1723
|
+
help="JSON task and response configuration for the selected profile",
|
|
1724
|
+
)
|
|
1725
|
+
p_rlvr.set_defaults(func=cmd_export_rlvr)
|
|
1726
|
+
|
|
1727
|
+
p_dpo = e_sub.add_parser("dpo", help="dpo.jsonl — chosen/rejected pairs")
|
|
1728
|
+
p_dpo.add_argument("--out", required=True, help="output directory")
|
|
1729
|
+
p_dpo.add_argument(
|
|
1730
|
+
"--recipe",
|
|
1731
|
+
choices=("dpo_human", "dpo_outcome"),
|
|
1732
|
+
default="dpo_human",
|
|
1733
|
+
help="evidence recipe (default: dpo_human; dpo_outcome requires opt-in)",
|
|
1734
|
+
)
|
|
1735
|
+
p_dpo.add_argument(
|
|
1736
|
+
"--from", dest="from_bundle", help="validated derived bundle directory"
|
|
1737
|
+
)
|
|
1738
|
+
p_dpo.add_argument(
|
|
1739
|
+
"--profile",
|
|
1740
|
+
choices=tuple(p.id for p in PROFILES if p.objective == "dpo"),
|
|
1741
|
+
type=_dpo_profile_name,
|
|
1742
|
+
help="exact optional consumer compatibility profile",
|
|
1743
|
+
)
|
|
1744
|
+
p_dpo.set_defaults(func=cmd_export_dpo)
|
|
1745
|
+
|
|
1746
|
+
p_sft = e_sub.add_parser("sft", help="sft.jsonl — supervised training targets")
|
|
1747
|
+
p_sft.add_argument("--out", required=True, help="output directory")
|
|
1748
|
+
p_sft.add_argument(
|
|
1749
|
+
"--recipe",
|
|
1750
|
+
choices=("sft_curated", "sft_verified"),
|
|
1751
|
+
default="sft_curated",
|
|
1752
|
+
help="evidence recipe (default: sft_curated; sft_verified requires opt-in)",
|
|
1753
|
+
)
|
|
1754
|
+
p_sft.add_argument(
|
|
1755
|
+
"--from", dest="from_bundle", help="validated derived bundle directory"
|
|
1756
|
+
)
|
|
1757
|
+
p_sft.add_argument(
|
|
1758
|
+
"--profile",
|
|
1759
|
+
choices=tuple(p.id for p in PROFILES if p.objective == "sft"),
|
|
1760
|
+
help="exact optional consumer compatibility profile",
|
|
1761
|
+
)
|
|
1762
|
+
p_sft.set_defaults(func=cmd_export_sft)
|
|
1763
|
+
|
|
1764
|
+
p_diff = e_sub.add_parser(
|
|
1765
|
+
"diff-sft", help="diff_sft.jsonl — per-commit, per-file training rows"
|
|
1766
|
+
)
|
|
1767
|
+
p_diff.add_argument("--out", required=True, help="output directory")
|
|
1768
|
+
p_diff.add_argument(
|
|
1769
|
+
"--recipe",
|
|
1770
|
+
choices=("sft_curated", "sft_verified"),
|
|
1771
|
+
default="sft_curated",
|
|
1772
|
+
help="SFT evidence recipe (default: sft_curated; sft_verified requires opt-in)",
|
|
1773
|
+
)
|
|
1774
|
+
p_diff.add_argument(
|
|
1775
|
+
"--from", dest="from_bundle", help="validated derived bundle directory"
|
|
1776
|
+
)
|
|
1777
|
+
p_diff.set_defaults(func=cmd_export_diff_sft)
|
|
1778
|
+
|
|
1779
|
+
p_rec = e_sub.add_parser(
|
|
1780
|
+
"recovery",
|
|
1781
|
+
help="recovery.jsonl — red-to-green CI-transition recovery pairs",
|
|
1782
|
+
)
|
|
1783
|
+
p_rec.add_argument("--out", required=True, help="output directory")
|
|
1784
|
+
p_rec.add_argument(
|
|
1785
|
+
"--recipe",
|
|
1786
|
+
choices=("recovery_ci",),
|
|
1787
|
+
default="recovery_ci",
|
|
1788
|
+
help="evidence recipe (default: recovery_ci)",
|
|
1789
|
+
)
|
|
1790
|
+
p_rec.set_defaults(func=cmd_export_recovery)
|
|
1791
|
+
|
|
1792
|
+
_propagate_descriptions(parser)
|
|
1793
|
+
return parser
|
|
1794
|
+
|
|
1795
|
+
|
|
1796
|
+
# Remote verbs speak HTTP through the client seam; they never open the fact
|
|
1797
|
+
# store and never construct Settings.
|
|
1798
|
+
_REMOTE_VERBS = {"login", "logout", "commit", "facts", "demo"}
|
|
1799
|
+
|
|
1800
|
+
# Dispatched pre-argparse to the stdlib-only attribution module.
|
|
1801
|
+
# install/uninstall/doctor get help stubs; the hook-plumbing verbs are
|
|
1802
|
+
# execution-only (hidden from --help).
|
|
1803
|
+
_ATTRIBUTION_VERBS = {
|
|
1804
|
+
"cursor-hook",
|
|
1805
|
+
"install",
|
|
1806
|
+
"uninstall",
|
|
1807
|
+
"doctor",
|
|
1808
|
+
"mark",
|
|
1809
|
+
"stamp",
|
|
1810
|
+
"union-squash-notes",
|
|
1811
|
+
"push-notes",
|
|
1812
|
+
"repair-notes",
|
|
1813
|
+
}
|
|
1814
|
+
|
|
1815
|
+
|
|
1816
|
+
def main(argv: list[str] | None = None) -> int:
|
|
1817
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
1818
|
+
if argv[:1] == ["delivery"]:
|
|
1819
|
+
module = importlib.import_module("sediment_cli.delivery")
|
|
1820
|
+
if argv[1:] in (["-h"], ["--help"]):
|
|
1821
|
+
return _print_dispatched_help(
|
|
1822
|
+
module.build_parser(), prog="sediment delivery"
|
|
1823
|
+
)
|
|
1824
|
+
if len(argv) == 3 and argv[2] in {"-h", "--help"}:
|
|
1825
|
+
parser = module.build_parser(prog="sediment delivery")
|
|
1826
|
+
commands = next(
|
|
1827
|
+
action.choices
|
|
1828
|
+
for action in parser._actions
|
|
1829
|
+
if isinstance(action, argparse._SubParsersAction)
|
|
1830
|
+
)
|
|
1831
|
+
if argv[1] in commands:
|
|
1832
|
+
return _print_dispatched_help(
|
|
1833
|
+
commands[argv[1]], prog=f"sediment delivery {argv[1]}"
|
|
1834
|
+
)
|
|
1835
|
+
return _public_status(module.main(argv[1:], prog="sediment delivery"))
|
|
1836
|
+
# Dispatched before argparse: report/mirror-gc forward argv verbatim to
|
|
1837
|
+
# the module's own parser and never construct Settings (see _REPORTS).
|
|
1838
|
+
if argv[:1] == ["report"]:
|
|
1839
|
+
return _run_report(argv[1:])
|
|
1840
|
+
if argv[:1] == ["mirror-gc"]:
|
|
1841
|
+
module = importlib.import_module("sediment_api.mirror_gc")
|
|
1842
|
+
if argv[1:] in (["-h"], ["--help"]):
|
|
1843
|
+
return _print_dispatched_help(
|
|
1844
|
+
module.build_parser(), prog="sediment mirror-gc"
|
|
1845
|
+
)
|
|
1846
|
+
try:
|
|
1847
|
+
return _public_status(module.main(argv[1:]))
|
|
1848
|
+
except (DatabaseOperationError, OSError, ValueError) as exc:
|
|
1849
|
+
return _fail(str(exc))
|
|
1850
|
+
if argv[:1] == ["transcript"]:
|
|
1851
|
+
module = importlib.import_module("sediment_cli.transcript")
|
|
1852
|
+
return _public_status(module.main(argv[1:]))
|
|
1853
|
+
if argv[:1] and argv[0] in _ATTRIBUTION_VERBS:
|
|
1854
|
+
# Forwarded verbatim to the attribution module's own parser —
|
|
1855
|
+
# stdlib-only, never constructs Settings. The plumbing verbs (mark,
|
|
1856
|
+
# stamp, union-squash-notes, push-notes, repair-notes) get no help
|
|
1857
|
+
# stub: hooks invoke them, humans don't.
|
|
1858
|
+
module = importlib.import_module("sediment_cli.attribution")
|
|
1859
|
+
if argv[0] in {"install", "uninstall", "doctor"} and argv[1:] in (
|
|
1860
|
+
["-h"],
|
|
1861
|
+
["--help"],
|
|
1862
|
+
):
|
|
1863
|
+
action = _subparsers_action(module.build_parser())
|
|
1864
|
+
return _print_dispatched_help(
|
|
1865
|
+
action.choices[argv[0]], prog=f"sediment {argv[0]}"
|
|
1866
|
+
)
|
|
1867
|
+
return _public_status(module.main(argv))
|
|
1868
|
+
|
|
1869
|
+
args = build_parser().parse_args(argv)
|
|
1870
|
+
|
|
1871
|
+
if args.command == "export" and getattr(args, "from_bundle", None):
|
|
1872
|
+
try:
|
|
1873
|
+
args._mirror_path = os.environ.get("SEDIMENT_MIRROR_PATH")
|
|
1874
|
+
return args.func(None, None, args)
|
|
1875
|
+
except (OSError, ValueError) as exc:
|
|
1876
|
+
return _fail(str(exc))
|
|
1877
|
+
|
|
1878
|
+
if args.command == "server":
|
|
1879
|
+
# Pre-Settings dispatch: provision before the API imports its settings.
|
|
1880
|
+
try:
|
|
1881
|
+
return args.func(args)
|
|
1882
|
+
except (DatabaseOperationError, OSError, ValueError) as exc:
|
|
1883
|
+
return _fail(str(exc))
|
|
1884
|
+
|
|
1885
|
+
if args.command == "db":
|
|
1886
|
+
from sediment_core.postgres_migrations import MigrationError
|
|
1887
|
+
|
|
1888
|
+
try:
|
|
1889
|
+
return args.func(args)
|
|
1890
|
+
except (DatabaseOperationError, MigrationError, ValueError) as exc:
|
|
1891
|
+
return _fail(str(exc))
|
|
1892
|
+
|
|
1893
|
+
if args.command in _REMOTE_VERBS:
|
|
1894
|
+
try:
|
|
1895
|
+
return args.func(args)
|
|
1896
|
+
except (ClientError, DatabaseOperationError, ValueError) as exc:
|
|
1897
|
+
return _fail(str(exc))
|
|
1898
|
+
|
|
1899
|
+
try:
|
|
1900
|
+
_prepare_store_command(args)
|
|
1901
|
+
|
|
1902
|
+
# Deferred so --help needs no SEDIMENT_ORG_ID. Keep construction
|
|
1903
|
+
# inside this boundary because Pydantic settings validation is an
|
|
1904
|
+
# expected operator-input failure.
|
|
1905
|
+
from sediment_api.config import settings
|
|
1906
|
+
from sediment_api.database import one_shot_fact_store
|
|
1907
|
+
|
|
1908
|
+
args._mirror_path = settings.mirror_path
|
|
1909
|
+
with one_shot_fact_store(
|
|
1910
|
+
settings.database_url.get_secret_value(), operation=f"run {args.command}"
|
|
1911
|
+
) as store:
|
|
1912
|
+
return args.func(store, settings.org_id, args)
|
|
1913
|
+
except (DatabaseOperationError, OSError, ValueError) as exc:
|
|
1914
|
+
# One net for every validation error below (empty reason via the
|
|
1915
|
+
# QuarantineRecord validator, naive --between bounds, pydantic
|
|
1916
|
+
# settings): clean `error:` line, not a
|
|
1917
|
+
# traceback.
|
|
1918
|
+
return _fail(str(exc))
|
|
1919
|
+
|
|
1920
|
+
|
|
1921
|
+
if __name__ == "__main__": # pragma: no cover
|
|
1922
|
+
sys.exit(main())
|