techtree 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- techtree/__init__.py +35 -0
- techtree/__main__.py +14 -0
- techtree/canonical.py +239 -0
- techtree/catalog/__init__.py +25 -0
- techtree/catalog/repository.py +400 -0
- techtree/catalog/service.py +419 -0
- techtree/cli/__init__.py +1 -0
- techtree/cli/app.py +416 -0
- techtree/cli/commands/__init__.py +1 -0
- techtree/cli/commands/climb.py +1223 -0
- techtree/cli/commands/doctor.py +147 -0
- techtree/cli/commands/engine.py +207 -0
- techtree/cli/commands/proof.py +556 -0
- techtree/cli/commands/publish.py +447 -0
- techtree/cli/commands/release.py +303 -0
- techtree/cli/commands/run.py +1067 -0
- techtree/cli/commands/setup.py +181 -0
- techtree/cli/commands/skill.py +221 -0
- techtree/cli/commands/uplift.py +698 -0
- techtree/cli/commands/withdraw.py +212 -0
- techtree/cli/confirm.py +47 -0
- techtree/cli/context.py +96 -0
- techtree/cli/invoke.py +220 -0
- techtree/cli/output.py +280 -0
- techtree/constants.py +138 -0
- techtree/crypto.py +128 -0
- techtree/doctor/__init__.py +1 -0
- techtree/doctor/checks.py +675 -0
- techtree/doctor/execution_checks.py +435 -0
- techtree/doctor/service.py +326 -0
- techtree/drafts/__init__.py +32 -0
- techtree/drafts/source.py +146 -0
- techtree/drafts/store.py +992 -0
- techtree/engines/__init__.py +1 -0
- techtree/engines/bundle.py +251 -0
- techtree/engines/installer.py +679 -0
- techtree/engines/registry.py +235 -0
- techtree/engines/runner.py +170 -0
- techtree/errors.py +262 -0
- techtree/fs.py +234 -0
- techtree/harness.py +108 -0
- techtree/identity/__init__.py +41 -0
- techtree/identity/models.py +113 -0
- techtree/identity/service.py +199 -0
- techtree/identity/store.py +263 -0
- techtree/ids.py +85 -0
- techtree/manifests/__init__.py +39 -0
- techtree/manifests/builder.py +433 -0
- techtree/manifests/compare.py +376 -0
- techtree/models/__init__.py +282 -0
- techtree/models/base.py +201 -0
- techtree/models/campaign.py +484 -0
- techtree/models/catalog.py +227 -0
- techtree/models/cli.py +151 -0
- techtree/models/climb.py +254 -0
- techtree/models/data_policy.py +130 -0
- techtree/models/engine.py +156 -0
- techtree/models/episode_receipt.py +130 -0
- techtree/models/evaluation_backend.py +113 -0
- techtree/models/experiment.py +154 -0
- techtree/models/run.py +214 -0
- techtree/models/skill.py +156 -0
- techtree/models/uplift_report.py +158 -0
- techtree/models/validation.py +299 -0
- techtree/paths.py +116 -0
- techtree/presentation/__init__.py +31 -0
- techtree/presentation/build.py +1242 -0
- techtree/presentation/compact.py +246 -0
- techtree/presentation/evidence.py +169 -0
- techtree/presentation/models.py +358 -0
- techtree/presentation/rich.py +312 -0
- techtree/presentation/sanitize.py +156 -0
- techtree/publication/__init__.py +44 -0
- techtree/publication/address.py +180 -0
- techtree/publication/coordinates.py +26 -0
- techtree/publication/journal.py +212 -0
- techtree/publication/keccak.py +183 -0
- techtree/publication/models.py +209 -0
- techtree/publication/offer.py +35 -0
- techtree/publication/service.py +618 -0
- techtree/publication/transport.py +296 -0
- techtree/publication/verify.py +242 -0
- techtree/publication/withdraw.py +156 -0
- techtree/py.typed +0 -0
- techtree/receipts/__init__.py +52 -0
- techtree/receipts/bundle.py +578 -0
- techtree/receipts/compare.py +1065 -0
- techtree/receipts/episode.py +672 -0
- techtree/receipts/execution.py +630 -0
- techtree/receipts/observed.py +474 -0
- techtree/receipts/set.py +336 -0
- techtree/receipts/uplift.py +655 -0
- techtree/receipts/verify.py +1055 -0
- techtree/release/__init__.py +9 -0
- techtree/release/bootstrap.py +509 -0
- techtree/release/checks.py +376 -0
- techtree/release/document.py +125 -0
- techtree/release/generate.py +221 -0
- techtree/release/models.py +293 -0
- techtree/release/provenance.py +109 -0
- techtree/resources/catalog/campaigns/hello-world-climb.json +1 -0
- techtree/resources/catalog/catalog.json +32 -0
- techtree/resources/catalog/climbs/hello-world-climb.json +1 -0
- techtree/resources/catalog/data-policies/hello-world-climb.json +1 -0
- techtree/resources/catalog/taskset-validations/hello-world-climb.json +1 -0
- techtree/resources/catalog/validation-evidence/hello-world-climb.json +1 -0
- techtree/resources/engines/default/engine.json +20 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/__init__.py +7 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/algorithm.py +136 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/dataset.py +156 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/env.py +48 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/procedure_transfer_v1/taskset.py +163 -0
- techtree/resources/engines/default/packages/procedure-transfer-v1/pyproject.toml +13 -0
- techtree/resources/engines/default/pyproject.toml +23 -0
- techtree/resources/engines/default/tools/inspect_taskset.py +124 -0
- techtree/resources/engines/default/tools/normalize_eval_output.py +470 -0
- techtree/resources/engines/default/tools/normalize_validation.py +222 -0
- techtree/resources/engines/default/uv.lock +1758 -0
- techtree/resources/harness/hermes-agent-0.19.0.json +69 -0
- techtree/resources/release/build-provenance.json +4 -0
- techtree/resources/release/release-core.json +24 -0
- techtree/runs/__init__.py +31 -0
- techtree/runs/artifacts.py +750 -0
- techtree/runs/child_registry.py +228 -0
- techtree/runs/events.py +478 -0
- techtree/runs/executor.py +140 -0
- techtree/runs/fake.py +741 -0
- techtree/runs/launcher.py +253 -0
- techtree/runs/machine.py +489 -0
- techtree/runs/real.py +789 -0
- techtree/runs/service.py +616 -0
- techtree/runs/store.py +555 -0
- techtree/runs/validation.py +259 -0
- techtree/runs/variants.py +684 -0
- techtree/settings.py +143 -0
- techtree/skills/__init__.py +14 -0
- techtree/skills/archive.py +282 -0
- techtree/skills/policy.py +62 -0
- techtree/skills/scanner.py +394 -0
- techtree/skills/service.py +752 -0
- techtree/skills/starter.py +434 -0
- techtree/tasksets/__init__.py +1 -0
- techtree/tasksets/membership.py +269 -0
- techtree/tasksets/provider.py +207 -0
- techtree/tasksets/resolver.py +311 -0
- techtree/tasksets/service.py +484 -0
- techtree/tasksets/verifiers_cli.py +538 -0
- techtree/uplift/__init__.py +20 -0
- techtree/uplift/context.py +544 -0
- techtree/uplift/derive.py +203 -0
- techtree/uplift/public_tasks.py +151 -0
- techtree/uplift/service.py +719 -0
- techtree/uplift/source.py +160 -0
- techtree/verifiers/__init__.py +31 -0
- techtree/verifiers/budget.py +219 -0
- techtree/verifiers/child.py +633 -0
- techtree/verifiers/compiler.py +432 -0
- techtree/verifiers/config.py +365 -0
- techtree/verifiers/credentials.py +321 -0
- techtree/verifiers/image.py +126 -0
- techtree/verifiers/models.py +527 -0
- techtree/verifiers/outputs.py +368 -0
- techtree/verifiers/progress.py +192 -0
- techtree/verifiers/supervisor.py +341 -0
- techtree/verifiers/verify.py +782 -0
- techtree/version.py +39 -0
- techtree/worker/__init__.py +18 -0
- techtree/worker/execute.py +487 -0
- techtree/worker/main.py +57 -0
- techtree-0.1.0.dist-info/METADATA +344 -0
- techtree-0.1.0.dist-info/RECORD +174 -0
- techtree-0.1.0.dist-info/WHEEL +4 -0
- techtree-0.1.0.dist-info/entry_points.txt +3 -0
- techtree-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,538 @@
|
|
|
1
|
+
"""Driving the pinned Verifiers validator. Spec section 21.4.
|
|
2
|
+
|
|
3
|
+
This module is the whole of Techtree's contact with the upstream ``validate``
|
|
4
|
+
command, and every line of it is shaped by what the PI0 preflight actually
|
|
5
|
+
observed (``docs/verifiers-pin-0.3.1.md``) rather than by what the
|
|
6
|
+
specification assumed.
|
|
7
|
+
|
|
8
|
+
*A run has a directory of its own* (deviation D2). ``--output-dir`` names the
|
|
9
|
+
directory runs are *grouped* under, not the directory a run writes into: the
|
|
10
|
+
run lands in ``<output-dir>/<run.dir>``, and an unnamed run auto-generates that
|
|
11
|
+
directory with a random suffix. So the pinned invocation always passes
|
|
12
|
+
``--run.name`` and everything here reads the artifacts from
|
|
13
|
+
:func:`validation_run_dir`.
|
|
14
|
+
|
|
15
|
+
*A run directory is written once* (deviation D3). The validator refuses a run
|
|
16
|
+
directory that already holds results, so :meth:`VerifiersValidationRunner.run`
|
|
17
|
+
requires a directory nothing has written into yet rather than resuming or
|
|
18
|
+
clearing one.
|
|
19
|
+
|
|
20
|
+
*The exit code is not the verdict* (finding C2). ``validate`` exits ``0`` when
|
|
21
|
+
every task in a taskset fails, because its status reports runner health. So
|
|
22
|
+
:meth:`VerifiersValidationRunner.run` returns the process result as data and
|
|
23
|
+
refuses to interpret it; validity is read from ``summary.json`` and nowhere
|
|
24
|
+
else.
|
|
25
|
+
|
|
26
|
+
*The summary is nested* (finding C1). The five outcome counts the protocol
|
|
27
|
+
models at the top level live under ``outcomes`` upstream, and the document
|
|
28
|
+
carries ``owed``, ``terminal``, and ``checks`` besides. The file is therefore
|
|
29
|
+
*projected* into :class:`~techtree.models.validation.UpstreamValidationSummary`
|
|
30
|
+
field by field and never validated as one.
|
|
31
|
+
|
|
32
|
+
*Per-check counts are optional* (finding C1 again). ``checks`` appears only when
|
|
33
|
+
both checks ran. Techtree always runs both, so its absence is an anomaly worth
|
|
34
|
+
reporting rather than a shape to tolerate — but it is read through an accessor
|
|
35
|
+
that can say "the validator reported none" instead of raising a parse error a
|
|
36
|
+
caller cannot act on.
|
|
37
|
+
|
|
38
|
+
*Row order is completion order* (finding C0). ``results.jsonl`` is appended as
|
|
39
|
+
each isolated check finishes, so its bytes differ between two identical runs.
|
|
40
|
+
Nothing here reads it: turning those rows into deterministic evidence is the
|
|
41
|
+
engine helper ``normalize_validation.py``'s job, invoked through
|
|
42
|
+
:meth:`VerifiersValidationRunner.normalize`, which joins on ``task_position``
|
|
43
|
+
and ``task_key`` and sorts by position.
|
|
44
|
+
|
|
45
|
+
*Scripts are addressed absolutely* (finding C3). The pinned build installs a
|
|
46
|
+
console script called plain ``validate``; :class:`EngineRunner` resolves it
|
|
47
|
+
inside the engine's own environment and never through ``PATH``.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
from __future__ import annotations
|
|
51
|
+
|
|
52
|
+
import json
|
|
53
|
+
import tempfile
|
|
54
|
+
from collections.abc import Mapping, Sequence
|
|
55
|
+
from pathlib import Path
|
|
56
|
+
from typing import Any, Final
|
|
57
|
+
|
|
58
|
+
from pydantic import ValidationError as PydanticValidationError
|
|
59
|
+
|
|
60
|
+
from techtree.canonical import sha256_digest_bytes
|
|
61
|
+
from techtree.engines.registry import EngineRegistry
|
|
62
|
+
from techtree.engines.runner import EngineProcessResult, EngineRunner
|
|
63
|
+
from techtree.errors import EngineError, ValidationError
|
|
64
|
+
from techtree.fs import ensure_private_directory
|
|
65
|
+
from techtree.models.base import ArtifactRef, Digest
|
|
66
|
+
from techtree.models.validation import (
|
|
67
|
+
UpstreamValidationSummary,
|
|
68
|
+
ValidationEvidence,
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
__all__ = [
|
|
72
|
+
"NORMALIZE_VALIDATION_TOOL",
|
|
73
|
+
"VALIDATE_EXECUTABLE",
|
|
74
|
+
"VALIDATION_FILENAMES",
|
|
75
|
+
"VALIDATION_RUN_NAME",
|
|
76
|
+
"VALIDATION_TIMEOUT_SECONDS",
|
|
77
|
+
"UpstreamCheckCounts",
|
|
78
|
+
"VerifiersValidationRunner",
|
|
79
|
+
"validate_argv",
|
|
80
|
+
"validation_run_dir",
|
|
81
|
+
]
|
|
82
|
+
|
|
83
|
+
#: The console script the pinned Verifiers build installs. Generic enough to
|
|
84
|
+
#: collide with anything on ``PATH``, which is why it is only ever run through
|
|
85
|
+
#: the engine's absolute path (``docs/verifiers-pin-0.3.1.md``, finding C3).
|
|
86
|
+
VALIDATE_EXECUTABLE: Final = "validate"
|
|
87
|
+
|
|
88
|
+
#: The engine helper that turns raw validator output into deterministic
|
|
89
|
+
#: evidence. It lives inside the digested bundle (decisions document 0003 A3).
|
|
90
|
+
NORMALIZE_VALIDATION_TOOL: Final = "normalize_validation.py"
|
|
91
|
+
|
|
92
|
+
#: What Techtree calls the one run it groups under an output directory. Naming
|
|
93
|
+
#: it is what makes the run's own directory predictable and its resolved
|
|
94
|
+
#: configuration the same from run to run (deviation D2).
|
|
95
|
+
VALIDATION_RUN_NAME: Final = "run"
|
|
96
|
+
|
|
97
|
+
#: Exactly the files one validation run writes, relative to the run's own
|
|
98
|
+
#: directory, in the order a receipt lists them
|
|
99
|
+
#: (``docs/verifiers-pin-0.3.1.md``, claim 7 and deviation D1). Nothing else
|
|
100
|
+
#: appears, and a missing one is a broken run rather than a smaller one.
|
|
101
|
+
VALIDATION_FILENAMES: Final[tuple[str, ...]] = (
|
|
102
|
+
"configs/resolved/validate.json",
|
|
103
|
+
"logs/validate.log",
|
|
104
|
+
"results.jsonl",
|
|
105
|
+
"summary.json",
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
#: Model-free validation of the reference taskset finishes in seconds. The
|
|
109
|
+
#: limit exists to end a hung validator, not to police performance.
|
|
110
|
+
VALIDATION_TIMEOUT_SECONDS: Final = 1800.0
|
|
111
|
+
|
|
112
|
+
#: How the validator spells the mode in which both checks run.
|
|
113
|
+
_BOTH_CHECKS: Final = "all"
|
|
114
|
+
|
|
115
|
+
#: The two per-task checks a ``mode = "all"`` run performs.
|
|
116
|
+
GOLD_CHECK: Final = "gold"
|
|
117
|
+
SETUP_CHECK: Final = "setup"
|
|
118
|
+
|
|
119
|
+
#: The five outcome counts, at whatever depth they appear.
|
|
120
|
+
_OUTCOME_KEYS: Final[tuple[str, ...]] = (
|
|
121
|
+
"valid",
|
|
122
|
+
"invalid",
|
|
123
|
+
"error",
|
|
124
|
+
"timeout",
|
|
125
|
+
"missing",
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
_SUMMARY_FILE: Final = "summary.json"
|
|
129
|
+
_JSON_MEDIA_TYPE: Final = "application/json"
|
|
130
|
+
_MEDIA_TYPES: Final[Mapping[str, str]] = {
|
|
131
|
+
"configs/resolved/validate.json": _JSON_MEDIA_TYPE,
|
|
132
|
+
"logs/validate.log": "text/plain",
|
|
133
|
+
"results.jsonl": "application/x-ndjson",
|
|
134
|
+
"summary.json": _JSON_MEDIA_TYPE,
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
_EVIDENCE_FILENAME: Final = "evidence.json"
|
|
138
|
+
_TEMPORARY_PREFIX: Final = "techtree-normalize-"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class UpstreamCheckCounts:
|
|
142
|
+
"""The five outcome counts for one of the validator's two checks.
|
|
143
|
+
|
|
144
|
+
A plain value object rather than a protocol model: these counts are read
|
|
145
|
+
to build a :class:`~techtree.models.validation.ValidationCheck` and are
|
|
146
|
+
never written anywhere, so giving them a schema version would claim a
|
|
147
|
+
stability nothing depends on.
|
|
148
|
+
"""
|
|
149
|
+
|
|
150
|
+
__slots__ = ("error", "invalid", "missing", "timeout", "valid")
|
|
151
|
+
|
|
152
|
+
def __init__(
|
|
153
|
+
self,
|
|
154
|
+
*,
|
|
155
|
+
valid: int,
|
|
156
|
+
invalid: int,
|
|
157
|
+
error: int,
|
|
158
|
+
timeout: int,
|
|
159
|
+
missing: int,
|
|
160
|
+
) -> None:
|
|
161
|
+
self.valid = valid
|
|
162
|
+
self.invalid = invalid
|
|
163
|
+
self.error = error
|
|
164
|
+
self.timeout = timeout
|
|
165
|
+
self.missing = missing
|
|
166
|
+
|
|
167
|
+
@property
|
|
168
|
+
def total(self) -> int:
|
|
169
|
+
"""Return how many verdicts this check accounted for."""
|
|
170
|
+
return self.valid + self.invalid + self.error + self.timeout + self.missing
|
|
171
|
+
|
|
172
|
+
@property
|
|
173
|
+
def passed(self) -> bool:
|
|
174
|
+
"""Return whether every task this check saw was valid."""
|
|
175
|
+
return self.total > 0 and self.valid == self.total
|
|
176
|
+
|
|
177
|
+
def summary(self) -> str:
|
|
178
|
+
"""Describe the counts in the order a reader cares about them."""
|
|
179
|
+
return (
|
|
180
|
+
f"{self.valid} valid, {self.invalid} invalid, {self.error} errored, "
|
|
181
|
+
f"{self.timeout} timed out, {self.missing} missing"
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def validation_run_dir(output_dir: Path) -> Path:
|
|
186
|
+
"""Return the directory the run itself writes into.
|
|
187
|
+
|
|
188
|
+
``output_dir`` is the directory runs are grouped under, which is all
|
|
189
|
+
``--output-dir`` has meant since v0.3.1; the run's own files are one level
|
|
190
|
+
below it, under the name :func:`validate_argv` pins (deviation D2).
|
|
191
|
+
"""
|
|
192
|
+
return output_dir / VALIDATION_RUN_NAME
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def validate_argv(*, taskset_id: str, num_tasks: int, output_dir: Path) -> list[str]:
|
|
196
|
+
"""Return the pinned validate arguments, without the executable.
|
|
197
|
+
|
|
198
|
+
Spelled once, here, because this exact form is what the PI0 preflight
|
|
199
|
+
proved and what a :class:`~techtree.models.validation.ValidationMethod`
|
|
200
|
+
claims. ``--runtime.type subprocess`` is mandatory — upstream defaults to
|
|
201
|
+
prime sandboxes — ``--run.name`` is what stops the run directory from
|
|
202
|
+
carrying a random suffix nobody can name afterwards, and ``--rich false``
|
|
203
|
+
replaces a live dashboard with one log line per task, which is what a
|
|
204
|
+
captured subprocess wants.
|
|
205
|
+
"""
|
|
206
|
+
return [
|
|
207
|
+
taskset_id,
|
|
208
|
+
"--num-tasks",
|
|
209
|
+
str(num_tasks),
|
|
210
|
+
"--runtime.type",
|
|
211
|
+
"subprocess",
|
|
212
|
+
"--output-dir",
|
|
213
|
+
str(output_dir),
|
|
214
|
+
"--run.name",
|
|
215
|
+
VALIDATION_RUN_NAME,
|
|
216
|
+
"--rich",
|
|
217
|
+
"false",
|
|
218
|
+
]
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
class VerifiersValidationRunner:
|
|
222
|
+
"""Runs and reads one model-free validation inside the managed engine.
|
|
223
|
+
|
|
224
|
+
Constructed from a registry and an engine digest rather than from a bare
|
|
225
|
+
:class:`~techtree.engines.runner.EngineRunner`, matching
|
|
226
|
+
:class:`~techtree.tasksets.resolver.TasksetResolver`: normalizing the
|
|
227
|
+
output needs a bundled helper, and only the registry can say where a
|
|
228
|
+
bundled helper lives (decisions document 0003 A3).
|
|
229
|
+
"""
|
|
230
|
+
|
|
231
|
+
def __init__(
|
|
232
|
+
self,
|
|
233
|
+
registry: EngineRegistry,
|
|
234
|
+
engine_digest: Digest,
|
|
235
|
+
*,
|
|
236
|
+
timeout_seconds: float = VALIDATION_TIMEOUT_SECONDS,
|
|
237
|
+
) -> None:
|
|
238
|
+
self._registry = registry
|
|
239
|
+
self._engine_digest = engine_digest
|
|
240
|
+
self._runner = EngineRunner(registry, engine_digest)
|
|
241
|
+
self._timeout_seconds = timeout_seconds
|
|
242
|
+
|
|
243
|
+
# -- running -----------------------------------------------------------
|
|
244
|
+
|
|
245
|
+
def run(
|
|
246
|
+
self,
|
|
247
|
+
*,
|
|
248
|
+
taskset_id: str,
|
|
249
|
+
num_tasks: int,
|
|
250
|
+
output_dir: Path,
|
|
251
|
+
) -> EngineProcessResult:
|
|
252
|
+
"""Invoke the pinned validate command and return what the process did.
|
|
253
|
+
|
|
254
|
+
``output_dir`` is the directory this run is grouped under; the run
|
|
255
|
+
writes into :func:`validation_run_dir` below it, and that directory has
|
|
256
|
+
to be one nothing has written into yet — the validator refuses a second
|
|
257
|
+
run in the same place (deviation D3).
|
|
258
|
+
|
|
259
|
+
The result is data. A zero exit code means the validator ran, not that
|
|
260
|
+
the taskset is valid (``docs/verifiers-pin-0.3.1.md``, finding C2), so
|
|
261
|
+
the only failure this raises is one where no summary could exist at all.
|
|
262
|
+
"""
|
|
263
|
+
ensure_private_directory(output_dir)
|
|
264
|
+
run_dir = validation_run_dir(output_dir)
|
|
265
|
+
if run_dir.exists():
|
|
266
|
+
raise EngineError(
|
|
267
|
+
f"a validation run has already written into {run_dir}; the "
|
|
268
|
+
"pinned validator writes a run directory once, so every "
|
|
269
|
+
"validation needs one of its own",
|
|
270
|
+
code="validation_run_directory_reused",
|
|
271
|
+
details={"taskset_id": taskset_id, "path": str(run_dir)},
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
result = self._runner.run(
|
|
275
|
+
VALIDATE_EXECUTABLE,
|
|
276
|
+
validate_argv(
|
|
277
|
+
taskset_id=taskset_id,
|
|
278
|
+
num_tasks=num_tasks,
|
|
279
|
+
output_dir=output_dir,
|
|
280
|
+
),
|
|
281
|
+
timeout=self._timeout_seconds,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
if not (run_dir / _SUMMARY_FILE).is_file():
|
|
285
|
+
raise EngineError(
|
|
286
|
+
f"the validator produced no summary for taskset {taskset_id}: "
|
|
287
|
+
f"{_last_line(result.stderr)}",
|
|
288
|
+
code="taskset_validation_failed",
|
|
289
|
+
details={
|
|
290
|
+
"taskset_id": taskset_id,
|
|
291
|
+
"engine_digest": self._engine_digest,
|
|
292
|
+
"exit_code": result.exit_code,
|
|
293
|
+
},
|
|
294
|
+
)
|
|
295
|
+
return result
|
|
296
|
+
|
|
297
|
+
def normalize(
|
|
298
|
+
self,
|
|
299
|
+
output_dir: Path,
|
|
300
|
+
*,
|
|
301
|
+
taskset_lock_digest: Digest,
|
|
302
|
+
) -> ValidationEvidence:
|
|
303
|
+
"""Return the deterministic evidence the engine derived from a run.
|
|
304
|
+
|
|
305
|
+
The transformation happens inside the engine because only the engine
|
|
306
|
+
can name the Verifiers commit that actually produced the output. The
|
|
307
|
+
helper writes readable JSON to a scratch file; the caller decides where
|
|
308
|
+
the canonical bytes finally live.
|
|
309
|
+
"""
|
|
310
|
+
script = self._registry.tool_path(
|
|
311
|
+
self._engine_digest, NORMALIZE_VALIDATION_TOOL
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
with tempfile.TemporaryDirectory(prefix=_TEMPORARY_PREFIX) as directory:
|
|
315
|
+
destination = Path(directory) / _EVIDENCE_FILENAME
|
|
316
|
+
result = self._runner.run_python_script(
|
|
317
|
+
script,
|
|
318
|
+
[
|
|
319
|
+
"--results-dir",
|
|
320
|
+
str(validation_run_dir(output_dir)),
|
|
321
|
+
"--taskset-lock-digest",
|
|
322
|
+
taskset_lock_digest,
|
|
323
|
+
"--output",
|
|
324
|
+
str(destination),
|
|
325
|
+
],
|
|
326
|
+
timeout=self._timeout_seconds,
|
|
327
|
+
)
|
|
328
|
+
if result.exit_code != 0:
|
|
329
|
+
raise EngineError(
|
|
330
|
+
"the engine could not normalize this validation run: "
|
|
331
|
+
f"{_last_line(result.stderr or result.stdout)}",
|
|
332
|
+
code="validation_evidence_unavailable",
|
|
333
|
+
details={
|
|
334
|
+
"engine_digest": self._engine_digest,
|
|
335
|
+
"exit_code": result.exit_code,
|
|
336
|
+
},
|
|
337
|
+
)
|
|
338
|
+
raw = destination.read_bytes()
|
|
339
|
+
|
|
340
|
+
try:
|
|
341
|
+
return ValidationEvidence.model_validate_json(raw)
|
|
342
|
+
except PydanticValidationError as error:
|
|
343
|
+
raise ValidationError(
|
|
344
|
+
"the engine's normalized validation evidence is not a valid "
|
|
345
|
+
f"evidence document: {error.errors()[0]['msg']}",
|
|
346
|
+
code="validation_evidence_invalid",
|
|
347
|
+
details={"engine_digest": self._engine_digest},
|
|
348
|
+
) from error
|
|
349
|
+
|
|
350
|
+
# -- reading -----------------------------------------------------------
|
|
351
|
+
|
|
352
|
+
def parse_summary(self, output_dir: Path) -> UpstreamValidationSummary:
|
|
353
|
+
"""Project ``summary.json`` into the protocol's flat summary.
|
|
354
|
+
|
|
355
|
+
Field by field, because every one of the five outcome counts sits at
|
|
356
|
+
the wrong depth upstream and two of the document's keys have no
|
|
357
|
+
protocol meaning at all (``docs/verifiers-pin-0.3.1.md``, finding C1).
|
|
358
|
+
"""
|
|
359
|
+
run_dir = validation_run_dir(output_dir)
|
|
360
|
+
document = self._summary_document(run_dir)
|
|
361
|
+
outcomes = _require_object(document, "outcomes", run_dir)
|
|
362
|
+
|
|
363
|
+
try:
|
|
364
|
+
return UpstreamValidationSummary(
|
|
365
|
+
mode=_require_str(document, "mode", run_dir), # type: ignore[arg-type]
|
|
366
|
+
total=_require_count(document, "total", run_dir),
|
|
367
|
+
recorded=_require_count(document, "recorded", run_dir),
|
|
368
|
+
valid=_require_count(outcomes, "valid", run_dir),
|
|
369
|
+
invalid=_require_count(outcomes, "invalid", run_dir),
|
|
370
|
+
error=_require_count(outcomes, "error", run_dir),
|
|
371
|
+
timeout=_require_count(outcomes, "timeout", run_dir),
|
|
372
|
+
missing=_require_count(outcomes, "missing", run_dir),
|
|
373
|
+
valid_rate=_optional_rate(document, run_dir),
|
|
374
|
+
)
|
|
375
|
+
except PydanticValidationError as error:
|
|
376
|
+
raise ValidationError(
|
|
377
|
+
"the validator's summary does not describe a coherent run: "
|
|
378
|
+
f"{error.errors()[0]['msg']}",
|
|
379
|
+
code="validation_summary_invalid",
|
|
380
|
+
details={"path": str(run_dir / _SUMMARY_FILE)},
|
|
381
|
+
) from error
|
|
382
|
+
|
|
383
|
+
def parse_check_counts(self, output_dir: Path) -> dict[str, UpstreamCheckCounts]:
|
|
384
|
+
"""Return the per-check breakdown, empty when the validator reported none.
|
|
385
|
+
|
|
386
|
+
Upstream writes ``checks`` only for a run in which both checks ran.
|
|
387
|
+
Techtree always asks for both, so an empty result is something the
|
|
388
|
+
caller reports as a failed check rather than something this hides.
|
|
389
|
+
"""
|
|
390
|
+
run_dir = validation_run_dir(output_dir)
|
|
391
|
+
document = self._summary_document(run_dir)
|
|
392
|
+
checks = document.get("checks")
|
|
393
|
+
if checks is None:
|
|
394
|
+
return {}
|
|
395
|
+
if not isinstance(checks, dict):
|
|
396
|
+
raise ValidationError(
|
|
397
|
+
"the validator's summary records a checks breakdown that is "
|
|
398
|
+
"not an object",
|
|
399
|
+
code="validation_summary_invalid",
|
|
400
|
+
details={"path": str(run_dir / _SUMMARY_FILE)},
|
|
401
|
+
)
|
|
402
|
+
|
|
403
|
+
counts: dict[str, UpstreamCheckCounts] = {}
|
|
404
|
+
for name, reported in checks.items():
|
|
405
|
+
if not isinstance(reported, dict):
|
|
406
|
+
raise ValidationError(
|
|
407
|
+
f"the validator's summary records the {name} check as "
|
|
408
|
+
"something other than a set of counts",
|
|
409
|
+
code="validation_summary_invalid",
|
|
410
|
+
details={"path": str(run_dir / _SUMMARY_FILE)},
|
|
411
|
+
)
|
|
412
|
+
counts[str(name)] = UpstreamCheckCounts(
|
|
413
|
+
**{key: _require_count(reported, key, run_dir) for key in _OUTCOME_KEYS}
|
|
414
|
+
)
|
|
415
|
+
return counts
|
|
416
|
+
|
|
417
|
+
def validation_artifacts(self, output_dir: Path) -> list[ArtifactRef]:
|
|
418
|
+
"""Digest the four files one validation run leaves behind.
|
|
419
|
+
|
|
420
|
+
These are raw execution outputs, not evidence: ``results.jsonl`` is
|
|
421
|
+
written in completion order and the log carries wall-clock times, so
|
|
422
|
+
two identical runs digest differently by construction (finding C0).
|
|
423
|
+
They belong to the local
|
|
424
|
+
:class:`~techtree.models.validation.ValidationExecutionRecord` and are
|
|
425
|
+
never referenced by a receipt or shipped in a catalog.
|
|
426
|
+
"""
|
|
427
|
+
run_dir = validation_run_dir(output_dir)
|
|
428
|
+
artifacts: list[ArtifactRef] = []
|
|
429
|
+
for name in VALIDATION_FILENAMES:
|
|
430
|
+
path = run_dir / name
|
|
431
|
+
try:
|
|
432
|
+
data = path.read_bytes()
|
|
433
|
+
except OSError as error:
|
|
434
|
+
raise ValidationError(
|
|
435
|
+
f"the validation run left no {name} behind",
|
|
436
|
+
code="validation_artifact_missing",
|
|
437
|
+
details={"path": str(path)},
|
|
438
|
+
) from error
|
|
439
|
+
artifacts.append(
|
|
440
|
+
ArtifactRef(
|
|
441
|
+
digest=sha256_digest_bytes(data),
|
|
442
|
+
media_type=_MEDIA_TYPES[name],
|
|
443
|
+
size=len(data),
|
|
444
|
+
relative_path=name,
|
|
445
|
+
)
|
|
446
|
+
)
|
|
447
|
+
return artifacts
|
|
448
|
+
|
|
449
|
+
# -- one read of one small file ----------------------------------------
|
|
450
|
+
|
|
451
|
+
def _summary_document(self, run_dir: Path) -> dict[str, Any]:
|
|
452
|
+
path = run_dir / _SUMMARY_FILE
|
|
453
|
+
try:
|
|
454
|
+
raw = path.read_bytes()
|
|
455
|
+
except OSError as error:
|
|
456
|
+
raise ValidationError(
|
|
457
|
+
f"the validation run wrote no summary at {path}",
|
|
458
|
+
code="validation_summary_missing",
|
|
459
|
+
details={"path": str(path)},
|
|
460
|
+
) from error
|
|
461
|
+
|
|
462
|
+
try:
|
|
463
|
+
document = json.loads(raw)
|
|
464
|
+
except ValueError as error:
|
|
465
|
+
raise ValidationError(
|
|
466
|
+
f"the validator's summary is not JSON: {path}",
|
|
467
|
+
code="validation_summary_invalid",
|
|
468
|
+
details={"path": str(path)},
|
|
469
|
+
) from error
|
|
470
|
+
|
|
471
|
+
if not isinstance(document, dict):
|
|
472
|
+
raise ValidationError(
|
|
473
|
+
f"the validator's summary is not a JSON object: {path}",
|
|
474
|
+
code="validation_summary_invalid",
|
|
475
|
+
details={"path": str(path)},
|
|
476
|
+
)
|
|
477
|
+
if document.get("mode") != _BOTH_CHECKS:
|
|
478
|
+
raise ValidationError(
|
|
479
|
+
f"this validation ran in mode {document.get('mode')!r}; Techtree "
|
|
480
|
+
f"pins mode {_BOTH_CHECKS!r}, which runs both checks",
|
|
481
|
+
code="validation_summary_invalid",
|
|
482
|
+
details={"path": str(path), "mode": str(document.get("mode"))},
|
|
483
|
+
)
|
|
484
|
+
return document
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def _require_object(
|
|
488
|
+
document: Mapping[str, Any], key: str, run_dir: Path
|
|
489
|
+
) -> dict[str, Any]:
|
|
490
|
+
value = document.get(key)
|
|
491
|
+
if not isinstance(value, dict):
|
|
492
|
+
raise ValidationError(
|
|
493
|
+
f"the validator's summary has no {key} object",
|
|
494
|
+
code="validation_summary_invalid",
|
|
495
|
+
details={"path": str(run_dir / _SUMMARY_FILE), "field": key},
|
|
496
|
+
)
|
|
497
|
+
return value
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def _require_str(document: Mapping[str, Any], key: str, run_dir: Path) -> str:
|
|
501
|
+
value = document.get(key)
|
|
502
|
+
if not isinstance(value, str) or not value:
|
|
503
|
+
raise ValidationError(
|
|
504
|
+
f"the validator's summary field {key} is not a name: {value!r}",
|
|
505
|
+
code="validation_summary_invalid",
|
|
506
|
+
details={"path": str(run_dir / _SUMMARY_FILE), "field": key},
|
|
507
|
+
)
|
|
508
|
+
return value
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
def _require_count(document: Mapping[str, Any], key: str, run_dir: Path) -> int:
|
|
512
|
+
value = document.get(key)
|
|
513
|
+
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
|
|
514
|
+
raise ValidationError(
|
|
515
|
+
f"the validator's summary field {key} is not a count: {value!r}",
|
|
516
|
+
code="validation_summary_invalid",
|
|
517
|
+
details={"path": str(run_dir / _SUMMARY_FILE), "field": key},
|
|
518
|
+
)
|
|
519
|
+
return value
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _optional_rate(document: Mapping[str, Any], run_dir: Path) -> float | None:
|
|
523
|
+
value = document.get("valid_rate")
|
|
524
|
+
if value is None:
|
|
525
|
+
return None
|
|
526
|
+
if isinstance(value, bool) or not isinstance(value, int | float):
|
|
527
|
+
raise ValidationError(
|
|
528
|
+
f"the validator's summary field valid_rate is not a rate: {value!r}",
|
|
529
|
+
code="validation_summary_invalid",
|
|
530
|
+
details={"path": str(run_dir / _SUMMARY_FILE), "field": "valid_rate"},
|
|
531
|
+
)
|
|
532
|
+
return float(value)
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def _last_line(text: str) -> str:
|
|
536
|
+
"""Return the last meaningful line of a child process's diagnostics."""
|
|
537
|
+
lines: Sequence[str] = [line.strip() for line in text.splitlines() if line.strip()]
|
|
538
|
+
return lines[-1] if lines else "the engine reported no diagnostics"
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Turning a finished evaluation into a run's result. Spec section 7.20.
|
|
2
|
+
|
|
3
|
+
WP6 ends with two variants' evidence written into a run directory and a run
|
|
4
|
+
still in ``running_variants``. Everything between that and a completed run —
|
|
5
|
+
receipts, ordered commitments, the controlled comparison, the aggregation and
|
|
6
|
+
the report — is a sequence of pure functions over the run's own files, and this
|
|
7
|
+
package is where that sequence is written down once so the worker can call it.
|
|
8
|
+
|
|
9
|
+
``service``
|
|
10
|
+
The stage that closes a real run: build, commit, compare, aggregate,
|
|
11
|
+
report, record.
|
|
12
|
+
|
|
13
|
+
Spec section 7.20's ``UpliftService`` — sanitized improvement context and
|
|
14
|
+
Skill-replacement preparation — is a different object with a different job and
|
|
15
|
+
belongs to WP7d. It will live beside this one.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
__all__: list[str] = []
|