crucible-bench 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. crucible_bench-1.1.0/ARCHITECTURE.md +220 -0
  2. crucible_bench-1.1.0/CHANGELOG.md +290 -0
  3. crucible_bench-1.1.0/LICENSE +78 -0
  4. crucible_bench-1.1.0/MANIFEST.in +6 -0
  5. crucible_bench-1.1.0/PKG-INFO +400 -0
  6. crucible_bench-1.1.0/README.md +379 -0
  7. crucible_bench-1.1.0/docs/ENTERPRISE-READINESS.md +48 -0
  8. crucible_bench-1.1.0/docs/RELEASE-READINESS.md +84 -0
  9. crucible_bench-1.1.0/examples/batch-binary-search.json +14 -0
  10. crucible_bench-1.1.0/examples/demo.py +74 -0
  11. crucible_bench-1.1.0/examples/measurements-binary-search.json +18 -0
  12. crucible_bench-1.1.0/examples/refine-discovery-loop.json +38 -0
  13. crucible_bench-1.1.0/examples/substrate-binary-search.json +17 -0
  14. crucible_bench-1.1.0/examples/thesis-binary-search.json +18 -0
  15. crucible_bench-1.1.0/pyproject.toml +45 -0
  16. crucible_bench-1.1.0/requirements-release.txt +3 -0
  17. crucible_bench-1.1.0/setup.cfg +4 -0
  18. crucible_bench-1.1.0/src/crucible/__init__.py +98 -0
  19. crucible_bench-1.1.0/src/crucible/__main__.py +8 -0
  20. crucible_bench-1.1.0/src/crucible/assess.py +295 -0
  21. crucible_bench-1.1.0/src/crucible/batch_cmd.py +142 -0
  22. crucible_bench-1.1.0/src/crucible/claim.py +59 -0
  23. crucible_bench-1.1.0/src/crucible/cli.py +215 -0
  24. crucible_bench-1.1.0/src/crucible/commands.py +293 -0
  25. crucible_bench-1.1.0/src/crucible/drift.py +105 -0
  26. crucible_bench-1.1.0/src/crucible/drift_cmd.py +63 -0
  27. crucible_bench-1.1.0/src/crucible/ecosystem_measure.py +255 -0
  28. crucible_bench-1.1.0/src/crucible/flagship.py +150 -0
  29. crucible_bench-1.1.0/src/crucible/gate.py +64 -0
  30. crucible_bench-1.1.0/src/crucible/mcp.py +71 -0
  31. crucible_bench-1.1.0/src/crucible/mcp_tools.py +284 -0
  32. crucible_bench-1.1.0/src/crucible/measure.py +100 -0
  33. crucible_bench-1.1.0/src/crucible/measurement_gate.py +292 -0
  34. crucible_bench-1.1.0/src/crucible/measurement_gate_cmd.py +45 -0
  35. crucible_bench-1.1.0/src/crucible/recheck_cmd.py +212 -0
  36. crucible_bench-1.1.0/src/crucible/refine.py +279 -0
  37. crucible_bench-1.1.0/src/crucible/refine_cmd.py +91 -0
  38. crucible_bench-1.1.0/src/crucible/registry.py +281 -0
  39. crucible_bench-1.1.0/src/crucible/registry_cmd.py +151 -0
  40. crucible_bench-1.1.0/src/crucible/registry_ops.py +196 -0
  41. crucible_bench-1.1.0/src/crucible/report.py +126 -0
  42. crucible_bench-1.1.0/src/crucible/report_cmd.py +39 -0
  43. crucible_bench-1.1.0/src/crucible/review_cmd.py +211 -0
  44. crucible_bench-1.1.0/src/crucible/review_contract.py +144 -0
  45. crucible_bench-1.1.0/src/crucible/run_cmd.py +175 -0
  46. crucible_bench-1.1.0/src/crucible/steelman.py +80 -0
  47. crucible_bench-1.1.0/src/crucible/subprocess_edges.py +263 -0
  48. crucible_bench-1.1.0/src/crucible/telos_measure.py +148 -0
  49. crucible_bench-1.1.0/src/crucible/thesis.py +79 -0
  50. crucible_bench-1.1.0/src/crucible/verdict.py +111 -0
  51. crucible_bench-1.1.0/src/crucible_bench.egg-info/PKG-INFO +400 -0
  52. crucible_bench-1.1.0/src/crucible_bench.egg-info/SOURCES.txt +85 -0
  53. crucible_bench-1.1.0/src/crucible_bench.egg-info/dependency_links.txt +1 -0
  54. crucible_bench-1.1.0/src/crucible_bench.egg-info/entry_points.txt +2 -0
  55. crucible_bench-1.1.0/src/crucible_bench.egg-info/requires.txt +8 -0
  56. crucible_bench-1.1.0/src/crucible_bench.egg-info/top_level.txt +1 -0
  57. crucible_bench-1.1.0/tests/test_assess.py +242 -0
  58. crucible_bench-1.1.0/tests/test_claim.py +50 -0
  59. crucible_bench-1.1.0/tests/test_cli.py +303 -0
  60. crucible_bench-1.1.0/tests/test_cli_batch.py +251 -0
  61. crucible_bench-1.1.0/tests/test_cli_drift.py +119 -0
  62. crucible_bench-1.1.0/tests/test_cli_export.py +49 -0
  63. crucible_bench-1.1.0/tests/test_cli_measure.py +92 -0
  64. crucible_bench-1.1.0/tests/test_cli_measurement_gate.py +71 -0
  65. crucible_bench-1.1.0/tests/test_cli_recheck.py +168 -0
  66. crucible_bench-1.1.0/tests/test_cli_refine.py +51 -0
  67. crucible_bench-1.1.0/tests/test_cli_registry_ops.py +78 -0
  68. crucible_bench-1.1.0/tests/test_cli_report.py +77 -0
  69. crucible_bench-1.1.0/tests/test_cli_run.py +287 -0
  70. crucible_bench-1.1.0/tests/test_cli_validation.py +60 -0
  71. crucible_bench-1.1.0/tests/test_drift.py +88 -0
  72. crucible_bench-1.1.0/tests/test_ecosystem_measure.py +164 -0
  73. crucible_bench-1.1.0/tests/test_flagship_cli.py +37 -0
  74. crucible_bench-1.1.0/tests/test_gate.py +47 -0
  75. crucible_bench-1.1.0/tests/test_mcp.py +297 -0
  76. crucible_bench-1.1.0/tests/test_measure.py +71 -0
  77. crucible_bench-1.1.0/tests/test_measurement_gate.py +133 -0
  78. crucible_bench-1.1.0/tests/test_readiness.py +193 -0
  79. crucible_bench-1.1.0/tests/test_refine.py +108 -0
  80. crucible_bench-1.1.0/tests/test_registry.py +205 -0
  81. crucible_bench-1.1.0/tests/test_registry_ops.py +132 -0
  82. crucible_bench-1.1.0/tests/test_report.py +75 -0
  83. crucible_bench-1.1.0/tests/test_steelman.py +60 -0
  84. crucible_bench-1.1.0/tests/test_subprocess_edges.py +135 -0
  85. crucible_bench-1.1.0/tests/test_telos_measure.py +113 -0
  86. crucible_bench-1.1.0/tests/test_thesis.py +79 -0
  87. crucible_bench-1.1.0/tests/test_verdict.py +104 -0
@@ -0,0 +1,220 @@
1
+ # crucible: architecture
2
+
3
+ crucible is the cognition organ of the constellation: it tests a thesis against evidence and emits a
4
+ verdict you can re-check. This document is the map of how it is built and why. It grows as the organ
5
+ does; sections describe what is shipped.
6
+
7
+ ## The one shape: a grounded `Verdict`
8
+
9
+ Every judgment crucible makes reduces to one shape: a verdict per claim, computed from a measurement,
10
+ never asserted.
11
+
12
+ ```python
13
+ def verdict_for(claim: Claim, measurement: Measurement | None) -> Verdict: ...
14
+ ```
15
+
16
+ A measurement records a deviation from what the claim predicts and a tolerance. The verdict is a
17
+ pure function of that record: within tolerance is MATCH, outside is DRIFT, absent or unmeasurable is
18
+ UNVERIFIABLE. There is no model in this step, so the verdict recomputes from the stored record and a
19
+ confident assertion has no effect on the rechecked result. UNVERIFIABLE is fail-closed: an axis that
20
+ cannot be measured is never read as holding.
21
+
22
+ ## The receipt: `Claim`
23
+
24
+ Every claim carries a content hash, so a tampered claim is caught by re-hashing.
25
+
26
+ ```python
27
+ @dataclass(frozen=True, slots=True)
28
+ class Claim:
29
+ id: str # defaults to sha256[:16]
30
+ text: str # the assertion
31
+ falsification: str # the observation that would refute it
32
+ sha256: str # content hash binding (text, falsification)
33
+ ```
34
+
35
+ `make_claim` computes the hash; `Claim.verify()` re-hashes and confirms it still matches. A claim
36
+ with no falsification condition can only ever be UNVERIFIABLE: nothing would settle it.
37
+
38
+ ## The thesis and its seal
39
+
40
+ A `Thesis` is a set of claims with a re-checkable seal over them (sorted, canonical JSON, sha256),
41
+ mirroring Gather's digest seal. The seal folds in the title, the disposition, and each claim's id and
42
+ content hash, so swapping a claim's content, relabelling its id, or flipping the disposition breaks
43
+ the seal; reordering does not, because a thesis is a set. The disposition being sealed is what lets
44
+ the publication gate trust the label.
45
+
46
+ ## The witnessed assessment
47
+
48
+ An assessment folds the per-claim verdicts into a record with its own seal, and it persists the
49
+ verdicts and the measurements alongside, so the seal has a preimage on disk rather than fingerprinting
50
+ discarded data. `verify_assessment` recomputes every seal from the stored arrays (the verdict seal
51
+ from the stored verdicts, the measurement seal from the stored measurements, the record seal from the
52
+ fields that bind both); it is not a tautology. `recheck_assessment` goes further: from the thesis and
53
+ the stored measurements it re-derives each verdict via `verdict_for` and confirms the stored verdicts
54
+ are exactly what the pure function yields, so a verdict that was asserted rather than computed is
55
+ exposed even when its seals are internally consistent. The CLI surfaces this as `crucible verdicts
56
+ --verify`. The seal proves integrity, not authorship: a fully consistent re-forge (every field and
57
+ seal rewritten together) is out of scope without a signature, and the docs say so where a user meets it.
58
+ Summary counts are not trusted as labels; verification re-derives them from the verdict rows.
59
+ Assessment and verdict rows also carry the thesis disposition, and the disposition participates in
60
+ the assessment and verdict seals so publication posture is visible in witnessed outputs.
61
+
62
+ Measurement rows persist and seal claim binding, deviation, tolerance, method, `measured_at`,
63
+ evidence, and optional `recheck` descriptors. When a descriptor is present, it is included in the
64
+ measurement seal; when absent, legacy rows keep the original replay shape and are skipped by
65
+ oracle-level replay. `recheck_measurements` is the oracle-level hook: a caller provides replay
66
+ functions keyed by descriptor `oracle`, and crucible compares the replayed measurement inputs to the
67
+ stored row. The shipped CLI still re-derives verdicts from stored measurements; external callers can
68
+ now also re-run descriptor-bearing measurements. `crucible recheck REGISTRY` exposes the descriptor
69
+ plan for the latest assessment, `--template FILE` writes a replay pack skeleton for the verifier to
70
+ fill, and `--pack FILE` checks a finished oracle replay pack against those sealed measurement rows
71
+ without creating a second verdict path. If a replay pack carries the template assessment block, the
72
+ block must match the selected thesis id, assessment seal, and measurement seal before replay starts.
73
+
74
+ `render_assessment_report` turns the same sealed record into a deterministic Markdown artifact. It
75
+ does not change the verdict contract or decide anything new; it gives an operator a readable surface
76
+ over the counts, seals, integrity checks, verdicts, evidence, and recheck descriptors. The CLI exposes
77
+ this as `crucible report REGISTRY`, defaulting to the latest assessment.
78
+
79
+ `crucible batch` is a runner over the same primitives, not a second judgment path. A manifest lists
80
+ thesis jobs and chooses either explicit measurements or a substrate oracle per job; the command
81
+ records each assessment into one registry and can write one Markdown report per job. The batch layer
82
+ coordinates throughput while leaving claim receipts, measurements, verdicts, assessment seals, and
83
+ report rendering unchanged.
84
+
85
+ `crucible run` is the single-thesis operator surface over the same path. It runs the null steelman,
86
+ loads either explicit measurements or a table substrate, records the witnessed assessment, reloads the
87
+ latest registry record for a disk recheck, and can write both the Markdown report and a JSON run
88
+ record. With `--bundle DIR`, it writes `DIR/spec.json`, `DIR/run.json`, `DIR/report.md`, and
89
+ `DIR/review.md` as a cleanroom review packet. The run record names packet artifacts relative to the
90
+ packet root, not by local workstation path. Review verifies those path fields before verifier
91
+ handoff. It does not introduce a second source of truth: the
92
+ assessment and verdict rows remain the authority, and the run record is the session envelope around
93
+ them.
94
+
95
+ ## The seams (the impure and the optional)
96
+
97
+ crucible's core is pure standard library. Optional edges live behind a Protocol seam with a Null
98
+ default, exactly as Gather isolates its synthesizer.
99
+
100
+ - **Steelman** (`Steelman` protocol): independent adversaries propose refutations. The default
101
+ `NullSteelman` surfaces the claim's own stated falsification as the standing test and invents
102
+ nothing; custom edges plug in to generate independent refutations. Adversaries propose what to
103
+ measure; the measurement decides. `steelman_thesis` stamps each refutation's source from the
104
+ producing steelman's name, so the label is the producer's.
105
+ - **Measure** (`Measure` protocol): the sound-oracle edge that decides a claim against a substrate.
106
+ The default `NullMeasure` measures nothing (UNVERIFIABLE); the deterministic `TableMeasure` computes
107
+ a claim's deviation from a predicted value over a provided substrate (offline, no model). A real
108
+ oracle (the Telos verifier, or a symbolic or proof oracle for abstract math) plugs in through the
109
+ same shape, and `verdict_for` decides from the `Measurement` it produces. This is where the verdict
110
+ is grounded.
111
+ - **Refine** (`refine_thesis`): the continuous loop that measures a thesis, grades each claim by its
112
+ normalized margin, computes harmonic-mean cohesion, reflects the weakest claim, and re-iterates via
113
+ a caller-provided adjuster. It never returns `correct` unless every claim has enough margin and the
114
+ margins are cohesive; if the budget is spent it reports the weakest claim.
115
+
116
+ The core imports neither the Null nor any model, so the package keeps zero third-party dependencies.
117
+
118
+ ## Subprocess seam adapters
119
+
120
+ `SubprocessSteelman` and `SubprocessMeasure` are optional stdlib adapters for configured commands.
121
+ They exchange one bounded JSON request/response over stdin/stdout, enforce a timeout, and reject shell
122
+ strings so arguments are not re-parsed by a shell. By default they pass only a minimal environment,
123
+ discard stderr, and terminate children whose stdout grows past the response cap. A child process may
124
+ propose a challenge or report a deviation, but crucible stamps the claim identity, claim hash, and
125
+ producer name locally. The verdict still follows from `verdict_for`; a subprocess cannot assert MATCH.
126
+
127
+ ## Telos artifact interop
128
+
129
+ `TelosMeasure` consumes the Telos engine's `telos.witnessed-artifact/v1` protocol without importing
130
+ Telos. The artifact carries a compact certificate and a `recheck` descriptor naming a verifier. A
131
+ caller supplies a verifier registry; crucible re-runs the named verifier, compares the reproduced
132
+ verdict to the carried verdict, and turns that live result into a normal Measurement. Reproduced
133
+ `verified` becomes a MATCH input, reproduced `refuted` or a drifted carried verdict becomes a DRIFT
134
+ input, and a missing/unregistered/unverifiable proof becomes an UNVERIFIABLE input. That keeps the
135
+ interop on the same spine: trust the proof, not the emitter. When the Telos artifact is well-shaped,
136
+ the produced Measurement persists a `telos:<verifier>` recheck descriptor so later assessment replay
137
+ can re-run the same oracle from the stored row.
138
+
139
+ ## Gather/index interop
140
+
141
+ `GatherDigestMeasure` consumes Gather's digest contract directly: a list of evidence receipts and a
142
+ seal recomputed from those receipts. A caller maps a crucible claim to the receipt fields it expects
143
+ to exist. A verified digest with that receipt becomes a MATCH input; a verified digest without it
144
+ becomes a DRIFT input; a malformed or missing digest becomes UNVERIFIABLE. This makes "show me the
145
+ evidence receipt" a measurable claim.
146
+
147
+ `IndexMeasure` consumes `index.verification/1` records directly. The record carries an index claim,
148
+ the canonical hash of the graph pack it was computed against, and a carried verdict. crucible is
149
+ given the graph pack by hash, replays the supported structural claim (`exists` or `depends`), and
150
+ maps the reproduced result into the same Measurement spine. It does not import index, execute index,
151
+ or trust the carried verdict without replaying the pack.
152
+
153
+ ## Drift tracking
154
+
155
+ The continuous loop needs an honest account of what changed between rounds. `drift_track(previous,
156
+ current)` compares two witnessed assessments of the same thesis and classifies each claim as held,
157
+ moved, improved, or regressed. Numeric margins decide improvement and regression. Unrankable
158
+ transitions, such as UNVERIFIABLE to MATCH, are reported as moved rather than silently promoted.
159
+ The CLI exposes this as `crucible drift REGISTRY`, comparing the latest two stored assessments.
160
+
161
+ ## Publication gate
162
+
163
+ Assessment and export are deliberately separate. A fenced thesis may be registered and assessed
164
+ locally, but the public export edge applies `gate_check` and refuses anything with a fenced
165
+ disposition or an explicit fenced/restricted marker in the title, claim text, or falsification. The
166
+ exported contract omits runtime metadata and carries only the title, disposition, thesis seal, and
167
+ content-hashed claims needed for public re-checking. This gate is a mechanical disposition and marker
168
+ guard, not a semantic content classifier.
169
+
170
+ ## The registry
171
+
172
+ A `Registry` is durable, content-addressed storage for theses and their assessments, mirroring
173
+ Gather's corpus: claim bodies at `objects/ab/cdef...` keyed by the claim hash, a `theses.jsonl`
174
+ catalog, and an `assessments.jsonl` history. `verify()` re-hashes every stored body and reports
175
+ MATCH / MISSING / CORRUPT, so the verdict's proof stays durable over a growing registry.
176
+
177
+ `registry_ops` reads across that store without changing the storage contract. `registry_stats`
178
+ summarizes thesis counts, claim bodies, dispositions, assessment history, skipped invalid latest
179
+ rows, and the latest verified verdict posture per thesis. `search_theses` recalls theses by scope
180
+ text, thesis status, and latest verified verdict status, falling back past invalid tail rows instead
181
+ of trusting or hiding history. `prune_objects` identifies orphaned claim bodies and is dry-run by
182
+ default; deletion requires an explicit apply path and validates the object root plus each object path
183
+ with the registry realpath guard before unlinking it. The registry rejects duplicate thesis ids with
184
+ different seals and refuses symlinked storage paths, so content-addressed writes stay inside the
185
+ registry root.
186
+
187
+ ## Verifier separation
188
+
189
+ The review loop is intentionally clean. A verifier receives the original spec/readiness docs and the
190
+ artifact under review. It does not receive the worker's context, reasoning trace, or intermediate
191
+ steps. If success cannot be evaluated from that minimal state, the spec is not checkable yet and the
192
+ readiness artifact needs work before release. For run packets, `crucible review BUNDLE` makes that
193
+ rule executable: extra context files fail the packet before any verifier judgment begins, and
194
+ `run.json` must declare passing embedded integrity checks, `report.md` must match the assessment
195
+ artifact rendered from `run.json`, `run.json` artifact path fields must remain packet-relative,
196
+ and `review.md` must match the canonical cleanroom instructions.
197
+
198
+ ## Determinism and the zero-dependency core
199
+
200
+ Clocks are injected everywhere time is recorded; iteration is sorted; JSON is canonical
201
+ (`sort_keys`, `ensure_ascii=False`). So an assessment replays and a seal recomputes. The core is pure
202
+ standard library. A custom edge may pull in whatever it needs, but only behind a seam, and only at
203
+ the impure edge.
204
+
205
+ ## Protocol interoperability (the dual mandate)
206
+
207
+ crucible stands alone and serves the constellation at once. At the 1.0 flagship floor, the
208
+ standing-alone half and the published contract are shipped: a sealed assessment and its verdicts,
209
+ re-checkable from disk, that a downstream organ reads to learn a thesis's standing. It also includes
210
+ protocol adapters for Telos witnessed artifacts, Gather witnessed digests, and index verification
211
+ records. These adapters consume documented JSON contracts without importing sibling internals. Shared
212
+ primitives, such as the `refine` loop, are integrated natively rather than taken as third-party
213
+ dependencies, so reuse never costs standing alone. Seams default to Null, so the absence of a peer is
214
+ a quieter capability, never a failure.
215
+
216
+ ## Peer composition
217
+
218
+ crucible composes with the rest of the constellation (telos, index, forum, gather) through clean
219
+ protocol seams; it does not absorb or get absorbed. The sealed assessment is the contract a
220
+ downstream reader consumes. This is why crucible is a peer organ, not a feature of index or of refine.
@@ -0,0 +1,290 @@
1
+ # Changelog
2
+
3
+ All notable changes to crucible. Versions follow semantic versioning; each minor release is built
4
+ behind a feature branch and reviewed before merge.
5
+
6
+ ## Unreleased
7
+
8
+ - CLI compatibility: `python -m crucible` now dispatches the normal Crucible CLI, so source
9
+ checkouts, MCP hosts, IDE harnesses, and automation runners can use the same command surface as
10
+ the installed `crucible` script.
11
+ - Creative measurement gate: adds `crucible measurement-gate PACKET [--criteria FILE]` and the
12
+ `crucible.measurement_gate` MCP tool for verifying Telos histogram, dither, Gaussian-splat,
13
+ clustered-lighting, and audio-spectral measurement packets without exporting raw pixels, assets,
14
+ prompts, tool arguments, or full payloads.
15
+ - Gate verdicts: keeps `decision_outcome` (`allow`, `require_review`, `block`) separate from
16
+ `verification_verdict` (MATCH, DRIFT, UNVERIFIABLE) and emits normalized failure codes for operator
17
+ alerting, including raw payload leaks, cluster budget overruns, dither-pattern gaps, provenance
18
+ gaps, pixel-dimension mismatches, and audio-spectrum gaps.
19
+ - MCP parity: expands the stdio MCP server beyond status/doctor/assess/recheck to host-call the run, review, report, batch, registry, drift, refine, and verdicts workflows through the existing CLI contract.
20
+
21
+ - Enterprise readiness: adds `docs/ENTERPRISE-READINESS.md` for context envelopes, action receipts, readability gates, and host-neutral operation.
22
+ - Operator surface: the status payload now advertises shared Project Telos CLI/MCP/plugin/IDE/TUI/app contracts for enterprise, research, creative, scientific, and education workflows.
23
+
24
+ Presentation and operator-surface housekeeping for Project Telos parity.
25
+
26
+ - README: brings Crucible up to the shared five-flagship presentation shape with title-case product naming, current CI badge, consistent navigation, and a current-status block.
27
+ - Status copy: updates the visible status from the old 1.0 flagship floor to the 1.1 operator floor.
28
+ - Status payload: exposes the primary workflow commands, current operator commands, integration surfaces, presentation freshness, MCP tool names, and the 1.1 operator-floor summary under `native`.
29
+ - MCP tools: records native availability for `crucible.status`, `crucible.doctor`, `crucible.assess`, `crucible.measurement_gate`, `crucible.recheck`, `crucible.run`, `crucible.review`, `crucible.report`, `crucible.batch`, `crucible.registry`, `crucible.drift`, `crucible.refine`, and `crucible.verdicts`.
30
+
31
+ ## 1.1.0
32
+
33
+ Operator run surface.
34
+
35
+ - CLI: `crucible run THESIS --registry DIR (--measurements FILE | --substrate FILE)` runs the
36
+ steelman, measurement, witnessed assessment, and disk recheck path as one session.
37
+ - `crucible run --json` emits a complete run record with thesis metadata, refutations, assessment,
38
+ verdicts, disk recheck status, and optional report path.
39
+ - `crucible run --report FILE` writes the deterministic Markdown assessment report for the same
40
+ witnessed assessment, while `--out FILE` writes the JSON run record with exclusive creation.
41
+ - `crucible run --bundle DIR` creates a self-contained cleanroom review packet containing
42
+ `spec.json`, `run.json`, `report.md`, and `review.md`, refusing pre-existing packet directories.
43
+ - Bundle run records carry a machine-readable verifier boundary: cleanroom mode, allowed packet
44
+ inputs, excluded worker context, and the checkability rule for underspecified specs.
45
+ - CLI: `crucible review BUNDLE` validates the cleanroom packet before verifier handoff, including
46
+ required files, absence of extra context, verifier boundary metadata, and spec/run agreement.
47
+ - CLI: `crucible recheck REGISTRY [--template FILE] [--pack FILE]` lists descriptor-bearing
48
+ measurement rows, writes replay pack templates, and validates finished oracle replay packs without
49
+ creating a second verdict path.
50
+ - Replay packs that return the template assessment block are checked against the selected thesis id,
51
+ assessment seal, and measurement seal before oracle replay starts.
52
+ - Cleanroom review now recomputes the expected `report.md` artifact from `spec.json` plus `run.json`
53
+ and fails closed when the human report is tampered or stale.
54
+ - Cleanroom review now also checks that `review.md` is the canonical verifier instruction sheet, so
55
+ packet-local instructions cannot widen the verifier's context.
56
+ - Bundle run records now use packet-relative artifact names, avoiding local workspace paths inside
57
+ cleanroom verifier packets.
58
+ - Cleanroom review now validates those `run.json` artifact path fields and fails closed if they are
59
+ absolute, local, or renamed away from the packet-relative contract.
60
+ - Cleanroom review now requires `run.json` to declare passing embedded integrity checks before
61
+ verifier handoff.
62
+
63
+ ## 1.0.0
64
+
65
+ Stable flagship floor.
66
+
67
+ - Assessment integrity now seals verdict margin and grounds, and `recheck_assessment` rejects stored
68
+ verdict rows or summary counts that do not re-derive from the thesis and measurements.
69
+ - Measurement rows now persist and seal `measured_at`, and Telos-backed measurements persist a
70
+ `telos:<verifier>` replay descriptor so oracle-level reassessment checks a real stored row.
71
+ - Drift and registry status/search now use the latest verified assessments, falling back past invalid
72
+ tail rows instead of trusting or hiding history.
73
+ - Registry body verification now reports non-file or unreadable object paths as CORRUPT instead of
74
+ raising, and registration rejects pre-existing non-file object paths.
75
+ - Registry prune now applies the registry realpath guard before scanning or deleting object paths, so
76
+ escaped, symlinked, or junctioned object roots are refused before unlink.
77
+ - Assessment and verdict records, CLI JSON, and Markdown reports now carry the sealed thesis
78
+ disposition, keeping publication posture visible in witnessed outputs.
79
+ - `refine.margin()` is public and reused by grading, matching the documented normalized-margin
80
+ contract.
81
+ - Index/Gather interop canonical hashing now uses unescaped sorted JSON (`ensure_ascii=False`) for
82
+ non-ASCII graph-pack content.
83
+ - The registry rejects duplicate thesis ids with different seals, refuses symlinked storage paths,
84
+ and keeps object writes on unique temp files inside the registry root.
85
+ - Batch manifests keep thesis, measurement, and substrate paths inside the manifest bundle; path-like
86
+ missing refs fail closed without rejecting dotted registry ids; report writes use index-prefixed
87
+ filenames and exclusive creation.
88
+ - Subprocess-backed edges use a clean default environment, discard unbounded stderr, write stdout to a
89
+ temporary file, actively terminate children that exceed the response cap, and still reject shell
90
+ strings.
91
+ - CLI JSON loaders reject non-object top-level payloads with clean errors, ambiguous claim text refs
92
+ are rejected, and refine thresholds reject negative or non-finite values cleanly.
93
+ - Release workflows remove manual PyPI dispatch, pin external GitHub Actions by commit SHA, and
94
+ install pinned build tooling plus the pinned build backend from `requirements-release.txt`.
95
+ - README and readiness docs record the clean verifier rule: verifier receives only the original spec
96
+ and artifact, never the worker context or reasoning trace.
97
+
98
+ ## 0.14.1
99
+
100
+ Batch hardening patch.
101
+
102
+ - Batch report filenames now include the manifest job index, so duplicate or colliding job IDs do not
103
+ overwrite earlier reports.
104
+ - Batch thesis, measurement, and substrate references now fail closed when a manifest-relative or
105
+ absolute file is missing instead of falling back to the caller's current working directory.
106
+
107
+ ## 0.14.0
108
+
109
+ Batch assessment workflow.
110
+
111
+ - `crucible.batch_cmd`: adds a manifest runner that assesses multiple thesis jobs into one
112
+ content-addressed registry.
113
+ - Manifest jobs accept either explicit measurements or a substrate oracle file, preserving the same
114
+ grounded measurement -> verdict spine as the single-thesis commands.
115
+ - CLI: `crucible batch MANIFEST --registry DIR [--reports DIR] [--json]` emits a row per job and can
116
+ write one deterministic Markdown report per assessment.
117
+ - Readiness coverage now runs the batch command through the bundled example thesis surface.
118
+
119
+ ## 0.13.0
120
+
121
+ Markdown assessment reports.
122
+
123
+ - `crucible.report`: adds `render_assessment_report`, a deterministic Markdown renderer for one
124
+ witnessed assessment and its thesis.
125
+ - Reports include assessment/thesis seals, outcome counts, integrity checks, per-claim verdicts,
126
+ measurement evidence, optional recheck descriptors, and unmeasured claims.
127
+ - CLI: `crucible report REGISTRY [--index N] [--out FILE]` renders the latest assessment by default
128
+ and can write the Markdown body to disk.
129
+ - Public API: exports `render_assessment_report`.
130
+
131
+ ## 0.12.0
132
+
133
+ Measurement recheck descriptors.
134
+
135
+ - `Measurement` now accepts an optional `recheck` descriptor, preserved at the end of the dataclass so
136
+ existing positional construction stays compatible.
137
+ - Assessments persist and seal `measured_at` plus `recheck` descriptors when present; legacy
138
+ assessment rows without descriptors keep their previous replay shape and still verify.
139
+ - `recheck_measurements` replays descriptor-bearing measurement rows through a caller-supplied oracle
140
+ registry and reports checked, skipped, missing, mismatched, and failed replays.
141
+ - `recheck_assessment(..., measurement_replayers=...)` can include an oracle-level measurement replay
142
+ result alongside the existing seal/thesis/verdict re-derivation checks.
143
+ - Public API: exports `recheck_measurements`.
144
+
145
+ ## 0.11.0
146
+
147
+ Gather/index protocol interop preview.
148
+
149
+ - `crucible.ecosystem_measure`: adds `verify_gather_digest`, `receipt_matches`,
150
+ `verify_index_verification`, and `canonical_sha` for consuming sibling-product JSON contracts
151
+ without importing sibling packages.
152
+ - `GatherDigestMeasure` maps a verified Gather digest plus receipt selector into a crucible
153
+ `Measurement`: receipt present -> MATCH input, receipt absent -> DRIFT input, malformed or missing
154
+ digest -> UNVERIFIABLE input.
155
+ - `IndexMeasure` maps an `index.verification/1` record plus supplied graph pack into a crucible
156
+ `Measurement`: reproduced MATCH -> MATCH input, reproduced REFUTED/DRIFT -> DRIFT input, missing
157
+ or unreplayable pack -> UNVERIFIABLE input.
158
+ - Public API: exports `GatherDigestMeasure`, `IndexMeasure`, `verify_gather_digest`,
159
+ `verify_index_verification`, `receipt_matches`, and `canonical_sha`.
160
+
161
+ ## 0.10.0
162
+
163
+ Telos witnessed-artifact interop preview.
164
+
165
+ - `crucible.telos_measure`: adds `verify_telos_artifact`, `is_telos_artifact`, `check_content`, and
166
+ `TelosMeasure` for consuming `telos.witnessed-artifact/v1` envelopes through the Measure seam.
167
+ - `TelosMeasure` maps a re-run Telos verifier result into a crucible `Measurement`: verified -> MATCH
168
+ input, refuted or drifted -> DRIFT input, absent or unregistered proof -> UNVERIFIABLE input.
169
+ - Public API: exports `TelosMeasure`, `verify_telos_artifact`, `is_telos_artifact`, and
170
+ `check_content`.
171
+
172
+ ## 0.9.0
173
+
174
+ 1.0-readiness hardening.
175
+
176
+ - Examples: the bundled demo and JSON examples are now regression-tested through the public CLI.
177
+ - CLI surface: help output is covered for the shipped commands and registry actions.
178
+ - Docs: adds a release-readiness checklist for the 1.0 review gate.
179
+
180
+ ## 0.8.0
181
+
182
+ Optional subprocess-backed seam adapters.
183
+
184
+ - `crucible.subprocess_edges`: adds `SubprocessSteelman` and `SubprocessMeasure`, stdlib-only adapters
185
+ for configured commands that exchange bounded JSON over stdin/stdout.
186
+ - Safety posture: commands must be argv sequences rather than shell strings; request and response
187
+ sizes are bounded; timeouts are enforced; claim identity and producer labels are stamped locally.
188
+ - Public API: exports `SubprocessSteelman` and `SubprocessMeasure`.
189
+
190
+ ## 0.7.0
191
+
192
+ Registry operations for a growing corpus.
193
+
194
+ - `crucible.registry_ops`: adds `registry_stats`, `search_theses`, and `prune_objects` for registry
195
+ health, recall, and object-store hygiene without changing the durable storage contract.
196
+ - CLI: `crucible registry stats`, `crucible registry search`, and `crucible registry prune` summarize
197
+ the corpus, search by scope/status/latest verdict, and report orphaned claim bodies. Prune is
198
+ dry-run by default and deletes only with `--apply`.
199
+ - Public API: exports `registry_stats`, `search_theses`, and `prune_objects`.
200
+
201
+ ## 0.6.0
202
+
203
+ Publication-gated thesis export.
204
+
205
+ - `crucible.gate`: adds `gate_check`, `export_guard`, and `export_thesis`. Fenced disposition and
206
+ explicit fenced/restricted markers fail closed at the public export edge.
207
+ - CLI: `crucible export THESIS` emits the public thesis contract for publishable theses, resolves
208
+ thesis ids from a registry with `--registry`, and refuses fenced theses with a clean error.
209
+ - Public API: exports `gate_check`, `export_guard`, and `export_thesis`.
210
+
211
+ ## 0.5.0
212
+
213
+ Drift tracking across witnessed assessment rounds.
214
+
215
+ - `crucible.drift`: compares two assessments of the same thesis and classifies each claim as held,
216
+ moved, improved, or regressed. Numeric margins decide direction; unrankable transitions such as
217
+ UNVERIFIABLE to MATCH are reported as moved.
218
+ - CLI: `crucible drift REGISTRY` compares the latest two stored assessments, with human and JSON
219
+ output and clean errors for too-short or mixed-thesis histories.
220
+ - Public API: exports `DriftRow`, `DriftReport`, and `drift_track`.
221
+
222
+ ## 0.4.0
223
+
224
+ The refine loop: margin, cohesion, reflection, and re-iteration over a thesis.
225
+
226
+ - `crucible.refine`: a zero-dependency refinement primitive with graded criteria, harmonic-mean
227
+ cohesion, reflect-weakest feedback, and fail-closed handling for broken generators, adjusters, and
228
+ non-numeric measurements.
229
+ - `refine_thesis`: points the primitive at a thesis and a `Measure` oracle, grading each claim by its
230
+ normalized measurement margin and returning a `RefineReport` with final verdicts and the cohesion
231
+ trajectory.
232
+ - CLI: `crucible refine <config.json>` runs ordered substrate rounds, stops on a cohesively verified
233
+ thesis, or reports the weakest claim when the budget is spent.
234
+ - Public API: exports `GradedCriterion`, `Reflection`, `RefineOutcome`, `RefineReport`, `cohesion`,
235
+ `refine`, and `refine_thesis`.
236
+
237
+ ## 0.3.0
238
+
239
+ The measure seam: a sound oracle decides a claim against a substrate. This is where the verdict is
240
+ grounded, the half the differentiator names.
241
+
242
+ - `crucible.measure`: a `Measure` protocol and a `MetricSpec` (the value a claim predicts, the
243
+ tolerance, the substrate key to observe, and the metric). `verdict_for` already consumes the
244
+ `Measurement` an oracle produces, so there is still no model in the verdict step.
245
+ - `NullMeasure`: the standing default, produces no measurement (UNVERIFIABLE); it invents nothing.
246
+ - `TableMeasure`: a deterministic, offline oracle. The deviation is the absolute or relative
247
+ difference between the observed value (from the substrate) and the value the claim predicts. An
248
+ unknown claim or a missing observation is UNVERIFIABLE, fail-closed. The shape a real oracle (the
249
+ Telos verifier, a proof or type checker) plugs into.
250
+ - `measure_thesis` runs an oracle over every claim in a thesis.
251
+ - CLI: `crucible measure <thesis> --substrate FILE` measures each claim and witnesses the verdicts; the
252
+ oracle-produced measurements re-derive from disk via `verdicts --verify`.
253
+ - Refactor: the registry-inspecting commands moved to `crucible.registry_cmd` so no module exceeds the
254
+ size budget.
255
+
256
+ ## 0.2.0
257
+
258
+ The steelman seam: adversarial refutation as a pluggable shape, the conjecture-and-attack half of
259
+ judgment.
260
+
261
+ - `crucible.steelman`: a `Steelman` protocol and a `Refutation` (the proposed attack plus the
262
+ measurable test that would settle it). Adversaries propose what to test; they do not decide.
263
+ - `NullSteelman`: the standing default. Deterministic and invents nothing: it surfaces the claim's own
264
+ stated falsification as the test, or flags a claim that states no falsification as unrefutable.
265
+ Custom refuters plug in through the same protocol.
266
+ - `steelman_thesis` runs a steelman over every claim in a thesis, returning the proposed tests in
267
+ claim order, ready to feed the measurement step.
268
+ - CLI: `crucible steelman <thesis>` prints each claim's challenge and the test it proposes.
269
+
270
+ ## 0.1.0
271
+
272
+ The P1 foundation: the verdict spine and the witnessed, re-derivable record.
273
+
274
+ - `crucible.claim`: a `Claim` carrying a content-hash receipt over its assertion and falsification
275
+ condition; `make_claim` computes it and `Claim.verify()` re-hashes to catch tampering.
276
+ - `crucible.thesis`: a `Thesis` with a seal binding its claims, title, and disposition (so a fenced
277
+ thesis cannot be relabelled publishable undetected); the seal is over the set, so reordering is not
278
+ a change.
279
+ - `crucible.verdict`: the differentiator. `verdict_for` is a pure MATCH / DRIFT / UNVERIFIABLE
280
+ decision from a `Measurement`, with no model in the verdict step. UNVERIFIABLE is fail-closed
281
+ (no measurement, an unmeasurable deviation, a non-positive tolerance, an unfalsifiable claim, or a
282
+ measurement bound to a different claim all read as UNVERIFIABLE, never as holding).
283
+ - `crucible.registry`: a content-addressed registry. Claim bodies are deduped and traversal-guarded;
284
+ `verify` reports MATCH / MISSING / CORRUPT; `verify_seals` catches a swapped or relabelled claim a
285
+ body check would miss; a tampered thesis is refused on load; re-registration is idempotent.
286
+ - `crucible.assess`: a witnessed `Assessment` that persists its verdicts and measurements.
287
+ `verify_assessment` recomputes the seals from the stored data; `recheck_assessment` re-derives each
288
+ verdict from the thesis and the measurements, so a verdict cannot be asserted, only computed.
289
+ - `crucible.cli`: `register`, `assess`, `registry list|verify`, and `verdicts [--verify]`.
290
+ - Zero third-party dependencies; an offline `examples/demo.py`; ruff and mypy clean.
@@ -0,0 +1,78 @@
1
+ crucible Fair-Source License, Version 1.0
2
+
3
+ Copyright (c) 2026 Zain Dana Harper. All rights reserved.
4
+
5
+ This license governs use of the accompanying software ("the Software", the crucible judgment organ). By using,
6
+ copying, modifying, or distributing the Software, you accept these terms. The
7
+ Software is source-available, not open source: the source is published so you can
8
+ read it, run it, and build on it, while commercial use that competes with the
9
+ project is reserved so the project can fund its own continued development.
10
+
11
+ 1. Definitions
12
+
13
+ "Licensor" means Zain Dana Harper, the copyright holder.
14
+
15
+ "You" means the individual or entity exercising rights under this license.
16
+
17
+ "Competing Use" means making the Software, or a modified version of it,
18
+ available to a third party as a commercial product or service that
19
+ substitutes for, or offers substantially the same functionality as, the
20
+ Software or any product or service the Licensor offers using the Software.
21
+
22
+ 2. Grant
23
+
24
+ Subject to your compliance with this license, the Licensor grants you a
25
+ worldwide, royalty-free, non-exclusive, non-transferable license to read,
26
+ run, copy, modify, create derivative works of, and redistribute the Software
27
+ for any Permitted Purpose.
28
+
29
+ 3. Permitted Purpose
30
+
31
+ A Permitted Purpose is any purpose other than a Competing Use. Permitted
32
+ Purposes include, without limitation: internal use within your organization;
33
+ personal use; evaluation; non-commercial education and research; and use in
34
+ providing professional services to a party that is itself using the Software
35
+ under this license.
36
+
37
+ 4. Reserved Commercial Use
38
+
39
+ A Competing Use is reserved to the Licensor and requires a separate
40
+ commercial license. This reservation is what funds the project's continued
41
+ development. To obtain a commercial license, contact the Licensor (see
42
+ Contact below).
43
+
44
+ 5. Conditions
45
+
46
+ You must retain, in all copies and derivative works you distribute, this
47
+ license, the copyright notice, and all attribution notices. You may add your
48
+ own notices to changes you make, so long as the origin of the Software is not
49
+ misrepresented.
50
+
51
+ 6. Trademarks
52
+
53
+ This license does not grant any right to use the Licensor's names, logos, or
54
+ trademarks.
55
+
56
+ 7. Disclaimer of Warranty
57
+
58
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
59
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
60
+ FITNESS FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT.
61
+
62
+ 8. Limitation of Liability
63
+
64
+ IN NO EVENT SHALL THE LICENSOR BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER
65
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM,
66
+ OUT OF, OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
67
+ SOFTWARE.
68
+
69
+ 9. Termination
70
+
71
+ If you breach this license, your rights under it terminate automatically. They
72
+ may be reinstated by the Licensor in writing.
73
+
74
+ 10. Contact
75
+
76
+ For commercial licensing or to report a security issue, open a private
77
+ security advisory at https://github.com/HarperZ9/crucible/security or reach the
78
+ Licensor via https://github.com/HarperZ9.
@@ -0,0 +1,6 @@
1
+ include ARCHITECTURE.md
2
+ include CHANGELOG.md
3
+ include requirements-release.txt
4
+ include docs/*.md
5
+ include examples/*.json
6
+ include examples/*.py