constraintloop 0.4.1__tar.gz → 0.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. {constraintloop-0.4.1 → constraintloop-0.5.1}/CHANGELOG.md +42 -0
  2. {constraintloop-0.4.1 → constraintloop-0.5.1}/CONTRIBUTING.md +2 -2
  3. {constraintloop-0.4.1 → constraintloop-0.5.1}/PKG-INFO +20 -8
  4. {constraintloop-0.4.1 → constraintloop-0.5.1}/README.md +19 -7
  5. constraintloop-0.5.1/docs/completion-policy.md +74 -0
  6. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/configuration.md +29 -1
  7. constraintloop-0.5.1/docs/convergence-loops.md +374 -0
  8. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/faq.md +15 -2
  9. constraintloop-0.5.1/docs/pre-release-review-2026-09-06.md +236 -0
  10. constraintloop-0.5.1/docs/release-readiness.md +85 -0
  11. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/threat-model.md +9 -0
  12. constraintloop-0.5.1/docs/worktrees.md +73 -0
  13. {constraintloop-0.4.1 → constraintloop-0.5.1}/pyproject.toml +1 -1
  14. {constraintloop-0.4.1 → constraintloop-0.5.1}/schema/constraintloop.schema.json +54 -0
  15. constraintloop-0.5.1/scripts/wheel_failure_smoke.py +173 -0
  16. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/__init__.py +2 -2
  17. constraintloop-0.5.1/src/constraintloop/_numbers.py +18 -0
  18. constraintloop-0.5.1/src/constraintloop/_process.py +121 -0
  19. constraintloop-0.5.1/src/constraintloop/challenges.py +248 -0
  20. constraintloop-0.5.1/src/constraintloop/checkout.py +88 -0
  21. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/cli.py +101 -14
  22. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/config.py +23 -2
  23. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/digest.py +24 -33
  24. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/engine.py +37 -12
  25. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/hooks.py +156 -66
  26. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/loops.py +211 -13
  27. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/models.py +133 -2
  28. constraintloop-0.5.1/src/constraintloop/redaction.py +38 -0
  29. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/runners.py +11 -8
  30. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/state.py +13 -5
  31. constraintloop-0.5.1/tests/test_challenges.py +601 -0
  32. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_cli.py +15 -0
  33. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_cli_commands.py +1 -1
  34. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_digest.py +22 -0
  35. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_failure_lab.py +2 -1
  36. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_hooks.py +29 -3
  37. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_loops.py +2 -2
  38. constraintloop-0.5.1/tests/test_process.py +99 -0
  39. constraintloop-0.5.1/tests/test_review_regressions.py +519 -0
  40. constraintloop-0.5.1/tests/test_worktrees.py +346 -0
  41. constraintloop-0.4.1/docs/convergence-loops.md +0 -219
  42. constraintloop-0.4.1/docs/release-readiness.md +0 -271
  43. constraintloop-0.4.1/scripts/wheel_failure_smoke.py +0 -47
  44. constraintloop-0.4.1/src/constraintloop/_process.py +0 -63
  45. constraintloop-0.4.1/tests/test_process.py +0 -50
  46. {constraintloop-0.4.1 → constraintloop-0.5.1}/.gitignore +0 -0
  47. {constraintloop-0.4.1 → constraintloop-0.5.1}/GOVERNANCE.md +0 -0
  48. {constraintloop-0.4.1 → constraintloop-0.5.1}/LICENSE +0 -0
  49. {constraintloop-0.4.1 → constraintloop-0.5.1}/RELEASE.md +0 -0
  50. {constraintloop-0.4.1 → constraintloop-0.5.1}/SECURITY.md +0 -0
  51. {constraintloop-0.4.1 → constraintloop-0.5.1}/SUPPORT.md +0 -0
  52. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/native-cli-evaluators.md +0 -0
  53. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/openai-evaluation.md +0 -0
  54. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/provider-privacy.md +0 -0
  55. {constraintloop-0.4.1 → constraintloop-0.5.1}/docs/recipes.md +0 -0
  56. {constraintloop-0.4.1 → constraintloop-0.5.1}/scripts/check_anthropic_sdk.py +0 -0
  57. {constraintloop-0.4.1 → constraintloop-0.5.1}/scripts/check_coverage.py +0 -0
  58. {constraintloop-0.4.1 → constraintloop-0.5.1}/scripts/check_openai_sdk.py +0 -0
  59. {constraintloop-0.4.1 → constraintloop-0.5.1}/scripts/check_sdist_contents.py +0 -0
  60. {constraintloop-0.4.1 → constraintloop-0.5.1}/scripts/generate_schema.py +0 -0
  61. {constraintloop-0.4.1 → constraintloop-0.5.1}/scripts/openai_eval_canary.py +0 -0
  62. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/__main__.py +0 -0
  63. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/diagnostics.py +0 -0
  64. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/environment.py +0 -0
  65. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/eval_corpus.py +0 -0
  66. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/evaluators.py +0 -0
  67. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/hygiene.py +0 -0
  68. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/native_cli_evaluator.py +0 -0
  69. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/py.typed +0 -0
  70. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/scaffold.py +0 -0
  71. {constraintloop-0.4.1 → constraintloop-0.5.1}/src/constraintloop/setup_hooks.py +0 -0
  72. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/__init__.py +0 -0
  73. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/failure_lab.py +0 -0
  74. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/fixtures/openai_eval_corpus_v1.yml +0 -0
  75. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_diagnostics.py +0 -0
  76. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_engine.py +0 -0
  77. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_environment.py +0 -0
  78. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_eval_corpus.py +0 -0
  79. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_evaluators.py +0 -0
  80. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_models.py +0 -0
  81. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_native_cli_evaluator.py +0 -0
  82. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_release_metadata.py +0 -0
  83. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_runners.py +0 -0
  84. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_security.py +0 -0
  85. {constraintloop-0.4.1 → constraintloop-0.5.1}/tests/test_state.py +0 -0
@@ -5,6 +5,48 @@ Versioning, with the usual initial-development flexibility for `0.y.z`.
5
5
 
6
6
  ## [Unreleased]
7
7
 
8
+ ## [0.5.1] - 2026-09-08
9
+
10
+ - Isolate local evidence, sessions, waivers, acknowledgments, journals, and leases
11
+ by resolved project/worktree and branch (commit for detached HEAD). Bind input
12
+ digests to checkout identity and HEAD, keeping branch retry budgets across commits.
13
+ - Reject checkout changes during gate evaluation and supervision. Report project,
14
+ worktree, branch, HEAD, and state directory in `doctor` and `explain`.
15
+ - Migration: Git projects start fresh state under `.constraintloop/state/checkouts/`;
16
+ prior unscoped state is retained but never imported into a branch. All prior input
17
+ digests become stale. Run checks again and start a fresh challenge cycle if used.
18
+
19
+ ## [0.5.0] - 2026-09-06
20
+
21
+ - Add optional session challenge gates for Claude Code, Codex, and Gemini CLI:
22
+ generate configurable domain-grounded scenarios, record immutable plans,
23
+ verify outcomes against fresh deterministic evidence, and repair defects
24
+ without launching a model evaluator.
25
+ - Add challenge show/submit commands, challenge/verify cycle states, Gemini
26
+ loop prompts, and persisted continuation budgets for all three adapters.
27
+ - Preserve task goals through generated continuation prompts and stop Gemini
28
+ sessions on exhausted budgets instead of requesting another retry.
29
+ - Fix all ten pre-release findings: frame content hashes unambiguously; bound
30
+ ordinary recursive Stop hooks and advisory continuations; fail closed on
31
+ malformed evidence and unexpected hook errors; reject push/CI waivers in
32
+ every engine entry point; report repairable dependency failures by root
33
+ cause; renew budgets for completed tasks; refresh pending hook evidence on
34
+ the shared cycle interval; reject non-finite metrics, thresholds, and
35
+ baselines; redact retained evidence; and deny observed agent baseline
36
+ weakening and direct baseline edits.
37
+ - Reject disabled or phase-incompatible prerequisites, keep CI loops free of
38
+ local overlays, bound subprocess collection to 8 MiB, and renew supervisor
39
+ leases during long evaluations and polling waits.
40
+ - Add regression matrices for adapters, phase policy, malformed numbers, digest
41
+ boundaries, lifecycle sequences, and installed-package challenge workflows.
42
+ - Cover every changed executable line, including submission rejection,
43
+ missing requests, unreadable files, and lost leases. Report malformed YAML
44
+ by location without raw source excerpts that could disclose credentials.
45
+ - Migration: existing evidence and waivers become stale under the new input
46
+ digest. Re-run checks; review incompatible prerequisite phases and invalid
47
+ numeric baselines. Challenge gates remain opt-in. Re-run hook setup after
48
+ upgrading if hooks use a pinned ephemeral executable.
49
+
8
50
  ## [0.4.1] - 2026-09-04
9
51
 
10
52
  - Raise enforced statement coverage from 90% to 95% and branch coverage from
@@ -5,7 +5,7 @@ By participating, you agree to follow the
5
5
  the contact method documented there.
6
6
 
7
7
  ConstraintLoop accepts focused changes that strengthen evidence, state,
8
- budgeting, locks, and stopping. It is not a general agent runtime and v0.1 must
8
+ budgeting, locks, and stopping. It is not a general agent runtime and must
9
9
  not launch provider CLIs for repair turns or offer unbounded repair loops.
10
10
  Opt-in native CLI evaluators are limited to isolated, tool-disabled reviews.
11
11
 
@@ -43,5 +43,5 @@ Releases are prepared through a focused release pull request and published only
43
43
  through GitHub Trusted Publishing. See `RELEASE.md`. Contributors and agents
44
44
  must not run local package upload commands or add long-lived registry tokens.
45
45
 
46
- The supported v0.1 platforms are macOS and Linux. Windows is not supported
46
+ The supported platforms are macOS and Linux. Windows is not supported
47
47
  until hook command generation and clean-wheel tests are implemented there.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: constraintloop
3
- Version: 0.4.1
3
+ Version: 0.5.1
4
4
  Summary: Evidence-based completion gates for AI coding agents
5
5
  Project-URL: Homepage, https://github.com/mauhpr/constraintloop
6
6
  Project-URL: Documentation, https://github.com/mauhpr/constraintloop/tree/main/docs
@@ -69,11 +69,17 @@ The central distinction is deliberate:
69
69
  ConstraintLoop supports Claude Code, Codex, and Gemini CLI through their hook
70
70
  lifecycles. CI is the final authority: it ignores local caches and human waivers.
71
71
 
72
- Bounded convergence loops are included in the v0.1 release scope. The design
72
+ Bounded convergence loops are implemented. The design
73
73
  keeps ConstraintLoop in control of evidence, budgets, and stopping while native
74
74
  Claude or Codex loops perform at most one requested repair per transition. See
75
75
  [docs/convergence-loops.md](docs/convergence-loops.md).
76
76
 
77
+ Completion loops can also require the current coding session to generate and
78
+ investigate N domain-grounded failure scenarios before stopping. Claude Code,
79
+ Codex, and Gemini CLI use their existing session context and tools; this
80
+ challenge gate requires no separate model evaluator. See
81
+ [session challenge gates](docs/convergence-loops.md#session-challenge-gates).
82
+
77
83
  ## At a glance
78
84
 
79
85
  | Question | ConstraintLoop answer |
@@ -308,10 +314,13 @@ Commands run from their configured project-contained `cwd`, with the selected
308
314
  project root prepended to `PYTHONPATH`.
309
315
 
310
316
  Evidence is keyed by the constraint definition and the bytes of every file
311
- matched by `watch`. A source change therefore makes old evidence and waivers
312
- stale without a mutable invalidation list. Local state lives under the
317
+ matched by `watch`, plus the resolved project/worktree, branch, and HEAD. A source
318
+ change or new HEAD therefore makes old evidence and waivers stale without a
319
+ mutable invalidation list. Local state lives under the
313
320
  gitignored `.constraintloop/state` directory; set `CONSTRAINTLOOP_CACHE_DIR` to
314
- override it.
321
+ override it. Git projects use a separate state subdirectory per branch in each
322
+ worktree. Use one worktree per concurrent line of work; `doctor` and `explain`
323
+ show the active scope. See [worktrees and concurrent sessions](docs/worktrees.md).
315
324
 
316
325
  For stronger machine-local gates, create a gitignored
317
326
  `constraintloop.local.yml`. ConstraintLoop recursively merges mappings over the
@@ -351,8 +360,11 @@ reactivate ConstraintLoop; setup clears the tombstone.
351
360
  - `constraintloop cycle NAME --json` — execute one journaled loop transition.
352
361
  - `constraintloop supervise NAME` — poll pending evidence under a recoverable
353
362
  single-writer lease and exit whenever repair or termination is required.
354
- - `constraintloop loop-prompt NAME --adapter claude|codex` — print the bounded
355
- native-agent repair protocol without launching an agent.
363
+ - `constraintloop loop-prompt NAME --adapter claude|codex|gemini` — print the bounded
364
+ native-agent repair and challenge protocol without launching an agent.
365
+ - `constraintloop challenge show NAME` — inspect saved scenarios and the submission schema.
366
+ - `constraintloop challenge submit NAME --file PATH` — record session-authored
367
+ discovery or verification for the current request and input snapshot.
356
368
  - `constraintloop status` — inspect evidence without executing commands.
357
369
  - `constraintloop explain --phase change|stop|push|ci` — show why each constraint
358
370
  runs or is skipped, including matched and changed watch paths, cache state,
@@ -375,7 +387,7 @@ reactivate ConstraintLoop; setup clears the tombstone.
375
387
  - `constraintloop author` — write a review-only QA/test-authoring proposal.
376
388
 
377
389
  `enhance` and `author` intentionally do not install dependencies or modify the
378
- active contract in v0.1. Their proposal files make the future self-improvement
390
+ active contract. Their proposal files make the future self-improvement
379
391
  path auditable.
380
392
 
381
393
  ## Documentation
@@ -26,11 +26,17 @@ The central distinction is deliberate:
26
26
  ConstraintLoop supports Claude Code, Codex, and Gemini CLI through their hook
27
27
  lifecycles. CI is the final authority: it ignores local caches and human waivers.
28
28
 
29
- Bounded convergence loops are included in the v0.1 release scope. The design
29
+ Bounded convergence loops are implemented. The design
30
30
  keeps ConstraintLoop in control of evidence, budgets, and stopping while native
31
31
  Claude or Codex loops perform at most one requested repair per transition. See
32
32
  [docs/convergence-loops.md](docs/convergence-loops.md).
33
33
 
34
+ Completion loops can also require the current coding session to generate and
35
+ investigate N domain-grounded failure scenarios before stopping. Claude Code,
36
+ Codex, and Gemini CLI use their existing session context and tools; this
37
+ challenge gate requires no separate model evaluator. See
38
+ [session challenge gates](docs/convergence-loops.md#session-challenge-gates).
39
+
34
40
  ## At a glance
35
41
 
36
42
  | Question | ConstraintLoop answer |
@@ -265,10 +271,13 @@ Commands run from their configured project-contained `cwd`, with the selected
265
271
  project root prepended to `PYTHONPATH`.
266
272
 
267
273
  Evidence is keyed by the constraint definition and the bytes of every file
268
- matched by `watch`. A source change therefore makes old evidence and waivers
269
- stale without a mutable invalidation list. Local state lives under the
274
+ matched by `watch`, plus the resolved project/worktree, branch, and HEAD. A source
275
+ change or new HEAD therefore makes old evidence and waivers stale without a
276
+ mutable invalidation list. Local state lives under the
270
277
  gitignored `.constraintloop/state` directory; set `CONSTRAINTLOOP_CACHE_DIR` to
271
- override it.
278
+ override it. Git projects use a separate state subdirectory per branch in each
279
+ worktree. Use one worktree per concurrent line of work; `doctor` and `explain`
280
+ show the active scope. See [worktrees and concurrent sessions](docs/worktrees.md).
272
281
 
273
282
  For stronger machine-local gates, create a gitignored
274
283
  `constraintloop.local.yml`. ConstraintLoop recursively merges mappings over the
@@ -308,8 +317,11 @@ reactivate ConstraintLoop; setup clears the tombstone.
308
317
  - `constraintloop cycle NAME --json` — execute one journaled loop transition.
309
318
  - `constraintloop supervise NAME` — poll pending evidence under a recoverable
310
319
  single-writer lease and exit whenever repair or termination is required.
311
- - `constraintloop loop-prompt NAME --adapter claude|codex` — print the bounded
312
- native-agent repair protocol without launching an agent.
320
+ - `constraintloop loop-prompt NAME --adapter claude|codex|gemini` — print the bounded
321
+ native-agent repair and challenge protocol without launching an agent.
322
+ - `constraintloop challenge show NAME` — inspect saved scenarios and the submission schema.
323
+ - `constraintloop challenge submit NAME --file PATH` — record session-authored
324
+ discovery or verification for the current request and input snapshot.
313
325
  - `constraintloop status` — inspect evidence without executing commands.
314
326
  - `constraintloop explain --phase change|stop|push|ci` — show why each constraint
315
327
  runs or is skipped, including matched and changed watch paths, cache state,
@@ -332,7 +344,7 @@ reactivate ConstraintLoop; setup clears the tombstone.
332
344
  - `constraintloop author` — write a review-only QA/test-authoring proposal.
333
345
 
334
346
  `enhance` and `author` intentionally do not install dependencies or modify the
335
- active contract in v0.1. Their proposal files make the future self-improvement
347
+ active contract. Their proposal files make the future self-improvement
336
348
  path auditable.
337
349
 
338
350
  ## Documentation
@@ -0,0 +1,74 @@
1
+ # Completion policy and v0.5 migration
2
+
3
+ ## Entry points
4
+
5
+ | Entry point | Local overlays | Local waivers | Pending refresh | Task completion |
6
+ | --- | --- | --- | --- | --- |
7
+ | `run --phase change/stop` | Strengthening only | Deterministic gates only | Use `--no-cache` for a fresh observation | Constraint report only |
8
+ | `run --phase push` | Strengthening only | Never | Use `--no-cache` | Constraint report only |
9
+ | `ci` / `run --phase ci` | Never | Never | Always uncached | Independent CI evidence |
10
+ | `cycle` / `supervise` | Except CI phase | Except push/CI | Refresh after loop interval | Persisted bounded transition; includes configured challenge work |
11
+ | Stop / AfterAgent hooks | Strengthening only | Deterministic gates only | Shared cycle interval, or refresh each event without a loop | Required evidence, configured challenges, and advisory dispositions |
12
+
13
+ The engine enforces waiver and CI-cache policy regardless of caller defaults.
14
+ `run` does not finish a session challenge gate. Use `cycle` and the native Stop
15
+ hook for completion. CI verifies committed deterministic/rubric constraints;
16
+ it does not trust a local challenge journal or start an interactive session.
17
+
18
+ Failed prerequisites produce blocked dependents with `blocked_by` IDs, not
19
+ spurious evaluation errors. A cycle's `blocking_constraints` names the root
20
+ causes, including advisory prerequisites needed by required gates. Missing
21
+ tools, parser failures, and uncertain evaluations remain blocking errors.
22
+
23
+ ## Task lifecycle
24
+
25
+ After a loop passes, a changed watched input or task goal starts a new run with
26
+ a fresh budget. Restarting an unfinished run, changing its code, or changing
27
+ its goal does not reset its limits. Unchanged completed evidence stays passed
28
+ even after the old time budget elapses. One project loop has one journal;
29
+ use separate worktrees for independent concurrent tasks.
30
+
31
+ Recursive Stop hooks are completion boundaries too. Ordinary loops use their
32
+ repair and duration budgets. Challenge loops also bound session continuations.
33
+ Without a loop, `max_auto_retries` bounds repair and advisory-disposition
34
+ continuations across changing evidence until completion succeeds. Exhaustion
35
+ returns `continue: false`, including Gemini, rather than silently allowing
36
+ completion or asking the agent to retry indefinitely.
37
+
38
+ `supervise` renews its lease during checks and waits, and checks ownership
39
+ before yielding a transition. It does not launch an agent or perform repairs.
40
+
41
+ ## Evidence and redaction
42
+
43
+ The input-digest format is versioned and length-framed: filename/content
44
+ boundaries, unreadable files, and missing baselines cannot alias ordinary
45
+ content. Upgrading invalidates old evidence and snapshot-bound waivers.
46
+
47
+ Known sensitive environment values of at least eight characters are scrubbed,
48
+ as are credential assignments such as `password=...`, `api_key=...`, and
49
+ `access_token=...`. Structured sensitive keys are scrubbed recursively.
50
+ Scrubbing applies to evidence messages, findings, structured artifact fields,
51
+ retained command output, cached reads, and native-hook feedback. Command output
52
+ is scrubbed before tail truncation so truncation cannot sever the credential
53
+ label from its value.
54
+
55
+ This is best-effort redaction, not a data-loss-prevention boundary. Unknown,
56
+ encoded, split, or unlabeled secrets may escape detection. Raw command output
57
+ exists transiently in memory for parsing, bounded to 8 MiB per subprocess.
58
+ Pre-existing files from older releases are not retroactively erased. Avoid
59
+ printing credentials and restrict access to local state. Repository artifacts
60
+ and challenge submissions remain author-controlled data; do not put secrets
61
+ in them. No new remote model API is used by the session challenge gate.
62
+
63
+ ## Upgrade checklist
64
+
65
+ 1. Upgrade ConstraintLoop to 0.5.0 and re-run `constraintloop setup --adapter all`
66
+ in projects whose hooks pin an ephemeral package version.
67
+ 2. Re-run checks to replace stale cached evidence. Revisit any intentional
68
+ local waiver against the new exact evidence; CI and push cannot use it.
69
+ 3. Ensure prerequisites are enabled in every dependent phase. Correct invalid
70
+ numeric baselines; finite numeric strings are still accepted.
71
+ 4. Keep noisy tool output below 8 MiB or write detailed reports to artifacts
72
+ and emit a compact command summary.
73
+ 5. Enable `loops.NAME.challenge` explicitly where domain-driven self-review is
74
+ desired. Omitted challenge configuration preserves ordinary gating.
@@ -41,6 +41,10 @@ Every constraint supports `description`, `enforcement` (`required` or
41
41
  `advisory`), `phases` (`change`, `stop`, `push`, `ci`), `watch` globs, dependency IDs
42
42
  in `needs`, `timeout_seconds`, and `enabled`. Dependencies must exist and the
43
43
  graph must be acyclic.
44
+ Every enabled dependent must have its prerequisites enabled in all of its
45
+ phases. A failed prerequisite blocks its dependents without running them;
46
+ cycles report the root cause for repair, even if that prerequisite is advisory.
47
+ Genuine evaluation errors still require human inspection.
44
48
  Identifiers may contain letters, numbers, dots, underscores, and hyphens.
45
49
  `watch` and `include` values must be nonempty project-relative POSIX globs.
46
50
 
@@ -75,11 +79,18 @@ retries require an explicit `total_timeout_seconds` greater than the per-attempt
75
79
  timeout. This keeps the total bound visible while leaving enough budget for a
76
80
  second attempt. Command and command-evaluator processes run from the selected
77
81
  project root by default, and ConstraintLoop prepends that root to `PYTHONPATH`.
82
+ Collection of combined stdout/stderr has a separate hard 8 MiB limit per
83
+ process. Exceeding it terminates the process group and produces an error;
84
+ truncated output is never treated as a complete metric or evaluator response.
85
+ `evidence_output_limit` controls the smaller, redacted tail retained afterward.
78
86
 
79
87
  Metric constraints add `parser` and `threshold`. A parser has type `json` or
80
88
  `regex`, reads `stdout`, `stderr`, or a project-contained `file`, and selects a
81
89
  dotted JSON `path` or regex `pattern` and `group`. Threshold operators are
82
90
  `gt`, `gte`, `lt`, `lte`, and `eq`.
91
+ Measurements, thresholds, and baselines must be finite numbers. Finite numeric
92
+ strings remain supported; booleans, nulls, containers, NaN, and infinities are
93
+ rejected. `--allow-regression` never overrides numeric validation.
83
94
 
84
95
  Ratchet constraints use `kind: ratchet` with the same command and parser fields
85
96
  as a metric. Their default `mode: must_not_increase` compares the current value
@@ -94,7 +105,11 @@ constraintloop baseline update --all
94
105
  Updates that would weaken an existing baseline are rejected. Use
95
106
  `--allow-regression` only for a reviewed, intentional reset, then commit the
96
107
  baseline artifact with the contract. `baseline_file` can select another
97
- project-relative JSON file. Each baseline entry records both the numeric value
108
+ project-relative JSON file. Observed agent commands using `--allow-regression`
109
+ and direct edits to baseline files are denied by pre-tool hooks; ask the human
110
+ to perform intentional policy changes outside the hooked session. Strengthening
111
+ through the ordinary baseline-update command remains allowed. Each baseline
112
+ entry records both the numeric value
98
113
  and the SHA-256 digest of the parsed evidence source, replacing the separate
99
114
  count-and-hash bookkeeping commonly used for migration inventories.
100
115
 
@@ -142,3 +157,16 @@ Hook responses use `hook_output_limit` to retain failing test names and the firs
142
157
  useful traceback line without injecting the complete test log. The unabridged
143
158
  retained tail remains available with `constraintloop debug CONSTRAINT`.
144
159
  See `docs/convergence-loops.md` for the cycle protocol and stable exit codes.
160
+
161
+ Stop-phase loops optionally accept `challenge` for discovery and verification
162
+ performed in the active Claude Code, Codex, or Gemini CLI session. Defaults are
163
+ `count: 10` (1–100), `max_rounds: 2` (1–10), `max_continuations: 8` (1–100),
164
+ `watch: ["**/*"]`, and `domain_context: []`. Context and watch entries are
165
+ project-relative globs. The existing loop duration and repair budgets also
166
+ apply; challenge work uses no model evaluator configuration. Omit `challenge`
167
+ to retain the ordinary completion behavior. See
168
+ [session challenge gates](convergence-loops.md#session-challenge-gates) for
169
+ submission examples, evidence requirements, and adapter behavior.
170
+
171
+ See [completion policy](completion-policy.md) for the cross-entry-point matrix,
172
+ task lifecycle, redaction policy, and upgrade notes.