tdd-cli 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/CHANGELOG.md +26 -0
  2. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/PKG-INFO +66 -3
  3. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/README.md +65 -2
  4. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/__init__.py +1 -1
  5. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/xctest_adapter.py +11 -11
  6. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/advance.py +10 -0
  7. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/cli.py +68 -17
  8. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/config.py +29 -7
  9. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/ledger.py +63 -20
  10. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/machine.py +20 -5
  11. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/conftest.py +88 -1
  12. tdd_cli-0.7.0/tests/test_baseline_integrity.py +321 -0
  13. tdd_cli-0.7.0/tests/test_concurrent_advance.py +178 -0
  14. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_config_and_staging.py +103 -0
  15. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_heartbeat.py +20 -1
  16. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_xctest_adapter.py +40 -49
  17. tdd_cli-0.6.0/tests/test_baseline_integrity.py +0 -148
  18. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/.gitignore +0 -0
  19. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/LICENSE +0 -0
  20. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/SECURITY.md +0 -0
  21. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/claude-code-hooks/README.md +0 -0
  22. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/claude-code-hooks/bash_hook.py +0 -0
  23. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/claude-code-hooks/stop_hook.py +0 -0
  24. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/plan.md +0 -0
  25. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-drive/README.md +0 -0
  26. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-drive/SKILL.md +0 -0
  27. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-handoff/README.md +0 -0
  28. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
  29. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/pyproject.toml +0 -0
  30. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/__init__.py +0 -0
  31. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/base.py +0 -0
  32. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/exec_adapter.py +0 -0
  33. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/gradle_adapter.py +0 -0
  34. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/pytest_adapter.py +0 -0
  35. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/vitest_adapter.py +0 -0
  36. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/contract.py +0 -0
  37. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/envelope.py +0 -0
  38. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/fleet.py +0 -0
  39. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/gitutil.py +0 -0
  40. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/identity.py +0 -0
  41. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/leases.py +0 -0
  42. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/render.py +0 -0
  43. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/snapshot.py +0 -0
  44. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/staging.py +0 -0
  45. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_artifact_regeneration.py +0 -0
  46. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_batch_collection.py +0 -0
  47. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_config_drift.py +0 -0
  48. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_contract.py +0 -0
  49. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_doctor_attribution.py +0 -0
  50. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_doctor_blockers.py +0 -0
  51. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_end_to_end.py +0 -0
  52. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_example_plan.py +0 -0
  53. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_exec_adapter.py +0 -0
  54. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_failure_clipping.py +0 -0
  55. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_fleet.py +0 -0
  56. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_gradle_adapter.py +0 -0
  57. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_init_detection.py +0 -0
  58. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_named_leases_and_timeout.py +0 -0
  59. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_pin_cycles.py +0 -0
  60. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_progress.py +0 -0
  61. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_project_commands.py +0 -0
  62. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_project_env.py +0 -0
  63. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_python_env_managers.py +0 -0
  64. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_refactor_cycles.py +0 -0
  65. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_release_surface.py +0 -0
  66. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_run_claim.py +0 -0
  67. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_single_project_repo.py +0 -0
  68. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_snapshot_and_identity.py +0 -0
  69. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_stub_hint.py +0 -0
  70. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_suite_overrides.py +0 -0
  71. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_target_validation.py +0 -0
  72. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_timing_visibility.py +0 -0
  73. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_vitest_adapter.py +0 -0
  74. {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_worker_leases.py +0 -0
@@ -4,6 +4,32 @@ All notable changes to this project are documented here.
4
4
  The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
5
5
  and the project adheres to [Semantic Versioning](https://semver.org/).
6
6
 
7
+ ## [Unreleased]
8
+
9
+ ## [0.7.0] - 2026-08-23
10
+
11
+ ### Fixed
12
+
13
+ - **Concurrent `tdd advance` no longer corrupts a run.** Two `advance` processes
14
+ racing on the same worktree could both close the same cycle row. Because
15
+ `close_cycle` unconditionally opened the next ordinal, a double-close forked the
16
+ run into two parallel cycle chains; every remaining cycle ran twice, the run
17
+ "completed" with one chain's last row permanently open, and the doubling was
18
+ invisible from the agent's perspective. `close_cycle` now re-reads `closed_at`
19
+ before acting; if the row is already closed it returns the currently-open cycle
20
+ without transitioning or opening anything. `open_cycle` returns the existing open
21
+ row for an ordinal rather than inserting a duplicate.
22
+
23
+ ### Added
24
+
25
+ - **Per-worktree advance claim.** `tdd advance` now acquires a `advance_claim` row
26
+ before dispatching. A second concurrent `advance` is refused immediately with
27
+ `ok: false`, `reason: "advance_in_flight"`, and metadata (`pid`, `started_at`,
28
+ `elapsed_s`) that lets the agent confirm the holder is still alive. The claim is
29
+ released in a `finally` so a raising handler cannot wedge the worktree; a claim
30
+ held by a dead pid is reclaimed automatically on the next call. Schema version
31
+ bumped to 3.
32
+
7
33
  ## [0.6.0] - 2026-08-19
8
34
 
9
35
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: tdd-cli
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: Ledger-backed TDD process controller for autonomous coding agents
5
5
  Project-URL: Homepage, https://github.com/geuben/tdd-cli
6
6
  Project-URL: Repository, https://github.com/geuben/tdd-cli
@@ -314,6 +314,34 @@ test_paths = ["src/test/"]
314
314
  test_command = "./gradlew testDebugUnitTest"
315
315
  ```
316
316
 
317
+ The `xctest` adapter drives Swift and Objective-C projects through `xcodebuild`. Test ids
318
+ use xcodebuild's own `-only-testing:` format — `BundleName/ClassName/testMethodName` — so
319
+ a single XCTest can be driven through RED → GREEN without any id translation. Collection
320
+ prefers `xcodebuild test -enumerate-tests` (Xcode 16+) and falls back to grepping
321
+ `class Foo: XCTestCase` / `func testBar()` out of the Swift sources under `test_paths`,
322
+ deriving the bundle name from `-scheme`. As with gradle, a *build* failure maps to
323
+ `not_collected` rather than `failed`, so a stub referencing a missing symbol is not
324
+ mistaken for RED.
325
+
326
+ `test_command` is required here — the adapter cannot guess the scheme, destination, or
327
+ derived-data path. It appends `-only-testing:` for targeted runs and `-enumerate-tests`
328
+ for collection, and changes nothing else. Simulator runs are serial in practice: give
329
+ them a `lease` so two projects don't drive the same simulator at once.
330
+
331
+ ```toml
332
+ [project.native-ios]
333
+ root = "native-ios"
334
+ adapter = "xctest"
335
+ test_paths = ["AppTests/"]
336
+ test_command = """xcodebuild test \
337
+ -project App.xcodeproj \
338
+ -scheme AppTests \
339
+ -destination 'platform=iOS Simulator,name=App-Unit' \
340
+ -derivedDataPath /tmp/app-unit-dd"""
341
+ lease = "ios-simulator"
342
+ timeout = 900
343
+ ```
344
+
317
345
  Third-party adapters register under the
318
346
  `tddcli.adapters` entry-point group:
319
347
 
@@ -375,14 +403,49 @@ and is never reclassified as a pin.
375
403
  | `tdd metrics` | fidelity, attempts, violations, interventions |
376
404
  | `tdd fleet [--json]` | all active runs across every worktree; read-only |
377
405
 
406
+ ## Scoped baseline capture (R9.5c)
407
+
408
+ `run start` probes only the projects the plan can actually reach: the declared cycle projects
409
+ plus the transitive `consumed_by` closure of artifacts whose producer is in that set. Projects
410
+ outside the reachable set never run during the plan, so their baseline is never subtracted from
411
+ anything — probing them is pure overhead. A `baseline_scoped` integrity event records which
412
+ projects were skipped, so the scoping is auditable.
413
+
414
+ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
415
+
416
+ tdd run start --plan tasks/plan.md --baseline-all
417
+
418
+ Use this when a cycle may edit files outside the predicted reachable set and you want every
419
+ project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
420
+
421
+ ### When a sweep reaches an un-baselined project (R9.5d)
422
+
423
+ If an edit during a run touches a file owned by an artifact that was outside the predicted
424
+ reachable set, the close sweep may pull in a project that was never baselined. Its failures are
425
+ unattributable — no baseline exists to subtract — so the sweep replies `resolve_blocker` with
426
+ kind `no_baseline_for_project` rather than mislabelling them as regressions. Recovery:
427
+
428
+ tdd blocker --kind no_baseline_for_project --detail "svc pulled in unexpectedly"
429
+ tdd resume --unblock --accept-failures --note "folding svc sweep failures into baseline"
430
+
431
+ `--accept-failures` inserts a fresh baseline row for the un-baselined project (recording those
432
+ failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
433
+ with the project properly baselined.
434
+
378
435
  ## Running a long baseline
379
436
 
380
- `run start` probes every project's suite before a run exists (R9.5a), and on a real project
437
+ `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
381
438
  that can take minutes — well past an agent harness's default Bash timeout. If the command
382
439
  appears to hang or time out, **do not re-run it**: the probe is still making progress in the
383
440
  background, and a second `run start` against the same worktree is refused with
384
441
  `reason: "baseline_in_progress"` — retrying on timeout just stacks refusals on top of a
385
- baseline that was never stuck. In order of preference:
442
+ baseline that was never stuck.
443
+
444
+ `tdd advance` is similarly single-flight per worktree. A close sweep (artifact regeneration, full
445
+ suite, lint, typecheck) can run for several minutes. If an advance appears to hang, **do not
446
+ re-run it**: a second `advance` against the same worktree is refused with
447
+ `reason: "advance_in_flight"`. The refusal carries `pid`, `started_at`, and `elapsed_s` so the
448
+ agent can confirm the first process is still alive. Run `tdd status` to see the current run state. In order of preference:
386
449
 
387
450
  1. **Background it.** Run `tdd run start` in the background if your harness supports it. The
388
451
  heartbeat (`baseline_captured` / `project_completed` lines on stderr) lands in the task log
@@ -286,6 +286,34 @@ test_paths = ["src/test/"]
286
286
  test_command = "./gradlew testDebugUnitTest"
287
287
  ```
288
288
 
289
+ The `xctest` adapter drives Swift and Objective-C projects through `xcodebuild`. Test ids
290
+ use xcodebuild's own `-only-testing:` format — `BundleName/ClassName/testMethodName` — so
291
+ a single XCTest can be driven through RED → GREEN without any id translation. Collection
292
+ prefers `xcodebuild test -enumerate-tests` (Xcode 16+) and falls back to grepping
293
+ `class Foo: XCTestCase` / `func testBar()` out of the Swift sources under `test_paths`,
294
+ deriving the bundle name from `-scheme`. As with gradle, a *build* failure maps to
295
+ `not_collected` rather than `failed`, so a stub referencing a missing symbol is not
296
+ mistaken for RED.
297
+
298
+ `test_command` is required here — the adapter cannot guess the scheme, destination, or
299
+ derived-data path. It appends `-only-testing:` for targeted runs and `-enumerate-tests`
300
+ for collection, and changes nothing else. Simulator runs are serial in practice: give
301
+ them a `lease` so two projects don't drive the same simulator at once.
302
+
303
+ ```toml
304
+ [project.native-ios]
305
+ root = "native-ios"
306
+ adapter = "xctest"
307
+ test_paths = ["AppTests/"]
308
+ test_command = """xcodebuild test \
309
+ -project App.xcodeproj \
310
+ -scheme AppTests \
311
+ -destination 'platform=iOS Simulator,name=App-Unit' \
312
+ -derivedDataPath /tmp/app-unit-dd"""
313
+ lease = "ios-simulator"
314
+ timeout = 900
315
+ ```
316
+
289
317
  Third-party adapters register under the
290
318
  `tddcli.adapters` entry-point group:
291
319
 
@@ -347,14 +375,49 @@ and is never reclassified as a pin.
347
375
  | `tdd metrics` | fidelity, attempts, violations, interventions |
348
376
  | `tdd fleet [--json]` | all active runs across every worktree; read-only |
349
377
 
378
+ ## Scoped baseline capture (R9.5c)
379
+
380
+ `run start` probes only the projects the plan can actually reach: the declared cycle projects
381
+ plus the transitive `consumed_by` closure of artifacts whose producer is in that set. Projects
382
+ outside the reachable set never run during the plan, so their baseline is never subtracted from
383
+ anything — probing them is pure overhead. A `baseline_scoped` integrity event records which
384
+ projects were skipped, so the scoping is auditable.
385
+
386
+ Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
387
+
388
+ tdd run start --plan tasks/plan.md --baseline-all
389
+
390
+ Use this when a cycle may edit files outside the predicted reachable set and you want every
391
+ project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
392
+
393
+ ### When a sweep reaches an un-baselined project (R9.5d)
394
+
395
+ If an edit during a run touches a file owned by an artifact that was outside the predicted
396
+ reachable set, the close sweep may pull in a project that was never baselined. Its failures are
397
+ unattributable — no baseline exists to subtract — so the sweep replies `resolve_blocker` with
398
+ kind `no_baseline_for_project` rather than mislabelling them as regressions. Recovery:
399
+
400
+ tdd blocker --kind no_baseline_for_project --detail "svc pulled in unexpectedly"
401
+ tdd resume --unblock --accept-failures --note "folding svc sweep failures into baseline"
402
+
403
+ `--accept-failures` inserts a fresh baseline row for the un-baselined project (recording those
404
+ failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
405
+ with the project properly baselined.
406
+
350
407
  ## Running a long baseline
351
408
 
352
- `run start` probes every project's suite before a run exists (R9.5a), and on a real project
409
+ `run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
353
410
  that can take minutes — well past an agent harness's default Bash timeout. If the command
354
411
  appears to hang or time out, **do not re-run it**: the probe is still making progress in the
355
412
  background, and a second `run start` against the same worktree is refused with
356
413
  `reason: "baseline_in_progress"` — retrying on timeout just stacks refusals on top of a
357
- baseline that was never stuck. In order of preference:
414
+ baseline that was never stuck.
415
+
416
+ `tdd advance` is similarly single-flight per worktree. A close sweep (artifact regeneration, full
417
+ suite, lint, typecheck) can run for several minutes. If an advance appears to hang, **do not
418
+ re-run it**: a second `advance` against the same worktree is refused with
419
+ `reason: "advance_in_flight"`. The refusal carries `pid`, `started_at`, and `elapsed_s` so the
420
+ agent can confirm the first process is still alive. Run `tdd status` to see the current run state. In order of preference:
358
421
 
359
422
  1. **Background it.** Run `tdd run start` in the background if your harness supports it. The
360
423
  heartbeat (`baseline_captured` / `project_completed` lines on stderr) lands in the task log
@@ -3,4 +3,4 @@
3
3
  State is derived from observed test execution, never asserted by the caller.
4
4
  """
5
5
 
6
- __version__ = "0.6.0"
6
+ __version__ = "0.7.0"
@@ -5,7 +5,7 @@ compose without translation:
5
5
 
6
6
  BundleName/ClassName/testMethodName
7
7
 
8
- e.g. CoParentTests/PollCadenceTests/testPollIntervalScalesOnlyUnderE2E
8
+ e.g. AppTests/PollCadenceTests/testPollIntervalScalesOnlyUnderE2E
9
9
 
10
10
  A xcodebuild build failure maps to `not_collected` rather than `failed`.
11
11
  Swift has no separate collection phase — a test referencing a missing symbol
@@ -27,12 +27,12 @@ Config (`tdd.toml`):
27
27
  [project.native-ios]
28
28
  root = "native-ios"
29
29
  adapter = "xctest"
30
- test_paths = ["CoParentTests/"] # Swift source directories
30
+ test_paths = ["AppTests/"] # Swift source directories
31
31
  test_command = "xcodebuild test \\
32
- -project CoParent.xcodeproj \\
33
- -scheme CoParentTests \\
34
- -destination 'platform=iOS Simulator,name=CoParent-Unit' \\
35
- -derivedDataPath /tmp/coparent-unit-dd"
32
+ -project App.xcodeproj \\
33
+ -scheme AppTests \\
34
+ -destination 'platform=iOS Simulator,name=App-Unit' \\
35
+ -derivedDataPath /tmp/app-unit-dd"
36
36
 
37
37
  `test_command` is required — the adapter cannot know the scheme, destination,
38
38
  or derived-data path otherwise. The adapter appends `-only-testing:` for
@@ -60,14 +60,14 @@ from .base import (
60
60
  # Patterns applied to xcodebuild stdout
61
61
  # ---------------------------------------------------------------------------
62
62
 
63
- # "Test Case '-[CoParentTests.PollCadenceTests testPollIntervalScalesOnlyUnderE2E]' passed (0.001 seconds)."
63
+ # "Test Case '-[AppTests.PollCadenceTests testPollIntervalScalesOnlyUnderE2E]' passed (0.001 seconds)."
64
64
  _CASE_RE = re.compile(r"Test Case '-\[(\w+)\.(\w+) (\w+)\]' (passed|failed)")
65
65
 
66
66
  # End-of-build marker for a failed build (no tests ran)
67
67
  _BUILD_FAILED_RE = re.compile(r"\*\* BUILD FAILED \*\*")
68
68
 
69
69
  # Lines emitted by xcodebuild test -enumerate-tests (Xcode 16+):
70
- # CoParentTests/PollCadenceTests/testPollIntervalScalesOnlyUnderE2E
70
+ # AppTests/PollCadenceTests/testPollIntervalScalesOnlyUnderE2E
71
71
  _ENUMERATE_LINE_RE = re.compile(r"^[ \t]*(\w+)/(\w+)/(test\w+)[ \t]*$", re.MULTILINE)
72
72
 
73
73
  # Swift source patterns for the per-file grep fallback
@@ -105,7 +105,7 @@ class XCTestAdapter(Adapter):
105
105
  """Best-effort: extract the test bundle name from `-scheme <Name>`.
106
106
 
107
107
  xcodebuild schemes typically share a name with their test bundle
108
- (e.g. `-scheme CoParentTests` → bundle `CoParentTests`). When the
108
+ (e.g. `-scheme AppTests` → bundle `AppTests`). When the
109
109
  scheme cannot be parsed, fall back to `"Unknown"` — per-file ids will
110
110
  still be formed, just unaddressable by `-only-testing:` without the
111
111
  real bundle name.
@@ -284,9 +284,9 @@ class XCTestAdapter(Adapter):
284
284
  """Capture the assertion lines between 'started' and 'failed' for one test.
285
285
 
286
286
  xcodebuild interleaves test output like:
287
- Test Case '-[CoParentTests.PollCadenceTests testSomething]' started.
287
+ Test Case '-[AppTests.PollCadenceTests testSomething]' started.
288
288
  /path/test.swift:42: error: … XCTAssertEqual failed: …
289
- Test Case '-[CoParentTests.PollCadenceTests testSomething]' failed (0.001 seconds).
289
+ Test Case '-[AppTests.PollCadenceTests testSomething]' failed (0.001 seconds).
290
290
  """
291
291
  parts = native_id.split("/")
292
292
  if len(parts) != 3:
@@ -310,6 +310,16 @@ def _handle_refactor(engine: Engine, cycle, retried: bool) -> Envelope:
310
310
  outcome = engine.sweep(cycle, touched, skip_own=skip_own)
311
311
 
312
312
  if not outcome.ok:
313
+ if outcome.unbaselined:
314
+ projects_list = ", ".join(sorted(outcome.unbaselined))
315
+ return _reply(
316
+ engine, cycle, Verb.RESOLVE_BLOCKER,
317
+ f"Close sweep found failures in un-baselined project(s): {projects_list}. "
318
+ "These are unattributable — no baseline exists to subtract. "
319
+ "File: tdd blocker --kind no_baseline_for_project --detail '...', "
320
+ "then resume --unblock --accept-failures to fold them into the baseline.",
321
+ unbaselined=outcome.unbaselined, commit=sha,
322
+ )
313
323
  if outcome.failures:
314
324
  return _reply(
315
325
  engine, cycle, Verb.FIX_REGRESSION,
@@ -48,6 +48,9 @@ BLOCKER_KINDS = {
48
48
  # Failing, but not caused by this run — a flake, or something the baseline missed.
49
49
  # Distinct from `regression`, which records a defect the run introduced.
50
50
  "pre_existing_failure",
51
+ # Failing in a project that was never baselined — not attributable as a regression
52
+ # (no baseline to subtract) vs. `pre_existing_failure` (baseline exists but missed it).
53
+ "no_baseline_for_project",
51
54
  }
52
55
 
53
56
 
@@ -489,14 +492,13 @@ def cmd_plan_register(args) -> Envelope:
489
492
  )
490
493
 
491
494
 
492
- def _probe_projects(cfg, worktree, ledger, on_progress):
493
- """Probe every project's baseline (R9.5a): run + collect, timing each, emitting a
494
- `baseline_captured` heartbeat, and calling `on_progress(done, name)` — extracted
495
- from `cmd_run_start`, which carried claiming, timing, heartbeating and progress
496
- updates inline past the point of legibility. Returns `{name: (verdict, collection)}`.
495
+ def _probe_projects(projects, worktree, ledger, on_progress):
496
+ """Probe a mapping of projects' baselines (R9.5a): run + collect, timing each,
497
+ emitting a `baseline_captured` heartbeat, and calling `on_progress(done, name)`.
498
+ Returns `{name: (verdict, collection)}`.
497
499
  """
498
500
  probes = {}
499
- for done, (name, project) in enumerate(cfg.projects.items(), start=1):
501
+ for done, (name, project) in enumerate(projects.items(), start=1):
500
502
  adapter = adapters.build(project, worktree)
501
503
  started = time.monotonic()
502
504
  verdict = adapter.run(None)
@@ -561,6 +563,15 @@ def cmd_run_start(args) -> Envelope:
561
563
  if contract_row["status"] == "undeclared" and not args.allow_undeclared:
562
564
  return failure("contract is undeclared; pass --allow-undeclared")
563
565
 
566
+ # R9.5c — scope baseline capture to plan-reachable projects unless opted out.
567
+ declared_cycles = contract_mod.cycles_from_json(contract_row["declared_cycles"])
568
+ if declared_cycles and not args.baseline_all:
569
+ declared_names = [p for c in declared_cycles for p in c.projects]
570
+ reachable_names = cfg.reachable_projects(declared_names)
571
+ probe_projects = {n: cfg.projects[n] for n in reachable_names if n in cfg.projects}
572
+ else:
573
+ probe_projects = cfg.projects
574
+
564
575
  # Claim the worktree before probing: two `run start` calls against
565
576
  # one worktree must not both pass the baseline window. `Ledger.claim`'s `UNIQUE`
566
577
  # insert is the lock — do not read-then-write, which is the race this
@@ -576,7 +587,7 @@ def cmd_run_start(args) -> Envelope:
576
587
  str(worktree),
577
588
  hostname=socket.gethostname(),
578
589
  pid=os.getpid(),
579
- projects_total=len(cfg.projects),
590
+ projects_total=len(probe_projects),
580
591
  )
581
592
  except sqlite3.IntegrityError:
582
593
  return failure(
@@ -594,7 +605,7 @@ def cmd_run_start(args) -> Envelope:
594
605
  # attempt — and must release the claim too, or the retry it invites is itself
595
606
  # refused.
596
607
  probes = _probe_projects(
597
- cfg,
608
+ probe_projects,
598
609
  worktree,
599
610
  ledger,
600
611
  on_progress=lambda done, name: ledger.update_claim(
@@ -640,6 +651,9 @@ def cmd_run_start(args) -> Envelope:
640
651
  run = ledger.one("SELECT * FROM run WHERE id = ?", (run_id,))
641
652
  if blob_changed:
642
653
  ledger.event(run_id, None, "plan_blob_changed", rel)
654
+ skipped = sorted(set(cfg.projects) - set(probe_projects))
655
+ if skipped:
656
+ ledger.event(run_id, None, "baseline_scoped", json.dumps(skipped))
643
657
 
644
658
  # Baselines and the collection snapshot, per project (R9.5, R8.9) — from the
645
659
  # probe above, so the suite is not run twice.
@@ -723,7 +737,31 @@ def cmd_advance(args) -> Envelope:
723
737
  run={"id": run["id"], "phase": CLOSED},
724
738
  next_action=NextAction(Verb.COMPLETE, "All cycles complete."),
725
739
  )
726
- return do_advance(engine, cycle, retry=args.retry)
740
+ existing_advance = ledger.active_advance_claim(str(worktree))
741
+ if existing_advance is not None and existing_advance["stale"]:
742
+ ledger.release_advance_claim(str(worktree))
743
+
744
+ try:
745
+ ledger.claim_advance(str(worktree), hostname=socket.gethostname(), pid=os.getpid())
746
+ except sqlite3.IntegrityError:
747
+ held = ledger.one("SELECT * FROM advance_claim WHERE worktree_path = ?", (str(worktree),))
748
+ started = datetime.fromisoformat(held["started_at"])
749
+ if started.tzinfo is None:
750
+ started = started.replace(tzinfo=timezone.utc)
751
+ elapsed_s = int((datetime.now(timezone.utc) - started).total_seconds())
752
+ return failure(
753
+ "an advance is already in flight for this worktree; do not re-run or kill it"
754
+ " — wait for its envelope, then run `tdd status` to see where the run stands",
755
+ reason="advance_in_flight",
756
+ pid=held["pid"],
757
+ started_at=held["started_at"],
758
+ elapsed_s=elapsed_s,
759
+ )
760
+ try:
761
+ result = do_advance(engine, cycle, retry=args.retry)
762
+ finally:
763
+ ledger.release_advance_claim(str(worktree))
764
+ return result
727
765
 
728
766
 
729
767
  def cmd_cycle_skip(args) -> Envelope:
@@ -818,15 +856,27 @@ def _accept_failures_into_baseline(ledger: Ledger, run_id: int) -> dict[str, lis
818
856
  }
819
857
  accepted: dict[str, list[str]] = {}
820
858
  for sweep in latest:
821
- row = rows.get(sweep["project"])
859
+ project = sweep["project"]
860
+ row = rows.get(project)
861
+ sweep_failed = sorted(json.loads(sweep["other_failures"]))
822
862
  if row is None:
823
- continue
824
- known = set(json.loads(row["failing"]))
825
- new = sorted(set(json.loads(sweep["other_failures"])) - known)
826
- if not new:
827
- continue
828
- ledger.update("baseline", row["id"], failing=json.dumps(sorted(known | set(new))))
829
- accepted[sweep["project"]] = new
863
+ if not sweep_failed:
864
+ continue
865
+ ledger.insert(
866
+ "baseline",
867
+ run_id=run_id,
868
+ project=project,
869
+ failing=json.dumps(sweep_failed),
870
+ captured_at=now(),
871
+ )
872
+ accepted[project] = sweep_failed
873
+ else:
874
+ known = set(json.loads(row["failing"]))
875
+ new = sorted(set(sweep_failed) - known)
876
+ if not new:
877
+ continue
878
+ ledger.update("baseline", row["id"], failing=json.dumps(sorted(known | set(new))))
879
+ accepted[project] = new
830
880
  if accepted:
831
881
  ledger.event(run_id, None, "baseline_amended", json.dumps(accepted))
832
882
  return accepted
@@ -1140,6 +1190,7 @@ def build_parser() -> argparse.ArgumentParser:
1140
1190
  s.add_argument("--executor", help="human-supplied label; agents must not use this")
1141
1191
  s.add_argument("--allow-dirty", action="store_true")
1142
1192
  s.add_argument("--allow-undeclared", action="store_true")
1193
+ s.add_argument("--baseline-all", action="store_true", help="probe all projects, skipping reachability scoping")
1143
1194
  s.set_defaults(fn=cmd_run_start)
1144
1195
 
1145
1196
  s = sub.add_parser("status")
@@ -212,6 +212,32 @@ class Config:
212
212
  seen.add(up)
213
213
  return chain
214
214
 
215
+ def _root_project(self, produced_by: str) -> str | None:
216
+ """Resolve a produced_by value (project name or artifact.<name> chain) to a root project name."""
217
+ if produced_by.startswith("artifact."):
218
+ upstream = self.artifacts.get(produced_by.split(".", 1)[1])
219
+ if upstream is None:
220
+ return None
221
+ return self._root_project(upstream.produced_by)
222
+ return produced_by if produced_by in self.projects else None
223
+
224
+ def reachable_projects(self, declared: list[str]) -> list[str]:
225
+ declared_set = set(declared)
226
+ reachable = set(declared)
227
+ changed = True
228
+ while changed:
229
+ changed = False
230
+ for art in self.artifacts.values():
231
+ root = self._root_project(art.produced_by)
232
+ if root in reachable:
233
+ for consumer in art.consumed_by:
234
+ if consumer not in reachable:
235
+ proj = self.projects.get(consumer)
236
+ if consumer in declared_set or (proj and proj.in_close_sweep):
237
+ reachable.add(consumer)
238
+ changed = True
239
+ return sorted(reachable)
240
+
215
241
  def close_sweep_projects(self, cycle_projects: list[str], touched: set[str]) -> list[str]:
216
242
  """R9.2 — the cycle's own projects, plus anything downstream of an artifact it touched."""
217
243
  names = {n for n in cycle_projects}
@@ -221,14 +247,10 @@ class Config:
221
247
  return [n for n in sorted(names) if self.projects[n].in_close_sweep]
222
248
 
223
249
  def _artifact_touched(self, art: Artifact, touched: set[str]) -> bool:
224
- producer = art.produced_by
225
- if producer.startswith("artifact."):
226
- upstream = self.artifacts.get(producer.split(".", 1)[1])
227
- return bool(upstream and self._artifact_touched(upstream, touched))
228
- proj = self.projects.get(producer)
229
- if proj is None:
250
+ root = self._root_project(art.produced_by)
251
+ if root is None:
230
252
  return False
231
- return any(proj.owns(p) for p in touched)
253
+ return any(self.projects[root].owns(p) for p in touched)
232
254
 
233
255
  def full_sweep_projects(self) -> list[str]:
234
256
  return sorted(self.projects)
@@ -13,7 +13,7 @@ import sqlite3
13
13
  from datetime import datetime, timedelta, timezone
14
14
  from pathlib import Path
15
15
 
16
- SCHEMA_VERSION = 2
16
+ SCHEMA_VERSION = 3
17
17
 
18
18
 
19
19
  class LedgerVersionError(RuntimeError):
@@ -29,6 +29,8 @@ class LedgerVersionError(RuntimeError):
29
29
  MIGRATIONS: dict[int, str] = {
30
30
  # v1 -> v2 added the baseline_claim table; CREATE TABLE IF NOT EXISTS covers it.
31
31
  1: "",
32
+ # v2 -> v3 added the advance_claim table; CREATE TABLE IF NOT EXISTS covers it.
33
+ 2: "",
32
34
  }
33
35
 
34
36
  SCHEMA = """
@@ -45,6 +47,14 @@ CREATE TABLE IF NOT EXISTS baseline_claim (
45
47
  started_at TEXT NOT NULL
46
48
  );
47
49
 
50
+ CREATE TABLE IF NOT EXISTS advance_claim (
51
+ id INTEGER PRIMARY KEY,
52
+ worktree_path TEXT NOT NULL UNIQUE,
53
+ hostname TEXT NOT NULL,
54
+ pid INTEGER NOT NULL,
55
+ started_at TEXT NOT NULL
56
+ );
57
+
48
58
  CREATE TABLE IF NOT EXISTS plan_contract (
49
59
  id INTEGER PRIMARY KEY,
50
60
  plan_path TEXT NOT NULL,
@@ -412,6 +422,54 @@ class Ledger:
412
422
  )
413
423
  self.db.commit()
414
424
 
425
+ def claim_advance(self, worktree: str, hostname: str, pid: int) -> int:
426
+ """Insert the advance claim row. The insert is the lock: `worktree_path`
427
+ carries `UNIQUE`, so a second claim on the same worktree raises
428
+ `sqlite3.IntegrityError` rather than racing a read-then-write check."""
429
+ return self.insert(
430
+ "advance_claim",
431
+ worktree_path=worktree,
432
+ hostname=hostname,
433
+ pid=pid,
434
+ started_at=now(),
435
+ )
436
+
437
+ def release_advance_claim(self, worktree: str) -> None:
438
+ self.db.execute("DELETE FROM advance_claim WHERE worktree_path = ?", (worktree,))
439
+ self.db.commit()
440
+
441
+ @staticmethod
442
+ def _claim_is_stale(hostname: str, pid: int, started_at: str) -> bool:
443
+ """Staleness rule shared by all claim kinds.
444
+
445
+ Same host → liveness via os.kill(pid, 0). Cross-host → age > 60 min
446
+ (a pid is meaningless from another host; reused pids would make a
447
+ liveness check actively wrong, so we fall back to age).
448
+ """
449
+ if hostname == socket.gethostname():
450
+ try:
451
+ os.kill(pid, 0)
452
+ return False
453
+ except ProcessLookupError:
454
+ # Same host, pid no longer running.
455
+ return True
456
+ except PermissionError:
457
+ # Pid exists but owned by someone else — alive.
458
+ return False
459
+ started = datetime.fromisoformat(started_at)
460
+ if started.tzinfo is None:
461
+ started = started.replace(tzinfo=timezone.utc)
462
+ return datetime.now(timezone.utc) - started > timedelta(minutes=60)
463
+
464
+ def active_advance_claim(self, worktree: str) -> dict | None:
465
+ """Read-only observer — staleness computed but no row deleted."""
466
+ row = self.one("SELECT * FROM advance_claim WHERE worktree_path = ?", (worktree,))
467
+ if row is None:
468
+ return None
469
+ claim = dict(row)
470
+ claim["stale"] = self._claim_is_stale(claim["hostname"], claim["pid"], claim["started_at"])
471
+ return claim
472
+
415
473
  def active_claim(self, worktree: str) -> dict | None:
416
474
  """Read-only, per the store's append-only contract — `cmd_progress` and
417
475
  `cmd_status` call this as pure observers. Only `cmd_run_start` acts on the
@@ -420,23 +478,8 @@ class Ledger:
420
478
  if row is None:
421
479
  return None
422
480
  claim = dict(row)
423
- if claim["hostname"] == socket.gethostname():
424
- try:
425
- os.kill(claim["pid"], 0)
426
- stale = False
427
- except ProcessLookupError:
428
- # Same host, pid no longer running (e.g. a `SIGKILL`ed `run start`).
429
- stale = True
430
- except PermissionError:
431
- # Pid exists but is owned by someone else — alive.
432
- stale = False
433
- else:
434
- # A pid is meaningless from another host, and reused pids would make a
435
- # host-crossing liveness check actively wrong. Fall back to age: a false
436
- # "alive" bricks the worktree, a false "dead" reopens the bug (Decisions).
437
- started = datetime.fromisoformat(claim["started_at"])
438
- if started.tzinfo is None:
439
- started = started.replace(tzinfo=timezone.utc)
440
- stale = datetime.now(timezone.utc) - started > timedelta(minutes=60)
441
- claim["stale"] = stale
481
+ # A pid is meaningless from another host, and reused pids would make a
482
+ # host-crossing liveness check actively wrong. Fall back to age: a false
483
+ # "alive" bricks the worktree, a false "dead" reopens the bug (Decisions).
484
+ claim["stale"] = self._claim_is_stale(claim["hostname"], claim["pid"], claim["started_at"])
442
485
  return claim