tdd-cli 0.6.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/CHANGELOG.md +26 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/PKG-INFO +66 -3
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/README.md +65 -2
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/__init__.py +1 -1
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/xctest_adapter.py +11 -11
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/advance.py +10 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/cli.py +68 -17
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/config.py +29 -7
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/ledger.py +63 -20
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/machine.py +20 -5
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/conftest.py +88 -1
- tdd_cli-0.7.0/tests/test_baseline_integrity.py +321 -0
- tdd_cli-0.7.0/tests/test_concurrent_advance.py +178 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_config_and_staging.py +103 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_heartbeat.py +20 -1
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_xctest_adapter.py +40 -49
- tdd_cli-0.6.0/tests/test_baseline_integrity.py +0 -148
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/.gitignore +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/LICENSE +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/SECURITY.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/claude-code-hooks/README.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/claude-code-hooks/bash_hook.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/claude-code-hooks/stop_hook.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/plan.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-drive/README.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-drive/SKILL.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-handoff/README.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/examples/skills/tdd-handoff/SKILL.md +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/pyproject.toml +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/__init__.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/base.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/exec_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/gradle_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/pytest_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/adapters/vitest_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/contract.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/envelope.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/fleet.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/gitutil.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/identity.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/leases.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/render.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/snapshot.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/src/tddcli/staging.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_artifact_regeneration.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_batch_collection.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_config_drift.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_contract.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_doctor_attribution.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_doctor_blockers.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_end_to_end.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_example_plan.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_exec_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_failure_clipping.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_fleet.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_gradle_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_init_detection.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_named_leases_and_timeout.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_pin_cycles.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_progress.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_project_commands.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_project_env.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_python_env_managers.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_refactor_cycles.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_release_surface.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_run_claim.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_single_project_repo.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_snapshot_and_identity.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_stub_hint.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_suite_overrides.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_target_validation.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_timing_visibility.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_vitest_adapter.py +0 -0
- {tdd_cli-0.6.0 → tdd_cli-0.7.0}/tests/test_worker_leases.py +0 -0
|
@@ -4,6 +4,32 @@ All notable changes to this project are documented here.
|
|
|
4
4
|
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
5
|
and the project adheres to [Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.7.0] - 2026-08-23
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- **Concurrent `tdd advance` no longer corrupts a run.** Two `advance` processes
|
|
14
|
+
racing on the same worktree could both close the same cycle row. Because
|
|
15
|
+
`close_cycle` unconditionally opened the next ordinal, a double-close forked the
|
|
16
|
+
run into two parallel cycle chains; every remaining cycle ran twice, the run
|
|
17
|
+
"completed" with one chain's last row permanently open, and the doubling was
|
|
18
|
+
invisible from the agent's perspective. `close_cycle` now re-reads `closed_at`
|
|
19
|
+
before acting; if the row is already closed it returns the currently-open cycle
|
|
20
|
+
without transitioning or opening anything. `open_cycle` returns the existing open
|
|
21
|
+
row for an ordinal rather than inserting a duplicate.
|
|
22
|
+
|
|
23
|
+
### Added
|
|
24
|
+
|
|
25
|
+
- **Per-worktree advance claim.** `tdd advance` now acquires a `advance_claim` row
|
|
26
|
+
before dispatching. A second concurrent `advance` is refused immediately with
|
|
27
|
+
`ok: false`, `reason: "advance_in_flight"`, and metadata (`pid`, `started_at`,
|
|
28
|
+
`elapsed_s`) that lets the agent confirm the holder is still alive. The claim is
|
|
29
|
+
released in a `finally` so a raising handler cannot wedge the worktree; a claim
|
|
30
|
+
held by a dead pid is reclaimed automatically on the next call. Schema version
|
|
31
|
+
bumped to 3.
|
|
32
|
+
|
|
7
33
|
## [0.6.0] - 2026-08-19
|
|
8
34
|
|
|
9
35
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: tdd-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Ledger-backed TDD process controller for autonomous coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/geuben/tdd-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/geuben/tdd-cli
|
|
@@ -314,6 +314,34 @@ test_paths = ["src/test/"]
|
|
|
314
314
|
test_command = "./gradlew testDebugUnitTest"
|
|
315
315
|
```
|
|
316
316
|
|
|
317
|
+
The `xctest` adapter drives Swift and Objective-C projects through `xcodebuild`. Test ids
|
|
318
|
+
use xcodebuild's own `-only-testing:` format — `BundleName/ClassName/testMethodName` — so
|
|
319
|
+
a single XCTest can be driven through RED → GREEN without any id translation. Collection
|
|
320
|
+
prefers `xcodebuild test -enumerate-tests` (Xcode 16+) and falls back to grepping
|
|
321
|
+
`class Foo: XCTestCase` / `func testBar()` out of the Swift sources under `test_paths`,
|
|
322
|
+
deriving the bundle name from `-scheme`. As with gradle, a *build* failure maps to
|
|
323
|
+
`not_collected` rather than `failed`, so a stub referencing a missing symbol is not
|
|
324
|
+
mistaken for RED.
|
|
325
|
+
|
|
326
|
+
`test_command` is required here — the adapter cannot guess the scheme, destination, or
|
|
327
|
+
derived-data path. It appends `-only-testing:` for targeted runs and `-enumerate-tests`
|
|
328
|
+
for collection, and changes nothing else. Simulator runs are serial in practice: give
|
|
329
|
+
them a `lease` so two projects don't drive the same simulator at once.
|
|
330
|
+
|
|
331
|
+
```toml
|
|
332
|
+
[project.native-ios]
|
|
333
|
+
root = "native-ios"
|
|
334
|
+
adapter = "xctest"
|
|
335
|
+
test_paths = ["AppTests/"]
|
|
336
|
+
test_command = """xcodebuild test \
|
|
337
|
+
-project App.xcodeproj \
|
|
338
|
+
-scheme AppTests \
|
|
339
|
+
-destination 'platform=iOS Simulator,name=App-Unit' \
|
|
340
|
+
-derivedDataPath /tmp/app-unit-dd"""
|
|
341
|
+
lease = "ios-simulator"
|
|
342
|
+
timeout = 900
|
|
343
|
+
```
|
|
344
|
+
|
|
317
345
|
Third-party adapters register under the
|
|
318
346
|
`tddcli.adapters` entry-point group:
|
|
319
347
|
|
|
@@ -375,14 +403,49 @@ and is never reclassified as a pin.
|
|
|
375
403
|
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
376
404
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
377
405
|
|
|
406
|
+
## Scoped baseline capture (R9.5c)
|
|
407
|
+
|
|
408
|
+
`run start` probes only the projects the plan can actually reach: the declared cycle projects
|
|
409
|
+
plus the transitive `consumed_by` closure of artifacts whose producer is in that set. Projects
|
|
410
|
+
outside the reachable set never run during the plan, so their baseline is never subtracted from
|
|
411
|
+
anything — probing them is pure overhead. A `baseline_scoped` integrity event records which
|
|
412
|
+
projects were skipped, so the scoping is auditable.
|
|
413
|
+
|
|
414
|
+
Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
415
|
+
|
|
416
|
+
tdd run start --plan tasks/plan.md --baseline-all
|
|
417
|
+
|
|
418
|
+
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
419
|
+
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
420
|
+
|
|
421
|
+
### When a sweep reaches an un-baselined project (R9.5d)
|
|
422
|
+
|
|
423
|
+
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
424
|
+
reachable set, the close sweep may pull in a project that was never baselined. Its failures are
|
|
425
|
+
unattributable — no baseline exists to subtract — so the sweep replies `resolve_blocker` with
|
|
426
|
+
kind `no_baseline_for_project` rather than mislabelling them as regressions. Recovery:
|
|
427
|
+
|
|
428
|
+
tdd blocker --kind no_baseline_for_project --detail "svc pulled in unexpectedly"
|
|
429
|
+
tdd resume --unblock --accept-failures --note "folding svc sweep failures into baseline"
|
|
430
|
+
|
|
431
|
+
`--accept-failures` inserts a fresh baseline row for the un-baselined project (recording those
|
|
432
|
+
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
433
|
+
with the project properly baselined.
|
|
434
|
+
|
|
378
435
|
## Running a long baseline
|
|
379
436
|
|
|
380
|
-
`run start` probes
|
|
437
|
+
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
381
438
|
that can take minutes — well past an agent harness's default Bash timeout. If the command
|
|
382
439
|
appears to hang or time out, **do not re-run it**: the probe is still making progress in the
|
|
383
440
|
background, and a second `run start` against the same worktree is refused with
|
|
384
441
|
`reason: "baseline_in_progress"` — retrying on timeout just stacks refusals on top of a
|
|
385
|
-
baseline that was never stuck.
|
|
442
|
+
baseline that was never stuck.
|
|
443
|
+
|
|
444
|
+
`tdd advance` is similarly single-flight per worktree. A close sweep (artifact regeneration, full
|
|
445
|
+
suite, lint, typecheck) can run for several minutes. If an advance appears to hang, **do not
|
|
446
|
+
re-run it**: a second `advance` against the same worktree is refused with
|
|
447
|
+
`reason: "advance_in_flight"`. The refusal carries `pid`, `started_at`, and `elapsed_s` so the
|
|
448
|
+
agent can confirm the first process is still alive. Run `tdd status` to see the current run state. In order of preference:
|
|
386
449
|
|
|
387
450
|
1. **Background it.** Run `tdd run start` in the background if your harness supports it. The
|
|
388
451
|
heartbeat (`baseline_captured` / `project_completed` lines on stderr) lands in the task log
|
|
@@ -286,6 +286,34 @@ test_paths = ["src/test/"]
|
|
|
286
286
|
test_command = "./gradlew testDebugUnitTest"
|
|
287
287
|
```
|
|
288
288
|
|
|
289
|
+
The `xctest` adapter drives Swift and Objective-C projects through `xcodebuild`. Test ids
|
|
290
|
+
use xcodebuild's own `-only-testing:` format — `BundleName/ClassName/testMethodName` — so
|
|
291
|
+
a single XCTest can be driven through RED → GREEN without any id translation. Collection
|
|
292
|
+
prefers `xcodebuild test -enumerate-tests` (Xcode 16+) and falls back to grepping
|
|
293
|
+
`class Foo: XCTestCase` / `func testBar()` out of the Swift sources under `test_paths`,
|
|
294
|
+
deriving the bundle name from `-scheme`. As with gradle, a *build* failure maps to
|
|
295
|
+
`not_collected` rather than `failed`, so a stub referencing a missing symbol is not
|
|
296
|
+
mistaken for RED.
|
|
297
|
+
|
|
298
|
+
`test_command` is required here — the adapter cannot guess the scheme, destination, or
|
|
299
|
+
derived-data path. It appends `-only-testing:` for targeted runs and `-enumerate-tests`
|
|
300
|
+
for collection, and changes nothing else. Simulator runs are serial in practice: give
|
|
301
|
+
them a `lease` so two projects don't drive the same simulator at once.
|
|
302
|
+
|
|
303
|
+
```toml
|
|
304
|
+
[project.native-ios]
|
|
305
|
+
root = "native-ios"
|
|
306
|
+
adapter = "xctest"
|
|
307
|
+
test_paths = ["AppTests/"]
|
|
308
|
+
test_command = """xcodebuild test \
|
|
309
|
+
-project App.xcodeproj \
|
|
310
|
+
-scheme AppTests \
|
|
311
|
+
-destination 'platform=iOS Simulator,name=App-Unit' \
|
|
312
|
+
-derivedDataPath /tmp/app-unit-dd"""
|
|
313
|
+
lease = "ios-simulator"
|
|
314
|
+
timeout = 900
|
|
315
|
+
```
|
|
316
|
+
|
|
289
317
|
Third-party adapters register under the
|
|
290
318
|
`tddcli.adapters` entry-point group:
|
|
291
319
|
|
|
@@ -347,14 +375,49 @@ and is never reclassified as a pin.
|
|
|
347
375
|
| `tdd metrics` | fidelity, attempts, violations, interventions |
|
|
348
376
|
| `tdd fleet [--json]` | all active runs across every worktree; read-only |
|
|
349
377
|
|
|
378
|
+
## Scoped baseline capture (R9.5c)
|
|
379
|
+
|
|
380
|
+
`run start` probes only the projects the plan can actually reach: the declared cycle projects
|
|
381
|
+
plus the transitive `consumed_by` closure of artifacts whose producer is in that set. Projects
|
|
382
|
+
outside the reachable set never run during the plan, so their baseline is never subtracted from
|
|
383
|
+
anything — probing them is pure overhead. A `baseline_scoped` integrity event records which
|
|
384
|
+
projects were skipped, so the scoping is auditable.
|
|
385
|
+
|
|
386
|
+
Pass `--baseline-all` to probe every project in `tdd.toml` regardless:
|
|
387
|
+
|
|
388
|
+
tdd run start --plan tasks/plan.md --baseline-all
|
|
389
|
+
|
|
390
|
+
Use this when a cycle may edit files outside the predicted reachable set and you want every
|
|
391
|
+
project baselined up front rather than hitting the `no_baseline_for_project` escape hatch later.
|
|
392
|
+
|
|
393
|
+
### When a sweep reaches an un-baselined project (R9.5d)
|
|
394
|
+
|
|
395
|
+
If an edit during a run touches a file owned by an artifact that was outside the predicted
|
|
396
|
+
reachable set, the close sweep may pull in a project that was never baselined. Its failures are
|
|
397
|
+
unattributable — no baseline exists to subtract — so the sweep replies `resolve_blocker` with
|
|
398
|
+
kind `no_baseline_for_project` rather than mislabelling them as regressions. Recovery:
|
|
399
|
+
|
|
400
|
+
tdd blocker --kind no_baseline_for_project --detail "svc pulled in unexpectedly"
|
|
401
|
+
tdd resume --unblock --accept-failures --note "folding svc sweep failures into baseline"
|
|
402
|
+
|
|
403
|
+
`--accept-failures` inserts a fresh baseline row for the un-baselined project (recording those
|
|
404
|
+
failures as pre-existing) and records a `baseline_amended` event. The next advance proceeds
|
|
405
|
+
with the project properly baselined.
|
|
406
|
+
|
|
350
407
|
## Running a long baseline
|
|
351
408
|
|
|
352
|
-
`run start` probes
|
|
409
|
+
`run start` probes the reachable project set (R9.5c) before a run exists, and on a real project
|
|
353
410
|
that can take minutes — well past an agent harness's default Bash timeout. If the command
|
|
354
411
|
appears to hang or time out, **do not re-run it**: the probe is still making progress in the
|
|
355
412
|
background, and a second `run start` against the same worktree is refused with
|
|
356
413
|
`reason: "baseline_in_progress"` — retrying on timeout just stacks refusals on top of a
|
|
357
|
-
baseline that was never stuck.
|
|
414
|
+
baseline that was never stuck.
|
|
415
|
+
|
|
416
|
+
`tdd advance` is similarly single-flight per worktree. A close sweep (artifact regeneration, full
|
|
417
|
+
suite, lint, typecheck) can run for several minutes. If an advance appears to hang, **do not
|
|
418
|
+
re-run it**: a second `advance` against the same worktree is refused with
|
|
419
|
+
`reason: "advance_in_flight"`. The refusal carries `pid`, `started_at`, and `elapsed_s` so the
|
|
420
|
+
agent can confirm the first process is still alive. Run `tdd status` to see the current run state. In order of preference:
|
|
358
421
|
|
|
359
422
|
1. **Background it.** Run `tdd run start` in the background if your harness supports it. The
|
|
360
423
|
heartbeat (`baseline_captured` / `project_completed` lines on stderr) lands in the task log
|
|
@@ -5,7 +5,7 @@ compose without translation:
|
|
|
5
5
|
|
|
6
6
|
BundleName/ClassName/testMethodName
|
|
7
7
|
|
|
8
|
-
e.g.
|
|
8
|
+
e.g. AppTests/PollCadenceTests/testPollIntervalScalesOnlyUnderE2E
|
|
9
9
|
|
|
10
10
|
A xcodebuild build failure maps to `not_collected` rather than `failed`.
|
|
11
11
|
Swift has no separate collection phase — a test referencing a missing symbol
|
|
@@ -27,12 +27,12 @@ Config (`tdd.toml`):
|
|
|
27
27
|
[project.native-ios]
|
|
28
28
|
root = "native-ios"
|
|
29
29
|
adapter = "xctest"
|
|
30
|
-
test_paths = ["
|
|
30
|
+
test_paths = ["AppTests/"] # Swift source directories
|
|
31
31
|
test_command = "xcodebuild test \\
|
|
32
|
-
-project
|
|
33
|
-
-scheme
|
|
34
|
-
-destination 'platform=iOS Simulator,name=
|
|
35
|
-
-derivedDataPath /tmp/
|
|
32
|
+
-project App.xcodeproj \\
|
|
33
|
+
-scheme AppTests \\
|
|
34
|
+
-destination 'platform=iOS Simulator,name=App-Unit' \\
|
|
35
|
+
-derivedDataPath /tmp/app-unit-dd"
|
|
36
36
|
|
|
37
37
|
`test_command` is required — the adapter cannot know the scheme, destination,
|
|
38
38
|
or derived-data path otherwise. The adapter appends `-only-testing:` for
|
|
@@ -60,14 +60,14 @@ from .base import (
|
|
|
60
60
|
# Patterns applied to xcodebuild stdout
|
|
61
61
|
# ---------------------------------------------------------------------------
|
|
62
62
|
|
|
63
|
-
# "Test Case '-[
|
|
63
|
+
# "Test Case '-[AppTests.PollCadenceTests testPollIntervalScalesOnlyUnderE2E]' passed (0.001 seconds)."
|
|
64
64
|
_CASE_RE = re.compile(r"Test Case '-\[(\w+)\.(\w+) (\w+)\]' (passed|failed)")
|
|
65
65
|
|
|
66
66
|
# End-of-build marker for a failed build (no tests ran)
|
|
67
67
|
_BUILD_FAILED_RE = re.compile(r"\*\* BUILD FAILED \*\*")
|
|
68
68
|
|
|
69
69
|
# Lines emitted by xcodebuild test -enumerate-tests (Xcode 16+):
|
|
70
|
-
#
|
|
70
|
+
# AppTests/PollCadenceTests/testPollIntervalScalesOnlyUnderE2E
|
|
71
71
|
_ENUMERATE_LINE_RE = re.compile(r"^[ \t]*(\w+)/(\w+)/(test\w+)[ \t]*$", re.MULTILINE)
|
|
72
72
|
|
|
73
73
|
# Swift source patterns for the per-file grep fallback
|
|
@@ -105,7 +105,7 @@ class XCTestAdapter(Adapter):
|
|
|
105
105
|
"""Best-effort: extract the test bundle name from `-scheme <Name>`.
|
|
106
106
|
|
|
107
107
|
xcodebuild schemes typically share a name with their test bundle
|
|
108
|
-
(e.g. `-scheme
|
|
108
|
+
(e.g. `-scheme AppTests` → bundle `AppTests`). When the
|
|
109
109
|
scheme cannot be parsed, fall back to `"Unknown"` — per-file ids will
|
|
110
110
|
still be formed, just unaddressable by `-only-testing:` without the
|
|
111
111
|
real bundle name.
|
|
@@ -284,9 +284,9 @@ class XCTestAdapter(Adapter):
|
|
|
284
284
|
"""Capture the assertion lines between 'started' and 'failed' for one test.
|
|
285
285
|
|
|
286
286
|
xcodebuild interleaves test output like:
|
|
287
|
-
Test Case '-[
|
|
287
|
+
Test Case '-[AppTests.PollCadenceTests testSomething]' started.
|
|
288
288
|
/path/test.swift:42: error: … XCTAssertEqual failed: …
|
|
289
|
-
Test Case '-[
|
|
289
|
+
Test Case '-[AppTests.PollCadenceTests testSomething]' failed (0.001 seconds).
|
|
290
290
|
"""
|
|
291
291
|
parts = native_id.split("/")
|
|
292
292
|
if len(parts) != 3:
|
|
@@ -310,6 +310,16 @@ def _handle_refactor(engine: Engine, cycle, retried: bool) -> Envelope:
|
|
|
310
310
|
outcome = engine.sweep(cycle, touched, skip_own=skip_own)
|
|
311
311
|
|
|
312
312
|
if not outcome.ok:
|
|
313
|
+
if outcome.unbaselined:
|
|
314
|
+
projects_list = ", ".join(sorted(outcome.unbaselined))
|
|
315
|
+
return _reply(
|
|
316
|
+
engine, cycle, Verb.RESOLVE_BLOCKER,
|
|
317
|
+
f"Close sweep found failures in un-baselined project(s): {projects_list}. "
|
|
318
|
+
"These are unattributable — no baseline exists to subtract. "
|
|
319
|
+
"File: tdd blocker --kind no_baseline_for_project --detail '...', "
|
|
320
|
+
"then resume --unblock --accept-failures to fold them into the baseline.",
|
|
321
|
+
unbaselined=outcome.unbaselined, commit=sha,
|
|
322
|
+
)
|
|
313
323
|
if outcome.failures:
|
|
314
324
|
return _reply(
|
|
315
325
|
engine, cycle, Verb.FIX_REGRESSION,
|
|
@@ -48,6 +48,9 @@ BLOCKER_KINDS = {
|
|
|
48
48
|
# Failing, but not caused by this run — a flake, or something the baseline missed.
|
|
49
49
|
# Distinct from `regression`, which records a defect the run introduced.
|
|
50
50
|
"pre_existing_failure",
|
|
51
|
+
# Failing in a project that was never baselined — not attributable as a regression
|
|
52
|
+
# (no baseline to subtract) vs. `pre_existing_failure` (baseline exists but missed it).
|
|
53
|
+
"no_baseline_for_project",
|
|
51
54
|
}
|
|
52
55
|
|
|
53
56
|
|
|
@@ -489,14 +492,13 @@ def cmd_plan_register(args) -> Envelope:
|
|
|
489
492
|
)
|
|
490
493
|
|
|
491
494
|
|
|
492
|
-
def _probe_projects(
|
|
493
|
-
"""Probe
|
|
494
|
-
`baseline_captured` heartbeat, and calling `on_progress(done, name)
|
|
495
|
-
|
|
496
|
-
updates inline past the point of legibility. Returns `{name: (verdict, collection)}`.
|
|
495
|
+
def _probe_projects(projects, worktree, ledger, on_progress):
|
|
496
|
+
"""Probe a mapping of projects' baselines (R9.5a): run + collect, timing each,
|
|
497
|
+
emitting a `baseline_captured` heartbeat, and calling `on_progress(done, name)`.
|
|
498
|
+
Returns `{name: (verdict, collection)}`.
|
|
497
499
|
"""
|
|
498
500
|
probes = {}
|
|
499
|
-
for done, (name, project) in enumerate(
|
|
501
|
+
for done, (name, project) in enumerate(projects.items(), start=1):
|
|
500
502
|
adapter = adapters.build(project, worktree)
|
|
501
503
|
started = time.monotonic()
|
|
502
504
|
verdict = adapter.run(None)
|
|
@@ -561,6 +563,15 @@ def cmd_run_start(args) -> Envelope:
|
|
|
561
563
|
if contract_row["status"] == "undeclared" and not args.allow_undeclared:
|
|
562
564
|
return failure("contract is undeclared; pass --allow-undeclared")
|
|
563
565
|
|
|
566
|
+
# R9.5c — scope baseline capture to plan-reachable projects unless opted out.
|
|
567
|
+
declared_cycles = contract_mod.cycles_from_json(contract_row["declared_cycles"])
|
|
568
|
+
if declared_cycles and not args.baseline_all:
|
|
569
|
+
declared_names = [p for c in declared_cycles for p in c.projects]
|
|
570
|
+
reachable_names = cfg.reachable_projects(declared_names)
|
|
571
|
+
probe_projects = {n: cfg.projects[n] for n in reachable_names if n in cfg.projects}
|
|
572
|
+
else:
|
|
573
|
+
probe_projects = cfg.projects
|
|
574
|
+
|
|
564
575
|
# Claim the worktree before probing: two `run start` calls against
|
|
565
576
|
# one worktree must not both pass the baseline window. `Ledger.claim`'s `UNIQUE`
|
|
566
577
|
# insert is the lock — do not read-then-write, which is the race this
|
|
@@ -576,7 +587,7 @@ def cmd_run_start(args) -> Envelope:
|
|
|
576
587
|
str(worktree),
|
|
577
588
|
hostname=socket.gethostname(),
|
|
578
589
|
pid=os.getpid(),
|
|
579
|
-
projects_total=len(
|
|
590
|
+
projects_total=len(probe_projects),
|
|
580
591
|
)
|
|
581
592
|
except sqlite3.IntegrityError:
|
|
582
593
|
return failure(
|
|
@@ -594,7 +605,7 @@ def cmd_run_start(args) -> Envelope:
|
|
|
594
605
|
# attempt — and must release the claim too, or the retry it invites is itself
|
|
595
606
|
# refused.
|
|
596
607
|
probes = _probe_projects(
|
|
597
|
-
|
|
608
|
+
probe_projects,
|
|
598
609
|
worktree,
|
|
599
610
|
ledger,
|
|
600
611
|
on_progress=lambda done, name: ledger.update_claim(
|
|
@@ -640,6 +651,9 @@ def cmd_run_start(args) -> Envelope:
|
|
|
640
651
|
run = ledger.one("SELECT * FROM run WHERE id = ?", (run_id,))
|
|
641
652
|
if blob_changed:
|
|
642
653
|
ledger.event(run_id, None, "plan_blob_changed", rel)
|
|
654
|
+
skipped = sorted(set(cfg.projects) - set(probe_projects))
|
|
655
|
+
if skipped:
|
|
656
|
+
ledger.event(run_id, None, "baseline_scoped", json.dumps(skipped))
|
|
643
657
|
|
|
644
658
|
# Baselines and the collection snapshot, per project (R9.5, R8.9) — from the
|
|
645
659
|
# probe above, so the suite is not run twice.
|
|
@@ -723,7 +737,31 @@ def cmd_advance(args) -> Envelope:
|
|
|
723
737
|
run={"id": run["id"], "phase": CLOSED},
|
|
724
738
|
next_action=NextAction(Verb.COMPLETE, "All cycles complete."),
|
|
725
739
|
)
|
|
726
|
-
|
|
740
|
+
existing_advance = ledger.active_advance_claim(str(worktree))
|
|
741
|
+
if existing_advance is not None and existing_advance["stale"]:
|
|
742
|
+
ledger.release_advance_claim(str(worktree))
|
|
743
|
+
|
|
744
|
+
try:
|
|
745
|
+
ledger.claim_advance(str(worktree), hostname=socket.gethostname(), pid=os.getpid())
|
|
746
|
+
except sqlite3.IntegrityError:
|
|
747
|
+
held = ledger.one("SELECT * FROM advance_claim WHERE worktree_path = ?", (str(worktree),))
|
|
748
|
+
started = datetime.fromisoformat(held["started_at"])
|
|
749
|
+
if started.tzinfo is None:
|
|
750
|
+
started = started.replace(tzinfo=timezone.utc)
|
|
751
|
+
elapsed_s = int((datetime.now(timezone.utc) - started).total_seconds())
|
|
752
|
+
return failure(
|
|
753
|
+
"an advance is already in flight for this worktree; do not re-run or kill it"
|
|
754
|
+
" — wait for its envelope, then run `tdd status` to see where the run stands",
|
|
755
|
+
reason="advance_in_flight",
|
|
756
|
+
pid=held["pid"],
|
|
757
|
+
started_at=held["started_at"],
|
|
758
|
+
elapsed_s=elapsed_s,
|
|
759
|
+
)
|
|
760
|
+
try:
|
|
761
|
+
result = do_advance(engine, cycle, retry=args.retry)
|
|
762
|
+
finally:
|
|
763
|
+
ledger.release_advance_claim(str(worktree))
|
|
764
|
+
return result
|
|
727
765
|
|
|
728
766
|
|
|
729
767
|
def cmd_cycle_skip(args) -> Envelope:
|
|
@@ -818,15 +856,27 @@ def _accept_failures_into_baseline(ledger: Ledger, run_id: int) -> dict[str, lis
|
|
|
818
856
|
}
|
|
819
857
|
accepted: dict[str, list[str]] = {}
|
|
820
858
|
for sweep in latest:
|
|
821
|
-
|
|
859
|
+
project = sweep["project"]
|
|
860
|
+
row = rows.get(project)
|
|
861
|
+
sweep_failed = sorted(json.loads(sweep["other_failures"]))
|
|
822
862
|
if row is None:
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
863
|
+
if not sweep_failed:
|
|
864
|
+
continue
|
|
865
|
+
ledger.insert(
|
|
866
|
+
"baseline",
|
|
867
|
+
run_id=run_id,
|
|
868
|
+
project=project,
|
|
869
|
+
failing=json.dumps(sweep_failed),
|
|
870
|
+
captured_at=now(),
|
|
871
|
+
)
|
|
872
|
+
accepted[project] = sweep_failed
|
|
873
|
+
else:
|
|
874
|
+
known = set(json.loads(row["failing"]))
|
|
875
|
+
new = sorted(set(sweep_failed) - known)
|
|
876
|
+
if not new:
|
|
877
|
+
continue
|
|
878
|
+
ledger.update("baseline", row["id"], failing=json.dumps(sorted(known | set(new))))
|
|
879
|
+
accepted[project] = new
|
|
830
880
|
if accepted:
|
|
831
881
|
ledger.event(run_id, None, "baseline_amended", json.dumps(accepted))
|
|
832
882
|
return accepted
|
|
@@ -1140,6 +1190,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1140
1190
|
s.add_argument("--executor", help="human-supplied label; agents must not use this")
|
|
1141
1191
|
s.add_argument("--allow-dirty", action="store_true")
|
|
1142
1192
|
s.add_argument("--allow-undeclared", action="store_true")
|
|
1193
|
+
s.add_argument("--baseline-all", action="store_true", help="probe all projects, skipping reachability scoping")
|
|
1143
1194
|
s.set_defaults(fn=cmd_run_start)
|
|
1144
1195
|
|
|
1145
1196
|
s = sub.add_parser("status")
|
|
@@ -212,6 +212,32 @@ class Config:
|
|
|
212
212
|
seen.add(up)
|
|
213
213
|
return chain
|
|
214
214
|
|
|
215
|
+
def _root_project(self, produced_by: str) -> str | None:
|
|
216
|
+
"""Resolve a produced_by value (project name or artifact.<name> chain) to a root project name."""
|
|
217
|
+
if produced_by.startswith("artifact."):
|
|
218
|
+
upstream = self.artifacts.get(produced_by.split(".", 1)[1])
|
|
219
|
+
if upstream is None:
|
|
220
|
+
return None
|
|
221
|
+
return self._root_project(upstream.produced_by)
|
|
222
|
+
return produced_by if produced_by in self.projects else None
|
|
223
|
+
|
|
224
|
+
def reachable_projects(self, declared: list[str]) -> list[str]:
|
|
225
|
+
declared_set = set(declared)
|
|
226
|
+
reachable = set(declared)
|
|
227
|
+
changed = True
|
|
228
|
+
while changed:
|
|
229
|
+
changed = False
|
|
230
|
+
for art in self.artifacts.values():
|
|
231
|
+
root = self._root_project(art.produced_by)
|
|
232
|
+
if root in reachable:
|
|
233
|
+
for consumer in art.consumed_by:
|
|
234
|
+
if consumer not in reachable:
|
|
235
|
+
proj = self.projects.get(consumer)
|
|
236
|
+
if consumer in declared_set or (proj and proj.in_close_sweep):
|
|
237
|
+
reachable.add(consumer)
|
|
238
|
+
changed = True
|
|
239
|
+
return sorted(reachable)
|
|
240
|
+
|
|
215
241
|
def close_sweep_projects(self, cycle_projects: list[str], touched: set[str]) -> list[str]:
|
|
216
242
|
"""R9.2 — the cycle's own projects, plus anything downstream of an artifact it touched."""
|
|
217
243
|
names = {n for n in cycle_projects}
|
|
@@ -221,14 +247,10 @@ class Config:
|
|
|
221
247
|
return [n for n in sorted(names) if self.projects[n].in_close_sweep]
|
|
222
248
|
|
|
223
249
|
def _artifact_touched(self, art: Artifact, touched: set[str]) -> bool:
|
|
224
|
-
|
|
225
|
-
if
|
|
226
|
-
upstream = self.artifacts.get(producer.split(".", 1)[1])
|
|
227
|
-
return bool(upstream and self._artifact_touched(upstream, touched))
|
|
228
|
-
proj = self.projects.get(producer)
|
|
229
|
-
if proj is None:
|
|
250
|
+
root = self._root_project(art.produced_by)
|
|
251
|
+
if root is None:
|
|
230
252
|
return False
|
|
231
|
-
return any(
|
|
253
|
+
return any(self.projects[root].owns(p) for p in touched)
|
|
232
254
|
|
|
233
255
|
def full_sweep_projects(self) -> list[str]:
|
|
234
256
|
return sorted(self.projects)
|
|
@@ -13,7 +13,7 @@ import sqlite3
|
|
|
13
13
|
from datetime import datetime, timedelta, timezone
|
|
14
14
|
from pathlib import Path
|
|
15
15
|
|
|
16
|
-
SCHEMA_VERSION =
|
|
16
|
+
SCHEMA_VERSION = 3
|
|
17
17
|
|
|
18
18
|
|
|
19
19
|
class LedgerVersionError(RuntimeError):
|
|
@@ -29,6 +29,8 @@ class LedgerVersionError(RuntimeError):
|
|
|
29
29
|
MIGRATIONS: dict[int, str] = {
|
|
30
30
|
# v1 -> v2 added the baseline_claim table; CREATE TABLE IF NOT EXISTS covers it.
|
|
31
31
|
1: "",
|
|
32
|
+
# v2 -> v3 added the advance_claim table; CREATE TABLE IF NOT EXISTS covers it.
|
|
33
|
+
2: "",
|
|
32
34
|
}
|
|
33
35
|
|
|
34
36
|
SCHEMA = """
|
|
@@ -45,6 +47,14 @@ CREATE TABLE IF NOT EXISTS baseline_claim (
|
|
|
45
47
|
started_at TEXT NOT NULL
|
|
46
48
|
);
|
|
47
49
|
|
|
50
|
+
CREATE TABLE IF NOT EXISTS advance_claim (
|
|
51
|
+
id INTEGER PRIMARY KEY,
|
|
52
|
+
worktree_path TEXT NOT NULL UNIQUE,
|
|
53
|
+
hostname TEXT NOT NULL,
|
|
54
|
+
pid INTEGER NOT NULL,
|
|
55
|
+
started_at TEXT NOT NULL
|
|
56
|
+
);
|
|
57
|
+
|
|
48
58
|
CREATE TABLE IF NOT EXISTS plan_contract (
|
|
49
59
|
id INTEGER PRIMARY KEY,
|
|
50
60
|
plan_path TEXT NOT NULL,
|
|
@@ -412,6 +422,54 @@ class Ledger:
|
|
|
412
422
|
)
|
|
413
423
|
self.db.commit()
|
|
414
424
|
|
|
425
|
+
def claim_advance(self, worktree: str, hostname: str, pid: int) -> int:
|
|
426
|
+
"""Insert the advance claim row. The insert is the lock: `worktree_path`
|
|
427
|
+
carries `UNIQUE`, so a second claim on the same worktree raises
|
|
428
|
+
`sqlite3.IntegrityError` rather than racing a read-then-write check."""
|
|
429
|
+
return self.insert(
|
|
430
|
+
"advance_claim",
|
|
431
|
+
worktree_path=worktree,
|
|
432
|
+
hostname=hostname,
|
|
433
|
+
pid=pid,
|
|
434
|
+
started_at=now(),
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
def release_advance_claim(self, worktree: str) -> None:
|
|
438
|
+
self.db.execute("DELETE FROM advance_claim WHERE worktree_path = ?", (worktree,))
|
|
439
|
+
self.db.commit()
|
|
440
|
+
|
|
441
|
+
@staticmethod
|
|
442
|
+
def _claim_is_stale(hostname: str, pid: int, started_at: str) -> bool:
|
|
443
|
+
"""Staleness rule shared by all claim kinds.
|
|
444
|
+
|
|
445
|
+
Same host → liveness via os.kill(pid, 0). Cross-host → age > 60 min
|
|
446
|
+
(a pid is meaningless from another host; reused pids would make a
|
|
447
|
+
liveness check actively wrong, so we fall back to age).
|
|
448
|
+
"""
|
|
449
|
+
if hostname == socket.gethostname():
|
|
450
|
+
try:
|
|
451
|
+
os.kill(pid, 0)
|
|
452
|
+
return False
|
|
453
|
+
except ProcessLookupError:
|
|
454
|
+
# Same host, pid no longer running.
|
|
455
|
+
return True
|
|
456
|
+
except PermissionError:
|
|
457
|
+
# Pid exists but owned by someone else — alive.
|
|
458
|
+
return False
|
|
459
|
+
started = datetime.fromisoformat(started_at)
|
|
460
|
+
if started.tzinfo is None:
|
|
461
|
+
started = started.replace(tzinfo=timezone.utc)
|
|
462
|
+
return datetime.now(timezone.utc) - started > timedelta(minutes=60)
|
|
463
|
+
|
|
464
|
+
def active_advance_claim(self, worktree: str) -> dict | None:
|
|
465
|
+
"""Read-only observer — staleness computed but no row deleted."""
|
|
466
|
+
row = self.one("SELECT * FROM advance_claim WHERE worktree_path = ?", (worktree,))
|
|
467
|
+
if row is None:
|
|
468
|
+
return None
|
|
469
|
+
claim = dict(row)
|
|
470
|
+
claim["stale"] = self._claim_is_stale(claim["hostname"], claim["pid"], claim["started_at"])
|
|
471
|
+
return claim
|
|
472
|
+
|
|
415
473
|
def active_claim(self, worktree: str) -> dict | None:
|
|
416
474
|
"""Read-only, per the store's append-only contract — `cmd_progress` and
|
|
417
475
|
`cmd_status` call this as pure observers. Only `cmd_run_start` acts on the
|
|
@@ -420,23 +478,8 @@ class Ledger:
|
|
|
420
478
|
if row is None:
|
|
421
479
|
return None
|
|
422
480
|
claim = dict(row)
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
except ProcessLookupError:
|
|
428
|
-
# Same host, pid no longer running (e.g. a `SIGKILL`ed `run start`).
|
|
429
|
-
stale = True
|
|
430
|
-
except PermissionError:
|
|
431
|
-
# Pid exists but is owned by someone else — alive.
|
|
432
|
-
stale = False
|
|
433
|
-
else:
|
|
434
|
-
# A pid is meaningless from another host, and reused pids would make a
|
|
435
|
-
# host-crossing liveness check actively wrong. Fall back to age: a false
|
|
436
|
-
# "alive" bricks the worktree, a false "dead" reopens the bug (Decisions).
|
|
437
|
-
started = datetime.fromisoformat(claim["started_at"])
|
|
438
|
-
if started.tzinfo is None:
|
|
439
|
-
started = started.replace(tzinfo=timezone.utc)
|
|
440
|
-
stale = datetime.now(timezone.utc) - started > timedelta(minutes=60)
|
|
441
|
-
claim["stale"] = stale
|
|
481
|
+
# A pid is meaningless from another host, and reused pids would make a
|
|
482
|
+
# host-crossing liveness check actively wrong. Fall back to age: a false
|
|
483
|
+
# "alive" bricks the worktree, a false "dead" reopens the bug (Decisions).
|
|
484
|
+
claim["stale"] = self._claim_is_stale(claim["hostname"], claim["pid"], claim["started_at"])
|
|
442
485
|
return claim
|