@hecer/yoke 1.2.1 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +66 -24
  4. package/README.md +186 -104
  5. package/TODOS.md +0 -3
  6. package/canon/loop/loop-spec.md +53 -24
  7. package/canon/loop/prd.schema.md +30 -6
  8. package/canon/manifest.yaml +1 -1
  9. package/canon/skills/authoring-prd/SKILL.md +30 -31
  10. package/dist/agents/contracts.js +50 -0
  11. package/dist/agents/process-incarnation.js +15 -0
  12. package/dist/agents/process-record.js +65 -0
  13. package/dist/agents/process-streams.js +40 -0
  14. package/dist/agents/process.js +177 -0
  15. package/dist/agents/providers.js +10 -7
  16. package/dist/agents/telemetry.js +62 -0
  17. package/dist/change/inbox.js +279 -0
  18. package/dist/cli.js +90 -4
  19. package/dist/loop/candidate-boundaries.js +43 -0
  20. package/dist/loop/candidate-cleanup.js +98 -0
  21. package/dist/loop/candidate-contracts.js +1 -0
  22. package/dist/loop/candidate-selection.js +84 -0
  23. package/dist/loop/candidates.js +228 -0
  24. package/dist/loop/claim-lease.js +131 -0
  25. package/dist/loop/claims.js +177 -40
  26. package/dist/loop/cleanup.js +117 -15
  27. package/dist/loop/decision.js +31 -0
  28. package/dist/loop/dispatcher.js +334 -0
  29. package/dist/loop/evidence.js +31 -0
  30. package/dist/loop/gates.js +10 -1
  31. package/dist/loop/loop.js +236 -33
  32. package/dist/loop/merge-queue.js +12 -6
  33. package/dist/loop/parallel-adapters.js +185 -0
  34. package/dist/loop/parallel-command.js +287 -0
  35. package/dist/loop/parallel.js +2 -4
  36. package/dist/loop/prd.js +63 -2
  37. package/dist/loop/reporter.js +86 -5
  38. package/dist/loop/run-command.js +227 -53
  39. package/dist/loop/runner.js +77 -34
  40. package/dist/loop/verify.js +11 -0
  41. package/dist/loop/watchdog.js +67 -8
  42. package/dist/loop/worker-cancellation.js +17 -0
  43. package/dist/loop/worker-cleanup.js +23 -0
  44. package/dist/loop/worker-contracts.js +1 -0
  45. package/dist/loop/worker.js +254 -0
  46. package/dist/prd/command.js +18 -5
  47. package/dist/quality/artifacts.js +59 -0
  48. package/dist/quality/candidate-comparison.js +130 -0
  49. package/dist/quality/command.js +316 -0
  50. package/dist/quality/loop.js +86 -0
  51. package/dist/quality/process-command.js +57 -0
  52. package/dist/quality/reference.js +187 -0
  53. package/dist/quality/repair.js +11 -0
  54. package/dist/quality/runner.js +66 -0
  55. package/dist/quality/types.js +60 -0
  56. package/dist/quality/verdict.js +142 -0
  57. package/dist/retrofit/config.js +13 -4
  58. package/dist/retrofit/gitignore.js +4 -0
  59. package/dist/review/command.js +27 -38
  60. package/dist/review/verdict.js +38 -7
  61. package/dist/routing/router.js +4 -1
  62. package/docs/MIGRATING-TO-1.4.md +70 -0
  63. package/docs/PUBLISHING.md +16 -2
  64. package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -0
  65. package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -0
  66. package/gemini-extension.json +1 -1
  67. package/package.json +6 -6
@@ -1,36 +1,65 @@
1
1
  # Loop Specification (Ralph + GSD)
2
2
 
3
- The autonomous loop is OPTIONAL and toggle-able:
3
+ The autonomous loop is optional and toggle-able:
4
4
 
5
- - `yoke loop on` / `yoke loop off` — enable/disable (recorded in `.yoke/config.yaml`, default off).
6
- - `yoke loop status` — show enabled state + PRD progress.
7
- - `yoke loop run [--max=N] [--isolate] [--decision-policy=auto|critical]` — run the loop (default cap 25 iterations).
8
- - `yoke loop decision` / `yoke loop answer --choice=<id>` — inspect and answer a structured critical stop; answering records a human-owned, decision-file-only commit and resumes by default with the original runner, isolation, review, permission, timeout, and policy settings. `yoke loop resume` retries a restart that could not begin without weakening those settings.
5
+ - `yoke loop on` / `yoke loop off` — enable or disable it in `.yoke/config.yaml`.
6
+ - `yoke loop status` — show enabled state and backlog progress.
7
+ - `yoke loop run [--max=N] [--parallel=N] [--isolate] [--decision-policy=auto|critical] [--quality|--no-quality] [--quality-rounds=N] [--quality-minutes=N] [--quality-policy=blocking|advisory] [--quality-unbounded] [--candidates=N]` — run until the current backlog is green or a gate blocks.
8
+ - `yoke change add --idea="..."` — queue a product change at any time, including while the loop is running.
9
+ - `yoke loop decision` / `yoke loop answer --choice=<id>` — inspect and answer a structured critical stop.
9
10
 
10
- Pass `--isolate` to run each iteration in a fresh git worktree: the agent works on a throwaway checkout, and only a verified, committed story is fast-forwarded back into the main tree. A failed iteration never touches your working tree. Requires `.yoke/prd.yaml` to be committed to git, since the worktree is a checkout of HEAD.
11
+ Pass `--isolate` to implement each story in a fresh git worktree. Only a verified, committed
12
+ story is fast-forwarded to the main tree. Pass `--review` or `--reviewer=<provider>` to require
13
+ a separate, schema-validated review. Pass `--parallel=N` to dispatch ready, non-colliding stories
14
+ concurrently. Pass `--json` for NDJSON status on stdout.
11
15
 
12
- Pass `--review` (or `--reviewer=<claude|codex|gemini>` for a different agent) to add a role-separated review step: after the tests pass, an independent reviewer agent must approve the change before the story is committed and marked done. A rejection blocks the story (no commit). The reviewer is a fresh agent pass — the implementer never reviews its own work.
16
+ Stories may declare a reference, candidate artifact, rubric, and blocking/advisory quality policy.
17
+ `--quality` runs a read-only blind critic plus bounded repair before review; every repair reruns the
18
+ mechanical gates. `--candidates=N` requires quality declarations and dispatches multiple isolated
19
+ implementations, rejects mechanically red candidates, selects one green candidate through opaque
20
+ pairwise handles, and retains every candidate's terminal proof before cleanup. Parallel/candidate
21
+ runs do not combine with adaptive routing.
13
22
 
14
- Pass `--json` for machine mode: each status transition is emitted as one NDJSON line on stdout (the `.yoke/loop-status.json` shape, tagged `"type":"status"`) instead of the human narrative, so a supervisor can consume the stream instead of polling the file.
23
+ At every story boundary, Yoke consumes at most one queued change. The configured Claude,
24
+ Codex, or Gemini provider may propose only new stories in a separate runtime file. A fresh
25
+ coverage-review pass must account for every distinct requested outcome before Yoke validates
26
+ strict criterion evidence, appends the stories itself, commits only the PRD, and leaves existing
27
+ stories untouched. The request stays pending on any failure or uncovered outcome.
15
28
 
16
- When enabled and run, each iteration:
29
+ For each story:
17
30
 
18
- 1. Pre-dispatch gate: the git worktree must be clean, else `blocked`.
19
- 2. Pick the highest-priority unfinished PRD story (`.yoke/prd.yaml`).
20
- 3. Stop-the-Line gate: the story must have acceptance criteria, else `blocked`.
21
- 4. Run a fresh agent to implement ONE story. Runner precedence is explicit `--runner`, configured `runner.agent`, active agent host, then the first configured agent. The loop refuses to start if that CLI is not installed.
22
- With `decisionPolicy: auto`, routine ambiguity is resolved from the approved plan and project conventions. With `critical`, only high-impact architecture, security/privacy, destructive data, material-cost, compliance, or irreversible choices may produce `.yoke/decision-request.yaml`; the loop validates its bounded single-line fields, unique options, and active story ID, blocks before verify, and preserves it for `yoke loop answer`.
23
- 5. Run the project's verify command (config `verify.command`, or detected `npm test`).
24
- **Verify is the source of truth** — the agent's exit code is advisory, so a spurious
25
- non-zero exit (e.g. a Windows `.cmd` wrapper) cannot block a story whose tests are green.
26
- A failing verify is retried up to `verify.retries` times (default 1) so a transient flake
27
- self-heals; a real failure still fails. Only if verify passes is the story marked
28
- `passes: true`, committed atomically, and a decision logged. If verify fails: `blocked`.
29
- 6. Stop when all stories `passes: true` (`complete`), or the iteration cap is reached (`cap-reached`).
31
+ 1. Require a clean git worktree.
32
+ 2. Pick the highest-priority ready unfinished story, or claim multiple dependency-ready stories
33
+ whose collision areas do not overlap when parallel dispatch is enabled.
34
+ 3. Stop the line if acceptance is empty. With `verify.requireCriteria: true`, every criterion
35
+ must be structured. Every structured criterion, including in compatible legacy projects,
36
+ must use a single approved test command containing its criterion ID and no shell operators.
37
+ 4. Run a fresh configured provider to implement exactly one story. Under the `critical`
38
+ decision policy, only high-impact architecture, security/privacy, destructive data,
39
+ material cost, compliance, or irreversible choices may pause for a human decision.
40
+ 5. Run every structured criterion's targeted commands and write
41
+ `.yoke/proof/<story>/evidence.json`. Then run project-wide `verify.command` (or detected
42
+ `npm test`). Performance and audit follow when configured. If quality is enabled, collect the
43
+ declared artifact, run the blind critic, and repair within the configured round/time bounds.
44
+ Independent review follows. Any blocking failure stops the candidate.
45
+ 6. Parallel workers enqueue green candidate commits. The integrator applies one at a time and
46
+ reruns mechanical gates, fresh quality, and review against the integrated tree. Failed
47
+ integration reopens the story and retains proof.
48
+ 7. Only after all gates pass, mark the story `passes: true`, log the decision, and commit
49
+ atomically. A failed commit restores the PRD state.
50
+ 8. When all current stories pass, run optional `completion.command` against the integrated
51
+ system. Only a green result reports `complete`; otherwise the loop blocks. This readiness
52
+ result is ephemeral, not a release and not a freeze on future changes.
30
53
 
31
- A supervisor can pause the loop by creating `.yoke/loop.pause`: at the next story boundary (before the next story is selected — the running story always finishes) the loop consumes the file, records `paused` in the status file and log, and exits with code `3`. Running `yoke loop run` again resumes.
54
+ A supervisor can pause the loop by creating `.yoke/loop.pause`. The running story finishes;
55
+ the dispatcher latches the signal, stops launching new workers, lets active workers reach safe
56
+ terminal proof/cleanup, and exits with code `3` before another story is integrated.
32
57
 
33
- State lives outside the model context: the PRD file + git. The agent runner is pluggable. The PRD is re-read from disk at every story boundary, so stories appended to `.yoke/prd.yaml` mid-run are picked up at the next iteration without a restart.
58
+ State lives outside model context: PRD, git, and the ignored `.yoke/changes/` inbox. All are
59
+ re-read at story boundaries, so a request queued mid-run becomes additional stories without a
60
+ restart.
34
61
 
35
62
  ## Limitations
36
- - The loop verifies via the project's test command and an optional agent review; it has no formal merge-queue or multi-reviewer quorum.
63
+
64
+ Yoke cannot infer the correct end-to-end journey command. Projects that need integrated
65
+ readiness must configure `completion.command`, for example a Playwright journey suite.
@@ -1,6 +1,6 @@
1
1
  # PRD Schema
2
2
 
3
- The loop is driven by a versioned PRD file. Each story:
3
+ The loop is driven by a continuous PRD backlog. Each story:
4
4
 
5
5
  ```yaml
6
6
  - id: STORY-1
@@ -9,11 +9,35 @@ The loop is driven by a versioned PRD file. Each story:
9
9
  needs: [] # optional dependency IDs; no unknown IDs, self-links, or cycles
10
10
  area: api # optional collision domain for parallel scheduling
11
11
  agent: codex # optional claude|codex|gemini affinity
12
- acceptance: # Definition of Done (required before implementation)
13
- - The endpoint returns 200 for a valid request.
14
- passes: false # set true only when acceptance is met and tests are green
12
+ acceptance:
13
+ - id: valid-request-returns-200
14
+ text: The endpoint returns 200 for a valid request.
15
+ verify: [npm run test:valid-request-returns-200]
16
+ - id: invalid-request-returns-400
17
+ text: The endpoint returns 400 for an invalid request.
18
+ verify: [npm run test:invalid-request-returns-400]
19
+ passes: false # Yoke-owned; true only after all gates pass and the commit lands
15
20
  ```
16
21
 
17
- Stop condition: every story has `passes: true`.
22
+ Each new story has 2–5 acceptance criteria. Every criterion has a stable `id`, observable
23
+ behavioral `text`, and one or more executable `verify` commands. Each entry is one approved test
24
+ command, contains the normalized criterion ID, and contains no shell control operator. Yoke runs and records each criterion separately in
25
+ `.yoke/proof/<story>/evidence.json`; a broad green suite cannot stand in for an untested
26
+ criterion. Legacy string criteria remain readable, but `verify.requireCriteria: true` blocks
27
+ them.
18
28
 
19
- Stories without `needs`, `area`, or `agent` retain the serial pre-1.0 behavior. A story is ready only when every ID in `needs` passes. The scheduler orders ready work by priority, avoids simultaneously active areas, and uses `agent` as an affinity hint.
29
+ The binding is deliberately mechanical, not an oracle for product meaning: Yoke can require a
30
+ criterion-targeted test command and a separate coverage review, but it cannot prove that arbitrary
31
+ test code faithfully models the real world. Critical cross-component behavior therefore also
32
+ belongs in a trusted `completion.command` journey suite.
33
+
34
+ `sourceChange` is an optional Yoke-owned request ID. The change inbox uses it to append new
35
+ stories idempotently; authors normally omit it.
36
+
37
+ Stories without `needs`, `area`, or `agent` retain serial behavior. A story is ready only when
38
+ every ID in `needs` passes. The scheduler orders ready work by priority, avoids simultaneously
39
+ active areas, and uses `agent` as an affinity hint.
40
+
41
+ The backlog is continuous, not a release object. A momentary stop condition is every story
42
+ having `passes: true`; if configured, `completion.command` must then prove the integrated
43
+ system before the loop reports `complete`.
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.1.0
2
+ version: 1.4.0
3
3
  agents: [claude, codex, gemini]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology }
@@ -1,38 +1,27 @@
1
1
  ---
2
2
  name: authoring-prd
3
- description: Use when turning a product idea into a loop-ready .yoke/prd.yaml — slice the idea into small, independently shippable stories with testable behavioral acceptance criteria; greenfield STORY-1 scaffolds the project and wires verify.command.
3
+ description: Use when turning a product idea or change into a loop-ready continuous backlog with small stories and executable behavioral evidence.
4
4
  ---
5
5
 
6
6
  # Authoring a PRD
7
7
 
8
- The Yoke loop is only as good as its stories. Bad stories ("build the app") stall it;
9
- good stories (small, testable, ordered) let it run overnight.
8
+ The Yoke loop is only as good as its stories. Keep the backlog continuous: new requests become
9
+ new stories; they do not require a release object.
10
10
 
11
11
  ## Story rules
12
12
 
13
- 1. **One iteration per story.** If you can't imagine an agent finishing it in one sitting,
14
- split it. Prefer 5-12 stories over 3 epics.
15
- 2. **Independently shippable.** After any story, the project builds and tests pass.
16
- 3. **Acceptance = observable behavior**, never implementation:
17
- - Good: "GET /health returns 200", "the CLI prints the sum of its arguments"
18
- - Bad: "create a HealthController class", "use express"
19
- 2-5 criteria per story. Each must be checkable by a test or a command.
20
- 4. **Dense priorities from 1**; lower runs first. Order by dependency, then by risk.
21
- 5. **Greenfield: STORY-1 scaffolds.** Project skeleton + runnable test suite + a criterion
22
- that the verify command (`verify.command` in `.yoke/config.yaml`) exits 0. Every later
23
- story stands on a green pipeline.
24
- 6. **Performance requirements are acceptance criteria — with numbers.** "Should be fast" is
25
- a vibe the loop cannot gate; "imports 1M rows in < 2s (asserted by the bench test)" is a
26
- criterion. If the whole project has a budget, wire `perf.command` in `.yoke/config.yaml`
27
- (see the `performance` skill) instead of repeating it per story.
28
- 7. **Ask everything now.** Clarifying questions belong in this planning round — a loop run
29
- is unattended. A criterion that still contains `TBD` or another placeholder is not
30
- loop-ready, and `yoke prd check` rejects it. During implementation,
31
- `loop.decisionPolicy: auto` resolves routine ambiguity; `critical` pauses only for
32
- high-impact choices and resumes after `yoke loop answer` records the answer.
33
- 8. **Model real dependencies.** Add `needs` only for hard prerequisites, `area` for files or
34
- subsystems that must not be edited concurrently, and `agent` only as an affinity hint.
35
- Dependency IDs must exist; self-dependencies and cycles are invalid.
13
+ 1. One iteration per story. Prefer 5–12 small stories over a few epics.
14
+ 2. Each story leaves the project buildable and testable.
15
+ 3. Acceptance describes observable behavior, never implementation. Give each of 2–5 criteria
16
+ a stable `id`, behavioral `text`, and `verify` list with one or more real commands proving
17
+ that exact outcome.
18
+ 4. Use dense priorities from 1; order by dependency, then risk.
19
+ 5. Greenfield `STORY-1` creates the skeleton, runnable suite, and `verify.command`.
20
+ 6. Express performance with numbers and executable benchmarks, not words such as “fast”.
21
+ 7. Resolve planning questions before unattended execution. `yoke prd check` rejects unresolved
22
+ placeholders; critical irreversible choices use the structured decision channel.
23
+ 8. Use `needs` only for hard prerequisites, `area` for collision domains, and `agent` only as
24
+ a Claude/Codex/Gemini affinity hint.
36
25
 
37
26
  ## Format (`.yoke/prd.yaml`)
38
27
 
@@ -41,8 +30,12 @@ good stories (small, testable, ordered) let it run overnight.
41
30
  title: scaffold a TypeScript CLI with vitest
42
31
  priority: 1
43
32
  acceptance:
44
- - "npm test exits 0 with at least one passing test"
45
- - "verify.command is set in .yoke/config.yaml"
33
+ - id: cli-help-runs
34
+ text: the CLI help command exits 0 and prints usage
35
+ verify: [npm run test:cli-help-runs]
36
+ - id: test-runner-starts
37
+ text: the project test runner starts and reports at least one passing test
38
+ verify: [npm run test:test-runner-starts]
46
39
  passes: false
47
40
  - id: STORY-2
48
41
  title: add the sum command
@@ -51,9 +44,15 @@ good stories (small, testable, ordered) let it run overnight.
51
44
  area: cli
52
45
  agent: codex
53
46
  acceptance:
54
- - "cli sum 1 2 prints 3"
55
- - "non-numeric input exits 1 with an error message"
47
+ - id: sum-valid
48
+ text: cli sum 1 2 prints 3
49
+ verify: [npm run test:sum-valid]
50
+ - id: sum-invalid
51
+ text: non-numeric input exits 1 with an error message
52
+ verify: [npm run test:sum-invalid]
56
53
  passes: false
57
54
  ```
58
55
 
59
- `passes` is owned by the loop — always start `false`. Validate with `yoke prd check`.
56
+ Every story has 2–5 structured criteria. Each `verify` entry is one approved test command whose
57
+ normalized text contains its criterion ID; never use shell operators or a broad unrelated suite.
58
+ `passes` is owned by the loop and always starts false. Validate with `yoke prd check`.
@@ -0,0 +1,50 @@
1
+ import { z } from 'zod';
2
+ export const AgentSchema = z.enum(['claude', 'codex', 'gemini']);
3
+ export const PermissionProfileSchema = z.enum(['safe', 'unsafe', 'read-only']);
4
+ export const ModelSelectionSchema = z.object({
5
+ model: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}$/).optional(),
6
+ reasoningEffort: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$/).optional(),
7
+ nativeMultiAgent: z.boolean().optional(),
8
+ bare: z.boolean().optional(),
9
+ });
10
+ export const AgentInvocationSchema = z.object({
11
+ command: z.string().min(1),
12
+ args: z.array(z.string()),
13
+ input: z.string(),
14
+ cwd: z.string().min(1),
15
+ });
16
+ const ProviderTokenUsageSchema = z.object({
17
+ inputTokens: z.number().nonnegative(),
18
+ cachedInputTokens: z.number().nonnegative().optional(),
19
+ cacheWriteInputTokens: z.number().nonnegative().optional(),
20
+ outputTokens: z.number().nonnegative(),
21
+ reasoningOutputTokens: z.number().nonnegative().optional(),
22
+ totalCostUsd: z.number().nonnegative().optional(),
23
+ model: z.string().min(1).optional(),
24
+ });
25
+ export const ProviderTelemetrySchema = z.object({
26
+ usageAvailable: z.boolean(),
27
+ tokens: ProviderTokenUsageSchema.optional(),
28
+ }).superRefine((telemetry, ctx) => {
29
+ if (telemetry.usageAvailable && !telemetry.tokens) {
30
+ ctx.addIssue({ code: 'custom', path: ['tokens'], message: 'usageAvailable telemetry requires token totals' });
31
+ }
32
+ });
33
+ const MachineRoleSchema = z.enum([
34
+ 'route',
35
+ 'review',
36
+ 'quality',
37
+ 'decomposition',
38
+ 'candidate-selection',
39
+ 'telemetry',
40
+ ]);
41
+ export const MachineEnvelopeSchema = z.object({
42
+ schemaVersion: z.literal(1),
43
+ provider: AgentSchema,
44
+ model: z.string().min(1).optional(),
45
+ role: MachineRoleSchema,
46
+ durationMs: z.number().int().nonnegative(),
47
+ permissions: PermissionProfileSchema,
48
+ usage: ProviderTokenUsageSchema.optional(),
49
+ raw: z.record(z.unknown()).optional(),
50
+ });
@@ -0,0 +1,15 @@
1
+ import { execFileSync } from 'node:child_process';
2
+ const queryProcessIdentity = (command, args, options) => execFileSync(command, args, { stdio: 'pipe', ...options }).toString();
3
+ export function processIncarnation(pid, platform = process.platform, query = queryProcessIdentity) {
4
+ try {
5
+ if (platform === 'win32') {
6
+ const output = query('powershell.exe', ['-NoProfile', '-NonInteractive', '-Command', '(Get-CimInstance Win32_Process -Filter "ProcessId = $env:YOKE_PROCESS_PID").CreationDate'], { env: { ...process.env, YOKE_PROCESS_PID: String(pid) } }).trim();
7
+ return output ? `win32:${output}` : undefined;
8
+ }
9
+ const output = query('ps', ['-o', 'lstart=', '-p', String(pid)]).trim();
10
+ return output ? `posix:${output}` : undefined;
11
+ }
12
+ catch {
13
+ return undefined;
14
+ }
15
+ }
@@ -0,0 +1,65 @@
1
+ import { randomUUID } from 'node:crypto';
2
+ import { linkSync, mkdirSync, rmSync, writeFileSync } from 'node:fs';
3
+ import { join } from 'node:path';
4
+ export function createProviderProcessRecord(targetDir, childPid, workerId, startedAt = new Date().toISOString()) {
5
+ const worker = workerId?.replace(/[^A-Za-z0-9_-]/gu, '_') || 'call';
6
+ return {
7
+ path: join(targetDir, '.yoke', 'provider-processes', `${worker}-${randomUUID()}.json`),
8
+ version: 1,
9
+ owner: 'provider-process',
10
+ targetDir,
11
+ childPid,
12
+ startedAt,
13
+ ...(workerId ? { workerId } : {}),
14
+ };
15
+ }
16
+ export const filesystemProviderProcessRecordAdapter = {
17
+ publish(record) {
18
+ const directory = join(record.targetDir, '.yoke', 'provider-processes');
19
+ mkdirSync(directory, { recursive: true });
20
+ const temporary = `${record.path}.${randomUUID()}.tmp`;
21
+ writeFileSync(temporary, JSON.stringify({
22
+ version: record.version,
23
+ owner: record.owner,
24
+ targetDir: record.targetDir,
25
+ childPid: record.childPid,
26
+ startedAt: record.startedAt,
27
+ ...(record.workerId ? { workerId: record.workerId } : {}),
28
+ }), { flag: 'wx' });
29
+ try {
30
+ linkSync(temporary, record.path);
31
+ }
32
+ finally {
33
+ rmSync(temporary, { force: true });
34
+ }
35
+ },
36
+ remove(path) {
37
+ rmSync(path, { force: true });
38
+ },
39
+ };
40
+ function isRecord(value) {
41
+ return typeof value === 'object' && value !== null && !Array.isArray(value);
42
+ }
43
+ export function parseProviderProcessRecord(value) {
44
+ if (!isRecord(value))
45
+ return null;
46
+ const workerId = value.workerId;
47
+ if (value.version !== 1 ||
48
+ value.owner !== 'provider-process' ||
49
+ typeof value.targetDir !== 'string' ||
50
+ typeof value.childPid !== 'number' ||
51
+ !Number.isInteger(value.childPid) ||
52
+ value.childPid <= 0 ||
53
+ typeof value.startedAt !== 'string' ||
54
+ value.startedAt.length === 0 ||
55
+ (workerId !== undefined && typeof workerId !== 'string'))
56
+ return null;
57
+ return {
58
+ version: 1,
59
+ owner: 'provider-process',
60
+ targetDir: value.targetDir,
61
+ childPid: value.childPid,
62
+ startedAt: value.startedAt,
63
+ ...(typeof workerId === 'string' ? { workerId } : {}),
64
+ };
65
+ }
@@ -0,0 +1,40 @@
1
+ import { parseProviderTelemetry } from './telemetry.js';
2
+ export function createBoundedOutput(limitBytes) {
3
+ let text = '';
4
+ let truncated = false;
5
+ return {
6
+ append(next) {
7
+ const combined = `${text}${next}`;
8
+ if (Buffer.byteLength(combined) <= limitBytes) {
9
+ text = combined;
10
+ return;
11
+ }
12
+ text = Buffer.from(combined).subarray(-limitBytes).toString('utf8');
13
+ truncated = true;
14
+ },
15
+ get text() { return text; },
16
+ get truncated() { return truncated; },
17
+ };
18
+ }
19
+ export function createTelemetryAccumulator(agent) {
20
+ let trailing = '';
21
+ let telemetry = { usageAvailable: false };
22
+ const update = (lines) => {
23
+ const next = parseProviderTelemetry(agent, [...lines]);
24
+ if (next.usageAvailable)
25
+ telemetry = next;
26
+ };
27
+ return {
28
+ append(text) {
29
+ const parts = `${trailing}${text}`.split(/\r?\n/u);
30
+ trailing = parts.pop() ?? '';
31
+ update(parts);
32
+ },
33
+ finish() {
34
+ if (trailing)
35
+ update([trailing]);
36
+ trailing = '';
37
+ return telemetry;
38
+ },
39
+ };
40
+ }
@@ -0,0 +1,177 @@
1
+ import { spawn } from 'node:child_process';
2
+ import { resolve } from 'node:path';
3
+ import { killProcessTreeForCleanup } from '../loop/watchdog.js';
4
+ import { createProviderProcessRecord, filesystemProviderProcessRecordAdapter, } from './process-record.js';
5
+ import { createBoundedOutput, createTelemetryAccumulator } from './process-streams.js';
6
+ import { processIncarnation } from './process-incarnation.js';
7
+ function cancellationReason(signal) {
8
+ return typeof signal.reason === 'string' && signal.reason.length > 0
9
+ ? signal.reason
10
+ : 'provider process cancellation requested';
11
+ }
12
+ export function providerSpawnOptions(invocation, platform = process.platform) {
13
+ const windowsCommandShim = !/[\\/]/u.test(invocation.command) || /\.(?:bat|cmd)$/iu.test(invocation.command);
14
+ return {
15
+ command: invocation.command,
16
+ args: invocation.args,
17
+ cwd: invocation.cwd,
18
+ shell: platform === 'win32' && windowsCommandShim,
19
+ detached: platform !== 'win32',
20
+ };
21
+ }
22
+ export function startProviderProcess(agent, invocation, options = {}) {
23
+ const spawnOptions = providerSpawnOptions(invocation);
24
+ const child = spawn(spawnOptions.command, [...spawnOptions.args], {
25
+ cwd: spawnOptions.cwd,
26
+ shell: spawnOptions.shell,
27
+ stdio: ['pipe', 'pipe', 'pipe'],
28
+ detached: spawnOptions.detached,
29
+ });
30
+ const targetDir = resolve(invocation.cwd);
31
+ const pid = child.pid;
32
+ const startedAt = pid === undefined ? `unverified:${new Date().toISOString()}` : processIncarnation(pid) ?? `unverified:${new Date().toISOString()}`;
33
+ const record = createProviderProcessRecord(targetDir, pid ?? 0, options.workerId, startedAt);
34
+ const recordAdapter = options.recordAdapter ?? filesystemProviderProcessRecordAdapter;
35
+ const terminateProcessTree = options.terminateProcessTree ?? ((processPid) => killProcessTreeForCleanup(processPid));
36
+ let recordPublished = false;
37
+ const stdout = createBoundedOutput(options.outputLimitBytes ?? 1_048_576);
38
+ const stderr = createBoundedOutput(options.outputLimitBytes ?? 1_048_576);
39
+ const telemetry = createTelemetryAccumulator(agent);
40
+ const idleTimeoutMs = options.idleTimeoutMs ?? 0;
41
+ const terminationGraceMs = options.terminationGraceMs ?? 5_000;
42
+ let termination;
43
+ let idleTimer;
44
+ let forceTimer;
45
+ let recordFailure;
46
+ let terminationConfirmed = false;
47
+ let settled = false;
48
+ let resolveCompletion = () => { };
49
+ const completion = new Promise(resolveCompletionValue => {
50
+ resolveCompletion = resolveCompletionValue;
51
+ });
52
+ const removeRecord = () => {
53
+ if (recordPublished)
54
+ recordAdapter.remove(record.path);
55
+ };
56
+ const clearTimers = () => {
57
+ if (idleTimer)
58
+ clearTimeout(idleTimer);
59
+ if (forceTimer)
60
+ clearTimeout(forceTimer);
61
+ idleTimer = undefined;
62
+ forceTimer = undefined;
63
+ };
64
+ const finish = (result) => {
65
+ if (settled)
66
+ return;
67
+ settled = true;
68
+ clearTimers();
69
+ options.signal?.removeEventListener('abort', onAbort);
70
+ if (!termination || terminationConfirmed)
71
+ removeRecord();
72
+ resolveCompletion(result);
73
+ };
74
+ const evidence = () => ({
75
+ invocation,
76
+ pid,
77
+ stdout: stdout.text,
78
+ stderr: stderr.text,
79
+ stdoutTruncated: stdout.truncated,
80
+ stderrTruncated: stderr.truncated,
81
+ telemetry: telemetry.finish(),
82
+ });
83
+ const finalize = (exitCode) => {
84
+ const details = evidence();
85
+ if (recordFailure) {
86
+ finish({ ...details, kind: 'spawn-failed', error: recordFailure });
87
+ return;
88
+ }
89
+ if (termination?.kind === 'timed-out') {
90
+ finish({ ...details, kind: 'timed-out', reason: termination.reason });
91
+ return;
92
+ }
93
+ if (termination?.kind === 'cancelled') {
94
+ finish({ ...details, kind: 'cancelled', reason: termination.reason });
95
+ return;
96
+ }
97
+ if (exitCode === 0) {
98
+ finish({ ...details, kind: 'succeeded', exitCode });
99
+ return;
100
+ }
101
+ finish({ ...details, kind: 'failed', exitCode });
102
+ };
103
+ const terminate = (next) => {
104
+ if (termination || settled)
105
+ return false;
106
+ termination = next;
107
+ if (pid !== undefined)
108
+ terminationConfirmed = terminateProcessTree(pid, false);
109
+ forceTimer = setTimeout(() => {
110
+ if (pid !== undefined && !settled)
111
+ terminationConfirmed = terminateProcessTree(pid, true);
112
+ }, terminationGraceMs);
113
+ return true;
114
+ };
115
+ const armIdleTimer = () => {
116
+ if (idleTimeoutMs <= 0 || termination || settled)
117
+ return;
118
+ if (idleTimer)
119
+ clearTimeout(idleTimer);
120
+ idleTimer = setTimeout(() => {
121
+ terminate({ kind: 'timed-out', reason: 'provider process produced no output before its idle timeout' });
122
+ }, idleTimeoutMs);
123
+ };
124
+ const onAbort = () => {
125
+ if (options.signal)
126
+ terminate({ kind: 'cancelled', reason: cancellationReason(options.signal) });
127
+ };
128
+ const onOutput = (stream, chunk) => {
129
+ const text = String(chunk);
130
+ if (stream === 'stdout') {
131
+ stdout.append(text);
132
+ telemetry.append(text);
133
+ }
134
+ else {
135
+ stderr.append(text);
136
+ }
137
+ options.onOutput?.({ stream, text });
138
+ armIdleTimer();
139
+ };
140
+ child.stdout?.on('data', chunk => { onOutput('stdout', chunk); });
141
+ child.stderr?.on('data', chunk => { onOutput('stderr', chunk); });
142
+ child.stdin?.on('error', () => { });
143
+ child.on('close', code => { finalize(code); });
144
+ child.on('error', error => {
145
+ finish({ ...evidence(), kind: 'spawn-failed', error: recordFailure ?? error.message });
146
+ });
147
+ if (pid !== undefined) {
148
+ try {
149
+ recordAdapter.publish(record);
150
+ recordPublished = true;
151
+ }
152
+ catch (error) {
153
+ const reason = error instanceof Error ? error.message : String(error);
154
+ recordFailure = `process ownership record failure: ${reason}`;
155
+ child.stdin?.end();
156
+ terminate({ kind: 'cancelled', reason: recordFailure });
157
+ }
158
+ }
159
+ const handle = {
160
+ pid,
161
+ invocation,
162
+ recordPath: record.path,
163
+ completion,
164
+ cancel(reason) {
165
+ return terminate({ kind: 'cancelled', reason });
166
+ },
167
+ };
168
+ if (recordFailure)
169
+ return handle;
170
+ child.stdin?.end(invocation.input);
171
+ if (options.signal?.aborted)
172
+ onAbort();
173
+ else
174
+ options.signal?.addEventListener('abort', onAbort, { once: true });
175
+ armIdleTimer();
176
+ return handle;
177
+ }
@@ -1,3 +1,5 @@
1
+ import { ModelSelectionSchema } from './contracts.js';
2
+ export { providerSpawnOptions, startProviderProcess, } from './process.js';
1
3
  const argsFor = (agent, permissions) => {
2
4
  if (agent === 'claude') {
3
5
  const mode = permissions === 'unsafe' ? 'bypassPermissions' : permissions === 'read-only' ? 'plan' : 'auto';
@@ -19,18 +21,19 @@ const argsFor = (agent, permissions) => {
19
21
  return ['--approval-mode', approval, '--sandbox'];
20
22
  };
21
23
  export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}) {
24
+ const parsedSelection = ModelSelectionSchema.parse(selection);
22
25
  const args = argsFor(agent, permissions);
23
- if (selection.model)
24
- args.push('--model', selection.model);
25
- if (selection.reasoningEffort) {
26
+ if (parsedSelection.model)
27
+ args.push('--model', parsedSelection.model);
28
+ if (parsedSelection.reasoningEffort) {
26
29
  if (agent === 'claude')
27
- args.push('--effort', selection.reasoningEffort);
30
+ args.push('--effort', parsedSelection.reasoningEffort);
28
31
  else if (agent === 'codex')
29
- args.push('--config', `model_reasoning_effort=${selection.reasoningEffort}`);
32
+ args.push('--config', `model_reasoning_effort=${parsedSelection.reasoningEffort}`);
30
33
  }
31
- if (agent === 'codex' && selection.nativeMultiAgent === false)
34
+ if (agent === 'codex' && parsedSelection.nativeMultiAgent === false)
32
35
  args.push('--disable', 'multi_agent');
33
- if (selection.bare) {
36
+ if (parsedSelection.bare) {
34
37
  if (agent === 'codex')
35
38
  args.push('--ignore-user-config');
36
39
  else if (agent === 'claude')