@hecer/yoke 1.9.0 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +13 -13
- package/.codex-plugin/plugin.json +7 -7
- package/CHANGELOG.md +398 -358
- package/README.md +915 -913
- package/TODOS.md +5 -5
- package/agents/docs.toml +6 -6
- package/agents/implementer.toml +6 -6
- package/agents/reviewer.toml +6 -6
- package/agents/security.toml +6 -6
- package/bench/README.md +86 -86
- package/bench/RESULTS.md +35 -35
- package/bench/output-compaction.mjs +65 -65
- package/bench/result-schema.mjs +12 -12
- package/bench/results/claude-2026-07-27T18-03-26.json +50 -50
- package/bench/results/codex-unavailable-1785175418318.json +15 -15
- package/bench/results/gemini-2026-07-27T18-03-44.json +46 -46
- package/bench/run-matrix.mjs +26 -26
- package/bench/run.mjs +106 -106
- package/canon/AGENTS.md +30 -30
- package/canon/context/DECISIONS.md +4 -4
- package/canon/context/GLOSSARY.md +11 -11
- package/canon/context/KNOWLEDGE.md +4 -4
- package/canon/context/PROJECT.md +15 -15
- package/canon/loop/loop-spec.md +65 -65
- package/canon/loop/prd.schema.md +46 -40
- package/canon/manifest.yaml +59 -59
- package/canon/policy/gates.md +7 -7
- package/canon/policy/roles.md +9 -9
- package/canon/skills/ATTRIBUTION.md +99 -99
- package/canon/skills/authoring-prd/SKILL.md +56 -56
- package/canon/skills/brainstorming/SKILL.md +164 -164
- package/canon/skills/codebase-design/DEEPENING.md +15 -15
- package/canon/skills/codebase-design/DESIGN-IT-TWICE.md +12 -12
- package/canon/skills/codebase-design/SKILL.md +39 -39
- package/canon/skills/dispatching-parallel-agents/SKILL.md +182 -182
- package/canon/skills/document-release/SKILL.md +302 -302
- package/canon/skills/domain-modeling/ADR-FORMAT.md +19 -19
- package/canon/skills/domain-modeling/CONTEXT-FORMAT.md +39 -39
- package/canon/skills/domain-modeling/SKILL.md +35 -35
- package/canon/skills/executing-plans/SKILL.md +70 -70
- package/canon/skills/finishing-a-development-branch/SKILL.md +200 -200
- package/canon/skills/health/SKILL.md +177 -177
- package/canon/skills/maintaining-context/SKILL.md +34 -34
- package/canon/skills/minimal-code/SKILL.md +21 -21
- package/canon/skills/no-ai-slop/SKILL.md +103 -103
- package/canon/skills/no-ai-slop/eval.md +43 -43
- package/canon/skills/plan-ceo-review/SKILL.md +541 -541
- package/canon/skills/plan-eng-review/SKILL.md +362 -362
- package/canon/skills/receiving-code-review/SKILL.md +213 -213
- package/canon/skills/requesting-code-review/SKILL.md +105 -105
- package/canon/skills/resolving-merge-conflicts/SKILL.md +18 -18
- package/canon/skills/retro/SKILL.md +397 -397
- package/canon/skills/review/SKILL.md +246 -246
- package/canon/skills/ship/SKILL.md +691 -691
- package/canon/skills/subagent-driven-development/SKILL.md +277 -277
- package/canon/skills/systematic-debugging/SKILL.md +296 -296
- package/canon/skills/tdd/SKILL.md +371 -371
- package/canon/skills/unslop-ui/SKILL.md +34 -34
- package/canon/skills/using-git-worktrees/SKILL.md +218 -218
- package/canon/skills/verification-before-completion/SKILL.md +139 -139
- package/canon/skills/visual-verification/SKILL.md +54 -54
- package/canon/skills/workflow/SKILL.md +22 -22
- package/canon/skills/writing-for-agents/SKILL-MECHANICS.md +27 -27
- package/canon/skills/writing-for-agents/SKILL.md +42 -42
- package/canon/skills/writing-plans/SKILL.md +152 -152
- package/canon/skills/writing-skills/SKILL.md +655 -655
- package/canon/skills/yoke-retrofit/SKILL.md +26 -26
- package/canon/skills/yoke-workflow/SKILL.md +20 -20
- package/canon/tools/codex-rtk-hook.mjs +35 -35
- package/canon/tools/gemini-rtk-hook.mjs +25 -25
- package/canon/tools/graphify.md +3 -3
- package/canon/tools/playwright-mcp.md +3 -3
- package/canon/tools/rtk.md +7 -7
- package/canon/tools/serena.md +6 -6
- package/dist/agents/contracts.js +1 -1
- package/dist/agents/host.js +4 -0
- package/dist/agents/process-incarnation.js +1 -1
- package/dist/agents/process.js +74 -6
- package/dist/agents/providers.js +13 -0
- package/dist/agents/supervision.js +153 -0
- package/dist/agents/telemetry.js +33 -0
- package/dist/agents/windows-launch.js +80 -0
- package/dist/canon/manifest.js +1 -1
- package/dist/change/inbox.js +21 -5
- package/dist/cli.js +19 -10
- package/dist/dashboard/discovery.js +73 -0
- package/dist/dashboard/page.js +122 -28
- package/dist/dashboard/panels.js +91 -15
- package/dist/goals/command.js +4 -2
- package/dist/loop/claims.js +1 -1
- package/dist/loop/decision.js +2 -2
- package/dist/loop/git.js +12 -4
- package/dist/loop/loop.js +8 -4
- package/dist/loop/parallel-adapters.js +2 -3
- package/dist/loop/parallel-command.js +5 -0
- package/dist/loop/prd.js +3 -1
- package/dist/loop/reporter.js +4 -1
- package/dist/loop/run-command.js +11 -2
- package/dist/loop/runner.js +22 -26
- package/dist/loop/watchdog.js +87 -11
- package/dist/loop/worker.js +5 -3
- package/dist/prd/assess.js +145 -0
- package/dist/prd/command.js +76 -38
- package/dist/quality/types.js +1 -1
- package/dist/retrofit/config.js +11 -0
- package/dist/retrofit/plan.js +2 -0
- package/dist/retrofit/planners/claude.js +14 -14
- package/dist/retrofit/planners/qwen.js +73 -0
- package/dist/retrofit/preserve.js +2 -2
- package/dist/retrofit/skill-actions.js +1 -0
- package/dist/review/command.js +1 -1
- package/dist/routing/assessment.js +1 -1
- package/dist/routing/capability.js +25 -13
- package/dist/routing/contracts.js +60 -0
- package/dist/routing/planning.js +12 -0
- package/dist/routing/router.js +51 -16
- package/dist/setup/command.js +11 -3
- package/docs/BATCH-PLANNING-VALIDATION.md +67 -0
- package/docs/CAPABILITY-ROUTING.md +78 -50
- package/docs/DASHBOARD-EVOLUTION.md +33 -0
- package/docs/MIGRATING-TO-1.0.md +33 -33
- package/docs/MIGRATING-TO-1.1.md +27 -27
- package/docs/MIGRATING-TO-1.4.md +70 -70
- package/docs/PRODUCT-DIRECTION-2026-09-05.md +218 -200
- package/docs/PUBLISHING.md +114 -114
- package/docs/VERIFIED-PROJECTS-VALIDATION.md +29 -29
- package/docs/VERIFIED-PROJECTS.md +167 -167
- package/docs/WINDOWS-RUNNER-VALIDATION.md +104 -0
- package/docs/assets/yoke-logo.png +0 -0
- package/docs/community-outreach-2026-08-20.md +85 -0
- package/docs/launch-copy-2026-08-21.md +193 -0
- package/docs/superpowers/plans/2026-06-28-baustein-e-context-layer.md +981 -981
- package/docs/superpowers/plans/2026-06-29-baustein-f-routing.md +258 -258
- package/docs/superpowers/plans/2026-06-29-baustein-g-loop-observability.md +1006 -1006
- package/docs/superpowers/plans/2026-06-29-baustein-h-loop-robustness.md +374 -374
- package/docs/superpowers/plans/2026-06-30-baustein-i-visual-design-verification.md +450 -450
- package/docs/superpowers/plans/2026-07-02-baustein-k-zero-to-100-bootstrap.md +1024 -1024
- package/docs/superpowers/plans/2026-07-02-baustein-m-flow-smoke-proofs.md +574 -574
- package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -537
- package/docs/superpowers/plans/2026-08-16-artifact-backed-output-compaction.md +329 -329
- package/docs/superpowers/plans/2026-09-05-verified-projects.md +83 -83
- package/docs/superpowers/specs/2026-06-28-baustein-e-context-layer-design.md +146 -146
- package/docs/superpowers/specs/2026-06-29-baustein-f-routing-design.md +106 -106
- package/docs/superpowers/specs/2026-06-29-baustein-g-loop-observability-design.md +186 -186
- package/docs/superpowers/specs/2026-06-29-baustein-h-loop-robustness-design.md +113 -113
- package/docs/superpowers/specs/2026-06-30-baustein-i-visual-design-verification-design.md +98 -98
- package/docs/superpowers/specs/2026-07-02-baustein-k-zero-to-100-bootstrap-design.md +200 -200
- package/docs/superpowers/specs/2026-07-02-baustein-m-flow-smoke-proofs-design.md +155 -155
- package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -422
- package/docs/superpowers/specs/2026-08-16-artifact-backed-output-compaction-design.md +166 -166
- package/gemini-extension.json +6 -6
- package/hooks/hooks.json +19 -19
- package/package.json +87 -87
|
@@ -1,181 +1,181 @@
|
|
|
1
|
-
# Artifact-backed output compaction design
|
|
2
|
-
|
|
3
|
-
**Date:** 2026-08-16
|
|
1
|
+
# Artifact-backed output compaction design
|
|
2
|
+
|
|
3
|
+
**Date:** 2026-08-16
|
|
4
4
|
**Status:** implemented, hardened, and verified on `main`
|
|
5
5
|
**Target:** Yoke 1.5.0
|
|
6
|
-
|
|
7
|
-
## Problem
|
|
8
|
-
|
|
9
|
-
Yoke already reduces shell noise through RTK and keeps loop prompts intentionally small. It does not, however, retain complete evidence from failed verify, criterion, performance, or completion commands. `commandVerifier` currently chooses stderr or stdout and keeps only the final five lines. That is token-cheap, but it can discard the first error, structured summaries, and the stdout half of a mixed failure.
|
|
10
|
-
|
|
11
|
-
Aphrodite demonstrates a useful pattern: keep a compact, type-aware preview in model-visible context and place the complete output behind a stable reference. Embedding Aphrodite itself is not a good fit because its primary integration is Hermes-specific, while Yoke must behave consistently across Claude Code, Codex, and Gemini and preserve its explicit fresh-context model.
|
|
12
|
-
|
|
13
|
-
## Goals
|
|
14
|
-
|
|
15
|
-
- Produce short, deterministic failure previews that retain the most actionable error and warning lines.
|
|
16
|
-
- Preserve complete stdout and stderr as a local artifact when output exceeds the inline budget.
|
|
17
|
-
- Put only the preview and a verifiable artifact reference into loop evidence and repair/review context.
|
|
18
|
-
- Use the same implementation for verify, executable acceptance criteria, performance, audit, and completion gates.
|
|
19
|
-
- Remain provider-independent and add no runtime dependency, daemon, database, proxy, or network request.
|
|
20
|
-
- Make savings and correctness measurable with Yoke's existing tests and benchmark schema.
|
|
21
|
-
|
|
22
|
-
## Non-goals
|
|
23
|
-
|
|
24
|
-
- Intercepting tool calls occurring inside Claude Code, Codex, or Gemini. Yoke cannot transparently alter those provider-internal streams.
|
|
25
|
-
- Replacing RTK, provider prompt caching, or Yoke's versioned context files.
|
|
26
|
-
- Semantic summarization by another model.
|
|
27
|
-
- Persisting interactive conversation memory or automatically injecting historical artifacts into later stories.
|
|
28
|
-
- Claiming Aphrodite's published compression ratios for Yoke.
|
|
29
|
-
|
|
30
|
-
## Chosen approach
|
|
31
|
-
|
|
32
|
-
Implement a small Yoke-native output subsystem with two isolated units:
|
|
33
|
-
|
|
34
|
-
1. `compactCommandOutput` is a pure deterministic function. It normalizes ANSI/control noise, classifies high-signal lines, removes repeated lines, and constructs a byte-bounded preview.
|
|
35
|
-
2. `writeOutputArtifact` stores the unmodified captured stdout and stderr in `.yoke/artifacts/` under a content-addressed filename and returns a relative path, byte count, and SHA-256 digest.
|
|
36
|
-
|
|
37
|
-
The gate runner combines both units. Small failures remain inline and create no artifact. Large failures include a compact preview followed by an artifact marker. Successful gates retain today's one-line summary and do not persist their output because success logs do not feed repair decisions.
|
|
38
|
-
|
|
39
|
-
This approach is preferred over:
|
|
40
|
-
|
|
41
|
-
- **Embedding Aphrodite:** stronger generic CCR machinery, but Hermes-oriented and operationally disproportionate for Yoke.
|
|
42
|
-
- **Adding a Chat Completions proxy:** could theoretically observe more traffic, but would couple Yoke to provider protocols, credentials, streaming semantics, and tool-call formats.
|
|
43
|
-
- **Only documenting Aphrodite as a companion:** zero maintenance, but provides no consistent behavior for Yoke's three supported providers and does not fix Yoke's current loss of gate evidence.
|
|
44
|
-
|
|
45
|
-
## Configuration
|
|
46
|
-
|
|
47
|
-
The feature is configured under an optional `output` block:
|
|
48
|
-
|
|
49
|
-
```yaml
|
|
50
|
-
output:
|
|
51
|
-
previewBytes: 2048
|
|
52
|
-
artifactThresholdBytes: 8192
|
|
53
|
-
```
|
|
54
|
-
|
|
55
|
-
Defaults apply when the block is absent:
|
|
56
|
-
|
|
57
|
-
- `previewBytes`: 2,048 bytes
|
|
58
|
-
- `artifactThresholdBytes`: 8,192 bytes
|
|
59
|
-
|
|
60
|
-
Both values are positive integers. `artifactThresholdBytes` must be greater than or equal to `previewBytes`. Existing configurations remain valid.
|
|
61
|
-
|
|
62
|
-
Artifact persistence is automatic only after the threshold is crossed. The artifacts directory is added to Yoke's managed `.gitignore` block. Files are created with user-only permissions where the platform supports POSIX modes. Yoke does not redact or transform the stored raw evidence, and documentation must therefore state that command output can contain secrets and must not be published blindly.
|
|
63
|
-
|
|
64
|
-
## Classification and preview rules
|
|
65
|
-
|
|
66
|
-
The compactor operates line-by-line without parsing project-specific formats:
|
|
67
|
-
|
|
68
|
-
1. Strip ANSI escape sequences and disallowed control characters from the preview only.
|
|
69
|
-
2. Mark case-insensitive error signals as highest priority: `error`, `failed`, `failure`, `fatal`, `panic`, `exception`, `traceback`, compiler error codes, and test failure markers.
|
|
70
|
-
3. Mark warnings as second priority: `warn`, `warning`, and deprecation notices.
|
|
71
|
-
4. Retain bounded context immediately around the first high-priority lines.
|
|
72
|
-
5. Retain the final non-empty lines because many test runners put totals and exit summaries at the end.
|
|
73
|
-
6. Remove duplicate preview lines while preserving their first selected order.
|
|
74
|
-
7. Enforce the preview byte budget at UTF-8 boundaries and append a deterministic omission line containing original line and byte counts.
|
|
75
|
-
|
|
6
|
+
|
|
7
|
+
## Problem
|
|
8
|
+
|
|
9
|
+
Yoke already reduces shell noise through RTK and keeps loop prompts intentionally small. It does not, however, retain complete evidence from failed verify, criterion, performance, or completion commands. `commandVerifier` currently chooses stderr or stdout and keeps only the final five lines. That is token-cheap, but it can discard the first error, structured summaries, and the stdout half of a mixed failure.
|
|
10
|
+
|
|
11
|
+
Aphrodite demonstrates a useful pattern: keep a compact, type-aware preview in model-visible context and place the complete output behind a stable reference. Embedding Aphrodite itself is not a good fit because its primary integration is Hermes-specific, while Yoke must behave consistently across Claude Code, Codex, and Gemini and preserve its explicit fresh-context model.
|
|
12
|
+
|
|
13
|
+
## Goals
|
|
14
|
+
|
|
15
|
+
- Produce short, deterministic failure previews that retain the most actionable error and warning lines.
|
|
16
|
+
- Preserve complete stdout and stderr as a local artifact when output exceeds the inline budget.
|
|
17
|
+
- Put only the preview and a verifiable artifact reference into loop evidence and repair/review context.
|
|
18
|
+
- Use the same implementation for verify, executable acceptance criteria, performance, audit, and completion gates.
|
|
19
|
+
- Remain provider-independent and add no runtime dependency, daemon, database, proxy, or network request.
|
|
20
|
+
- Make savings and correctness measurable with Yoke's existing tests and benchmark schema.
|
|
21
|
+
|
|
22
|
+
## Non-goals
|
|
23
|
+
|
|
24
|
+
- Intercepting tool calls occurring inside Claude Code, Codex, or Gemini. Yoke cannot transparently alter those provider-internal streams.
|
|
25
|
+
- Replacing RTK, provider prompt caching, or Yoke's versioned context files.
|
|
26
|
+
- Semantic summarization by another model.
|
|
27
|
+
- Persisting interactive conversation memory or automatically injecting historical artifacts into later stories.
|
|
28
|
+
- Claiming Aphrodite's published compression ratios for Yoke.
|
|
29
|
+
|
|
30
|
+
## Chosen approach
|
|
31
|
+
|
|
32
|
+
Implement a small Yoke-native output subsystem with two isolated units:
|
|
33
|
+
|
|
34
|
+
1. `compactCommandOutput` is a pure deterministic function. It normalizes ANSI/control noise, classifies high-signal lines, removes repeated lines, and constructs a byte-bounded preview.
|
|
35
|
+
2. `writeOutputArtifact` stores the unmodified captured stdout and stderr in `.yoke/artifacts/` under a content-addressed filename and returns a relative path, byte count, and SHA-256 digest.
|
|
36
|
+
|
|
37
|
+
The gate runner combines both units. Small failures remain inline and create no artifact. Large failures include a compact preview followed by an artifact marker. Successful gates retain today's one-line summary and do not persist their output because success logs do not feed repair decisions.
|
|
38
|
+
|
|
39
|
+
This approach is preferred over:
|
|
40
|
+
|
|
41
|
+
- **Embedding Aphrodite:** stronger generic CCR machinery, but Hermes-oriented and operationally disproportionate for Yoke.
|
|
42
|
+
- **Adding a Chat Completions proxy:** could theoretically observe more traffic, but would couple Yoke to provider protocols, credentials, streaming semantics, and tool-call formats.
|
|
43
|
+
- **Only documenting Aphrodite as a companion:** zero maintenance, but provides no consistent behavior for Yoke's three supported providers and does not fix Yoke's current loss of gate evidence.
|
|
44
|
+
|
|
45
|
+
## Configuration
|
|
46
|
+
|
|
47
|
+
The feature is configured under an optional `output` block:
|
|
48
|
+
|
|
49
|
+
```yaml
|
|
50
|
+
output:
|
|
51
|
+
previewBytes: 2048
|
|
52
|
+
artifactThresholdBytes: 8192
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Defaults apply when the block is absent:
|
|
56
|
+
|
|
57
|
+
- `previewBytes`: 2,048 bytes
|
|
58
|
+
- `artifactThresholdBytes`: 8,192 bytes
|
|
59
|
+
|
|
60
|
+
Both values are positive integers. `artifactThresholdBytes` must be greater than or equal to `previewBytes`. Existing configurations remain valid.
|
|
61
|
+
|
|
62
|
+
Artifact persistence is automatic only after the threshold is crossed. The artifacts directory is added to Yoke's managed `.gitignore` block. Files are created with user-only permissions where the platform supports POSIX modes. Yoke does not redact or transform the stored raw evidence, and documentation must therefore state that command output can contain secrets and must not be published blindly.
|
|
63
|
+
|
|
64
|
+
## Classification and preview rules
|
|
65
|
+
|
|
66
|
+
The compactor operates line-by-line without parsing project-specific formats:
|
|
67
|
+
|
|
68
|
+
1. Strip ANSI escape sequences and disallowed control characters from the preview only.
|
|
69
|
+
2. Mark case-insensitive error signals as highest priority: `error`, `failed`, `failure`, `fatal`, `panic`, `exception`, `traceback`, compiler error codes, and test failure markers.
|
|
70
|
+
3. Mark warnings as second priority: `warn`, `warning`, and deprecation notices.
|
|
71
|
+
4. Retain bounded context immediately around the first high-priority lines.
|
|
72
|
+
5. Retain the final non-empty lines because many test runners put totals and exit summaries at the end.
|
|
73
|
+
6. Remove duplicate preview lines while preserving their first selected order.
|
|
74
|
+
7. Enforce the preview byte budget at UTF-8 boundaries and append a deterministic omission line containing original line and byte counts.
|
|
75
|
+
|
|
76
76
|
The raw artifact contains the exact captured stdout and stderr with explicit stream headings.
|
|
77
77
|
Preview cleanup must never modify the artifact. Capture is bounded at 16 MiB per stream; if that
|
|
78
78
|
quota is exceeded, the command fails closed and the retained prefix is explicitly marked truncated.
|
|
79
|
-
|
|
80
|
-
## Artifact identity and layout
|
|
81
|
-
|
|
82
|
-
Artifacts use this layout:
|
|
83
|
-
|
|
84
|
-
```text
|
|
85
|
-
.yoke/artifacts/<story-or-session>/<phase>-<sha256-prefix>.log
|
|
86
|
-
```
|
|
87
|
-
|
|
88
|
-
- `story-or-session` is the sanitized `YOKE_STORY` value, falling back to `session`.
|
|
89
|
-
- `phase` is one of `criterion`, `verify`, `perf`, `audit`, or `completion`.
|
|
90
|
-
- The digest is computed from the complete artifact bytes; the filename uses a readable prefix while the marker records the full SHA-256.
|
|
91
|
-
- Repeating the same failure for the same story and phase resolves to the same path and content rather than generating timestamp noise.
|
|
92
|
-
- Paths returned to summaries are project-relative and use `/` separators so evidence is portable across Windows and Unix output.
|
|
93
|
-
|
|
94
|
-
The marker format is human-readable rather than a proprietary retrieval protocol:
|
|
95
|
-
|
|
96
|
-
```text
|
|
97
|
-
[full output: .yoke/artifacts/STORY/verify-0123abcd.log | 42,810 bytes | sha256:0123...]
|
|
98
|
-
```
|
|
99
|
-
|
|
100
|
-
Agents can retrieve it with their normal file-reading tools when the preview is insufficient.
|
|
101
|
-
|
|
102
|
-
## Data flow
|
|
103
|
-
|
|
104
|
-
1. A Yoke gate executes a configured command with stdout and stderr captured.
|
|
105
|
-
2. On exit zero, Yoke discards captured bytes and returns the existing compact success summary.
|
|
79
|
+
|
|
80
|
+
## Artifact identity and layout
|
|
81
|
+
|
|
82
|
+
Artifacts use this layout:
|
|
83
|
+
|
|
84
|
+
```text
|
|
85
|
+
.yoke/artifacts/<story-or-session>/<phase>-<sha256-prefix>.log
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
- `story-or-session` is the sanitized `YOKE_STORY` value, falling back to `session`.
|
|
89
|
+
- `phase` is one of `criterion`, `verify`, `perf`, `audit`, or `completion`.
|
|
90
|
+
- The digest is computed from the complete artifact bytes; the filename uses a readable prefix while the marker records the full SHA-256.
|
|
91
|
+
- Repeating the same failure for the same story and phase resolves to the same path and content rather than generating timestamp noise.
|
|
92
|
+
- Paths returned to summaries are project-relative and use `/` separators so evidence is portable across Windows and Unix output.
|
|
93
|
+
|
|
94
|
+
The marker format is human-readable rather than a proprietary retrieval protocol:
|
|
95
|
+
|
|
96
|
+
```text
|
|
97
|
+
[full output: .yoke/artifacts/STORY/verify-0123abcd.log | 42,810 bytes | sha256:0123...]
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Agents can retrieve it with their normal file-reading tools when the preview is insufficient.
|
|
101
|
+
|
|
102
|
+
## Data flow
|
|
103
|
+
|
|
104
|
+
1. A Yoke gate executes a configured command with stdout and stderr captured.
|
|
105
|
+
2. On exit zero, Yoke discards captured bytes and returns the existing compact success summary.
|
|
106
106
|
3. On timeout or non-zero exit, Yoke combines both streams with labels.
|
|
107
107
|
Capture quota overflow follows the same failure path but marks the retained prefix as truncated.
|
|
108
108
|
4. The pure compactor creates the bounded preview.
|
|
109
|
-
5. If the combined raw output crosses `artifactThresholdBytes`, Yoke writes the content-addressed artifact.
|
|
110
|
-
6. `VerifyResult.summary` carries the command, timeout state, preview, and optional artifact marker.
|
|
111
|
-
7. Existing worker evidence, status reporting, quality repair, and retry logic transport that bounded summary unchanged.
|
|
112
|
-
|
|
113
|
-
## Failure handling
|
|
114
|
-
|
|
115
|
-
- An artifact write failure must not hide the original gate failure or crash the loop. The summary keeps the compact preview and adds a bounded `artifact unavailable` reason.
|
|
116
|
-
- Invalid configuration is rejected by the existing Zod configuration boundary before any command runs.
|
|
117
|
-
- Empty command output produces the current command-only failure summary.
|
|
118
|
-
- stdout and stderr are both retained even when one is empty.
|
|
109
|
+
5. If the combined raw output crosses `artifactThresholdBytes`, Yoke writes the content-addressed artifact.
|
|
110
|
+
6. `VerifyResult.summary` carries the command, timeout state, preview, and optional artifact marker.
|
|
111
|
+
7. Existing worker evidence, status reporting, quality repair, and retry logic transport that bounded summary unchanged.
|
|
112
|
+
|
|
113
|
+
## Failure handling
|
|
114
|
+
|
|
115
|
+
- An artifact write failure must not hide the original gate failure or crash the loop. The summary keeps the compact preview and adds a bounded `artifact unavailable` reason.
|
|
116
|
+
- Invalid configuration is rejected by the existing Zod configuration boundary before any command runs.
|
|
117
|
+
- Empty command output produces the current command-only failure summary.
|
|
118
|
+
- stdout and stderr are both retained even when one is empty.
|
|
119
119
|
- Retried identical failures reuse the same artifact; a changed failure gets a different digest.
|
|
120
120
|
- Artifact paths are built from fixed directories and sanitized labels. User-controlled story IDs never become raw path segments.
|
|
121
121
|
- stdout and stderr capture is capped at 16 MiB per stream. Overflow fails closed, retains only the
|
|
122
122
|
bounded prefix, and uses a `truncated output` marker rather than a `full output` marker.
|
|
123
|
-
|
|
124
|
-
## Security and privacy
|
|
125
|
-
|
|
126
|
-
- `.yoke/artifacts/` is runtime state and must be gitignored by retrofit/setup.
|
|
127
|
-
- Artifacts never leave the local project through Yoke.
|
|
128
|
-
- No artifact content is injected automatically into prompts; only the bounded preview and reference are transported.
|
|
123
|
+
|
|
124
|
+
## Security and privacy
|
|
125
|
+
|
|
126
|
+
- `.yoke/artifacts/` is runtime state and must be gitignored by retrofit/setup.
|
|
127
|
+
- Artifacts never leave the local project through Yoke.
|
|
128
|
+
- No artifact content is injected automatically into prompts; only the bounded preview and reference are transported.
|
|
129
129
|
- The full digest lets reviewers verify that retrieved evidence matches the retained artifact bytes.
|
|
130
|
-
- Raw logs may contain credentials or personal data emitted by project commands. The README must warn users to inspect artifacts before sharing them.
|
|
131
|
-
- Absence of provenance metadata or detectable text markers must never be treated as evidence of human authorship.
|
|
132
|
-
|
|
133
|
-
## Testing strategy
|
|
134
|
-
|
|
135
|
-
Implementation follows red-green-refactor cycles:
|
|
136
|
-
|
|
137
|
-
- Pure compactor tests cover error prioritization, warning/context selection, duplicate removal, ANSI cleanup, UTF-8 byte bounds, tail summaries, empty input, and deterministic output.
|
|
138
|
-
- Artifact tests cover content fidelity, SHA-256 identity, stable paths, story sanitization, directory creation, and project-relative markers.
|
|
130
|
+
- Raw logs may contain credentials or personal data emitted by project commands. The README must warn users to inspect artifacts before sharing them.
|
|
131
|
+
- Absence of provenance metadata or detectable text markers must never be treated as evidence of human authorship.
|
|
132
|
+
|
|
133
|
+
## Testing strategy
|
|
134
|
+
|
|
135
|
+
Implementation follows red-green-refactor cycles:
|
|
136
|
+
|
|
137
|
+
- Pure compactor tests cover error prioritization, warning/context selection, duplicate removal, ANSI cleanup, UTF-8 byte bounds, tail summaries, empty input, and deterministic output.
|
|
138
|
+
- Artifact tests cover content fidelity, SHA-256 identity, stable paths, story sanitization, directory creation, and project-relative markers.
|
|
139
139
|
- Verifier integration tests prove that small failures remain inline, large failures create retrievable artifacts, mixed stdout/stderr is retained, successful commands create no artifacts, timeouts remain labelled, capture overflow fails closed as truncated, and artifact-write failures preserve the gate result.
|
|
140
|
-
- Configuration tests cover defaults, valid overrides, and the threshold invariant.
|
|
141
|
-
- Retrofit tests require `.yoke/artifacts/` in the managed ignore set.
|
|
142
|
-
- Existing loop, parallel-worker, quality, and retry suites must stay green.
|
|
143
|
-
|
|
144
|
-
## Measurement and release claims
|
|
145
|
-
|
|
146
|
-
Add a deterministic benchmark fixture that emits a large noisy failure containing an early actionable error and a final test summary. Record:
|
|
147
|
-
|
|
148
|
-
- raw byte and approximate token counts,
|
|
149
|
-
- preview byte and approximate token counts,
|
|
150
|
-
- compression ratio,
|
|
151
|
-
- whether the early error and final summary survived,
|
|
152
|
-
- whether the artifact digest round-trips.
|
|
153
|
-
|
|
154
|
-
The benchmark is for the Yoke-visible gate summary only. Release notes must not imply savings for provider-internal tool usage and must not reuse Aphrodite's reported ratios.
|
|
155
|
-
|
|
156
|
-
## Documentation and compatibility
|
|
157
|
-
|
|
158
|
-
- Add an output-compaction section to the README describing defaults, configuration, retrieval, privacy, and scope limitations.
|
|
159
|
-
- Add a changelog entry under the next unreleased section, without changing the already published `1.4.0` package version during implementation.
|
|
160
|
-
- Existing `Verifier` callers remain source-compatible. New options are threaded from the loaded Yoke config by `runLoopCommand`.
|
|
161
|
-
- No migration is required; projects receive the new gitignore line on their next setup/retrofit run.
|
|
162
|
-
|
|
163
|
-
## Acceptance criteria
|
|
164
|
-
|
|
165
|
-
1. A failed gate whose combined stdout/stderr is at most 8 KiB returns a deterministic preview and writes no artifact under default configuration.
|
|
140
|
+
- Configuration tests cover defaults, valid overrides, and the threshold invariant.
|
|
141
|
+
- Retrofit tests require `.yoke/artifacts/` in the managed ignore set.
|
|
142
|
+
- Existing loop, parallel-worker, quality, and retry suites must stay green.
|
|
143
|
+
|
|
144
|
+
## Measurement and release claims
|
|
145
|
+
|
|
146
|
+
Add a deterministic benchmark fixture that emits a large noisy failure containing an early actionable error and a final test summary. Record:
|
|
147
|
+
|
|
148
|
+
- raw byte and approximate token counts,
|
|
149
|
+
- preview byte and approximate token counts,
|
|
150
|
+
- compression ratio,
|
|
151
|
+
- whether the early error and final summary survived,
|
|
152
|
+
- whether the artifact digest round-trips.
|
|
153
|
+
|
|
154
|
+
The benchmark is for the Yoke-visible gate summary only. Release notes must not imply savings for provider-internal tool usage and must not reuse Aphrodite's reported ratios.
|
|
155
|
+
|
|
156
|
+
## Documentation and compatibility
|
|
157
|
+
|
|
158
|
+
- Add an output-compaction section to the README describing defaults, configuration, retrieval, privacy, and scope limitations.
|
|
159
|
+
- Add a changelog entry under the next unreleased section, without changing the already published `1.4.0` package version during implementation.
|
|
160
|
+
- Existing `Verifier` callers remain source-compatible. New options are threaded from the loaded Yoke config by `runLoopCommand`.
|
|
161
|
+
- No migration is required; projects receive the new gitignore line on their next setup/retrofit run.
|
|
162
|
+
|
|
163
|
+
## Acceptance criteria
|
|
164
|
+
|
|
165
|
+
1. A failed gate whose combined stdout/stderr is at most 8 KiB returns a deterministic preview and writes no artifact under default configuration.
|
|
166
166
|
2. A failed gate larger than 8 KiB but within the 16 MiB-per-stream capture quota writes the complete combined output below `.yoke/artifacts/` and returns a preview of at most 2 KiB plus a relative path, byte count, and full SHA-256 digest. Quota overflow fails closed and labels the bounded retained prefix as truncated.
|
|
167
|
-
3. The preview retains an early error and a final runner summary for the benchmark fixture.
|
|
168
|
-
4. Successful gates retain current summaries and write no artifacts.
|
|
169
|
-
5. Artifact failures do not change a gate's pass/fail result and cannot remove its inline preview.
|
|
170
|
-
6. `.yoke/artifacts/` is managed runtime state and is gitignored.
|
|
171
|
-
7. Configuration is validated and remains backward-compatible when `output` is absent.
|
|
172
|
-
8. The full test suite, lint, build, documentation check, package dry-run, audit, and benchmark verification complete successfully before release readiness is claimed.
|
|
173
|
-
|
|
174
|
-
## Implementation outcome
|
|
175
|
-
|
|
176
|
-
The implementation covers verify, executable-criterion, performance, completion, and configured
|
|
177
|
-
custom-audit commands. Yoke's structured built-in audit findings remain bounded by their existing
|
|
178
|
-
finding schema. The deterministic `gate-output-v1` fixture measured 26,699 raw bytes versus 470
|
|
179
|
-
bytes for the preview plus artifact reference (56.81×) while retaining its early compiler error,
|
|
180
|
-
final test summary, and SHA-256 round-trip. This is fixture-specific Yoke gate evidence, not a
|
|
181
|
-
provider-token or billing claim.
|
|
167
|
+
3. The preview retains an early error and a final runner summary for the benchmark fixture.
|
|
168
|
+
4. Successful gates retain current summaries and write no artifacts.
|
|
169
|
+
5. Artifact failures do not change a gate's pass/fail result and cannot remove its inline preview.
|
|
170
|
+
6. `.yoke/artifacts/` is managed runtime state and is gitignored.
|
|
171
|
+
7. Configuration is validated and remains backward-compatible when `output` is absent.
|
|
172
|
+
8. The full test suite, lint, build, documentation check, package dry-run, audit, and benchmark verification complete successfully before release readiness is claimed.
|
|
173
|
+
|
|
174
|
+
## Implementation outcome
|
|
175
|
+
|
|
176
|
+
The implementation covers verify, executable-criterion, performance, completion, and configured
|
|
177
|
+
custom-audit commands. Yoke's structured built-in audit findings remain bounded by their existing
|
|
178
|
+
finding schema. The deterministic `gate-output-v1` fixture measured 26,699 raw bytes versus 470
|
|
179
|
+
bytes for the preview plus artifact reference (56.81×) while retaining its early compiler error,
|
|
180
|
+
final test summary, and SHA-256 round-trip. This is fixture-specific Yoke gate evidence, not a
|
|
181
|
+
provider-token or billing claim.
|
package/gemini-extension.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
{
|
|
2
|
-
"name": "yoke",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
5
|
-
"contextFileName": "GEMINI-EXTENSION.md"
|
|
6
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"name": "yoke",
|
|
3
|
+
"version": "1.10.0",
|
|
4
|
+
"description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
5
|
+
"contextFileName": "GEMINI-EXTENSION.md"
|
|
6
|
+
}
|
package/hooks/hooks.json
CHANGED
|
@@ -1,19 +1,19 @@
|
|
|
1
|
-
{
|
|
2
|
-
"description": "Yoke command compression for Codex",
|
|
3
|
-
"hooks": {
|
|
4
|
-
"PreToolUse": [
|
|
5
|
-
{
|
|
6
|
-
"matcher": "^Bash$",
|
|
7
|
-
"hooks": [
|
|
8
|
-
{
|
|
9
|
-
"type": "command",
|
|
10
|
-
"command": "node \"$PLUGIN_ROOT/canon/tools/codex-rtk-hook.mjs\"",
|
|
11
|
-
"commandWindows": "powershell -NoProfile -ExecutionPolicy Bypass -Command \"node (Join-Path $env:PLUGIN_ROOT 'canon/tools/codex-rtk-hook.mjs')\"",
|
|
12
|
-
"timeout": 5,
|
|
13
|
-
"statusMessage": "Compressing command output with RTK"
|
|
14
|
-
}
|
|
15
|
-
]
|
|
16
|
-
}
|
|
17
|
-
]
|
|
18
|
-
}
|
|
19
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"description": "Yoke command compression for Codex",
|
|
3
|
+
"hooks": {
|
|
4
|
+
"PreToolUse": [
|
|
5
|
+
{
|
|
6
|
+
"matcher": "^Bash$",
|
|
7
|
+
"hooks": [
|
|
8
|
+
{
|
|
9
|
+
"type": "command",
|
|
10
|
+
"command": "node \"$PLUGIN_ROOT/canon/tools/codex-rtk-hook.mjs\"",
|
|
11
|
+
"commandWindows": "powershell -NoProfile -ExecutionPolicy Bypass -Command \"node (Join-Path $env:PLUGIN_ROOT 'canon/tools/codex-rtk-hook.mjs')\"",
|
|
12
|
+
"timeout": 5,
|
|
13
|
+
"statusMessage": "Compressing command output with RTK"
|
|
14
|
+
}
|
|
15
|
+
]
|
|
16
|
+
}
|
|
17
|
+
]
|
|
18
|
+
}
|
|
19
|
+
}
|
package/package.json
CHANGED
|
@@ -1,87 +1,87 @@
|
|
|
1
|
-
{
|
|
2
|
-
"name": "@hecer/yoke",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "One harness,
|
|
5
|
-
"type": "module",
|
|
6
|
-
"bin": {
|
|
7
|
-
"yoke": "dist/cli.js"
|
|
8
|
-
},
|
|
9
|
-
"files": [
|
|
10
|
-
"dist",
|
|
11
|
-
"canon",
|
|
12
|
-
".claude-plugin",
|
|
13
|
-
".codex-plugin",
|
|
14
|
-
"gemini-extension.json",
|
|
15
|
-
"agents",
|
|
16
|
-
"hooks",
|
|
17
|
-
"bench/README.md",
|
|
18
|
-
"bench/output-compaction.mjs",
|
|
19
|
-
"bench/RESULTS.md",
|
|
20
|
-
"bench/result-schema.mjs",
|
|
21
|
-
"bench/run.mjs",
|
|
22
|
-
"bench/run-large.mjs",
|
|
23
|
-
"bench/run-matrix.mjs",
|
|
24
|
-
"bench/analyze-routing-study.mjs",
|
|
25
|
-
"bench/fixtures",
|
|
26
|
-
"bench/results",
|
|
27
|
-
"docs",
|
|
28
|
-
"CHANGELOG.md",
|
|
29
|
-
"TODOS.md",
|
|
30
|
-
"README.md",
|
|
31
|
-
"LICENSE"
|
|
32
|
-
],
|
|
33
|
-
"engines": {
|
|
34
|
-
"node": ">=20"
|
|
35
|
-
},
|
|
36
|
-
"repository": {
|
|
37
|
-
"type": "git",
|
|
38
|
-
"url": "git+https://github.com/HECer/yoke.git"
|
|
39
|
-
},
|
|
40
|
-
"homepage": "https://github.com/HECer/yoke#readme",
|
|
41
|
-
"bugs": {
|
|
42
|
-
"url": "https://github.com/HECer/yoke/issues"
|
|
43
|
-
},
|
|
44
|
-
"keywords": [
|
|
45
|
-
"claude-code",
|
|
46
|
-
"codex",
|
|
47
|
-
"gemini-cli",
|
|
48
|
-
"agents",
|
|
49
|
-
"agentic-coding",
|
|
50
|
-
"harness",
|
|
51
|
-
"autonomous",
|
|
52
|
-
"ralph-loop",
|
|
53
|
-
"code-review",
|
|
54
|
-
"skills",
|
|
55
|
-
"agents-md",
|
|
56
|
-
"tdd",
|
|
57
|
-
"claude-plugin",
|
|
58
|
-
"claude-code-plugin",
|
|
59
|
-
"gemini-cli-extension"
|
|
60
|
-
],
|
|
61
|
-
"license": "MIT",
|
|
62
|
-
"publishConfig": {
|
|
63
|
-
"access": "public"
|
|
64
|
-
},
|
|
65
|
-
"scripts": {
|
|
66
|
-
"build": "tsc",
|
|
67
|
-
"lint": "tsc --noEmit",
|
|
68
|
-
"test": "vitest run",
|
|
69
|
-
"docs:check": "node scripts/release-metadata.mjs --check",
|
|
70
|
-
"docs:update": "node scripts/release-metadata.mjs --write",
|
|
71
|
-
"audit:ci": "npm audit --audit-level=high",
|
|
72
|
-
"package:check": "npm pack --dry-run",
|
|
73
|
-
"yoke": "tsx src/cli.ts",
|
|
74
|
-
"prepublishOnly": "npm run lint && npm run build && vitest run && npm run docs:check && npm run package:check"
|
|
75
|
-
},
|
|
76
|
-
"dependencies": {
|
|
77
|
-
"yaml": "^2.5.0",
|
|
78
|
-
"zod": "^3.23.8"
|
|
79
|
-
},
|
|
80
|
-
"devDependencies": {
|
|
81
|
-
"@types/node": "^22.7.0",
|
|
82
|
-
"smol-toml": "^1.7.0",
|
|
83
|
-
"tsx": "^4.19.1",
|
|
84
|
-
"typescript": "^5.6.2",
|
|
85
|
-
"vitest": "^4.1.10"
|
|
86
|
-
}
|
|
87
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"name": "@hecer/yoke",
|
|
3
|
+
"version": "1.11.0",
|
|
4
|
+
"description": "One harness, four agents, zero trust in \"done\" — cross-agent coding harness for Claude Code, Codex CLI, Gemini CLI, and Qwen Code: one skill canon, mechanical safety gates, an autonomous loop with screenshot/video proofs.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": {
|
|
7
|
+
"yoke": "dist/cli.js"
|
|
8
|
+
},
|
|
9
|
+
"files": [
|
|
10
|
+
"dist",
|
|
11
|
+
"canon",
|
|
12
|
+
".claude-plugin",
|
|
13
|
+
".codex-plugin",
|
|
14
|
+
"gemini-extension.json",
|
|
15
|
+
"agents",
|
|
16
|
+
"hooks",
|
|
17
|
+
"bench/README.md",
|
|
18
|
+
"bench/output-compaction.mjs",
|
|
19
|
+
"bench/RESULTS.md",
|
|
20
|
+
"bench/result-schema.mjs",
|
|
21
|
+
"bench/run.mjs",
|
|
22
|
+
"bench/run-large.mjs",
|
|
23
|
+
"bench/run-matrix.mjs",
|
|
24
|
+
"bench/analyze-routing-study.mjs",
|
|
25
|
+
"bench/fixtures",
|
|
26
|
+
"bench/results",
|
|
27
|
+
"docs",
|
|
28
|
+
"CHANGELOG.md",
|
|
29
|
+
"TODOS.md",
|
|
30
|
+
"README.md",
|
|
31
|
+
"LICENSE"
|
|
32
|
+
],
|
|
33
|
+
"engines": {
|
|
34
|
+
"node": ">=20"
|
|
35
|
+
},
|
|
36
|
+
"repository": {
|
|
37
|
+
"type": "git",
|
|
38
|
+
"url": "git+https://github.com/HECer/yoke.git"
|
|
39
|
+
},
|
|
40
|
+
"homepage": "https://github.com/HECer/yoke#readme",
|
|
41
|
+
"bugs": {
|
|
42
|
+
"url": "https://github.com/HECer/yoke/issues"
|
|
43
|
+
},
|
|
44
|
+
"keywords": [
|
|
45
|
+
"claude-code",
|
|
46
|
+
"codex",
|
|
47
|
+
"gemini-cli",
|
|
48
|
+
"agents",
|
|
49
|
+
"agentic-coding",
|
|
50
|
+
"harness",
|
|
51
|
+
"autonomous",
|
|
52
|
+
"ralph-loop",
|
|
53
|
+
"code-review",
|
|
54
|
+
"skills",
|
|
55
|
+
"agents-md",
|
|
56
|
+
"tdd",
|
|
57
|
+
"claude-plugin",
|
|
58
|
+
"claude-code-plugin",
|
|
59
|
+
"gemini-cli-extension"
|
|
60
|
+
],
|
|
61
|
+
"license": "MIT",
|
|
62
|
+
"publishConfig": {
|
|
63
|
+
"access": "public"
|
|
64
|
+
},
|
|
65
|
+
"scripts": {
|
|
66
|
+
"build": "tsc",
|
|
67
|
+
"lint": "tsc --noEmit",
|
|
68
|
+
"test": "vitest run",
|
|
69
|
+
"docs:check": "node scripts/release-metadata.mjs --check",
|
|
70
|
+
"docs:update": "node scripts/release-metadata.mjs --write",
|
|
71
|
+
"audit:ci": "npm audit --audit-level=high",
|
|
72
|
+
"package:check": "npm pack --dry-run",
|
|
73
|
+
"yoke": "tsx src/cli.ts",
|
|
74
|
+
"prepublishOnly": "npm run lint && npm run build && vitest run && npm run docs:check && npm run package:check"
|
|
75
|
+
},
|
|
76
|
+
"dependencies": {
|
|
77
|
+
"yaml": "^2.5.0",
|
|
78
|
+
"zod": "^3.23.8"
|
|
79
|
+
},
|
|
80
|
+
"devDependencies": {
|
|
81
|
+
"@types/node": "^22.7.0",
|
|
82
|
+
"smol-toml": "^1.7.0",
|
|
83
|
+
"tsx": "^4.19.1",
|
|
84
|
+
"typescript": "^5.6.2",
|
|
85
|
+
"vitest": "^4.1.10"
|
|
86
|
+
}
|
|
87
|
+
}
|