@frankzhang2026/opencode-android-orchestrator 0.6.1 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -1
- package/README.md +81 -22
- package/THIRD_PARTY_NOTICES.md +15 -11
- package/dist/cli.js +2 -0
- package/dist/cli.js.map +1 -1
- package/dist/config/verification-policy.d.ts +20 -0
- package/dist/config/verification-policy.d.ts.map +1 -0
- package/dist/config/verification-policy.js +59 -0
- package/dist/config/verification-policy.js.map +1 -0
- package/dist/doctor/installation.d.ts.map +1 -1
- package/dist/doctor/installation.js +46 -3
- package/dist/doctor/installation.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/installer/adaptive-templates.d.ts +9 -6
- package/dist/installer/adaptive-templates.d.ts.map +1 -1
- package/dist/installer/adaptive-templates.js +13 -0
- package/dist/installer/adaptive-templates.js.map +1 -1
- package/dist/installer/index.d.ts +1 -1
- package/dist/installer/index.d.ts.map +1 -1
- package/dist/installer/index.js +1 -1
- package/dist/installer/index.js.map +1 -1
- package/dist/installer/init.d.ts +1 -1
- package/dist/installer/init.d.ts.map +1 -1
- package/dist/installer/init.js.map +1 -1
- package/dist/installer/install-manifest.d.ts.map +1 -1
- package/dist/installer/install-manifest.js +13 -1
- package/dist/installer/install-manifest.js.map +1 -1
- package/dist/installer/opencode-config.d.ts +3 -4
- package/dist/installer/opencode-config.d.ts.map +1 -1
- package/dist/installer/opencode-config.js +1 -3
- package/dist/installer/opencode-config.js.map +1 -1
- package/dist/installer/upgrade.d.ts.map +1 -1
- package/dist/installer/upgrade.js +34 -2
- package/dist/installer/upgrade.js.map +1 -1
- package/dist/plugin/index.d.ts +2 -0
- package/dist/plugin/index.d.ts.map +1 -1
- package/dist/plugin/index.js +32 -0
- package/dist/plugin/index.js.map +1 -1
- package/docs/MIGRATION.md +27 -10
- package/docs/SECURITY.md +24 -7
- package/docs/TROUBLESHOOTING.md +13 -6
- package/package.json +3 -1
- package/resources/third-party/superpowers-v6.2.0/LICENSE +21 -0
- package/resources/third-party/superpowers-v6.2.0/PROVENANCE.md +21 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-brainstorming/SKILL.md +43 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-systematic-debugging/SKILL.md +288 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-systematic-debugging/references/condition-based-waiting-example.ts +158 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-systematic-debugging/references/condition-based-waiting.md +115 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-systematic-debugging/references/defense-in-depth.md +122 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-systematic-debugging/references/root-cause-tracing.md +169 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-systematic-debugging/scripts/find-polluter.sh +72 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-test-driven-development/SKILL.md +327 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-test-driven-development/references/writing-good-tests.md +197 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-verification-before-completion/SKILL.md +125 -0
- package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-writing-plans/SKILL.md +50 -0
- package/templates/.opencode/agents/scheduled-coder.md +3 -4
- package/templates/.opencode/agents/scheduled-planner.md +5 -5
- package/templates/.opencode/agents/scheduled-reviewer.md +1 -2
- package/templates/.opencode/skills/scheduled-quality-coder/SKILL.md +11 -9
- package/templates/.opencode/skills/scheduled-quality-orchestrator/SKILL.md +2 -1
- package/templates/.opencode/skills/scheduled-quality-reviewer/SKILL.md +4 -3
- package/templates/README.md +10 -3
- package/templates/automation/config.json +8 -9
- package/templates/automation/config.schema.json +6 -11
- package/templates/automation/task-contract.schema.json +7 -7
- package/templates/automation/tasks/TASK-TEMPLATE.json.example +6 -6
- package/templates/scripts/automation/claim-task.sh +23 -8
- package/templates/scripts/automation/lib.sh +37 -3
- package/templates/scripts/automation/orchestrate-task.sh +1 -1
- package/templates/scripts/automation/preflight.sh +0 -3
- package/templates/scripts/automation/tests/run-tests.sh +36 -15
- package/templates/scripts/automation/validate-contract.sh +7 -2
- package/templates/scripts/automation/verify-integration.sh +2 -9
- package/templates/scripts/automation/verify-task.sh +2 -9
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
# Writing Good Tests
|
|
2
|
+
|
|
3
|
+
**Load this reference when:** writing or changing tests, adding mocks, or
|
|
4
|
+
adding cleanup/helper methods for tests.
|
|
5
|
+
|
|
6
|
+
## Overview
|
|
7
|
+
|
|
8
|
+
A test exists to catch a specific break. Two principles govern everything
|
|
9
|
+
here:
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
1. Every test names the break it catches
|
|
13
|
+
2. Every test exercises the real thing
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Strict TDD produces both naturally: a test written first and watched
|
|
17
|
+
failing against real code has already proven it can fail, and only earns
|
|
18
|
+
a mock when the real dependency proves slow or external.
|
|
19
|
+
|
|
20
|
+
## Principle 1: Name the Break
|
|
21
|
+
|
|
22
|
+
Before writing the test body, answer: **what production change should
|
|
23
|
+
make this test fail — and is that change a bug or a decision?** A test
|
|
24
|
+
earns its place by catching a wrong branch, missing side effect, wrong
|
|
25
|
+
argument, boundary case, or broken contract.
|
|
26
|
+
|
|
27
|
+
**Derive expectations independently.** Use literals and hand-checked
|
|
28
|
+
fixtures; table-driven tests with literal `want` values are the preferred
|
|
29
|
+
shape. An expectation computed by the code under test — or its helpers —
|
|
30
|
+
passes no matter what that code does:
|
|
31
|
+
|
|
32
|
+
```typescript
|
|
33
|
+
// ❌ Mirror assertion: the same builder computes both sides — always true
|
|
34
|
+
const expected = buildSearchQuery({ tag: 'urgent' });
|
|
35
|
+
expect(buildSearchQuery({ tag: 'urgent' })).toBe(expected);
|
|
36
|
+
|
|
37
|
+
// ✅ Hand-derived literal
|
|
38
|
+
expect(buildSearchQuery({ tag: 'urgent' })).toBe('tag:"urgent"');
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
**No change detectors.** If only intentional decisions can fail a test —
|
|
42
|
+
a constant's value, exact message wording, private structure — it fires
|
|
43
|
+
on redesign and sleeps through bugs. Test the behavior that depends on
|
|
44
|
+
the decision: not `expect(MAX_RETRIES).toBe(5)` but "a failing call is
|
|
45
|
+
retried 5 times and the 6th attempt never happens."
|
|
46
|
+
|
|
47
|
+
**Behavior, not text.** Asserting that a script, skill, or config
|
|
48
|
+
contains an exact line proves only that the source is the source. Run
|
|
49
|
+
scripts against controlled inputs and assert outputs, side effects, or
|
|
50
|
+
exit codes. Documents that instruct agents are tested by the consuming
|
|
51
|
+
agent's behavior; prose for humans earns no test at all.
|
|
52
|
+
|
|
53
|
+
**Your code, not the framework.** Test the contract your code makes at
|
|
54
|
+
its boundaries — the route you register, the query you emit, the payload
|
|
55
|
+
you produce. Upstream mechanics are their maintainers' tests to write
|
|
56
|
+
(the classic: asserting your router invokes a registered handler — that
|
|
57
|
+
is the framework's test, not yours). When upstream behavior genuinely
|
|
58
|
+
surprised you, write one narrow characterization test naming the
|
|
59
|
+
assumption. The same boundary applies inside your code: constructors,
|
|
60
|
+
getters, constants, and trivial forwarding earn tests only when they
|
|
61
|
+
validate, normalize, default, derive, enforce, or cause side effects —
|
|
62
|
+
otherwise assert the first consumer-visible result that depends on them.
|
|
63
|
+
|
|
64
|
+
### Gate Function
|
|
65
|
+
|
|
66
|
+
```
|
|
67
|
+
BEFORE writing the test body:
|
|
68
|
+
Name the production change that would make this test fail.
|
|
69
|
+
|
|
70
|
+
Cannot name one → redesign around an observable behavior
|
|
71
|
+
"The source text changed" → run the artifact and assert its effects
|
|
72
|
+
Only intentional decisions → change detector; test the behavior
|
|
73
|
+
that depends on the decision
|
|
74
|
+
|
|
75
|
+
Confirm the expected value is derived without the code under test.
|
|
76
|
+
IF it reuses the code's logic or helpers:
|
|
77
|
+
Replace it with a literal or hand-checked fixture
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Principle 2: Exercise the Real Thing
|
|
81
|
+
|
|
82
|
+
**The mock earns no assertions.** A mock assertion passes when the mock
|
|
83
|
+
is present and fails when it is absent — it says nothing about the
|
|
84
|
+
component. Assert the real component's behavior; if the mock is what you
|
|
85
|
+
are checking, unmock it or delete the assertion.
|
|
86
|
+
|
|
87
|
+
```typescript
|
|
88
|
+
// ✅ Real behavior
|
|
89
|
+
expect(screen.getByRole('navigation')).toBeInTheDocument();
|
|
90
|
+
|
|
91
|
+
// ❌ Mock existence
|
|
92
|
+
expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument();
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
**your human partner's correction:** "Are we testing the behavior of a
|
|
96
|
+
mock?"
|
|
97
|
+
|
|
98
|
+
**Mock at the right level.** Learn every side effect of the real method
|
|
99
|
+
before replacing it; mock the slow or external operation and keep what
|
|
100
|
+
the test depends on real. When unsure, run the test against the real
|
|
101
|
+
implementation first and observe what actually needs to happen.
|
|
102
|
+
|
|
103
|
+
```typescript
|
|
104
|
+
// ❌ The mock swallows the config write that duplicate detection reads
|
|
105
|
+
vi.mock('ToolCatalog', () => ({
|
|
106
|
+
discoverAndCacheTools: vi.fn().mockResolvedValue(undefined)
|
|
107
|
+
}));
|
|
108
|
+
|
|
109
|
+
// ✅ Mock only the slow server startup; the config write stays real
|
|
110
|
+
vi.mock('MCPServerManager');
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
**Make doubles specific.** When arguments, call counts, or ordering are
|
|
114
|
+
part of the contract, assert them — a fake that accepts anything verifies
|
|
115
|
+
nothing. Give each branch (success, error, malformed) its own fixture or
|
|
116
|
+
spy, so the wrong branch cannot satisfy the expectation.
|
|
117
|
+
|
|
118
|
+
**Mirror real data completely.** Mock the complete structure as it exists
|
|
119
|
+
in reality — all documented fields — not just the ones your test reads.
|
|
120
|
+
Partial mocks fail silently when downstream code reads an omitted field:
|
|
121
|
+
the test passes while integration breaks.
|
|
122
|
+
|
|
123
|
+
**Production classes carry production methods only.** Cleanup that only
|
|
124
|
+
tests need lives in test utilities, never as a `destroy()` on the
|
|
125
|
+
production class. Ask: is this method called only from tests? Does this
|
|
126
|
+
class own this resource's lifecycle? Wrong answers → test utility.
|
|
127
|
+
|
|
128
|
+
**Prefer real components over complex mocks.** When mock setup outgrows
|
|
129
|
+
the test logic, mocks miss methods the real components have, or tests
|
|
130
|
+
break when the mock changes, switch to an integration test with real
|
|
131
|
+
components. **your human partner's question:** "Do we need to be using a
|
|
132
|
+
mock here?"
|
|
133
|
+
|
|
134
|
+
### Gate Function
|
|
135
|
+
|
|
136
|
+
```
|
|
137
|
+
BEFORE adding a mock or test helper:
|
|
138
|
+
List the real method's side effects; keep the ones the test
|
|
139
|
+
depends on real — mock the slow/external level below them.
|
|
140
|
+
|
|
141
|
+
Mock responses mirror the complete real structure.
|
|
142
|
+
|
|
143
|
+
A method only tests call lives in test utilities, not production.
|
|
144
|
+
|
|
145
|
+
About to assert on the mock itself?
|
|
146
|
+
Unmock it or delete the assertion.
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
## Tests Ship With the Implementation
|
|
150
|
+
|
|
151
|
+
The TDD cycle — failing test, minimal implementation, refactor — is what
|
|
152
|
+
"complete" means. Ship the tests the behavior needs and only those:
|
|
153
|
+
trivial code and human prose earn none, and a test written to satisfy
|
|
154
|
+
process costs maintenance forever.
|
|
155
|
+
|
|
156
|
+
## The Mutation Check
|
|
157
|
+
|
|
158
|
+
Before finishing, mentally mutate the production code; at least one test
|
|
159
|
+
should fail for each realistic mutation:
|
|
160
|
+
|
|
161
|
+
- Wrong constant or argument
|
|
162
|
+
- Wrong branch handler
|
|
163
|
+
- Missing state change or side effect
|
|
164
|
+
- Empty or default return
|
|
165
|
+
- Missing validation for zero, empty, nil, unauthorized, or malformed input
|
|
166
|
+
|
|
167
|
+
A mutation nothing catches marks the behavior as unprotected — or the
|
|
168
|
+
test as tautological.
|
|
169
|
+
|
|
170
|
+
## Quick Reference
|
|
171
|
+
|
|
172
|
+
| When you... | Do |
|
|
173
|
+
|-------------|-----|
|
|
174
|
+
| Write any test | Name the break it catches — a bug, not a decision |
|
|
175
|
+
| Build an expected value | Derive it by hand; never with the code under test |
|
|
176
|
+
| Test a script or document | Run it / pressure-test its consumer; never grep its text |
|
|
177
|
+
| Reach for a dependency test | Test your boundary contract, not their documented mechanics |
|
|
178
|
+
| Want to assert on a mocked element | Test the real component, or unmock it |
|
|
179
|
+
| Are about to mock a method | Learn its side effects; mock the slow/external level |
|
|
180
|
+
| Build a mock response | Mirror the real structure completely |
|
|
181
|
+
| Need cleanup only tests use | Put it in test utilities |
|
|
182
|
+
| Watch mock setup balloon | Switch to an integration test with real components |
|
|
183
|
+
| Finish a test file | Run the mutation check |
|
|
184
|
+
|
|
185
|
+
## Warning Signs
|
|
186
|
+
|
|
187
|
+
- Setup and assertion share the same object, guaranteeing equality
|
|
188
|
+
- The test can fail only through a panic, crash, or missing selector
|
|
189
|
+
- The test fails on every intentional change, never on accidental breakage
|
|
190
|
+
- Expected values are hidden behind loops, builders, or helpers
|
|
191
|
+
- The test greps source text, or asserts a removed symbol stays removed
|
|
192
|
+
- The test would still matter if only the framework remained
|
|
193
|
+
- The test exists for coverage, checking no side effect or outcome
|
|
194
|
+
- An assertion checks a `*-mock` test ID, or fails if you remove the mock
|
|
195
|
+
- A method is called only from test files
|
|
196
|
+
- Mock setup is more than half the test, or you can't explain why the mock is needed
|
|
197
|
+
- Mocking "just to be safe"
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: android-orchestrator-verification-before-completion
|
|
3
|
+
description: Require fresh command evidence before an Android Orchestrator Coder or Reviewer reports completion or approval
|
|
4
|
+
license: MIT
|
|
5
|
+
compatibility: opencode
|
|
6
|
+
metadata:
|
|
7
|
+
upstream: obra/superpowers@v6.2.0
|
|
8
|
+
workflow: scheduled-coding
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Verification Before Completion
|
|
12
|
+
|
|
13
|
+
## Overview
|
|
14
|
+
|
|
15
|
+
**Core principle:** Evidence before claims, always.
|
|
16
|
+
|
|
17
|
+
**Violating the letter of this rule is violating the spirit of this rule.**
|
|
18
|
+
|
|
19
|
+
## The Iron Law
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
NO COMPLETION CLAIMS WITHOUT FRESH VERIFICATION EVIDENCE
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
If you haven't run the verification command in this message, you cannot claim it passes.
|
|
26
|
+
|
|
27
|
+
## The Gate Function
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
BEFORE claiming any status or expressing satisfaction:
|
|
31
|
+
|
|
32
|
+
1. IDENTIFY: What command proves this claim?
|
|
33
|
+
2. RUN: Execute the FULL command (fresh, complete)
|
|
34
|
+
3. READ: Full output, check exit code, count failures
|
|
35
|
+
4. VERIFY: Does output confirm the claim?
|
|
36
|
+
- If NO: State actual status with evidence
|
|
37
|
+
- If YES: State claim WITH evidence
|
|
38
|
+
5. ONLY THEN: Make the claim
|
|
39
|
+
|
|
40
|
+
Skip any step = lying, not verifying
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Common Failures
|
|
44
|
+
|
|
45
|
+
| Claim | Requires | Not Sufficient |
|
|
46
|
+
|-------|----------|----------------|
|
|
47
|
+
| Tests pass | Test command output: 0 failures | Previous run, "should pass" |
|
|
48
|
+
| Linter clean | Linter output: 0 errors | Partial check, extrapolation |
|
|
49
|
+
| Build succeeds | Build command: exit 0 | Linter passing, logs look good |
|
|
50
|
+
| Bug fixed | Test original symptom: passes | Code changed, assumed fixed |
|
|
51
|
+
| Regression test works | Red-green cycle verified | Test passes once |
|
|
52
|
+
| Agent completed | VCS diff shows changes | Agent reports "success" |
|
|
53
|
+
| Requirements met | Line-by-line checklist | Tests passing |
|
|
54
|
+
|
|
55
|
+
## Red Flags - STOP
|
|
56
|
+
|
|
57
|
+
- Using "should", "probably", "seems to"
|
|
58
|
+
- Expressing satisfaction before verification ("Great!", "Perfect!", "Done!", etc.)
|
|
59
|
+
- About to commit/push/PR without verification
|
|
60
|
+
- Trusting agent success reports
|
|
61
|
+
- Relying on partial verification
|
|
62
|
+
- Thinking "just this once"
|
|
63
|
+
- Tired and wanting work over
|
|
64
|
+
- **ANY wording implying success without having run verification**
|
|
65
|
+
|
|
66
|
+
## Rationalization Prevention
|
|
67
|
+
|
|
68
|
+
| Excuse | Reality |
|
|
69
|
+
|--------|---------|
|
|
70
|
+
| "Should work now" | RUN the verification |
|
|
71
|
+
| "I'm confident" | Confidence ≠ evidence |
|
|
72
|
+
| "Just this once" | No exceptions |
|
|
73
|
+
| "Linter passed" | Linter ≠ compiler |
|
|
74
|
+
| "Agent said success" | Verify independently |
|
|
75
|
+
| "I'm tired" | Exhaustion ≠ excuse |
|
|
76
|
+
| "Partial check is enough" | Partial proves nothing |
|
|
77
|
+
| "Different words so rule doesn't apply" | Spirit over letter |
|
|
78
|
+
|
|
79
|
+
## Key Patterns
|
|
80
|
+
|
|
81
|
+
**Tests:**
|
|
82
|
+
```
|
|
83
|
+
✅ [Run test command] [See: 34/34 pass] "All tests pass"
|
|
84
|
+
❌ "Should pass now" / "Looks correct"
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
**Regression tests (TDD Red-Green):**
|
|
88
|
+
```
|
|
89
|
+
✅ Write → Run (pass) → Revert fix → Run (MUST FAIL) → Restore → Run (pass)
|
|
90
|
+
❌ "I've written a regression test" (without red-green verification)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
**Build:**
|
|
94
|
+
```
|
|
95
|
+
✅ [Run build] [See: exit 0] "Build passes"
|
|
96
|
+
❌ "Linter passed" (linter doesn't check compilation)
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
**Requirements:**
|
|
100
|
+
```
|
|
101
|
+
✅ Re-read plan → Create checklist → Verify each → Report gaps or completion
|
|
102
|
+
❌ "Tests pass, phase complete"
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
**Agent delegation:**
|
|
106
|
+
```
|
|
107
|
+
✅ Agent reports success → Check VCS diff → Verify changes → Report actual state
|
|
108
|
+
❌ Trust agent report
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## When To Apply
|
|
112
|
+
|
|
113
|
+
**ALWAYS before:**
|
|
114
|
+
- ANY variation of success/completion claims
|
|
115
|
+
- ANY expression of satisfaction
|
|
116
|
+
- ANY positive statement about work state
|
|
117
|
+
- Committing, PR creation, task completion
|
|
118
|
+
- Moving to next task
|
|
119
|
+
- Delegating to agents
|
|
120
|
+
|
|
121
|
+
**Rule applies to:**
|
|
122
|
+
- Exact phrases
|
|
123
|
+
- Paraphrases and synonyms
|
|
124
|
+
- Implications of success
|
|
125
|
+
- ANY communication suggesting completion/correctness
|
package/resources/third-party/superpowers-v6.2.0/skills/android-orchestrator-writing-plans/SKILL.md
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: android-orchestrator-writing-plans
|
|
3
|
+
description: Convert an approved Android Orchestrator proposal into an implementation-ready plan that matches its sealed task contract
|
|
4
|
+
license: MIT
|
|
5
|
+
compatibility: opencode
|
|
6
|
+
metadata:
|
|
7
|
+
upstream: obra/superpowers@v6.2.0
|
|
8
|
+
workflow: scheduled-coding
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# Android Orchestrator writing plans
|
|
12
|
+
|
|
13
|
+
Write a plan detailed enough for the restricted Coder to implement without
|
|
14
|
+
guessing. The approved proposal is the scope ceiling.
|
|
15
|
+
|
|
16
|
+
## Required output
|
|
17
|
+
|
|
18
|
+
Create only the plan path selected by `scheduled-quality-orchestrator`:
|
|
19
|
+
`docs/plans/<TASK-ID>.md`. The matching JSON contract is created separately at
|
|
20
|
+
`automation/tasks/<TASK-ID>.json`; both artifacts must describe the same task.
|
|
21
|
+
|
|
22
|
+
The plan must include:
|
|
23
|
+
|
|
24
|
+
- task ID and title;
|
|
25
|
+
- current and desired observable behavior;
|
|
26
|
+
- acceptance criteria and edge cases;
|
|
27
|
+
- exact files to create or modify;
|
|
28
|
+
- implementation sequence with concrete symbols and interfaces;
|
|
29
|
+
- the first failing behavior test and expected RED reason;
|
|
30
|
+
- focused and full verification commands;
|
|
31
|
+
- allowed paths, forbidden paths, maximum changed-file count, and non-goals;
|
|
32
|
+
- device/emulator policy and any residual risk.
|
|
33
|
+
|
|
34
|
+
## Plan quality
|
|
35
|
+
|
|
36
|
+
- Use small ordered steps: test, verify RED, minimal implementation, verify
|
|
37
|
+
GREEN, then the configured quality gate.
|
|
38
|
+
- Derive expected values independently from production code.
|
|
39
|
+
- Do not use placeholders such as TODO, TBD, “add suitable handling”, or
|
|
40
|
+
“similar to the previous step”.
|
|
41
|
+
- Keep names, signatures, resources, and paths consistent across every step
|
|
42
|
+
and with the task contract.
|
|
43
|
+
- Do not introduce unapproved refactors or dependencies.
|
|
44
|
+
|
|
45
|
+
## Handoff
|
|
46
|
+
|
|
47
|
+
Do not offer alternate execution modes, dispatch subagents, commit, or start
|
|
48
|
+
implementation. Return control to `scheduled-quality-orchestrator`, which
|
|
49
|
+
validates and seals the plan and contract before the separate Coder and
|
|
50
|
+
Reviewer sessions can run.
|
|
@@ -77,11 +77,10 @@ permission:
|
|
|
77
77
|
list: allow
|
|
78
78
|
skill:
|
|
79
79
|
"*": deny
|
|
80
|
-
"using-superpowers": allow
|
|
81
80
|
"scheduled-quality-coder": allow
|
|
82
|
-
"test-driven-development": allow
|
|
83
|
-
"systematic-debugging": allow
|
|
84
|
-
"verification-before-completion": allow
|
|
81
|
+
"android-orchestrator-test-driven-development": allow
|
|
82
|
+
"android-orchestrator-systematic-debugging": allow
|
|
83
|
+
"android-orchestrator-verification-before-completion": allow
|
|
85
84
|
schedule_job: deny
|
|
86
85
|
list_jobs: deny
|
|
87
86
|
get_version: deny
|
|
@@ -68,9 +68,8 @@ permission:
|
|
|
68
68
|
list: allow
|
|
69
69
|
skill:
|
|
70
70
|
"*": deny
|
|
71
|
-
"
|
|
72
|
-
"
|
|
73
|
-
"writing-plans": allow
|
|
71
|
+
"android-orchestrator-brainstorming": allow
|
|
72
|
+
"android-orchestrator-writing-plans": allow
|
|
74
73
|
"scheduled-quality-orchestrator": allow
|
|
75
74
|
question: allow
|
|
76
75
|
schedule_job: deny
|
|
@@ -96,8 +95,9 @@ user provides a natural-language coding request. Remain the conversational
|
|
|
96
95
|
coordinator through planning, contract approval, unattended execution, final
|
|
97
96
|
human acceptance, and local integration into the recorded original branch.
|
|
98
97
|
|
|
99
|
-
Load `scheduled-quality-orchestrator`, `brainstorming`,
|
|
100
|
-
before taking action, then follow the
|
|
98
|
+
Load `scheduled-quality-orchestrator`, `android-orchestrator-brainstorming`,
|
|
99
|
+
and `android-orchestrator-writing-plans` before taking action, then follow the
|
|
100
|
+
orchestrator skill literally. Run the
|
|
101
101
|
source preflight before planning. Inspect the current repository code and
|
|
102
102
|
tests, then interactively narrow the request to exactly one small, observable
|
|
103
103
|
behavior change. Ask for clarification when scope, acceptance behavior, edge
|
|
@@ -54,9 +54,8 @@ permission:
|
|
|
54
54
|
list: allow
|
|
55
55
|
skill:
|
|
56
56
|
"*": deny
|
|
57
|
-
"using-superpowers": allow
|
|
58
57
|
"scheduled-quality-reviewer": allow
|
|
59
|
-
"verification-before-completion": allow
|
|
58
|
+
"android-orchestrator-verification-before-completion": allow
|
|
60
59
|
schedule_job: deny
|
|
61
60
|
list_jobs: deny
|
|
62
61
|
get_version: deny
|
|
@@ -9,8 +9,8 @@ metadata:
|
|
|
9
9
|
|
|
10
10
|
# Scheduled quality coder
|
|
11
11
|
|
|
12
|
-
Execute one task contract
|
|
13
|
-
|
|
12
|
+
Execute one task contract through the bundled, deterministic, non-interactive
|
|
13
|
+
Android workflow. The scripts are the source of truth for state;
|
|
14
14
|
your prose is never proof of completion.
|
|
15
15
|
|
|
16
16
|
## Required input
|
|
@@ -28,17 +28,18 @@ it with `./scripts/automation/block-task.sh <TASK-ID> <reason>` before stopping.
|
|
|
28
28
|
|
|
29
29
|
## Mandatory sequence
|
|
30
30
|
|
|
31
|
-
1. Load `test-driven-development` and
|
|
32
|
-
`verification-before-completion`. Do not load any other implementation
|
|
31
|
+
1. Load `android-orchestrator-test-driven-development` and
|
|
32
|
+
`android-orchestrator-verification-before-completion`. Do not load any other implementation
|
|
33
33
|
workflow skill.
|
|
34
34
|
2. Run `./scripts/automation/status.sh <TASK-ID>` and read the contract.
|
|
35
35
|
3. Branch by deterministic state:
|
|
36
36
|
|
|
37
37
|
- For `PENDING`, run `./scripts/automation/claim-task.sh <TASK-ID>`. It
|
|
38
38
|
performs preflight, verifies that the only orchestration-visible initial
|
|
39
|
-
changes are the two sealed, uncommitted planning artifacts, captures
|
|
40
|
-
green baseline
|
|
41
|
-
|
|
39
|
+
changes are the two sealed, uncommitted planning artifacts, captures a
|
|
40
|
+
green unit-test baseline when `unitTestsEnabled` is true or records the
|
|
41
|
+
configured skip otherwise, and changes the task to `CODING`. The status
|
|
42
|
+
JSON's `runtime.effectiveWorktreeAllowlist` contains human-owned local paths that
|
|
42
43
|
are outside the task; do not edit, stage, report, or reason from those
|
|
43
44
|
paths or from `.automation-worktree-allowlist`. Never edit, stage, or
|
|
44
45
|
remove the planning artifacts; the integrator will include them in the
|
|
@@ -64,7 +65,7 @@ it with `./scripts/automation/block-task.sh <TASK-ID> <reason>` before stopping.
|
|
|
64
65
|
7. Run `./scripts/automation/quality-gate.sh <TASK-ID>`.
|
|
65
66
|
8. If the first gate attempt in the current coding cycle fails while state
|
|
66
67
|
remains `CODING`, load
|
|
67
|
-
`systematic-debugging`, diagnose the root cause, and make at most one fix
|
|
68
|
+
`android-orchestrator-systematic-debugging`, diagnose the root cause, and make at most one fix
|
|
68
69
|
loop. Then run the gate once more. If it fails again, stop in
|
|
69
70
|
`TEST_FAILED`.
|
|
70
71
|
9. When the gate succeeds, report the changed files and evidence paths. Do not
|
|
@@ -91,7 +92,8 @@ human can revise and requeue the contract.
|
|
|
91
92
|
|
|
92
93
|
## Forbidden capabilities
|
|
93
94
|
|
|
94
|
-
Do not invoke `
|
|
95
|
+
Do not invoke `android-orchestrator-brainstorming`,
|
|
96
|
+
`android-orchestrator-writing-plans`, `using-git-worktrees`,
|
|
95
97
|
`finishing-a-development-branch`, `requesting-code-review`, parallel agents, or
|
|
96
98
|
subagent-driven development. Planning and approval happen before this session;
|
|
97
99
|
review happens in a separate fresh read-only session.
|
|
@@ -20,7 +20,8 @@ approvals; state files, hashes, tests, and Git checks grant execution.
|
|
|
20
20
|
If `ANDROID_HOME` is missing, the working tree is dirty, the branch is detached,
|
|
21
21
|
Git identity is missing, or OpenCode discovery is unsafe, report the exact
|
|
22
22
|
blocker and stop.
|
|
23
|
-
2. Use brainstorming and
|
|
23
|
+
2. Use `android-orchestrator-brainstorming` and
|
|
24
|
+
`android-orchestrator-writing-plans` to produce one bounded proposal. After
|
|
24
25
|
displaying it, immediately call `question` once with `multiple: false` and
|
|
25
26
|
`custom: false`:
|
|
26
27
|
|
|
@@ -15,7 +15,7 @@ you review.
|
|
|
15
15
|
|
|
16
16
|
## Mandatory sequence
|
|
17
17
|
|
|
18
|
-
1. Load `verification-before-completion`.
|
|
18
|
+
1. Load `android-orchestrator-verification-before-completion`.
|
|
19
19
|
2. Require exactly one task ID or the compatibility token
|
|
20
20
|
`NEXT_REVIEWING`. For the selector token, first run
|
|
21
21
|
`./scripts/automation/select-task.sh REVIEWING` and continue only if
|
|
@@ -51,8 +51,9 @@ you review.
|
|
|
51
51
|
|
|
52
52
|
`./scripts/automation/submit-review.sh <TASK-ID> CHANGES_REQUESTED <summary>`
|
|
53
53
|
|
|
54
|
-
The script reruns the focused tests
|
|
55
|
-
|
|
54
|
+
The script reruns the focused tests and full unit suite when
|
|
55
|
+
`unitTestsEnabled` is true, always runs the debug build, runs Android lint
|
|
56
|
+
when `lintEnabled` is true, and records that fresh output. Therefore do not
|
|
56
57
|
run those Gradle commands separately before an approval submission. Run an
|
|
57
58
|
individual verification command only to diagnose a failed submission, and
|
|
58
59
|
always reserve a step for the final `submit-review.sh` call.
|
package/templates/README.md
CHANGED
|
@@ -14,7 +14,7 @@ Migrated template roots:
|
|
|
14
14
|
- `.opencode/skills`: the three `scheduled-quality-*` skills
|
|
15
15
|
- `scripts/automation`: all 29 deterministic V3 Bash transactions and their
|
|
16
16
|
test runner, preserved as executable files
|
|
17
|
-
- `automation`: the portable
|
|
17
|
+
- `automation`: the portable V4 configuration render source, both JSON Schemas,
|
|
18
18
|
and the task contract example
|
|
19
19
|
- `docs/plans/README.md`: the human-approved plan authoring contract
|
|
20
20
|
- `AGENTS.md.fragment`: a bounded managed block for non-destructive
|
|
@@ -38,7 +38,10 @@ adaptive render. This covers product flavors without shipping a project name,
|
|
|
38
38
|
an absolute configuration path, or a hand-maintained JSON file. The packaged
|
|
39
39
|
matrix remains the deterministic fallback for the read-only planning API;
|
|
40
40
|
callers may still pass an explicit matrix when they intentionally need a custom
|
|
41
|
-
policy.
|
|
41
|
+
policy. Verification switches live in `automation/config.json`:
|
|
42
|
+
`unitTestsEnabled` defaults to true and `lintEnabled` defaults to false.
|
|
43
|
+
Discovered task lists are retained even while their corresponding gate is
|
|
44
|
+
disabled.
|
|
42
45
|
|
|
43
46
|
After read-only planning succeeds, `init` creates a comment-only
|
|
44
47
|
`.automation-worktree-allowlist` if it is missing and preserves any existing
|
|
@@ -47,7 +50,11 @@ resource manifest so its exact-path entries can change without causing package
|
|
|
47
50
|
integrity drift; failed first-install verification removes a newly bootstrapped
|
|
48
51
|
file during rollback.
|
|
49
52
|
|
|
50
|
-
The legacy Scheduler field
|
|
53
|
+
The legacy Scheduler field and external Superpowers plugin dependency have
|
|
54
|
+
been removed. Five namespaced workflow skills are loaded directly from the
|
|
55
|
+
Orchestrator npm package; the optional Superpowers browser companion is not
|
|
56
|
+
distributed.
|
|
57
|
+
|
|
51
58
|
The scope gates consume generated source-set arrays, and the Shell test fixture
|
|
52
59
|
uses a neutral custom module and package. No shipped template contains a local
|
|
53
60
|
absolute path or project-specific package name.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"schemaVersion":
|
|
2
|
+
"schemaVersion": 4,
|
|
3
3
|
"enabled": true,
|
|
4
4
|
"mode": "orchestrated",
|
|
5
5
|
"workspaceStrategy": "inPlaceExclusive",
|
|
@@ -8,6 +8,8 @@
|
|
|
8
8
|
"maxFixLoops": 1,
|
|
9
9
|
"maxReviewCycles": 1,
|
|
10
10
|
"maxReviewerRestarts": 2,
|
|
11
|
+
"unitTestsEnabled": true,
|
|
12
|
+
"lintEnabled": false,
|
|
11
13
|
"longCommandTimeoutMs": 1800000,
|
|
12
14
|
"autoCleanupWorktrees": true,
|
|
13
15
|
"pushAfterAcceptance": false,
|
|
@@ -18,18 +20,15 @@
|
|
|
18
20
|
"abort": "中止任务,封存修改并恢复原分支。",
|
|
19
21
|
"resume": "恢复任务,重新捕获基线并继续自动执行。"
|
|
20
22
|
},
|
|
21
|
-
"plugins": {
|
|
22
|
-
"superpowers": "superpowers@git+https://github.com/obra/superpowers.git#v6.2.0"
|
|
23
|
-
},
|
|
24
23
|
"requiredSkills": [
|
|
25
|
-
"brainstorming",
|
|
26
|
-
"writing-plans",
|
|
24
|
+
"android-orchestrator-brainstorming",
|
|
25
|
+
"android-orchestrator-writing-plans",
|
|
27
26
|
"scheduled-quality-orchestrator",
|
|
28
27
|
"scheduled-quality-coder",
|
|
29
28
|
"scheduled-quality-reviewer",
|
|
30
|
-
"test-driven-development",
|
|
31
|
-
"systematic-debugging",
|
|
32
|
-
"verification-before-completion"
|
|
29
|
+
"android-orchestrator-test-driven-development",
|
|
30
|
+
"android-orchestrator-systematic-debugging",
|
|
31
|
+
"android-orchestrator-verification-before-completion"
|
|
33
32
|
],
|
|
34
33
|
"gradleVerification": {
|
|
35
34
|
"fullUnitTestTasks": [
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
-
"$id": "urn:frankzhang2026:opencode-android-orchestrator:automation-config:
|
|
3
|
+
"$id": "urn:frankzhang2026:opencode-android-orchestrator:automation-config:v4",
|
|
4
4
|
"title": "OpenCode task orchestration configuration",
|
|
5
5
|
"type": "object",
|
|
6
6
|
"additionalProperties": false,
|
|
@@ -36,18 +36,19 @@
|
|
|
36
36
|
"maxFixLoops",
|
|
37
37
|
"maxReviewCycles",
|
|
38
38
|
"maxReviewerRestarts",
|
|
39
|
+
"unitTestsEnabled",
|
|
40
|
+
"lintEnabled",
|
|
39
41
|
"longCommandTimeoutMs",
|
|
40
42
|
"autoCleanupWorktrees",
|
|
41
43
|
"pushAfterAcceptance",
|
|
42
44
|
"approvalPhrases",
|
|
43
|
-
"plugins",
|
|
44
45
|
"requiredSkills",
|
|
45
46
|
"gradleVerification",
|
|
46
47
|
"androidProject",
|
|
47
48
|
"protectedPaths"
|
|
48
49
|
],
|
|
49
50
|
"properties": {
|
|
50
|
-
"schemaVersion": { "const":
|
|
51
|
+
"schemaVersion": { "const": 4 },
|
|
51
52
|
"enabled": { "type": "boolean" },
|
|
52
53
|
"mode": { "enum": ["shadow", "orchestrated"] },
|
|
53
54
|
"workspaceStrategy": {
|
|
@@ -58,6 +59,8 @@
|
|
|
58
59
|
"maxFixLoops": { "type": "integer", "minimum": 0, "maximum": 1 },
|
|
59
60
|
"maxReviewCycles": { "type": "integer", "minimum": 0, "maximum": 2 },
|
|
60
61
|
"maxReviewerRestarts": { "type": "integer", "minimum": 0, "maximum": 3 },
|
|
62
|
+
"unitTestsEnabled": { "type": "boolean", "default": true },
|
|
63
|
+
"lintEnabled": { "type": "boolean", "default": false },
|
|
61
64
|
"longCommandTimeoutMs": {
|
|
62
65
|
"type": "integer",
|
|
63
66
|
"minimum": 120000,
|
|
@@ -77,14 +80,6 @@
|
|
|
77
80
|
"resume": { "type": "string", "minLength": 4 }
|
|
78
81
|
}
|
|
79
82
|
},
|
|
80
|
-
"plugins": {
|
|
81
|
-
"type": "object",
|
|
82
|
-
"additionalProperties": false,
|
|
83
|
-
"required": ["superpowers"],
|
|
84
|
-
"properties": {
|
|
85
|
-
"superpowers": { "type": "string", "minLength": 1 }
|
|
86
|
-
}
|
|
87
|
-
},
|
|
88
83
|
"requiredSkills": {
|
|
89
84
|
"type": "array",
|
|
90
85
|
"minItems": 6,
|