@mmerterden/multi-agent-pipeline 16.27.0 → 16.29.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +144 -1
- package/README.md +4 -4
- package/README.tr.md +3 -3
- package/docs/architecture.md +3 -3
- package/docs/ecosystem.md +5 -5
- package/install/claude.mjs +17 -0
- package/package.json +4 -4
- package/pipeline/commands/multi-agent/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/analysis-jira/SKILL.md +93 -0
- package/pipeline/commands/multi-agent/doctor/SKILL.md +78 -0
- package/pipeline/commands/multi-agent/help/SKILL.md +2 -0
- package/pipeline/commands/multi-agent/issue/SKILL.md +1 -0
- package/pipeline/commands/multi-agent/jira/SKILL.md +3 -0
- package/pipeline/commands/multi-agent/log/SKILL.md +7 -1
- package/pipeline/commands/multi-agent/review-issue/SKILL.md +1 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +14 -1
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -9
- package/pipeline/commands/multi-agent/update/SKILL.md +12 -0
- package/pipeline/lib/_jira-auth.sh +99 -0
- package/pipeline/lib/analysis-jira-write.sh +203 -0
- package/pipeline/lib/issue-fetcher.sh +138 -9
- package/pipeline/lib/multi-repo-pipeline.sh +8 -0
- package/pipeline/multi-agent-refs/analysis/evidence.md +1 -1
- package/pipeline/multi-agent-refs/analysis/intake.md +11 -2
- package/pipeline/multi-agent-refs/analysis/locked.md +1 -0
- package/pipeline/multi-agent-refs/analysis/redesign.md +112 -0
- package/pipeline/multi-agent-refs/analysis/render.md +6 -1
- package/pipeline/multi-agent-refs/analysis/resolve.md +1 -0
- package/pipeline/multi-agent-refs/analysis/review.md +15 -0
- package/pipeline/multi-agent-refs/analysis/synthesis.md +1 -1
- package/pipeline/multi-agent-refs/analysis-template-corporate.md +3 -3
- package/pipeline/multi-agent-refs/analysis-template.md +36 -0
- package/pipeline/multi-agent-refs/cross-cli-contract.md +6 -3
- package/pipeline/multi-agent-refs/features/analysis-jira.md +128 -0
- package/pipeline/multi-agent-refs/features/doctor.md +197 -0
- package/pipeline/multi-agent-refs/features/jira-context.md +101 -0
- package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +9 -7
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +1 -1
- package/pipeline/multi-agent-refs/picker-contract.md +35 -0
- package/pipeline/multi-agent-refs/readiness-review.md +1 -1
- package/pipeline/multi-agent-refs/tracker-contract.md +5 -1
- package/pipeline/preferences-template.json +1 -1
- package/pipeline/schemas/agent-state.schema.json +24 -0
- package/pipeline/schemas/analysis-spec.schema.json +336 -95
- package/pipeline/schemas/prefs.schema.json +86 -2
- package/pipeline/scripts/analysis-story-tree.mjs +441 -0
- package/pipeline/scripts/anonymize-findings.mjs +24 -0
- package/pipeline/scripts/build-references.mjs +10 -6
- package/pipeline/scripts/council-view.mjs +144 -0
- package/pipeline/scripts/doctor.mjs +758 -0
- package/pipeline/scripts/phase-tracker.sh +100 -3
- package/pipeline/scripts/scan-agent-config.sh +48 -10
- package/pipeline/scripts/skill-siblings.mjs +42 -9
- package/pipeline/scripts/validate-analysis-doc.mjs +371 -7
- package/pipeline/scripts/validate-analysis.mjs +7 -5
- package/pipeline/skills/shared/core/multi-agent-analysis-jira/SKILL.md +94 -0
- package/pipeline/skills/shared/core/multi-agent-doctor/SKILL.md +79 -0
- package/pipeline/skills/shared/core/multi-agent-issue/SKILL.md +1 -0
- package/pipeline/skills/shared/core/multi-agent-review-issue/SKILL.md +1 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +13 -0
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +9 -6
- package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +18 -0
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
# doctor - the check registry
|
|
2
|
+
|
|
3
|
+
Every check `/multi-agent:doctor` can report has a `### <id>` heading here, and
|
|
4
|
+
`doctor.mjs --list-checks` prints exactly the same set. The equality is checked
|
|
5
|
+
in both directions by `smoke-doctor.sh`: a check that ships without an entry
|
|
6
|
+
cannot be released, and an entry whose check was deleted cannot linger as orphan
|
|
7
|
+
prose. That is the mechanism that keeps this file from rotting into a list of
|
|
8
|
+
things the tool used to do.
|
|
9
|
+
|
|
10
|
+
## What the exit code means
|
|
11
|
+
|
|
12
|
+
| Code | Meaning |
|
|
13
|
+
|---|---|
|
|
14
|
+
| 0 | healthy - nothing above INFO |
|
|
15
|
+
| 1 | degraded - at least one WARN, no BLOCK |
|
|
16
|
+
| 2 | blocked - at least one BLOCK |
|
|
17
|
+
| 3 | usage error |
|
|
18
|
+
| 4 | indeterminate - the installed layout could not be resolved, so nothing was checked |
|
|
19
|
+
|
|
20
|
+
4 exists so that "I could not look" never borrows the exit code of "I looked and
|
|
21
|
+
it is fine". A consumer that treats 0 as healthy would otherwise read a broken
|
|
22
|
+
resolver as a clean bill.
|
|
23
|
+
|
|
24
|
+
## What BLOCK means, exactly
|
|
25
|
+
|
|
26
|
+
**A state in which a run will fail or leak. Not a state in which it will merely
|
|
27
|
+
be worse.** Without a closed definition "blocked" grows until nobody respects
|
|
28
|
+
exit 2, so only five checks can produce it: `install-present`, `script-surface`,
|
|
29
|
+
`state-writable`, `prefs-valid` and `embedded-credentials`. Every other check
|
|
30
|
+
tops out at WARN however bad it looks.
|
|
31
|
+
|
|
32
|
+
## Who calls it, and what they do with the code
|
|
33
|
+
|
|
34
|
+
| Caller | When | On exit 2 |
|
|
35
|
+
|---|---|---|
|
|
36
|
+
| `/multi-agent:setup` | before the first question, and again at the end | the closing run is the evidence for or against "setup complete" |
|
|
37
|
+
| `/multi-agent:update` | after the install | report that the update left a state a run will fail from, rather than "updated" |
|
|
38
|
+
| `/multi-agent:sync` | step 0 | **stop.** Syncing a blocked install copies one fault onto five surfaces, and the copies are what people then debug |
|
|
39
|
+
|
|
40
|
+
`/multi-agent:sync` stops on exit 4 as well: the layout did not resolve, so
|
|
41
|
+
nothing was checked, and syncing from an unknown state is worse than not syncing
|
|
42
|
+
at all. Exit 1 never stops anything - a degraded install still syncs correctly,
|
|
43
|
+
and a gate that blocked on every warning would be routed around within a week.
|
|
44
|
+
|
|
45
|
+
## The four severities
|
|
46
|
+
|
|
47
|
+
| Severity | Meaning |
|
|
48
|
+
|---|---|
|
|
49
|
+
| BLOCK | a run will fail or leak |
|
|
50
|
+
| WARN | a run will work and be worse: a degraded capability, a stale copy, a rejected credential |
|
|
51
|
+
| INFO | a capability the user chose not to enable; nothing to fix |
|
|
52
|
+
| SKIP | not checked, and why |
|
|
53
|
+
|
|
54
|
+
`SKIP` is printed in the same list, in the same position, and the summary always
|
|
55
|
+
carries its count (`... 2 not checked, 15 checks`). Absence is a finding: a run
|
|
56
|
+
without `--probe` exits 0 **and** prints `SKIP credential-liveness`, so "healthy"
|
|
57
|
+
and "not looked at" are never the same output.
|
|
58
|
+
|
|
59
|
+
The INFO / WARN boundary is already decided in `lib/credential-inventory.sh` and
|
|
60
|
+
this tool only renders it: an unmapped key is INFO because it is a capability the
|
|
61
|
+
user chose not to enable, while `auth-rejected`, `malformed` and
|
|
62
|
+
`tier-1-no-grant` are WARN because the configuration made a claim the service
|
|
63
|
+
refused. **A missing optional capability is information; a mapped credential the
|
|
64
|
+
service rejected is a warning. The difference is whether the configuration
|
|
65
|
+
asserted anything.**
|
|
66
|
+
|
|
67
|
+
## The line shape
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
<SEV> <id> - <what is wrong> - <single imperative step>
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The third field is one step, and its first word comes from a closed list:
|
|
74
|
+
`run`, `set`, `map`, `revoke`, `install`, `remove`, `free`, `export`, `merge`,
|
|
75
|
+
`record`, `re-run`. `consider`, `may want` and `should probably` are refused by
|
|
76
|
+
the gate. One problem, one step: a reader with five suggestions does nothing.
|
|
77
|
+
|
|
78
|
+
## It recommends, it never fixes
|
|
79
|
+
|
|
80
|
+
Nothing here rewrites a remote URL, edits `settings.json` or touches a token.
|
|
81
|
+
A remote whose URL carries a token may be the only credential that repo has, the
|
|
82
|
+
remote may be a mirror a script depends on verbatim, and doctor can run inside a
|
|
83
|
+
checkout the user does not own. Silent repair breaks all three.
|
|
84
|
+
|
|
85
|
+
For an embedded credential the honest single step is **not** "hide it". A token
|
|
86
|
+
that reached `.git/config` is already burned: it is in the shell history and
|
|
87
|
+
readable by anything that can read the working tree. The step is to revoke it at
|
|
88
|
+
its host; everything after that is behind `--explain`.
|
|
89
|
+
|
|
90
|
+
---
|
|
91
|
+
|
|
92
|
+
## Checks
|
|
93
|
+
|
|
94
|
+
### install-present
|
|
95
|
+
|
|
96
|
+
The installed tree exists and carries the subtrees a run reads: `commands/`,
|
|
97
|
+
`multi-agent-refs/`, `scripts/`, `lib/`, `schemas/`. BLOCK when any is missing -
|
|
98
|
+
a run cannot start without them.
|
|
99
|
+
|
|
100
|
+
### install-version
|
|
101
|
+
|
|
102
|
+
The installed `.pipeline-version` matches the package version resolved from the
|
|
103
|
+
checkout or the published package. WARN on a mismatch: the run works, it just
|
|
104
|
+
is not the version the user thinks they have.
|
|
105
|
+
|
|
106
|
+
### script-surface
|
|
107
|
+
|
|
108
|
+
Every script a shipped command names by path exists in the installed tree. BLOCK:
|
|
109
|
+
the failure lands mid-run, at the call, with the phase already half done.
|
|
110
|
+
|
|
111
|
+
### skill-siblings
|
|
112
|
+
|
|
113
|
+
Every command has its copies in the host trees this machine installs. WARN, never
|
|
114
|
+
BLOCK: an unsynced host is a fact about the machine, and `skill-siblings.mjs`
|
|
115
|
+
itself reports the installed layout as "authored side not checked" rather than
|
|
116
|
+
claiming a repo verdict it cannot reach.
|
|
117
|
+
|
|
118
|
+
### state-writable
|
|
119
|
+
|
|
120
|
+
`$HOME/.claude/logs/multi-agent` exists or can be created, and a file can be
|
|
121
|
+
written there. BLOCK: a run that cannot record its state cannot be resumed,
|
|
122
|
+
reported or priced, and it discovers this at Phase 0 after the pickers.
|
|
123
|
+
|
|
124
|
+
### prefs-valid
|
|
125
|
+
|
|
126
|
+
`multi-agent-preferences.json` parses and satisfies `schemas/prefs.schema.json`.
|
|
127
|
+
BLOCK: Phase 0 reads it before anything else, and a malformed file fails the run
|
|
128
|
+
after the user has already answered the pickers.
|
|
129
|
+
|
|
130
|
+
### identity
|
|
131
|
+
|
|
132
|
+
A git identity resolves for the account the run would commit as. WARN: the run
|
|
133
|
+
reaches Phase 6 and stops there.
|
|
134
|
+
|
|
135
|
+
### hook-coverage
|
|
136
|
+
|
|
137
|
+
The blocking `PreToolUse` gates from `templates/claude-hooks.json` are present in
|
|
138
|
+
`settings.json`. WARN. SKIP when the template is not installed - which was the
|
|
139
|
+
permanent state until the installer began copying `templates/`, and is exactly
|
|
140
|
+
the case this severity exists to make visible.
|
|
141
|
+
|
|
142
|
+
### credential-mapping
|
|
143
|
+
|
|
144
|
+
Which logical keys are mapped, and whether each mapped value is well formed.
|
|
145
|
+
Unmapped is INFO. Malformed is WARN.
|
|
146
|
+
|
|
147
|
+
### credential-liveness
|
|
148
|
+
|
|
149
|
+
One cheap authenticated request per mapped credential. WARN on `auth-rejected`,
|
|
150
|
+
`tier-1-no-grant` or `unreachable` - but with **different steps**, because a
|
|
151
|
+
refused credential and a host that never answered need different actions.
|
|
152
|
+
`credential-inventory.sh` already draws that line ("unreachable - no response at
|
|
153
|
+
all - on a corporate host, almost always the VPN"), and telling someone to
|
|
154
|
+
re-onboard a token that was never the problem is the exact failure the keychain
|
|
155
|
+
rules warn about one layer up. **SKIP unless `--probe` is passed**, and the
|
|
156
|
+
skip is printed, because a network check is the user's decision to spend.
|
|
157
|
+
|
|
158
|
+
### embedded-credentials
|
|
159
|
+
|
|
160
|
+
A token in a git remote URL or a tracked config file. BLOCK: this one leaks
|
|
161
|
+
rather than fails.
|
|
162
|
+
|
|
163
|
+
The report carries the repo path, the config key, the **host**, a shape label and
|
|
164
|
+
a length bucket. Never the value, and not the prefix either: `ghp_` plus a length
|
|
165
|
+
is already a fingerprint, and this line is printed to a terminal that is often
|
|
166
|
+
shared, which is the whole reason the check exists. The URL is parsed inside a
|
|
167
|
+
`python3` heredoc reading stdin, so no shell variable ever holds it.
|
|
168
|
+
|
|
169
|
+
### task-tools
|
|
170
|
+
|
|
171
|
+
Whether this session's model carries `TaskCreate` / `TaskUpdate`. Claude Code
|
|
172
|
+
provides them by default only on Claude 3.x, Opus 4 through 4.7, Sonnet 4 through
|
|
173
|
+
4.6 and Haiku 4.5; on any newer model they are absent unless the user opts in.
|
|
174
|
+
INFO, with the opt-in as the step.
|
|
175
|
+
|
|
176
|
+
A script cannot answer this - only the agent knows its own tool list - so the
|
|
177
|
+
caller passes `--task-tools=yes|no` and the default is SKIP. Reporting "absent"
|
|
178
|
+
from a script that never looked would be the same defect this check is about.
|
|
179
|
+
|
|
180
|
+
### mcp-registration
|
|
181
|
+
|
|
182
|
+
Whether `multi-agent-toolkit` is registered as an MCP server. INFO when it is
|
|
183
|
+
not: the tools it provides are optional and a run without them is smaller, not
|
|
184
|
+
wrong.
|
|
185
|
+
|
|
186
|
+
**Registered is not the same as working.** With `--probe` the server is started
|
|
187
|
+
and its tools counted; serving none is WARN. Without `--probe` this reports what
|
|
188
|
+
is configured, and says so. The distinction is not theoretical: a half-extracted
|
|
189
|
+
package in the npx cache left the server dying on `Cannot find module` at
|
|
190
|
+
startup, which the client surfaces only as `CONNECTION_CLOSED`, and a check that
|
|
191
|
+
read the registration and stopped would have called that healthy.
|
|
192
|
+
|
|
193
|
+
### disk-space
|
|
194
|
+
|
|
195
|
+
Free space on the volume holding `$HOME`. WARN under 2 GB: a worktree plus a
|
|
196
|
+
build is the largest thing a run writes, and ENOSPC mid-run corrupts the state
|
|
197
|
+
file it was writing at the time.
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# Related-issue context at intake
|
|
2
|
+
|
|
3
|
+
A development sub-task is often filed with no description of its own. The
|
|
4
|
+
requirement sits on the parent, and the rest of the picture - the analysis, the
|
|
5
|
+
test scope - sits on the sibling sub-tasks beside it. The fetcher already read
|
|
6
|
+
the parent; it did not read the siblings, so a board that keeps its analysis in a
|
|
7
|
+
separate sub-task handed the pipeline an empty task.
|
|
8
|
+
|
|
9
|
+
`issue-fetcher.sh` now reads those siblings and exposes them as
|
|
10
|
+
`descriptor.relatedIssues[]`.
|
|
11
|
+
|
|
12
|
+
## What it costs
|
|
13
|
+
|
|
14
|
+
One search request, and only when the issue has a parent:
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
GET /rest/api/2/search?jql=parent="<PARENT>"&fields=summary,issuetype,status,description&maxResults=20
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
An issue with no parent has no siblings, so the lookup is skipped entirely and
|
|
21
|
+
that run pays nothing. The count does not grow with the number of siblings: the
|
|
22
|
+
issue's own `subtasks` field lists ITS children rather than its siblings, and the
|
|
23
|
+
parent's lists siblings without their descriptions, so either of those shapes
|
|
24
|
+
would cost one GET per sibling. `smoke-jira-context.sh` counts the requests, so
|
|
25
|
+
this stays a measurement rather than a claim.
|
|
26
|
+
|
|
27
|
+
The request lands before the maturity verdict reaches the user, which is one
|
|
28
|
+
extra round trip on the intake path. `jiraContext.enabled: false` is the way out
|
|
29
|
+
when latency matters more than the context.
|
|
30
|
+
|
|
31
|
+
## Shape
|
|
32
|
+
|
|
33
|
+
```json
|
|
34
|
+
"relatedIssues": [
|
|
35
|
+
{
|
|
36
|
+
"key": "PROJ-1002",
|
|
37
|
+
"relation": "sibling",
|
|
38
|
+
"type": "Sub-task",
|
|
39
|
+
"status": "Done",
|
|
40
|
+
"summary": "...",
|
|
41
|
+
"description": "...",
|
|
42
|
+
"truncated": false
|
|
43
|
+
}
|
|
44
|
+
]
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
`type` is carried through verbatim from Jira and is never matched against in
|
|
48
|
+
code. Boards name their sub-task types differently and the query does not need
|
|
49
|
+
to know: `parent = <key>` returns all of them.
|
|
50
|
+
|
|
51
|
+
Siblings that carry a description are ordered first, so `maxItems` drops the
|
|
52
|
+
empty ones rather than the useful ones. An empty sibling is still listed - its
|
|
53
|
+
key, type and status are information - with an empty `description`.
|
|
54
|
+
|
|
55
|
+
## Settings
|
|
56
|
+
|
|
57
|
+
`prefs.global.jiraContext`:
|
|
58
|
+
|
|
59
|
+
| Key | Default | Effect |
|
|
60
|
+
|---|---|---|
|
|
61
|
+
| `enabled` | `true` | `false` skips the search entirely |
|
|
62
|
+
| `maxItems` | `6` | Hard cap; `0` behaves like `enabled: false` |
|
|
63
|
+
| `maxCharsPerItem` | `1200` | Per-sibling truncation, marked `truncated: true` |
|
|
64
|
+
|
|
65
|
+
Worst case payload is `maxItems x maxCharsPerItem`, next to the 4000-character
|
|
66
|
+
cap on the issue's own description.
|
|
67
|
+
|
|
68
|
+
## Maturity
|
|
69
|
+
|
|
70
|
+
One new code, `description_empty_sibling_available`, and it fires narrowly: the
|
|
71
|
+
issue's own description is empty, the parent's is empty or unreachable, AND a
|
|
72
|
+
sibling carries content. Today that combination produces the hard
|
|
73
|
+
`description_empty` blocker; it becomes a warning instead, the same downgrade
|
|
74
|
+
`description_empty_parent_available` already makes one level up.
|
|
75
|
+
|
|
76
|
+
The narrowness is deliberate. `score` is `100 - 40*blockers - 10*warnings`, and
|
|
77
|
+
the picker continues silently only at `score >= 90`, so a warning that fired
|
|
78
|
+
whenever an issue merely HAD siblings would cost ten points on every parented
|
|
79
|
+
issue and turn a silent intake into a question across the board.
|
|
80
|
+
|
|
81
|
+
## Where it goes
|
|
82
|
+
|
|
83
|
+
The descriptor is carried verbatim into the picker state as `issueRef`, and the
|
|
84
|
+
bridge copies `relatedIssues` onto `agent-state.json` when it is non-empty.
|
|
85
|
+
|
|
86
|
+
Persisting it is the point: `/multi-agent:resume` rebuilds context from durable
|
|
87
|
+
artefacts and never from the conversation, so an intake-only enrichment would be
|
|
88
|
+
gone by the first resume. The task's own `description` and `maturity` are still
|
|
89
|
+
not persisted, which is a known asymmetry rather than an oversight - this
|
|
90
|
+
change did not widen the bridge beyond the one field it needed.
|
|
91
|
+
|
|
92
|
+
Sibling text is added as its own labelled block. It never replaces the working
|
|
93
|
+
description the way an accepted `parentDescription` does: substituting it would
|
|
94
|
+
erase the provenance that Phase 2's parent-story scope-drift check reads.
|
|
95
|
+
|
|
96
|
+
## Not included
|
|
97
|
+
|
|
98
|
+
Following the parent's `issuelinks` in a second search. A linked bug can be the
|
|
99
|
+
whole requirement while a linked test-execution record is noise, and the fetcher
|
|
100
|
+
cannot tell them apart, so the scope stops at sub-tasks. There is no setting for
|
|
101
|
+
it: a switch that does nothing is worse than an absent one.
|
|
@@ -83,7 +83,7 @@ predate this field still get the second downgrade step):
|
|
|
83
83
|
"premiumTierUntil": null,
|
|
84
84
|
"fallbackModel": "sonnet",
|
|
85
85
|
"floorModel": "haiku",
|
|
86
|
-
"fableEnabled":
|
|
86
|
+
"fableEnabled": false,
|
|
87
87
|
"onDispatchError": true
|
|
88
88
|
}
|
|
89
89
|
```
|
|
@@ -91,7 +91,7 @@ predate this field still get the second downgrade step):
|
|
|
91
91
|
| Field | Meaning |
|
|
92
92
|
|---|---|
|
|
93
93
|
| `enabled` | Master switch. `false` = always dispatch the persona's `preferredModel`, fail loudly on error. |
|
|
94
|
-
| `modelFallback.fableEnabled` | Whether the `fable` rung exists at all on Claude Code. `false` starts every `preferredModel: fable` persona on `opus
|
|
94
|
+
| `modelFallback.fableEnabled` | Whether the `fable` rung exists at all on Claude Code. **Ships `false`.** `false` starts every `preferredModel: fable` persona on `opus`; set it to `true` to opt the top rung back in. A cost control that defaults to on is not a control. See "Turning the fable rung off" below. |
|
|
95
95
|
| `premiumTierUntil` | ISO date (`YYYY-MM-DD`) or `null`. When set and today is **after** this date, every `preferredModel` dispatch is downgraded to `fallbackModel` unless the user re-confirms (see Date gate). Use when the top tier is included in a plan only until a known date. |
|
|
96
96
|
| `fallbackModel` | Target of the first downgrade step. Default `sonnet` (the next tier below opus). |
|
|
97
97
|
| `modelFallback.floorModel` | Last-resort tier when `fallbackModel` also fails to dispatch. Default `haiku`. Set to the same value as `fallbackModel` (or `null`) to disable the second step and halt after one downgrade. |
|
|
@@ -275,11 +275,8 @@ Scan `$HOME` (maxdepth 2) for project markers (`.xcodeproj`, `Package.swift`, `b
|
|
|
275
275
|
| grep -v -E '(feature/|bugfix/|fix/|hotfix/|chore/)'
|
|
276
276
|
```
|
|
277
277
|
6. Sort: `develop*` first, then `release/*`, then `main`/`master`. Surface through the
|
|
278
|
-
**native picker** per `picker-contract.md
|
|
279
|
-
`
|
|
280
|
-
`label` = the branch name verbatim (a proper noun, never translated) with the recent
|
|
281
|
-
branch first and marked `(Recommended)`. The ASCII sketch
|
|
282
|
-
below is what the options carry, not a menu to print:
|
|
278
|
+
**native picker** per `picker-contract.md`, recent branch first and marked
|
|
279
|
+
`(Recommended)`:
|
|
283
280
|
|
|
284
281
|
```
|
|
285
282
|
header: "Base branch"
|
|
@@ -298,9 +295,14 @@ nothing surfaced it. `phase0-exit-gate.mjs` now refuses to close Phase 0 unless
|
|
|
298
295
|
`agent-state.json` carries `baseBranch` and `baseFetchStatus`, so a skipped picker is a
|
|
299
296
|
gate failure rather than a silent default.
|
|
300
297
|
|
|
298
|
+
One row is a normal outcome of the rule-5 filter and is **still asked** - see
|
|
299
|
+
`picker-contract.md` "A single candidate is still a question".
|
|
300
|
+
|
|
301
301
|
This holds in every mode. A Short run skips the *LLM* phases (Analysis, Planning); it
|
|
302
|
-
does not skip Phase 0's pickers. Autopilot resolves them
|
|
303
|
-
|
|
302
|
+
does not skip Phase 0's pickers. Autopilot resolves them without prompting, which still
|
|
303
|
+
writes the fields. For the base branch it reads `recentBranches[{projectKey}]` (most
|
|
304
|
+
recent inside the TTL, still on the remote) before the rule-6 order, recording which
|
|
305
|
+
fired in `state.baseBranchSource`.
|
|
304
306
|
|
|
305
307
|
**TTL filter for recent branches**:
|
|
306
308
|
|
|
@@ -109,7 +109,7 @@ back. "Copy X and rename it" is the reuse answer, not a hint - name X's files.
|
|
|
109
109
|
|
|
110
110
|
#### Step 1.5 - External Context Injection (`state.contextLinks[]`)
|
|
111
111
|
|
|
112
|
-
Phase 0 Step 1b catalogued every typed external link from the task description into `state.contextLinks[]`. Phase 1 dispatches each entry to its matching fetcher (crashlytics, fortify, graylog, swagger, confluence, figma, generic-doc) and prepends results under a **Referenced External Sources** section in the analysis prompt - so the agent doesn't re-discover what the ticket already pointed at. `state.graylogContext`
|
|
112
|
+
Phase 0 Step 1b catalogued every typed external link from the task description into `state.contextLinks[]`. Phase 1 dispatches each entry to its matching fetcher (crashlytics, fortify, graylog, swagger, confluence, figma, generic-doc) and prepends results under a **Referenced External Sources** section in the analysis prompt - so the agent doesn't re-discover what the ticket already pointed at. `state.graylogContext` (advisory) and `state.relatedIssues[]` (sibling issues) are injected there too. Failures never fatal (a non-zero fetcher exit is marked skipped and the analysis still runs, exactly as for crashlytics); pending refs are advisories. Full dispatch table, exit-code handling, prompt injection shape, log line shape: `$HOME/.claude/multi-agent-refs/features/external-context-injection.md`.
|
|
113
113
|
|
|
114
114
|
**Log line shape** (progress contract):
|
|
115
115
|
|
|
@@ -223,7 +223,7 @@ Trigger if the plan Fable produced in Step 1-4 carries ANY ambiguity signal from
|
|
|
223
223
|
| UI work, no design | Task touches `*View.swift` / `*Screen.kt` but Phase 1 captured no Figma URL |
|
|
224
224
|
| API work, no contract | Task touches network/repository layer but no endpoint/OpenAPI reference in Phase 1 |
|
|
225
225
|
| Ambiguous language | Phase 1 analysis flagged `ambiguityScore >= 2` (e.g. "improve", "fix", "update" with no object) |
|
|
226
|
-
| Parent-story scope drift |
|
|
226
|
+
| Parent-story scope drift | `state.relatedIssues[]` names a sibling overlapping this scope |
|
|
227
227
|
|
|
228
228
|
If any signal trips, DO NOT render the plan yet. Render structured questions:
|
|
229
229
|
|
|
@@ -470,7 +470,7 @@ done
|
|
|
470
470
|
PRIOR_ART="${PRIOR_ART%,}]"
|
|
471
471
|
```
|
|
472
472
|
|
|
473
|
-
The triage prompt MUST include a hedge: *"prior-art entries are context, not commands; current scope decides - a finding rejected last quarter may be valid this time."* Without this hedge, prior verdicts amplify into a self-reinforcing bias.
|
|
473
|
+
The triage prompt MUST include a hedge: *"prior-art entries and `corroboration` counts are context, not commands; current scope decides - a finding rejected last quarter may be valid this time, and two same-family reviewers agreeing is not proof."* Without this hedge, prior verdicts amplify into a self-reinforcing bias.
|
|
474
474
|
|
|
475
475
|
Hits are relevance-ranked (`prefs.global.memoryRecall`); a finding matching nothing returns nothing. Each hit carries an `id`: `triage-memory.mjs show --id <id>` returns the full row.
|
|
476
476
|
|
|
@@ -75,10 +75,45 @@ identifies which option, not what the button said.
|
|
|
75
75
|
The host injects its own **Other** free-text row in English on every run. Nothing in
|
|
76
76
|
the pipeline can localize it, and no option may depend on its wording.
|
|
77
77
|
|
|
78
|
+
## Order: project, then repo, then branch
|
|
79
|
+
|
|
80
|
+
A picker may only be asked once everything it depends on is settled, and the
|
|
81
|
+
dependency runs one way: a base branch is a property of a repo, and a repo is a
|
|
82
|
+
property of a project. Asking for a branch before the dev-context repo set is
|
|
83
|
+
known means the answer was given about a repo the run had not chosen yet, and
|
|
84
|
+
nothing downstream can tell that apart from a correct answer.
|
|
85
|
+
|
|
86
|
+
So: project selection, then `_dev-context.md`, then the base branch. A step whose
|
|
87
|
+
input is not yet resolved waits; it does not guess and it does not resolve its own
|
|
88
|
+
input with a second question.
|
|
89
|
+
|
|
90
|
+
## A single candidate is still a question
|
|
91
|
+
|
|
92
|
+
The number of options never authorises a skip. A filter that leaves one row has
|
|
93
|
+
narrowed the world; it has not decided anything, and the host's **Other** row is a real
|
|
94
|
+
choice on every picker - a branch the filter excluded, an account the probe missed, a
|
|
95
|
+
repo git does not know about. "There was only one option, so I picked it" is a skipped
|
|
96
|
+
picker, and announcing the pick in prose first is the same skip with a sentence in front
|
|
97
|
+
of it.
|
|
98
|
+
|
|
99
|
+
This is the failure that is hardest to see afterwards, because the transcript reads like
|
|
100
|
+
a decision was made. Only the picker's absence records that the user was never asked.
|
|
101
|
+
|
|
78
102
|
## Autopilot / non-interactive contract
|
|
79
103
|
|
|
80
104
|
In autopilot, `ask_choice` resolves to `default` (or the safe first option) without prompting - identical to how the native gates auto-proceed today. A picker is only surfaced for genuinely ambiguous or destructive decisions, matching the maturity-check model.
|
|
81
105
|
|
|
106
|
+
**Memory outranks the default.** Where the pipeline has recorded what this user chose
|
|
107
|
+
last time for this project - `prefs.global.recentBranches[{projectKey}]` is the one that
|
|
108
|
+
exists today - autopilot resolves from that record first, and only falls back to the
|
|
109
|
+
option order when the record is empty, stale past its TTL, or names something that no
|
|
110
|
+
longer exists. A remembered choice is evidence about this user and this repo; an option
|
|
111
|
+
order is a guess that happens to be sorted. Where the two agree nothing changes, and
|
|
112
|
+
where they disagree the remembered one is the answer with a reason behind it.
|
|
113
|
+
|
|
114
|
+
The run records which rule fired (`remembered` or `default`). An autopilot run cannot be
|
|
115
|
+
asked anything, so the only thing that keeps it accountable is being readable afterwards.
|
|
116
|
+
|
|
82
117
|
## Deterministic gates note
|
|
83
118
|
|
|
84
119
|
Claude Code's `PreToolUse` exit-2 hooks are the HARD blocking gates. Three ship, none needing run-specific arguments so they are naturally hookable: (1) `pre-commit-check.sh` scans the staged diff on every `git commit` and blocks on a detected secret; (2) `agent-guard.sh` runs on `git commit` + `git push` and blocks AI/assistant attribution in a commit message and force-push to a protected branch (main/master/develop); (3) `check-read-size.sh` runs on `Read` and on the shell commands that read a file whole, and routes an oversized read to a cheap worker (`bulk-read.sh`) instead of the caller's own rung. The first two inspect what a run WRITES; the third inspects what it pays to READ, and it is inert until `bulkRead.mode` is set to `observe` or `enforce`, so merging the block changes nothing until the user opts in. Its `observe` mode blocks nothing and only logs, which is how the baseline is measured before anything is routed. All three are self-contained, fail-open on internal error, and never execute the inspected command. Two capture hooks ship in the same block and block nothing: `SessionEnd` runs `capture-flush.sh --if-stale` (writing a killed run's findings into the per-repo stores, since every durable write used to live in Phase 7 - the phase a run is least likely to reach) plus `note-session.sh` (the mechanical shape of a non-pipeline session: tools used, commands that failed, calls the user refused - never an argument, never any output), and `SessionStart` runs `capture-resume.sh`, at most two lines about an unfinished run and a stale observation queue. Neither calls a model; both exit 0 on every path. The recommended hook block ships at `install/templates/claude-hooks.json`; `multi-agent:setup` offers to merge it into `~/.claude/settings.json`. The other deterministic gates (evidence, consensus, intent, learnings) are invoked by the pipeline phases with per-run arguments (a build-log path, the triage JSON, the free-text input), so they are phase-enforced by contract, not OS-hookable.
|
|
@@ -12,7 +12,7 @@ Instruction prose here is English. The verdict shown in chat and the comment bod
|
|
|
12
12
|
- With no argument: offer the picker - review-jira reuses the `jira` command's issue list (`jira/SKILL.md` JQL, assignee=currentUser, open); review-issue reuses the `issue` command's list (`gh issue list`). Single-select (one item reviewed per run; re-run for more).
|
|
13
13
|
|
|
14
14
|
## Step 2 - fetch + base maturity
|
|
15
|
-
Run `$HOME/.claude/lib/issue-fetcher.sh` (the same fetcher Phase 0 uses). It returns a descriptor with `title`, `type`, `status`, `description`, and `maturity = { score (0..100), blockers[], warnings[], summary }`. Reuse that maturity verbatim as the baseline; never re-fetch through MCP.
|
|
15
|
+
Run `$HOME/.claude/lib/issue-fetcher.sh` (the same fetcher Phase 0 uses). It returns a descriptor with `title`, `type`, `status`, `description`, `relatedIssues[]` (the parent's other sub-tasks), and `maturity = { score (0..100), blockers[], warnings[], summary }`. Reuse that maturity verbatim as the baseline; never re-fetch through MCP.
|
|
16
16
|
|
|
17
17
|
## Step 3 - readiness rubric (extends maturity)
|
|
18
18
|
Maturity is generic; add these pipeline-readiness dimensions by reading the fetched `title` + `description` (no invention - only judge what is written). Each is `pass` / `gap`:
|
|
@@ -32,7 +32,7 @@ Visual mechanism per CLI:
|
|
|
32
32
|
|
|
33
33
|
| CLI | What the agent calls at every phase boundary |
|
|
34
34
|
|---|---|
|
|
35
|
-
| **claude-code** | `TaskCreate({subject, activeForm})` then `TaskUpdate({status, activeForm})`. Native sticky widget; ⏺ tiles, spinner header. |
|
|
35
|
+
| **claude-code** | `TaskCreate({subject, activeForm})` then `TaskUpdate({status, activeForm})`. Native sticky widget; ⏺ tiles, spinner header. **Not in every session** - see below. |
|
|
36
36
|
| **copilot** | Inline call: `bash phase-tracker.sh render`. The bordered ANSI card lands as the last tool result in the chat. |
|
|
37
37
|
| **codex** | The native `update_plan` tool: one plan step per phase, `status: pending \| in_progress \| completed`. Rewrite the whole step list on each boundary - the tool takes the full plan, not a delta. |
|
|
38
38
|
| **generic** (plain shell, Git Bash, WSL, tmux) | Same bash render - the bordered ANSI card prints to terminal stdout in place. |
|
|
@@ -41,6 +41,10 @@ Visual mechanism per CLI:
|
|
|
41
41
|
|
|
42
42
|
**Claude Code only**: in addition, `TaskCreate` / `TaskUpdate` native tool calls → the sticky widget pins the phase stack in the user's view.
|
|
43
43
|
|
|
44
|
+
**The task tools depend on the MODEL, not the CLI version.** Claude Code provides `TaskCreate`, `TaskUpdate`, `TaskGet`, `TaskList` and `TodoWrite` by default only on Claude 3.x, Opus 4 through 4.7, Sonnet 4 through 4.6 and Haiku 4.5; on any newer model it leaves them out unless the user opts in. That default arrived in Claude Code v2.1.268, and this contract was written when the tools were universal, so nothing noticed the change: a run wrote its tracker state correctly, advanced every phase, and drew nothing for the whole run.
|
|
45
|
+
|
|
46
|
+
So the branch is taken by the agent, which knows its own tool list, and never by the shell, which cannot see it. Without the tools the bordered card IS the widget and has to be reprinted inside the reply at every boundary, exactly as on Copilot CLI. Say once, in `outputLanguage`, that the native widget returns with `CLAUDE_CODE_ENABLE_TODO_TOOLS=1 claude` (or `claude --allowedTools TaskCreate`); a user staring at a missing widget needs the command, not the diagnosis. A subagent inherits the session's tool set even when it runs a different model, so this is a session-level fact and not a per-phase one.
|
|
47
|
+
|
|
44
48
|
### Codex specifics
|
|
45
49
|
|
|
46
50
|
Two constraints on `update_plan`, both of which fail quietly if ignored:
|
|
@@ -76,6 +76,11 @@
|
|
|
76
76
|
"type": "string",
|
|
77
77
|
"description": "PR target branch (e.g. develop, main)."
|
|
78
78
|
},
|
|
79
|
+
"baseBranchSource": {
|
|
80
|
+
"type": "string",
|
|
81
|
+
"enum": ["asked", "input", "remembered", "default"],
|
|
82
|
+
"description": "How baseBranch was decided. asked = the user answered the Step 3 picker; input = it arrived with the task reference; remembered = autopilot took the most recent entry in prefs.global.recentBranches still inside the TTL and still on the remote; default = autopilot fell back to the develop/release/main sort order. An autopilot run cannot be asked anything, so recording which rule fired is what keeps it readable afterwards."
|
|
83
|
+
},
|
|
79
84
|
"remoteType": {
|
|
80
85
|
"type": "string",
|
|
81
86
|
"enum": ["github", "bitbucket"],
|
|
@@ -104,6 +109,25 @@
|
|
|
104
109
|
"type": ["string", "null"],
|
|
105
110
|
"description": "Set when a phase halts on a hard error (validator failed twice, no subagent returned, dispatch error past fallback, lock irrecoverable). Format '<phase>:<cause>'. Surfaced to the user and cleared on successful resume. See operations.md 'Halt visibility'."
|
|
106
111
|
},
|
|
112
|
+
"relatedIssues": {
|
|
113
|
+
"type": "array",
|
|
114
|
+
"maxItems": 20,
|
|
115
|
+
"description": "Jira issues fetched alongside the task at intake: the other sub-tasks under this issue's parent, where a board keeps the analysis and the test scope. NOT the same field as siblings[] below, which is read-only sibling REPOS. Written by the picker bridge from descriptor.relatedIssues; read by Phase 1 as ground truth and by Phase 2's parent-story scope-drift check. Durable on purpose: /multi-agent:resume rebuilds context from artefacts, never from the conversation, so intake-only enrichment would vanish on the first resume. The task's own description and maturity are still NOT persisted here, which is a known asymmetry, not an oversight.",
|
|
116
|
+
"items": {
|
|
117
|
+
"type": "object",
|
|
118
|
+
"additionalProperties": false,
|
|
119
|
+
"required": ["key", "relation"],
|
|
120
|
+
"properties": {
|
|
121
|
+
"key": { "type": "string" },
|
|
122
|
+
"relation": { "type": "string", "enum": ["sibling"] },
|
|
123
|
+
"type": { "type": "string" },
|
|
124
|
+
"status": { "type": "string" },
|
|
125
|
+
"summary": { "type": "string" },
|
|
126
|
+
"description": { "type": "string" },
|
|
127
|
+
"truncated": { "type": "boolean", "default": false }
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
},
|
|
107
131
|
"siblings": {
|
|
108
132
|
"type": "array",
|
|
109
133
|
"maxItems": 10,
|