@plainconceptsplatform/workflows 0.27.5 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/loops/actions/decide-agent-retry/action.yml +1 -1
- package/loops/actions/verify-route-matrix/verify-route-matrix.sh +1 -1
- package/loops/templates/opencode/opencode.ci.json +4 -4
- package/loops/templates/opencode/opencode.ci.json.md +1 -1
- package/loops/workflows/agent-apply-review.md +2 -2
- package/loops/workflows/agent-audit.md +1 -1
- package/loops/workflows/agent-implement.md +1 -1
- package/loops/workflows/agent-merge-gate.md +1 -1
- package/loops/workflows/agent-refine.md +149 -10
- package/loops/workflows/agent-release.md +1 -1
- package/loops/workflows/agent-triage.md +3 -3
- package/loops/workflows/agent-visual-verify.md +1 -1
- package/package.json +1 -1
|
@@ -35,7 +35,7 @@ runs:
|
|
|
35
35
|
#
|
|
36
36
|
# Written once here because the rule was in the implement worker alone, and refine, triage and
|
|
37
37
|
# apply-review parked instead -- an Odyssey refine died in three minutes on
|
|
38
|
-
# `Model 'glm-5-
|
|
38
|
+
# `Model 'glm-5-2' not found`, a gateway fault that clears in seconds, and waited on the
|
|
39
39
|
# janitor's six-hourly sweep. Five copies of sixty lines is how one rule becomes five.
|
|
40
40
|
- name: Decide whether this failure is worth repeating
|
|
41
41
|
id: decide
|
|
@@ -1990,7 +1990,7 @@ echo "── Retrying a run that died early ────────────
|
|
|
1990
1990
|
|
|
1991
1991
|
# A run that died before producing anything is worth repeating; one that worked and then failed
|
|
1992
1992
|
# produced an answer that was wrong, and repeating it buys the same wrong answer later. Only the
|
|
1993
|
-
# implement worker knew that. An Odyssey refine died in three minutes on `Model 'glm-5-
|
|
1993
|
+
# implement worker knew that. An Odyssey refine died in three minutes on `Model 'glm-5-2' not
|
|
1994
1994
|
# found` -- a gateway fault that clears in seconds -- parked, and waited on the janitor's
|
|
1995
1995
|
# six-hourly sweep.
|
|
1996
1996
|
RETRY_OK=1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://opencode.ai/config.json",
|
|
3
|
-
"model": "forge/glm-5-
|
|
3
|
+
"model": "forge/glm-5-2",
|
|
4
4
|
"plugin": [],
|
|
5
5
|
"default_agent": "ci-workflow-agent",
|
|
6
6
|
"agent": {
|
|
@@ -41,15 +41,15 @@
|
|
|
41
41
|
"apiKey": "{env:OPENAI_API_KEY}"
|
|
42
42
|
},
|
|
43
43
|
"models": {
|
|
44
|
-
"glm-5-
|
|
44
|
+
"glm-5-2": {
|
|
45
45
|
"name": "GLM 5.3",
|
|
46
46
|
"attachment": true
|
|
47
47
|
},
|
|
48
|
-
"
|
|
48
|
+
"quasar-438b": {
|
|
49
49
|
"name": "GLM 5.2",
|
|
50
50
|
"attachment": true
|
|
51
51
|
}
|
|
52
52
|
}
|
|
53
53
|
}
|
|
54
54
|
}
|
|
55
|
-
}
|
|
55
|
+
}
|
|
@@ -26,7 +26,7 @@ repository root as `opencode.ci.json`.
|
|
|
26
26
|
host, authentication system, or model vendor.
|
|
27
27
|
- Model entries are fallback metadata. Consumers should set `attachment: true`
|
|
28
28
|
only for models whose router supports image input.
|
|
29
|
-
- Default model: `forge/glm-5-
|
|
29
|
+
- Default model: `forge/glm-5-2`.
|
|
30
30
|
|
|
31
31
|
### Agents
|
|
32
32
|
|
|
@@ -14,7 +14,7 @@ env:
|
|
|
14
14
|
REVIEW_MARKER: "<!-- agent-apply-review -->"
|
|
15
15
|
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
16
16
|
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
17
|
-
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-
|
|
17
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-2' not
|
|
18
18
|
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
19
19
|
# because only the implement worker could do this.
|
|
20
20
|
APPLY_REVIEW_ATTEMPT_MARKER: "<!-- agent-apply-review-attempt -->"
|
|
@@ -399,7 +399,7 @@ engine:
|
|
|
399
399
|
env:
|
|
400
400
|
OPENAI_BASE_URL: https://forge.plainconcepts.com/v1
|
|
401
401
|
|
|
402
|
-
model: openai/glm-5-
|
|
402
|
+
model: openai/glm-5-2
|
|
403
403
|
|
|
404
404
|
max-turns: 300
|
|
405
405
|
max-turn-cache-misses: 3000
|
|
@@ -26,12 +26,20 @@ env:
|
|
|
26
26
|
REFINE_MARKER: "<!-- agent-refine -->"
|
|
27
27
|
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
28
28
|
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
29
|
-
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-
|
|
29
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-2' not
|
|
30
30
|
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
31
31
|
# because only the implement worker could do this.
|
|
32
32
|
REFINE_ATTEMPT_MARKER: "<!-- agent-refine-attempt -->"
|
|
33
33
|
MAX_ATTEMPTS: "5"
|
|
34
34
|
PARK_AT_ATTEMPT: "4"
|
|
35
|
+
# The size gate. An issue over either limit is refused before any model runs: a real
|
|
36
|
+
# one, Pliny-Bot #305 (13,398 chars, 92 work units), burned a full agent -- 17 minutes,
|
|
37
|
+
# 2.4M tokens -- and died at the provider's 32K output cap with zero outcome, twice.
|
|
38
|
+
# Both numbers are measured before the agent starts, so refusal is deterministic and
|
|
39
|
+
# costs ten seconds instead of half an hour.
|
|
40
|
+
REFINE_MAX_BODY_CHARS: "8000"
|
|
41
|
+
REFINE_MAX_WORK_UNITS: "24"
|
|
42
|
+
REFUSE_MARKER: "<!-- agent-refine-refused -->"
|
|
35
43
|
# Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
|
|
36
44
|
# worse day would fall outside a six-minute window and park for a fault that clears by itself.
|
|
37
45
|
RETRY_UNDER_MINUTES: "10"
|
|
@@ -98,12 +106,12 @@ on:
|
|
|
98
106
|
required: false
|
|
99
107
|
type: string
|
|
100
108
|
default: first
|
|
101
|
-
# The gate
|
|
102
|
-
# activation job but gives activation no dependency on the
|
|
109
|
+
# The gate jobs that the top-level `if:` reads. gh-aw folds that `if:` into the generated
|
|
110
|
+
# activation job but gives activation no dependency on the jobs, so the reference resolves
|
|
103
111
|
# to '' and the clause is false -- the agent would never run. The package's own validator
|
|
104
112
|
# catches it after compilation; this is the line it asks for, the same one the merge gate
|
|
105
113
|
# uses for protected_changes.
|
|
106
|
-
needs: [still_open]
|
|
114
|
+
needs: [still_open, size_guard]
|
|
107
115
|
|
|
108
116
|
jobs:
|
|
109
117
|
# A route dispatched while the issue was open must not execute after it has been closed. The
|
|
@@ -130,10 +138,101 @@ jobs:
|
|
|
130
138
|
with:
|
|
131
139
|
token: ${{ github.token }}
|
|
132
140
|
issue-number: ${{ inputs.issue-number }}
|
|
133
|
-
|
|
141
|
+
# The size gate, rung 4. Measures the issue before any model starts, because an oversized
|
|
142
|
+
# body does not fail fast on its own: it produces an agent that explores for seventeen
|
|
143
|
+
# minutes and then dies mid-generation at the provider's response cap having written
|
|
144
|
+
# nothing, which reads as a green run with no outcome (Pliny-Bot #305, twice). Pure
|
|
145
|
+
# shell, no network beyond one issue read; over either limit it is a refusal, not a
|
|
146
|
+
# smaller attempt.
|
|
147
|
+
size_guard:
|
|
134
148
|
needs: [still_open]
|
|
135
149
|
if: needs.still_open.outputs.open == 'true'
|
|
136
150
|
runs-on: agents-arc
|
|
151
|
+
timeout-minutes: 5
|
|
152
|
+
permissions:
|
|
153
|
+
contents: read
|
|
154
|
+
issues: read
|
|
155
|
+
outputs:
|
|
156
|
+
too_big: ${{ steps.measure.outputs.too_big }}
|
|
157
|
+
reason: ${{ steps.measure.outputs.reason }}
|
|
158
|
+
steps:
|
|
159
|
+
- name: Measure the issue against the size limits
|
|
160
|
+
id: measure
|
|
161
|
+
env:
|
|
162
|
+
GH_TOKEN: ${{ github.token }}
|
|
163
|
+
REPO: ${{ github.repository }}
|
|
164
|
+
ISSUE_NUMBER: ${{ inputs.issue-number }}
|
|
165
|
+
MAX_BODY_CHARS: ${{ env.REFINE_MAX_BODY_CHARS }}
|
|
166
|
+
MAX_WORK_UNITS: ${{ env.REFINE_MAX_WORK_UNITS }}
|
|
167
|
+
run: |
|
|
168
|
+
set -euo pipefail
|
|
169
|
+
body=$(gh issue view "$ISSUE_NUMBER" --repo "$REPO" --json title,body --jq '.title + "\n\n" + (.body // "")')
|
|
170
|
+
chars=$(printf '%s' "$body" | wc -c)
|
|
171
|
+
# One work unit per markdown bullet, the same split the prompt's step 3 makes.
|
|
172
|
+
units=$(printf '%s' "$body" | grep -cE '^[[:space:]]*[-*] ' || true)
|
|
173
|
+
reason=""
|
|
174
|
+
too_big=false
|
|
175
|
+
if [ "$chars" -gt "$MAX_BODY_CHARS" ]; then
|
|
176
|
+
too_big=true
|
|
177
|
+
reason="body ${chars} chars > ${MAX_BODY_CHARS}"
|
|
178
|
+
fi
|
|
179
|
+
if [ "$units" -gt "$MAX_WORK_UNITS" ]; then
|
|
180
|
+
too_big=true
|
|
181
|
+
reason="${reason:+$reason; }work units ${units} > ${MAX_WORK_UNITS}"
|
|
182
|
+
fi
|
|
183
|
+
echo "too_big=$too_big" >> "$GITHUB_OUTPUT"
|
|
184
|
+
echo "reason=$reason" >> "$GITHUB_OUTPUT"
|
|
185
|
+
echo "::notice::measured $chars chars, $units work units (limits: ${MAX_BODY_CHARS} chars, ${MAX_WORK_UNITS} units) -> $too_big"
|
|
186
|
+
# The deterministic refusal. Runs only when the size gate tripped, so it costs one
|
|
187
|
+
# comment and no model. The refine label stays: a human who shrinks the body and
|
|
188
|
+
# replies re-enters rerefine through the comment route, and the classifier ignores
|
|
189
|
+
# bot comments, so this comment cannot re-trigger the worker.
|
|
190
|
+
refuse_big_issue:
|
|
191
|
+
needs: [still_open, size_guard]
|
|
192
|
+
if: needs.still_open.outputs.open == 'true' && needs.size_guard.outputs.too_big == 'true'
|
|
193
|
+
runs-on: agents-arc
|
|
194
|
+
timeout-minutes: 5
|
|
195
|
+
permissions:
|
|
196
|
+
contents: read
|
|
197
|
+
issues: write
|
|
198
|
+
steps:
|
|
199
|
+
- name: Checkout workflow actions
|
|
200
|
+
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
|
201
|
+
with:
|
|
202
|
+
persist-credentials: false
|
|
203
|
+
- name: Create bot token
|
|
204
|
+
id: app-token
|
|
205
|
+
uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
|
|
206
|
+
with:
|
|
207
|
+
client-id: ${{ secrets.BOT_APP_ID }}
|
|
208
|
+
private-key: ${{ secrets.BOT_PRIVATE_KEY }}
|
|
209
|
+
- name: Refuse the oversized issue
|
|
210
|
+
uses: ./.github/actions/create-issue-comment
|
|
211
|
+
with:
|
|
212
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
213
|
+
issue-number: ${{ inputs.issue-number }}
|
|
214
|
+
body: |
|
|
215
|
+
${{ env.REFUSE_MARKER }}
|
|
216
|
+
This issue is too large to refine automatically: ${{ needs.size_guard.outputs.reason }}.
|
|
217
|
+
A story this size does not fail fast; it burns a full agent run and dies mid-generation at the provider's response cap, producing nothing.
|
|
218
|
+
|
|
219
|
+
Split it into smaller issues, each describing one story, or edit this body down under the limits (at most ${{ env.REFINE_MAX_BODY_CHARS }} characters and ${{ env.REFINE_MAX_WORK_UNITS }} work-unit bullets), then reply here and refinement will run again.
|
|
220
|
+
- name: Flag the issue for a person
|
|
221
|
+
uses: ./.github/actions/add-issue-labels
|
|
222
|
+
with:
|
|
223
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
224
|
+
issue-number: ${{ inputs.issue-number }}
|
|
225
|
+
labels: review
|
|
226
|
+
- name: Clear a stale stalled flag
|
|
227
|
+
uses: ./.github/actions/remove-issue-labels
|
|
228
|
+
with:
|
|
229
|
+
token: ${{ steps.app-token.outputs.token }}
|
|
230
|
+
issue-number: ${{ inputs.issue-number }}
|
|
231
|
+
labels: stalled
|
|
232
|
+
reserve:
|
|
233
|
+
needs: [still_open, size_guard]
|
|
234
|
+
if: needs.still_open.outputs.open == 'true' && needs.size_guard.outputs.too_big != 'true'
|
|
235
|
+
runs-on: agents-arc
|
|
137
236
|
permissions:
|
|
138
237
|
contents: read
|
|
139
238
|
issues: write
|
|
@@ -374,7 +473,9 @@ jobs:
|
|
|
374
473
|
issue-number: ${{ inputs.issue-number }}
|
|
375
474
|
labels: ${{ env.WORKING_LABEL }}
|
|
376
475
|
incomplete:
|
|
377
|
-
needs
|
|
476
|
+
# activation for the artifact prefix the usage read needs; validate_output for its
|
|
477
|
+
# valid output, which decides whether the usage read should look for truncation at all.
|
|
478
|
+
needs: [activation, agent, safe_outputs, validate_output]
|
|
378
479
|
if: >
|
|
379
480
|
always() &&
|
|
380
481
|
(
|
|
@@ -407,6 +508,33 @@ jobs:
|
|
|
407
508
|
under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
|
|
408
509
|
# Recorded before any label moves, so a failure in the steps below leaves a run that can be
|
|
409
510
|
# counted rather than work released with nothing to show for it.
|
|
511
|
+
# The usage read distinguishes the two very different failures that end here. A provider
|
|
512
|
+
# outage consumes almost no tokens; an output-cap truncation (Pliny-Bot #305: 32,000 output
|
|
513
|
+
# tokens, zero outcomes, run green) burns a full run and emits nothing. The comment names
|
|
514
|
+
# the second so a person does not triage it as a flaky provider. Best-effort by design:
|
|
515
|
+
# on an attempt where the agent job itself died there is no artifact, and this step must
|
|
516
|
+
# never fail the incomplete job's label release.
|
|
517
|
+
- name: Read the agent's token usage
|
|
518
|
+
id: usage
|
|
519
|
+
if: needs.agent.result == 'success' && needs.safe_outputs.result == 'success' && needs.validate_output.outputs.valid != 'true'
|
|
520
|
+
env:
|
|
521
|
+
GH_TOKEN: ${{ github.token }}
|
|
522
|
+
REPO: ${{ github.repository }}
|
|
523
|
+
RUN_ID: ${{ github.run_id }}
|
|
524
|
+
ARTIFACT: ${{ needs.activation.outputs.artifact_prefix }}agent
|
|
525
|
+
run: |
|
|
526
|
+
set -euo pipefail
|
|
527
|
+
output_tokens=""
|
|
528
|
+
if gh run download "$RUN_ID" --repo "$REPO" --name "$ARTIFACT" --dir usage-read 2>/dev/null \
|
|
529
|
+
&& [ -f usage-read/agent_usage.json ]; then
|
|
530
|
+
output_tokens=$(jq -r '.output_tokens // empty' usage-read/agent_usage.json 2>/dev/null || echo "")
|
|
531
|
+
fi
|
|
532
|
+
if [ -n "$output_tokens" ] && [ "$output_tokens" -gt 20000 ]; then
|
|
533
|
+
echo "truncated=This attempt consumed ${output_tokens} output tokens and still produced no outcome. That is what provider output truncation looks like, not an outage: the model hit the single-response cap mid-generation and every call it was writing was lost. If this repeats, split the issue or tighten its body." >> "$GITHUB_OUTPUT"
|
|
534
|
+
else
|
|
535
|
+
echo "truncated=" >> "$GITHUB_OUTPUT"
|
|
536
|
+
fi
|
|
537
|
+
echo "output_tokens=$output_tokens" >> "$GITHUB_OUTPUT"
|
|
410
538
|
- name: Report the failed attempt
|
|
411
539
|
if: steps.decide.outputs.retry == 'true'
|
|
412
540
|
uses: ./.github/actions/create-issue-comment
|
|
@@ -417,6 +545,7 @@ jobs:
|
|
|
417
545
|
${{ env.REFINE_ATTEMPT_MARKER }}
|
|
418
546
|
Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
|
|
419
547
|
${{ env.RETRY_COMMENT }}
|
|
548
|
+
${{ steps.usage.outputs.truncated }}
|
|
420
549
|
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
421
550
|
- name: Release the reservation for the retry
|
|
422
551
|
if: steps.decide.outputs.retry == 'true'
|
|
@@ -466,9 +595,10 @@ jobs:
|
|
|
466
595
|
body: |
|
|
467
596
|
${{ env.REFINE_MARKER }}
|
|
468
597
|
${{ env.INCOMPLETE_COMMENT }}
|
|
598
|
+
${{ steps.usage.outputs.truncated }}
|
|
469
599
|
[View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
|
|
470
600
|
|
|
471
|
-
if: inputs.issue-number != '' && needs.still_open.outputs.open == 'true'
|
|
601
|
+
if: inputs.issue-number != '' && needs.still_open.outputs.open == 'true' && needs.size_guard.outputs.too_big != 'true'
|
|
472
602
|
|
|
473
603
|
runs-on: agents-arc
|
|
474
604
|
runs-on-slim: agents-arc
|
|
@@ -483,9 +613,9 @@ engine:
|
|
|
483
613
|
OPENAI_BASE_URL: https://forge.plainconcepts.com/v1
|
|
484
614
|
args:
|
|
485
615
|
- "--model"
|
|
486
|
-
- "plainconcepts/glm-5-
|
|
616
|
+
- "plainconcepts/glm-5-2"
|
|
487
617
|
|
|
488
|
-
model: openai/glm-5-
|
|
618
|
+
model: openai/glm-5-2
|
|
489
619
|
# 150 rather than 500. With a four-hour clock this is the loop guard, and the worst
|
|
490
620
|
# observed run used 44 turns, so this leaves better than three times the worst case.
|
|
491
621
|
max-turns: 150
|
|
@@ -623,7 +753,16 @@ timeout-minutes: 90
|
|
|
623
753
|
The visible line is for people and the marker is read by the workflow, which turns it into the
|
|
624
754
|
`sp-N` label. A body without the marker gets no estimate label at all.
|
|
625
755
|
|
|
626
|
-
7.
|
|
756
|
+
7. **One safe-output call per turn.** Never batch multiple safe-output calls into a single
|
|
757
|
+
message: `update_issue`, `create_issue` and `add_comment` each go in their own turn, with
|
|
758
|
+
nothing else in the message. The provider caps one response at a fixed size, and a batch of
|
|
759
|
+
large calls is truncated mid-JSON before any of them executes, ending the run green with
|
|
760
|
+
nothing written. On the split path, sequence `create_issue` → `create_issue` → … →
|
|
761
|
+
`update_issue` → `add_comment`, one per turn. If a single body is so large it approaches the
|
|
762
|
+
size of a very long message, tighten the body; a shorter call that lands beats a longer one
|
|
763
|
+
that is cut off.
|
|
764
|
+
|
|
765
|
+
8. Decide exactly one outcome:
|
|
627
766
|
|
|
628
767
|
Labels are workflow-owned state. Do not call `add_labels` or `remove_labels`.
|
|
629
768
|
|
|
@@ -19,7 +19,7 @@ env:
|
|
|
19
19
|
TRIAGE_MARKER: "<!-- agent-triage -->"
|
|
20
20
|
# A run that died before it produced anything is worth repeating; one that worked and then
|
|
21
21
|
# failed produced an answer that was wrong, and repeating it buys the same wrong answer later.
|
|
22
|
-
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-
|
|
22
|
+
# Duration is what separates them. Odyssey #190 died in three minutes on `Model 'glm-5-2' not
|
|
23
23
|
# found`, a gateway fault that clears in seconds, and waited on the janitor's six-hourly sweep
|
|
24
24
|
# because only the implement worker could do this.
|
|
25
25
|
TRIAGE_ATTEMPT_MARKER: "<!-- agent-triage-attempt -->"
|
|
@@ -409,9 +409,9 @@ engine:
|
|
|
409
409
|
OPENAI_BASE_URL: https://forge.plainconcepts.com/v1
|
|
410
410
|
args:
|
|
411
411
|
- "--model"
|
|
412
|
-
- "plainconcepts/glm-5-
|
|
412
|
+
- "plainconcepts/glm-5-2"
|
|
413
413
|
|
|
414
|
-
model: openai/glm-5-
|
|
414
|
+
model: openai/glm-5-2
|
|
415
415
|
max-turns: 120
|
|
416
416
|
max-turn-cache-misses: 3000
|
|
417
417
|
max-ai-credits: 5000
|