@plainconceptsplatform/workflows 0.27.6 → 0.28.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -32,6 +32,14 @@ env:
32
32
  REFINE_ATTEMPT_MARKER: "<!-- agent-refine-attempt -->"
33
33
  MAX_ATTEMPTS: "5"
34
34
  PARK_AT_ATTEMPT: "4"
35
+ # The size gate. An issue over either limit is refused before any model runs: a real
36
+ # one, Pliny-Bot #305 (13,398 chars, 92 work units), burned a full agent -- 17 minutes,
37
+ # 2.4M tokens -- and died at the provider's 32K output cap with zero outcome, twice.
38
+ # Both numbers are measured before the agent starts, so refusal is deterministic and
39
+ # costs ten seconds instead of half an hour.
40
+ REFINE_MAX_BODY_CHARS: "8000"
41
+ REFINE_MAX_WORK_UNITS: "24"
42
+ REFUSE_MARKER: "<!-- agent-refine-refused -->"
35
43
  # Ten rather than six: the failure that prompted this took three minutes, and a slower one on a
36
44
  # worse day would fall outside a six-minute window and park for a fault that clears by itself.
37
45
  RETRY_UNDER_MINUTES: "10"
@@ -98,12 +106,12 @@ on:
98
106
  required: false
99
107
  type: string
100
108
  default: first
101
- # The gate job that the top-level `if:` reads. gh-aw folds that `if:` into the generated
102
- # activation job but gives activation no dependency on the job, so the reference resolves
109
+ # The gate jobs that the top-level `if:` reads. gh-aw folds that `if:` into the generated
110
+ # activation job but gives activation no dependency on the jobs, so the reference resolves
103
111
  # to '' and the clause is false -- the agent would never run. The package's own validator
104
112
  # catches it after compilation; this is the line it asks for, the same one the merge gate
105
113
  # uses for protected_changes.
106
- needs: [still_open]
114
+ needs: [still_open, size_guard]
107
115
 
108
116
  jobs:
109
117
  # A route dispatched while the issue was open must not execute after it has been closed. The
@@ -130,10 +138,101 @@ jobs:
130
138
  with:
131
139
  token: ${{ github.token }}
132
140
  issue-number: ${{ inputs.issue-number }}
133
- reserve:
141
+ # The size gate, rung 4. Measures the issue before any model starts, because an oversized
142
+ # body does not fail fast on its own: it produces an agent that explores for seventeen
143
+ # minutes and then dies mid-generation at the provider's response cap having written
144
+ # nothing, which reads as a green run with no outcome (Pliny-Bot #305, twice). Pure
145
+ # shell, no network beyond one issue read; over either limit it is a refusal, not a
146
+ # smaller attempt.
147
+ size_guard:
134
148
  needs: [still_open]
135
149
  if: needs.still_open.outputs.open == 'true'
136
150
  runs-on: agents-arc
151
+ timeout-minutes: 5
152
+ permissions:
153
+ contents: read
154
+ issues: read
155
+ outputs:
156
+ too_big: ${{ steps.measure.outputs.too_big }}
157
+ reason: ${{ steps.measure.outputs.reason }}
158
+ steps:
159
+ - name: Measure the issue against the size limits
160
+ id: measure
161
+ env:
162
+ GH_TOKEN: ${{ github.token }}
163
+ REPO: ${{ github.repository }}
164
+ ISSUE_NUMBER: ${{ inputs.issue-number }}
165
+ MAX_BODY_CHARS: ${{ env.REFINE_MAX_BODY_CHARS }}
166
+ MAX_WORK_UNITS: ${{ env.REFINE_MAX_WORK_UNITS }}
167
+ run: |
168
+ set -euo pipefail
169
+ body=$(gh issue view "$ISSUE_NUMBER" --repo "$REPO" --json title,body --jq '.title + "\n\n" + (.body // "")')
170
+ chars=$(printf '%s' "$body" | wc -c)
171
+ # One work unit per markdown bullet, the same split the prompt's step 3 makes.
172
+ units=$(printf '%s' "$body" | grep -cE '^[[:space:]]*[-*] ' || true)
173
+ reason=""
174
+ too_big=false
175
+ if [ "$chars" -gt "$MAX_BODY_CHARS" ]; then
176
+ too_big=true
177
+ reason="body ${chars} chars > ${MAX_BODY_CHARS}"
178
+ fi
179
+ if [ "$units" -gt "$MAX_WORK_UNITS" ]; then
180
+ too_big=true
181
+ reason="${reason:+$reason; }work units ${units} > ${MAX_WORK_UNITS}"
182
+ fi
183
+ echo "too_big=$too_big" >> "$GITHUB_OUTPUT"
184
+ echo "reason=$reason" >> "$GITHUB_OUTPUT"
185
+ echo "::notice::measured $chars chars, $units work units (limits: ${MAX_BODY_CHARS} chars, ${MAX_WORK_UNITS} units) -> $too_big"
186
+ # The deterministic refusal. Runs only when the size gate tripped, so it costs one
187
+ # comment and no model. The refine label stays: a human who shrinks the body and
188
+ # replies re-enters rerefine through the comment route, and the classifier ignores
189
+ # bot comments, so this comment cannot re-trigger the worker.
190
+ refuse_big_issue:
191
+ needs: [still_open, size_guard]
192
+ if: needs.still_open.outputs.open == 'true' && needs.size_guard.outputs.too_big == 'true'
193
+ runs-on: agents-arc
194
+ timeout-minutes: 5
195
+ permissions:
196
+ contents: read
197
+ issues: write
198
+ steps:
199
+ - name: Checkout workflow actions
200
+ uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
201
+ with:
202
+ persist-credentials: false
203
+ - name: Create bot token
204
+ id: app-token
205
+ uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
206
+ with:
207
+ client-id: ${{ secrets.BOT_APP_ID }}
208
+ private-key: ${{ secrets.BOT_PRIVATE_KEY }}
209
+ - name: Refuse the oversized issue
210
+ uses: ./.github/actions/create-issue-comment
211
+ with:
212
+ token: ${{ steps.app-token.outputs.token }}
213
+ issue-number: ${{ inputs.issue-number }}
214
+ body: |
215
+ ${{ env.REFUSE_MARKER }}
216
+ This issue is too large to refine automatically: ${{ needs.size_guard.outputs.reason }}.
217
+ A story this size does not fail fast; it burns a full agent run and dies mid-generation at the provider's response cap, producing nothing.
218
+
219
+ Split it into smaller issues, each describing one story, or edit this body down under the limits (at most ${{ env.REFINE_MAX_BODY_CHARS }} characters and ${{ env.REFINE_MAX_WORK_UNITS }} work-unit bullets), then reply here and refinement will run again.
220
+ - name: Flag the issue for a person
221
+ uses: ./.github/actions/add-issue-labels
222
+ with:
223
+ token: ${{ steps.app-token.outputs.token }}
224
+ issue-number: ${{ inputs.issue-number }}
225
+ labels: review
226
+ - name: Clear a stale stalled flag
227
+ uses: ./.github/actions/remove-issue-labels
228
+ with:
229
+ token: ${{ steps.app-token.outputs.token }}
230
+ issue-number: ${{ inputs.issue-number }}
231
+ labels: stalled
232
+ reserve:
233
+ needs: [still_open, size_guard]
234
+ if: needs.still_open.outputs.open == 'true' && needs.size_guard.outputs.too_big != 'true'
235
+ runs-on: agents-arc
137
236
  permissions:
138
237
  contents: read
139
238
  issues: write
@@ -374,7 +473,9 @@ jobs:
374
473
  issue-number: ${{ inputs.issue-number }}
375
474
  labels: ${{ env.WORKING_LABEL }}
376
475
  incomplete:
377
- needs: [agent, safe_outputs, validate_output]
476
+ # activation for the artifact prefix the usage read needs; validate_output for its
477
+ # valid output, which decides whether the usage read should look for truncation at all.
478
+ needs: [activation, agent, safe_outputs, validate_output]
378
479
  if: >
379
480
  always() &&
380
481
  (
@@ -407,6 +508,33 @@ jobs:
407
508
  under-minutes: ${{ env.RETRY_UNDER_MINUTES }}
408
509
  # Recorded before any label moves, so a failure in the steps below leaves a run that can be
409
510
  # counted rather than work released with nothing to show for it.
511
+ # The usage read distinguishes the two very different failures that end here. A provider
512
+ # outage consumes almost no tokens; an output-cap truncation (Pliny-Bot #305: 32,000 output
513
+ # tokens, zero outcomes, run green) burns a full run and emits nothing. The comment names
514
+ # the second so a person does not triage it as a flaky provider. Best-effort by design:
515
+ # on an attempt where the agent job itself died there is no artifact, and this step must
516
+ # never fail the incomplete job's label release.
517
+ - name: Read the agent's token usage
518
+ id: usage
519
+ if: needs.agent.result == 'success' && needs.safe_outputs.result == 'success' && needs.validate_output.outputs.valid != 'true'
520
+ env:
521
+ GH_TOKEN: ${{ github.token }}
522
+ REPO: ${{ github.repository }}
523
+ RUN_ID: ${{ github.run_id }}
524
+ ARTIFACT: ${{ needs.activation.outputs.artifact_prefix }}agent
525
+ run: |
526
+ set -euo pipefail
527
+ output_tokens=""
528
+ if gh run download "$RUN_ID" --repo "$REPO" --name "$ARTIFACT" --dir usage-read 2>/dev/null \
529
+ && [ -f usage-read/agent_usage.json ]; then
530
+ output_tokens=$(jq -r '.output_tokens // empty' usage-read/agent_usage.json 2>/dev/null || echo "")
531
+ fi
532
+ if [ -n "$output_tokens" ] && [ "$output_tokens" -gt 20000 ]; then
533
+ echo "truncated=This attempt consumed ${output_tokens} output tokens and still produced no outcome. That is what provider output truncation looks like, not an outage: the model hit the single-response cap mid-generation and every call it was writing was lost. If this repeats, split the issue or tighten its body." >> "$GITHUB_OUTPUT"
534
+ else
535
+ echo "truncated=" >> "$GITHUB_OUTPUT"
536
+ fi
537
+ echo "output_tokens=$output_tokens" >> "$GITHUB_OUTPUT"
410
538
  - name: Report the failed attempt
411
539
  if: steps.decide.outputs.retry == 'true'
412
540
  uses: ./.github/actions/create-issue-comment
@@ -417,6 +545,7 @@ jobs:
417
545
  ${{ env.REFINE_ATTEMPT_MARKER }}
418
546
  Attempt ${{ steps.decide.outputs.next }} of ${{ env.MAX_ATTEMPTS }} ended after ${{ steps.decide.outputs.minutes }} minutes, before the run could produce an answer.
419
547
  ${{ env.RETRY_COMMENT }}
548
+ ${{ steps.usage.outputs.truncated }}
420
549
  [View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
421
550
  - name: Release the reservation for the retry
422
551
  if: steps.decide.outputs.retry == 'true'
@@ -466,9 +595,10 @@ jobs:
466
595
  body: |
467
596
  ${{ env.REFINE_MARKER }}
468
597
  ${{ env.INCOMPLETE_COMMENT }}
598
+ ${{ steps.usage.outputs.truncated }}
469
599
  [View this workflow run](${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }})
470
600
 
471
- if: inputs.issue-number != '' && needs.still_open.outputs.open == 'true'
601
+ if: inputs.issue-number != '' && needs.still_open.outputs.open == 'true' && needs.size_guard.outputs.too_big != 'true'
472
602
 
473
603
  runs-on: agents-arc
474
604
  runs-on-slim: agents-arc
@@ -623,7 +753,16 @@ timeout-minutes: 90
623
753
  The visible line is for people and the marker is read by the workflow, which turns it into the
624
754
  `sp-N` label. A body without the marker gets no estimate label at all.
625
755
 
626
- 7. Decide exactly one outcome:
756
+ 7. **One safe-output call per turn.** Never batch multiple safe-output calls into a single
757
+ message: `update_issue`, `create_issue` and `add_comment` each go in their own turn, with
758
+ nothing else in the message. The provider caps one response at a fixed size, and a batch of
759
+ large calls is truncated mid-JSON before any of them executes, ending the run green with
760
+ nothing written. On the split path, sequence `create_issue` → `create_issue` → … →
761
+ `update_issue` → `add_comment`, one per turn. If a single body is so large it approaches the
762
+ size of a very long message, tighten the body; a shorter call that lands beats a longer one
763
+ that is cut off.
764
+
765
+ 8. Decide exactly one outcome:
627
766
 
628
767
  Labels are workflow-owned state. Do not call `add_labels` or `remove_labels`.
629
768
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@plainconceptsplatform/workflows",
3
- "version": "0.27.6",
3
+ "version": "0.28.0",
4
4
  "description": "Install and update Platform GitHub agentic workflows.",
5
5
  "keywords": [
6
6
  "github-actions",