@cxi-lmai/ci-agent-platform 3.1.3 → 3.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/payload/ci-templates/claude-pipeline.gitlab-ci.yml +17 -1
- package/payload/ci-templates/scripts/lib/pipeline-common.sh +98 -10
- package/payload/ci-templates/scripts/metrics-snapshot.sh +5 -1
- package/payload/ci-templates/scripts/review-fix.sh +3 -0
- package/payload/ci-templates/scripts/review.sh +8 -1
- package/payload/ci-templates/scripts/test-fix.sh +3 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cxi-lmai/ci-agent-platform",
|
|
3
|
-
"version": "3.1.
|
|
3
|
+
"version": "3.1.5",
|
|
4
4
|
"description": "Autonomous dev pipeline on plain GitLab CI or GitHub Actions, driven by Claude Code. A labeled issue goes in, an open merge request comes out.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude",
|
|
@@ -79,6 +79,18 @@ variables:
|
|
|
79
79
|
PIPE_ORCHESTRATE: "0" # set to "1" in the schedule to run orchestrate
|
|
80
80
|
PIPE_SPEC_TEMPLATE_PATH: ".gitlab/issue_templates/Spec.md"
|
|
81
81
|
|
|
82
|
+
# --- Agent deadline (omp harness) -----------------------------------------
|
|
83
|
+
# A job `timeout:` is a hard kill — the runner SIGKILLs the process tree and
|
|
84
|
+
# the runner script never posts its comment, metrics or marker file. So
|
|
85
|
+
# `pipe_run_omp` bounds the agent itself with `omp --max-time`, derived from
|
|
86
|
+
# GitLab's own CI_JOB_TIMEOUT minus the time already spent minus the reserve
|
|
87
|
+
# below, which leaves room for the post-agent steps. An overrun then takes
|
|
88
|
+
# the normal degraded path (failure-notice comment + metrics) instead of
|
|
89
|
+
# dying silently. Set PIPE_AGENT_MAX_TIME (e.g. "20m") to override the
|
|
90
|
+
# derivation for every job; GitHub Actions exposes no job-timeout variable,
|
|
91
|
+
# so there the flag is omitted and timeout-minutes stays the only bound.
|
|
92
|
+
PIPE_AGENT_TIME_RESERVE: "180" # seconds of the job budget kept for the post-agent steps
|
|
93
|
+
|
|
82
94
|
# --- Build / verify --------------------------------------------------------
|
|
83
95
|
PIPE_VERIFY_CMD: "" # compile + unit tests, run after a fix
|
|
84
96
|
PIPE_COMPILE_CMD: "" # compile only (defaults to PIPE_VERIFY_CMD)
|
|
@@ -212,11 +224,15 @@ variables:
|
|
|
212
224
|
|
|
213
225
|
# =============================================================================
|
|
214
226
|
# review: run the automated review on a non-draft MR that changes code.
|
|
227
|
+
# The 45m budget matches `code` and is not generous: the skill fans out to the
|
|
228
|
+
# code-reviewer plus up to three specialist subagents, and observed runs on a
|
|
229
|
+
# real project spread from 1m to over 20m with diff size. It was 20m until
|
|
230
|
+
# three consecutive cap-outs on large MRs were measured in production.
|
|
215
231
|
# =============================================================================
|
|
216
232
|
review:
|
|
217
233
|
extends: .claude-base
|
|
218
234
|
stage: review
|
|
219
|
-
timeout:
|
|
235
|
+
timeout: 45m
|
|
220
236
|
script:
|
|
221
237
|
- bash "$PIPE_SCRIPTS_DIR/review.sh"
|
|
222
238
|
artifacts:
|
|
@@ -277,10 +277,16 @@ pipe_run_claude() {
|
|
|
277
277
|
# $3 = allowed tools (optional), $4 = model for the main loop (optional,
|
|
278
278
|
# e.g. $PIPE_MODEL_REVIEW; subagents keep their frontmatter models),
|
|
279
279
|
# stdin = extra prompt (optional, usually empty)
|
|
280
|
+
#
|
|
281
|
+
# Always returns 0, so a caller running under `set -e` keeps control over what
|
|
282
|
+
# a failed agent means for its job. The agent's real exit code lands in
|
|
283
|
+
# PIPE_AGENT_RC. A caller that must not read a crashed agent as a clean run
|
|
284
|
+
# checks that variable, or calls pipe_require_agent_ran below.
|
|
280
285
|
local label="$1" skill="$2" tools="${3:-Agent,Read,Write,Edit,Glob,Grep,Bash}" model="${4:-}"
|
|
281
286
|
pipe_require_harness claude
|
|
282
287
|
local stdin_file; stdin_file=$(mktemp)
|
|
283
288
|
cat > "$stdin_file" || true
|
|
289
|
+
PIPE_AGENT_RC=0
|
|
284
290
|
|
|
285
291
|
if [ "$(id -u)" -eq 0 ]; then
|
|
286
292
|
local runuser="pipelinebot"
|
|
@@ -315,7 +321,7 @@ pipe_run_claude() {
|
|
|
315
321
|
} > "$runner"
|
|
316
322
|
chmod +x "$runner"
|
|
317
323
|
chown "$runuser:$runuser" "$runner"
|
|
318
|
-
su -m "$runuser" -s /bin/bash "$runner" ||
|
|
324
|
+
su -m "$runuser" -s /bin/bash "$runner" || PIPE_AGENT_RC=$?
|
|
319
325
|
chown -R root:root "$PWD" 2>/dev/null || true
|
|
320
326
|
rm -f "$runner"
|
|
321
327
|
else
|
|
@@ -327,9 +333,32 @@ pipe_run_claude() {
|
|
|
327
333
|
else
|
|
328
334
|
capture_claude "$label" -- --dangerously-skip-permissions --allowedTools "$tools" -p "$skill" < "$stdin_file"
|
|
329
335
|
fi
|
|
330
|
-
) ||
|
|
336
|
+
) || PIPE_AGENT_RC=$?
|
|
331
337
|
fi
|
|
332
338
|
rm -f "$stdin_file"
|
|
339
|
+
[ "$PIPE_AGENT_RC" -eq 0 ] || pipe_log "agent exited with code $PIPE_AGENT_RC"
|
|
340
|
+
return 0
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
pipe_require_agent_ran() {
|
|
344
|
+
# $1 = file the skill was supposed to write (empty to check the exit code
|
|
345
|
+
# only, for skills whose marker is legitimately optional), $2 = human label.
|
|
346
|
+
# Fails the job when the agent crashed or wrote nothing. Without this an
|
|
347
|
+
# agent that never started is indistinguishable from one that ran and found
|
|
348
|
+
# nothing: every caller reads its marker file, misses it, and falls back to
|
|
349
|
+
# the "all clear" default, so the job goes green without a review, fix, or
|
|
350
|
+
# triage having happened.
|
|
351
|
+
local marker="$1" what="${2:-agent}"
|
|
352
|
+
if [ "${PIPE_AGENT_RC:-0}" -ne 0 ]; then
|
|
353
|
+
pipe_log "ERROR: $what did not complete (exit code $PIPE_AGENT_RC). See the job log above."
|
|
354
|
+
return 1
|
|
355
|
+
fi
|
|
356
|
+
[ -n "$marker" ] || return 0
|
|
357
|
+
if [ ! -s "$marker" ]; then
|
|
358
|
+
pipe_log "ERROR: $what produced no output ($marker is missing or empty)."
|
|
359
|
+
return 1
|
|
360
|
+
fi
|
|
361
|
+
return 0
|
|
333
362
|
}
|
|
334
363
|
|
|
335
364
|
# --- omp invocation -----------------------------------------------------
|
|
@@ -369,17 +398,72 @@ pipe_translate_tools_to_omp() {
|
|
|
369
398
|
echo "${result#,}"
|
|
370
399
|
}
|
|
371
400
|
|
|
401
|
+
# --- Agent deadline -----------------------------------------------------
|
|
402
|
+
# A CI job timeout is a hard kill: the runner SIGKILLs the whole process tree,
|
|
403
|
+
# so the runner script never reaches the steps that post the comment, write
|
|
404
|
+
# the metrics record and emit the marker file. An agent that overruns
|
|
405
|
+
# therefore leaves a red job with no output at all (observed in production:
|
|
406
|
+
# three review timeouts in one week, each at the cap with the agent mid-run).
|
|
407
|
+
# Bounding the agent itself with `omp --max-time` turns an overrun into the
|
|
408
|
+
# platform's normal degraded path — failure-notice comment plus metrics —
|
|
409
|
+
# instead of a hard kill.
|
|
410
|
+
|
|
411
|
+
pipe_agent_max_time() {
|
|
412
|
+
# Echoes the --max-time value to pass to omp, or nothing when no deadline
|
|
413
|
+
# can be derived (GitHub Actions exposes no CI_JOB_TIMEOUT equivalent;
|
|
414
|
+
# there the workflow's timeout-minutes stays the only bound). An explicit
|
|
415
|
+
# PIPE_AGENT_MAX_TIME wins and is passed through verbatim, so a project can
|
|
416
|
+
# set "20m" or "1h" by hand. Otherwise the budget is the job timeout, minus
|
|
417
|
+
# the time already spent (image pull, checkout, before_script, API prep),
|
|
418
|
+
# minus PIPE_AGENT_TIME_RESERVE for the post-agent steps.
|
|
419
|
+
if [ -n "${PIPE_AGENT_MAX_TIME:-}" ]; then
|
|
420
|
+
printf '%s' "$PIPE_AGENT_MAX_TIME"
|
|
421
|
+
return 0
|
|
422
|
+
fi
|
|
423
|
+
local budget="${CI_JOB_TIMEOUT:-}"
|
|
424
|
+
case "$budget" in ''|*[!0-9]*) return 0 ;; esac
|
|
425
|
+
local reserve="${PIPE_AGENT_TIME_RESERVE:-180}" elapsed=0 started now
|
|
426
|
+
# Guard the emptiness explicitly: `date -d ""` does not fail, it silently
|
|
427
|
+
# resolves to midnight today, which would read as a many-hour-old job.
|
|
428
|
+
if [ -n "${CI_JOB_STARTED_AT:-}" ]; then
|
|
429
|
+
started=$(date -d "$CI_JOB_STARTED_AT" +%s 2>/dev/null || true)
|
|
430
|
+
if [ -n "$started" ]; then
|
|
431
|
+
now=$(date +%s)
|
|
432
|
+
elapsed=$(( now - started ))
|
|
433
|
+
[ "$elapsed" -lt 0 ] && elapsed=0
|
|
434
|
+
fi
|
|
435
|
+
fi
|
|
436
|
+
local left=$(( budget - elapsed - reserve ))
|
|
437
|
+
# A job already out of budget gets no flag: let the agent start and be cut
|
|
438
|
+
# by the runner exactly as it was before this function existed, rather than
|
|
439
|
+
# be killed one second in with a nonsensical deadline.
|
|
440
|
+
[ "$left" -lt 60 ] && return 0
|
|
441
|
+
printf '%s' "$left"
|
|
442
|
+
}
|
|
443
|
+
|
|
372
444
|
pipe_run_omp() {
|
|
373
445
|
# $1 = agent label for metrics, $2 = skill invocation (e.g. "/review-mr"),
|
|
374
446
|
# $3 = allowed tools in Claude Code naming (optional, translated
|
|
375
447
|
# internally), $4 = model for the main loop (optional, e.g.
|
|
376
448
|
# $PIPE_MODEL_REVIEW; subagents keep the models from their own
|
|
377
449
|
# frontmatter), stdin = extra prompt (optional, usually empty).
|
|
450
|
+
#
|
|
451
|
+
# Same contract as pipe_run_claude: always returns 0 and reports the agent's
|
|
452
|
+
# real exit code in PIPE_AGENT_RC, which pipe_require_agent_ran checks.
|
|
378
453
|
local label="$1" skill="$2" tools="${3:-Agent,Read,Write,Edit,Glob,Grep,Bash}" model="${4:-}"
|
|
379
454
|
local omp_tools; omp_tools=$(pipe_translate_tools_to_omp "$tools")
|
|
455
|
+
# Bound the run below the job's own hard kill (see pipe_agent_max_time).
|
|
456
|
+
local max_time; max_time=$(pipe_agent_max_time)
|
|
457
|
+
local -a max_time_args=()
|
|
458
|
+
local max_time_txt=""
|
|
459
|
+
if [ -n "$max_time" ]; then
|
|
460
|
+
max_time_args=(--max-time "$max_time")
|
|
461
|
+
max_time_txt=$(printf ' --max-time %q' "$max_time")
|
|
462
|
+
fi
|
|
380
463
|
pipe_require_harness omp
|
|
381
464
|
local stdin_file; stdin_file=$(mktemp)
|
|
382
465
|
cat > "$stdin_file" || true
|
|
466
|
+
PIPE_AGENT_RC=0
|
|
383
467
|
|
|
384
468
|
if [ "$(id -u)" -eq 0 ]; then
|
|
385
469
|
local runuser="pipelinebot"
|
|
@@ -402,17 +486,19 @@ pipe_run_omp() {
|
|
|
402
486
|
printf 'cd %q\n' "$PWD"
|
|
403
487
|
printf 'source %q\n' "$SCRIPT_LIB_DIR/usage-capture-omp.sh"
|
|
404
488
|
printf 'export PIPE_CAPTURE_MODEL=%q\n' "$model"
|
|
489
|
+
# The %s after `--` is the optional " --max-time <d>" fragment, already
|
|
490
|
+
# shell-quoted; it collapses to nothing when no deadline was derived.
|
|
405
491
|
if [ -n "$model" ]; then
|
|
406
|
-
printf 'capture_omp %q
|
|
407
|
-
"$label" "$model" "$omp_tools" "$skill" "$stdin_file"
|
|
492
|
+
printf 'capture_omp %q --%s --yolo --no-session --no-title --model %q --tools %q -p %q < %q\n' \
|
|
493
|
+
"$label" "$max_time_txt" "$model" "$omp_tools" "$skill" "$stdin_file"
|
|
408
494
|
else
|
|
409
|
-
printf 'capture_omp %q
|
|
410
|
-
"$label" "$omp_tools" "$skill" "$stdin_file"
|
|
495
|
+
printf 'capture_omp %q --%s --yolo --no-session --no-title --tools %q -p %q < %q\n' \
|
|
496
|
+
"$label" "$max_time_txt" "$omp_tools" "$skill" "$stdin_file"
|
|
411
497
|
fi
|
|
412
498
|
} > "$runner"
|
|
413
499
|
chmod +x "$runner"
|
|
414
500
|
chown "$runuser:$runuser" "$runner"
|
|
415
|
-
su -m "$runuser" -s /bin/bash "$runner" ||
|
|
501
|
+
su -m "$runuser" -s /bin/bash "$runner" || PIPE_AGENT_RC=$?
|
|
416
502
|
chown -R root:root "$PWD" 2>/dev/null || true
|
|
417
503
|
rm -f "$runner"
|
|
418
504
|
else
|
|
@@ -420,13 +506,15 @@ pipe_run_omp() {
|
|
|
420
506
|
pipe_scrub_agent_secrets
|
|
421
507
|
export PIPE_CAPTURE_MODEL="$model"
|
|
422
508
|
if [ -n "$model" ]; then
|
|
423
|
-
capture_omp "$label" -- --yolo --no-session --no-title --model "$model" --tools "$omp_tools" -p "$skill" < "$stdin_file"
|
|
509
|
+
capture_omp "$label" -- "${max_time_args[@]}" --yolo --no-session --no-title --model "$model" --tools "$omp_tools" -p "$skill" < "$stdin_file"
|
|
424
510
|
else
|
|
425
|
-
capture_omp "$label" -- --yolo --no-session --no-title --tools "$omp_tools" -p "$skill" < "$stdin_file"
|
|
511
|
+
capture_omp "$label" -- "${max_time_args[@]}" --yolo --no-session --no-title --tools "$omp_tools" -p "$skill" < "$stdin_file"
|
|
426
512
|
fi
|
|
427
|
-
) ||
|
|
513
|
+
) || PIPE_AGENT_RC=$?
|
|
428
514
|
fi
|
|
429
515
|
rm -f "$stdin_file"
|
|
516
|
+
[ "$PIPE_AGENT_RC" -eq 0 ] || pipe_log "agent exited with code $PIPE_AGENT_RC"
|
|
517
|
+
return 0
|
|
430
518
|
}
|
|
431
519
|
|
|
432
520
|
# --- Harness dispatch ---------------------------------------------------
|
|
@@ -186,9 +186,13 @@ MRS_VIEW=$(jq -c \
|
|
|
186
186
|
elif ($body | test($reintroduce; "i")) then "reintroduces"
|
|
187
187
|
else null end
|
|
188
188
|
),
|
|
189
|
+
# match/2 with "g", not scan/2: a Debian bookworm based runner image
|
|
190
|
+
# ships jq 1.6, which has no scan/2 and fails the whole program with
|
|
191
|
+
# a compile error.
|
|
189
192
|
references: [
|
|
190
193
|
$body
|
|
191
|
-
|
|
|
194
|
+
| match("(?:" + $revert + "|" + $regression + "|" + $reintroduce + ")[[:space:]]*[!#]?([0-9]+)"; "gi")
|
|
195
|
+
| .captures[].string
|
|
192
196
|
| select(. != null)
|
|
193
197
|
| tonumber
|
|
194
198
|
]
|
|
@@ -41,6 +41,9 @@ echo "$NOTES" | jq -r --arg u "${PIPE_BOT_USER:-}" --arg m "$REVIEW_MARKER" \
|
|
|
41
41
|
# verifies, writes the result JSON, and the review-fix.env marker. On cap it
|
|
42
42
|
# runs the shared postmortem flow (writing postmortem.md + postmortem.env).
|
|
43
43
|
pipe_run_agent coder "/fix-review-findings" "Agent,Read,Write,Edit,Glob,Grep,Bash" "$PIPE_MODEL_CODE" < /dev/null
|
|
44
|
+
# A crashed agent writes no marker, which step 6 would read as "nothing changed"
|
|
45
|
+
# and step 7 would report as a completed fix run that had nothing to do.
|
|
46
|
+
pipe_require_agent_ran "" "The review-fix agent" || exit 1
|
|
44
47
|
|
|
45
48
|
# 6. Read the marker the skill wrote.
|
|
46
49
|
RF_ENV="$PIPE_CONTEXT_DIR/review-fix.env"
|
|
@@ -87,7 +87,14 @@ pipe_metric_event code-reviewer review_findings \
|
|
|
87
87
|
--argjson inc "$([ "${PIPE_IS_INCREMENTAL:-false}" = "true" ] && echo true || echo false)" \
|
|
88
88
|
'{findings_total: $t, status: $s, has_fixable_bugs: $b, is_incremental: $inc}')"
|
|
89
89
|
|
|
90
|
-
# 7.
|
|
90
|
+
# 7. Fail the job when no review was produced. The comment above already told
|
|
91
|
+
# the humans; this tells the pipeline. PIPE_NO_GATE does not apply: it turns
|
|
92
|
+
# off the verdict gate, and this is a broken run, not a verdict. Without it a
|
|
93
|
+
# review that never ran (crashed agent, missing ANTHROPIC_API_KEY) reads as a
|
|
94
|
+
# review that ran and found nothing, and the MR goes green unreviewed.
|
|
95
|
+
pipe_require_agent_ran "$PIPE_RESULT_REVIEW" "The review agent" || exit 1
|
|
96
|
+
|
|
97
|
+
# 8. Gate. review.env is exported so the fix job can read REVIEW_HAS_BUGS at
|
|
91
98
|
# runtime. On GitLab the job exits non-zero to block the pipeline. On GitHub
|
|
92
99
|
# the workflow step owns the gate (it reads the output first), so it sets
|
|
93
100
|
# PIPE_NO_GATE=1 and this script leaves the exit to the workflow.
|
|
@@ -27,6 +27,9 @@ COVERAGE_BEFORE=$(pipe_count_fix_commits "$PIPE_COMMIT_COVERAGE")
|
|
|
27
27
|
# test-writer, verifies, and writes fix-tests.env. On cap it runs the shared
|
|
28
28
|
# postmortem flow (writing postmortem.md + postmortem.env).
|
|
29
29
|
pipe_run_agent test-fix "/fix-tests" "Agent,Read,Write,Edit,Glob,Grep,Bash" "$PIPE_MODEL_CODE" < /dev/null
|
|
30
|
+
# A crashed agent writes no marker, which step 4 would read as FIX_BRANCH=none
|
|
31
|
+
# and report as a false-alarm trigger, hiding the failure behind a green job.
|
|
32
|
+
pipe_require_agent_ran "" "The test-fix agent" || exit 1
|
|
30
33
|
|
|
31
34
|
# 4. Read the marker the skill wrote.
|
|
32
35
|
FT_ENV="$PIPE_CONTEXT_DIR/fix-tests.env"
|