@cxi-lmai/ci-agent-platform 3.1.3 → 3.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cxi-lmai/ci-agent-platform",
3
- "version": "3.1.3",
3
+ "version": "3.1.5",
4
4
  "description": "Autonomous dev pipeline on plain GitLab CI or GitHub Actions, driven by Claude Code. A labeled issue goes in, an open merge request comes out.",
5
5
  "keywords": [
6
6
  "claude",
@@ -79,6 +79,18 @@ variables:
79
79
  PIPE_ORCHESTRATE: "0" # set to "1" in the schedule to run orchestrate
80
80
  PIPE_SPEC_TEMPLATE_PATH: ".gitlab/issue_templates/Spec.md"
81
81
 
82
+ # --- Agent deadline (omp harness) -----------------------------------------
83
+ # A job `timeout:` is a hard kill — the runner SIGKILLs the process tree and
84
+ # the runner script never posts its comment, metrics or marker file. So
85
+ # `pipe_run_omp` bounds the agent itself with `omp --max-time`, derived from
86
+ # GitLab's own CI_JOB_TIMEOUT minus the time already spent minus the reserve
87
+ # below, which leaves room for the post-agent steps. An overrun then takes
88
+ # the normal degraded path (failure-notice comment + metrics) instead of
89
+ # dying silently. Set PIPE_AGENT_MAX_TIME (e.g. "20m") to override the
90
+ # derivation for every job; GitHub Actions exposes no job-timeout variable,
91
+ # so there the flag is omitted and timeout-minutes stays the only bound.
92
+ PIPE_AGENT_TIME_RESERVE: "180" # seconds of the job budget kept for the post-agent steps
93
+
82
94
  # --- Build / verify --------------------------------------------------------
83
95
  PIPE_VERIFY_CMD: "" # compile + unit tests, run after a fix
84
96
  PIPE_COMPILE_CMD: "" # compile only (defaults to PIPE_VERIFY_CMD)
@@ -212,11 +224,15 @@ variables:
212
224
 
213
225
  # =============================================================================
214
226
  # review: run the automated review on a non-draft MR that changes code.
227
+ # The 45m budget matches `code` and is not generous: the skill fans out to the
228
+ # code-reviewer plus up to three specialist subagents, and observed runs on a
229
+ # real project spread from 1m to over 20m with diff size. It was 20m until
230
+ # three consecutive cap-outs on large MRs were measured in production.
215
231
  # =============================================================================
216
232
  review:
217
233
  extends: .claude-base
218
234
  stage: review
219
- timeout: 20m
235
+ timeout: 45m
220
236
  script:
221
237
  - bash "$PIPE_SCRIPTS_DIR/review.sh"
222
238
  artifacts:
@@ -277,10 +277,16 @@ pipe_run_claude() {
277
277
  # $3 = allowed tools (optional), $4 = model for the main loop (optional,
278
278
  # e.g. $PIPE_MODEL_REVIEW; subagents keep their frontmatter models),
279
279
  # stdin = extra prompt (optional, usually empty)
280
+ #
281
+ # Always returns 0, so a caller running under `set -e` keeps control over what
282
+ # a failed agent means for its job. The agent's real exit code lands in
283
+ # PIPE_AGENT_RC. A caller that must not read a crashed agent as a clean run
284
+ # checks that variable, or calls pipe_require_agent_ran below.
280
285
  local label="$1" skill="$2" tools="${3:-Agent,Read,Write,Edit,Glob,Grep,Bash}" model="${4:-}"
281
286
  pipe_require_harness claude
282
287
  local stdin_file; stdin_file=$(mktemp)
283
288
  cat > "$stdin_file" || true
289
+ PIPE_AGENT_RC=0
284
290
 
285
291
  if [ "$(id -u)" -eq 0 ]; then
286
292
  local runuser="pipelinebot"
@@ -315,7 +321,7 @@ pipe_run_claude() {
315
321
  } > "$runner"
316
322
  chmod +x "$runner"
317
323
  chown "$runuser:$runuser" "$runner"
318
- su -m "$runuser" -s /bin/bash "$runner" || true
324
+ su -m "$runuser" -s /bin/bash "$runner" || PIPE_AGENT_RC=$?
319
325
  chown -R root:root "$PWD" 2>/dev/null || true
320
326
  rm -f "$runner"
321
327
  else
@@ -327,9 +333,32 @@ pipe_run_claude() {
327
333
  else
328
334
  capture_claude "$label" -- --dangerously-skip-permissions --allowedTools "$tools" -p "$skill" < "$stdin_file"
329
335
  fi
330
- ) || true
336
+ ) || PIPE_AGENT_RC=$?
331
337
  fi
332
338
  rm -f "$stdin_file"
339
+ [ "$PIPE_AGENT_RC" -eq 0 ] || pipe_log "agent exited with code $PIPE_AGENT_RC"
340
+ return 0
341
+ }
342
+
343
+ pipe_require_agent_ran() {
344
+ # $1 = file the skill was supposed to write (empty to check the exit code
345
+ # only, for skills whose marker is legitimately optional), $2 = human label.
346
+ # Fails the job when the agent crashed or wrote nothing. Without this an
347
+ # agent that never started is indistinguishable from one that ran and found
348
+ # nothing: every caller reads its marker file, misses it, and falls back to
349
+ # the "all clear" default, so the job goes green without a review, fix, or
350
+ # triage having happened.
351
+ local marker="$1" what="${2:-agent}"
352
+ if [ "${PIPE_AGENT_RC:-0}" -ne 0 ]; then
353
+ pipe_log "ERROR: $what did not complete (exit code $PIPE_AGENT_RC). See the job log above."
354
+ return 1
355
+ fi
356
+ [ -n "$marker" ] || return 0
357
+ if [ ! -s "$marker" ]; then
358
+ pipe_log "ERROR: $what produced no output ($marker is missing or empty)."
359
+ return 1
360
+ fi
361
+ return 0
333
362
  }
334
363
 
335
364
  # --- omp invocation -----------------------------------------------------
@@ -369,17 +398,72 @@ pipe_translate_tools_to_omp() {
369
398
  echo "${result#,}"
370
399
  }
371
400
 
401
+ # --- Agent deadline -----------------------------------------------------
402
+ # A CI job timeout is a hard kill: the runner SIGKILLs the whole process tree,
403
+ # so the runner script never reaches the steps that post the comment, write
404
+ # the metrics record and emit the marker file. An agent that overruns
405
+ # therefore leaves a red job with no output at all (observed in production:
406
+ # three review timeouts in one week, each at the cap with the agent mid-run).
407
+ # Bounding the agent itself with `omp --max-time` turns an overrun into the
408
+ # platform's normal degraded path — failure-notice comment plus metrics —
409
+ # instead of a hard kill.
410
+
411
+ pipe_agent_max_time() {
412
+ # Echoes the --max-time value to pass to omp, or nothing when no deadline
413
+ # can be derived (GitHub Actions exposes no CI_JOB_TIMEOUT equivalent;
414
+ # there the workflow's timeout-minutes stays the only bound). An explicit
415
+ # PIPE_AGENT_MAX_TIME wins and is passed through verbatim, so a project can
416
+ # set "20m" or "1h" by hand. Otherwise the budget is the job timeout, minus
417
+ # the time already spent (image pull, checkout, before_script, API prep),
418
+ # minus PIPE_AGENT_TIME_RESERVE for the post-agent steps.
419
+ if [ -n "${PIPE_AGENT_MAX_TIME:-}" ]; then
420
+ printf '%s' "$PIPE_AGENT_MAX_TIME"
421
+ return 0
422
+ fi
423
+ local budget="${CI_JOB_TIMEOUT:-}"
424
+ case "$budget" in ''|*[!0-9]*) return 0 ;; esac
425
+ local reserve="${PIPE_AGENT_TIME_RESERVE:-180}" elapsed=0 started now
426
+ # Guard the emptiness explicitly: `date -d ""` does not fail, it silently
427
+ # resolves to midnight today, which would read as a many-hour-old job.
428
+ if [ -n "${CI_JOB_STARTED_AT:-}" ]; then
429
+ started=$(date -d "$CI_JOB_STARTED_AT" +%s 2>/dev/null || true)
430
+ if [ -n "$started" ]; then
431
+ now=$(date +%s)
432
+ elapsed=$(( now - started ))
433
+ [ "$elapsed" -lt 0 ] && elapsed=0
434
+ fi
435
+ fi
436
+ local left=$(( budget - elapsed - reserve ))
437
+ # A job already out of budget gets no flag: let the agent start and be cut
438
+ # by the runner exactly as it was before this function existed, rather than
439
+ # be killed one second in with a nonsensical deadline.
440
+ [ "$left" -lt 60 ] && return 0
441
+ printf '%s' "$left"
442
+ }
443
+
372
444
  pipe_run_omp() {
373
445
  # $1 = agent label for metrics, $2 = skill invocation (e.g. "/review-mr"),
374
446
  # $3 = allowed tools in Claude Code naming (optional, translated
375
447
  # internally), $4 = model for the main loop (optional, e.g.
376
448
  # $PIPE_MODEL_REVIEW; subagents keep the models from their own
377
449
  # frontmatter), stdin = extra prompt (optional, usually empty).
450
+ #
451
+ # Same contract as pipe_run_claude: always returns 0 and reports the agent's
452
+ # real exit code in PIPE_AGENT_RC, which pipe_require_agent_ran checks.
378
453
  local label="$1" skill="$2" tools="${3:-Agent,Read,Write,Edit,Glob,Grep,Bash}" model="${4:-}"
379
454
  local omp_tools; omp_tools=$(pipe_translate_tools_to_omp "$tools")
455
+ # Bound the run below the job's own hard kill (see pipe_agent_max_time).
456
+ local max_time; max_time=$(pipe_agent_max_time)
457
+ local -a max_time_args=()
458
+ local max_time_txt=""
459
+ if [ -n "$max_time" ]; then
460
+ max_time_args=(--max-time "$max_time")
461
+ max_time_txt=$(printf ' --max-time %q' "$max_time")
462
+ fi
380
463
  pipe_require_harness omp
381
464
  local stdin_file; stdin_file=$(mktemp)
382
465
  cat > "$stdin_file" || true
466
+ PIPE_AGENT_RC=0
383
467
 
384
468
  if [ "$(id -u)" -eq 0 ]; then
385
469
  local runuser="pipelinebot"
@@ -402,17 +486,19 @@ pipe_run_omp() {
402
486
  printf 'cd %q\n' "$PWD"
403
487
  printf 'source %q\n' "$SCRIPT_LIB_DIR/usage-capture-omp.sh"
404
488
  printf 'export PIPE_CAPTURE_MODEL=%q\n' "$model"
489
+ # The %s after `--` is the optional " --max-time <d>" fragment, already
490
+ # shell-quoted; it collapses to nothing when no deadline was derived.
405
491
  if [ -n "$model" ]; then
406
- printf 'capture_omp %q -- --yolo --no-session --no-title --model %q --tools %q -p %q < %q\n' \
407
- "$label" "$model" "$omp_tools" "$skill" "$stdin_file"
492
+ printf 'capture_omp %q --%s --yolo --no-session --no-title --model %q --tools %q -p %q < %q\n' \
493
+ "$label" "$max_time_txt" "$model" "$omp_tools" "$skill" "$stdin_file"
408
494
  else
409
- printf 'capture_omp %q -- --yolo --no-session --no-title --tools %q -p %q < %q\n' \
410
- "$label" "$omp_tools" "$skill" "$stdin_file"
495
+ printf 'capture_omp %q --%s --yolo --no-session --no-title --tools %q -p %q < %q\n' \
496
+ "$label" "$max_time_txt" "$omp_tools" "$skill" "$stdin_file"
411
497
  fi
412
498
  } > "$runner"
413
499
  chmod +x "$runner"
414
500
  chown "$runuser:$runuser" "$runner"
415
- su -m "$runuser" -s /bin/bash "$runner" || true
501
+ su -m "$runuser" -s /bin/bash "$runner" || PIPE_AGENT_RC=$?
416
502
  chown -R root:root "$PWD" 2>/dev/null || true
417
503
  rm -f "$runner"
418
504
  else
@@ -420,13 +506,15 @@ pipe_run_omp() {
420
506
  pipe_scrub_agent_secrets
421
507
  export PIPE_CAPTURE_MODEL="$model"
422
508
  if [ -n "$model" ]; then
423
- capture_omp "$label" -- --yolo --no-session --no-title --model "$model" --tools "$omp_tools" -p "$skill" < "$stdin_file"
509
+ capture_omp "$label" -- "${max_time_args[@]}" --yolo --no-session --no-title --model "$model" --tools "$omp_tools" -p "$skill" < "$stdin_file"
424
510
  else
425
- capture_omp "$label" -- --yolo --no-session --no-title --tools "$omp_tools" -p "$skill" < "$stdin_file"
511
+ capture_omp "$label" -- "${max_time_args[@]}" --yolo --no-session --no-title --tools "$omp_tools" -p "$skill" < "$stdin_file"
426
512
  fi
427
- ) || true
513
+ ) || PIPE_AGENT_RC=$?
428
514
  fi
429
515
  rm -f "$stdin_file"
516
+ [ "$PIPE_AGENT_RC" -eq 0 ] || pipe_log "agent exited with code $PIPE_AGENT_RC"
517
+ return 0
430
518
  }
431
519
 
432
520
  # --- Harness dispatch ---------------------------------------------------
@@ -186,9 +186,13 @@ MRS_VIEW=$(jq -c \
186
186
  elif ($body | test($reintroduce; "i")) then "reintroduces"
187
187
  else null end
188
188
  ),
189
+ # match/2 with "g", not scan/2: a Debian bookworm based runner image
190
+ # ships jq 1.6, which has no scan/2 and fails the whole program with
191
+ # a compile error.
189
192
  references: [
190
193
  $body
191
- | scan("(?:" + $revert + "|" + $regression + "|" + $reintroduce + ")[[:space:]]*[!#]?([0-9]+)"; "i")[]?
194
+ | match("(?:" + $revert + "|" + $regression + "|" + $reintroduce + ")[[:space:]]*[!#]?([0-9]+)"; "gi")
195
+ | .captures[].string
192
196
  | select(. != null)
193
197
  | tonumber
194
198
  ]
@@ -41,6 +41,9 @@ echo "$NOTES" | jq -r --arg u "${PIPE_BOT_USER:-}" --arg m "$REVIEW_MARKER" \
41
41
  # verifies, writes the result JSON, and the review-fix.env marker. On cap it
42
42
  # runs the shared postmortem flow (writing postmortem.md + postmortem.env).
43
43
  pipe_run_agent coder "/fix-review-findings" "Agent,Read,Write,Edit,Glob,Grep,Bash" "$PIPE_MODEL_CODE" < /dev/null
44
+ # A crashed agent writes no marker, which step 6 would read as "nothing changed"
45
+ # and step 7 would report as a completed fix run that had nothing to do.
46
+ pipe_require_agent_ran "" "The review-fix agent" || exit 1
44
47
 
45
48
  # 6. Read the marker the skill wrote.
46
49
  RF_ENV="$PIPE_CONTEXT_DIR/review-fix.env"
@@ -87,7 +87,14 @@ pipe_metric_event code-reviewer review_findings \
87
87
  --argjson inc "$([ "${PIPE_IS_INCREMENTAL:-false}" = "true" ] && echo true || echo false)" \
88
88
  '{findings_total: $t, status: $s, has_fixable_bugs: $b, is_incremental: $inc}')"
89
89
 
90
- # 7. Gate. review.env is exported so the fix job can read REVIEW_HAS_BUGS at
90
+ # 7. Fail the job when no review was produced. The comment above already told
91
+ # the humans; this tells the pipeline. PIPE_NO_GATE does not apply: it turns
92
+ # off the verdict gate, and this is a broken run, not a verdict. Without it a
93
+ # review that never ran (crashed agent, missing ANTHROPIC_API_KEY) reads as a
94
+ # review that ran and found nothing, and the MR goes green unreviewed.
95
+ pipe_require_agent_ran "$PIPE_RESULT_REVIEW" "The review agent" || exit 1
96
+
97
+ # 8. Gate. review.env is exported so the fix job can read REVIEW_HAS_BUGS at
91
98
  # runtime. On GitLab the job exits non-zero to block the pipeline. On GitHub
92
99
  # the workflow step owns the gate (it reads the output first), so it sets
93
100
  # PIPE_NO_GATE=1 and this script leaves the exit to the workflow.
@@ -27,6 +27,9 @@ COVERAGE_BEFORE=$(pipe_count_fix_commits "$PIPE_COMMIT_COVERAGE")
27
27
  # test-writer, verifies, and writes fix-tests.env. On cap it runs the shared
28
28
  # postmortem flow (writing postmortem.md + postmortem.env).
29
29
  pipe_run_agent test-fix "/fix-tests" "Agent,Read,Write,Edit,Glob,Grep,Bash" "$PIPE_MODEL_CODE" < /dev/null
30
+ # A crashed agent writes no marker, which step 4 would read as FIX_BRANCH=none
31
+ # and report as a false-alarm trigger, hiding the failure behind a green job.
32
+ pipe_require_agent_ran "" "The test-fix agent" || exit 1
30
33
 
31
34
  # 4. Read the marker the skill wrote.
32
35
  FT_ENV="$PIPE_CONTEXT_DIR/fix-tests.env"