@cxi-lmai/ci-agent-platform 3.2.1 → 3.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cxi-lmai/ci-agent-platform",
3
- "version": "3.2.1",
3
+ "version": "3.2.3",
4
4
  "description": "Autonomous dev pipeline on plain GitLab CI or GitHub Actions, driven by Claude Code. A labeled issue goes in, an open merge request comes out.",
5
5
  "keywords": [
6
6
  "claude",
@@ -86,9 +86,12 @@ variables:
86
86
  # GitLab's own CI_JOB_TIMEOUT minus the time already spent minus the reserve
87
87
  # below, which leaves room for the post-agent steps. An overrun then takes
88
88
  # the normal degraded path (failure-notice comment + metrics) instead of
89
- # dying silently. Set PIPE_AGENT_MAX_TIME (e.g. "20m") to override the
90
- # derivation for every job; GitHub Actions exposes no job-timeout variable,
91
- # so there the flag is omitted and timeout-minutes stays the only bound.
89
+ # dying silently. This parent deadline bounds the complete omp session.
90
+ # Each invocation also supplies a CI-only omp config overlay: task execution
91
+ # is blocking (no parent polling), each child is capped at 15 minutes and 60
92
+ # requests, and no interactive project setting is changed. Set
93
+ # PIPE_AGENT_MAX_TIME (e.g. "20m") to override the parent derivation for
94
+ # every job.
92
95
  PIPE_AGENT_TIME_RESERVE: "180" # seconds of the job budget kept for the post-agent steps
93
96
 
94
97
  # --- Build / verify --------------------------------------------------------
@@ -441,6 +441,24 @@ pipe_agent_max_time() {
441
441
  printf '%s' "$left"
442
442
  }
443
443
 
444
+ pipe_omp_ci_config() {
445
+ # Omp's interactive defaults run tasks asynchronously and leave their
446
+ # runtime unbounded. CI skills consume task results synchronously, so an
447
+ # async reviewer makes the parent poll through repeated model turns and an
448
+ # unbounded child can consume the whole parent/job deadline. Keep these
449
+ # semantics scoped to this invocation: --config overlays the user's normal
450
+ # project/profile settings without modifying either.
451
+ local config="$PIPE_CONTEXT_DIR/omp-ci-config.yml"
452
+ cat > "$config" <<'YAML'
453
+ async:
454
+ enabled: false
455
+ task:
456
+ maxRuntimeMs: 900000
457
+ softRequestBudget: 60
458
+ YAML
459
+ printf '%s' "$config"
460
+ }
461
+
444
462
  pipe_run_omp() {
445
463
  # $1 = agent label for metrics, $2 = skill invocation (e.g. "/review-mr"),
446
464
  # $3 = allowed tools in Claude Code naming (optional, translated
@@ -452,6 +470,7 @@ pipe_run_omp() {
452
470
  # real exit code in PIPE_AGENT_RC, which pipe_require_agent_ran checks.
453
471
  local label="$1" skill="$2" tools="${3:-Agent,Read,Write,Edit,Glob,Grep,Bash}" model="${4:-}"
454
472
  local omp_tools; omp_tools=$(pipe_translate_tools_to_omp "$tools")
473
+ local omp_config; omp_config=$(pipe_omp_ci_config)
455
474
  # Bound the run below the job's own hard kill (see pipe_agent_max_time).
456
475
  local max_time; max_time=$(pipe_agent_max_time)
457
476
  local -a max_time_args=()
@@ -486,14 +505,15 @@ pipe_run_omp() {
486
505
  printf 'cd %q\n' "$PWD"
487
506
  printf 'source %q\n' "$SCRIPT_LIB_DIR/usage-capture-omp.sh"
488
507
  printf 'export PIPE_CAPTURE_MODEL=%q\n' "$model"
489
- # The %s after `--` is the optional " --max-time <d>" fragment, already
490
- # shell-quoted; it collapses to nothing when no deadline was derived.
508
+ # The %s after the config is the optional " --max-time <d>" fragment,
509
+ # already shell-quoted; it collapses to nothing when no parent deadline
510
+ # was derived.
491
511
  if [ -n "$model" ]; then
492
- printf 'capture_omp %q --%s --yolo --no-session --no-title --model %q --tools %q -p %q < %q\n' \
493
- "$label" "$max_time_txt" "$model" "$omp_tools" "$skill" "$stdin_file"
512
+ printf 'capture_omp %q -- --config %q%s --yolo --no-session --no-title --model %q --tools %q -p %q < %q\n' \
513
+ "$label" "$omp_config" "$max_time_txt" "$model" "$omp_tools" "$skill" "$stdin_file"
494
514
  else
495
- printf 'capture_omp %q --%s --yolo --no-session --no-title --tools %q -p %q < %q\n' \
496
- "$label" "$max_time_txt" "$omp_tools" "$skill" "$stdin_file"
515
+ printf 'capture_omp %q -- --config %q%s --yolo --no-session --no-title --tools %q -p %q < %q\n' \
516
+ "$label" "$omp_config" "$max_time_txt" "$omp_tools" "$skill" "$stdin_file"
497
517
  fi
498
518
  } > "$runner"
499
519
  chmod +x "$runner"
@@ -506,9 +526,9 @@ pipe_run_omp() {
506
526
  pipe_scrub_agent_secrets
507
527
  export PIPE_CAPTURE_MODEL="$model"
508
528
  if [ -n "$model" ]; then
509
- capture_omp "$label" -- "${max_time_args[@]}" --yolo --no-session --no-title --model "$model" --tools "$omp_tools" -p "$skill" < "$stdin_file"
529
+ capture_omp "$label" -- --config "$omp_config" "${max_time_args[@]}" --yolo --no-session --no-title --model "$model" --tools "$omp_tools" -p "$skill" < "$stdin_file"
510
530
  else
511
- capture_omp "$label" -- "${max_time_args[@]}" --yolo --no-session --no-title --tools "$omp_tools" -p "$skill" < "$stdin_file"
531
+ capture_omp "$label" -- --config "$omp_config" "${max_time_args[@]}" --yolo --no-session --no-title --tools "$omp_tools" -p "$skill" < "$stdin_file"
512
532
  fi
513
533
  ) || PIPE_AGENT_RC=$?
514
534
  fi
@@ -1,18 +1,37 @@
1
1
  # shellcheck shell=bash
2
2
  # Source-only library. Wraps an `omp --mode json` invocation, sums usage/cost
3
- # from the agent_end event's message transcript, writes a metrics record, and
4
- # surfaces the assistant text on stdout — the omp equivalent of
3
+ # across the whole spawn tree the invocation paid for, writes a metrics record,
4
+ # and surfaces the assistant text on stdout — the omp equivalent of
5
5
  # usage-capture.sh's capture_claude. Same function contract, same metrics
6
- # record shape (bar the harness-specific exit-code key), different event
7
- # parsing because omp's schema differs from Claude's stream-json.
6
+ # record shape (bar the harness-specific exit-code and completeness keys),
7
+ # different event parsing because omp's schema differs from Claude's
8
+ # stream-json.
8
9
  #
9
- # Schema verified live against a real omp install (18.0.6): the final
10
- # "agent_end" line carries {"type":"agent_end","messages":[...],
10
+ # Schema verified against omp 18.1.21 (minimum supported; the nested
11
+ # `extractedToolData.task` carrier below does not exist in older lines): the
12
+ # final "agent_end" line carries {"type":"agent_end","messages":[...],
11
13
  # "isTerminal":true}. Each assistant message in `messages[]` carries its own
12
14
  # `usage` object (per-message, not session-cumulative) and top-level
13
15
  # `model`/`provider`/`duration` fields. See
14
16
  # docs/superpowers/specs/2026-08-26-omp-openrouter-harness-design.md section 3
15
- # for the full verification record.
17
+ # for the original verification record and docs/metrics.md for the aggregation
18
+ # semantics.
19
+ #
20
+ # Subagents run as child sessions, so their model calls are never parent
21
+ # assistant messages. omp exposes them one level at a time instead:
22
+ #
23
+ # - a blocking `task` call's tool result carries `details.usage`, the merge
24
+ # of `details.results[].usage`, each of which counts only that child's own
25
+ # assistant messages;
26
+ # - that child's own `task` calls arrive under
27
+ # `details.results[].extractedToolData.task[]` as the same shape again
28
+ # (omp registers a subprocess handler for the tool), so the tree recurses
29
+ # with every level counted exactly once;
30
+ # - an async spawn exposes no usage at all: the delivery message carries
31
+ # jobs, not tokens. Such children are counted as unmeasured and the record
32
+ # marks itself incomplete rather than passing a parent-only figure off as
33
+ # the total. The CI config overlay in pipeline-common.sh keeps task
34
+ # execution blocking precisely so this stays the exceptional case.
16
35
  #
17
36
  # Usage: capture_omp <agent_name> -- <args to pass to `omp`...>
18
37
  # Stdin: forwarded to `omp` stdin
@@ -21,6 +40,62 @@
21
40
 
22
41
  set -u
23
42
 
43
+ # Aggregates one `agent_end` object into a tab-separated row:
44
+ # input, output, cacheWrite, cacheRead, cost, parent duration, subagent cost,
45
+ # measured subagents, unmeasured subagents, model. The model is deliberately
46
+ # last: it is the one field that can be empty (a run killed before its first
47
+ # assistant turn), and tab is an IFS *whitespace* character, so `read` would
48
+ # collapse an empty leading column and shift every number one place left.
49
+ # Kept to jq 1.6 features (Debian stable ships 1.6; see metrics-snapshot.sh),
50
+ # and assigned from a plain quoted string rather than a `cat` heredoc: sourcing
51
+ # this library must not need a single external binary, because callers run it
52
+ # with a deliberately minimal PATH.
53
+ _OMP_USAGE_JQ='
54
+ def usum(f): map(f // 0) | add // 0;
55
+ # One `task` tool result, recursing into the nested calls its children made.
56
+ def level:
57
+ ([.results[]? | select(.usage != null) | .usage]) as $own
58
+ | (if ($own | length) > 0 then $own
59
+ elif (.usage != null) then [.usage]
60
+ else [] end) as $usage
61
+ | (if ($own | length) > 0 then ($own | length)
62
+ elif (.usage != null) then 1
63
+ else 0 end) as $measured
64
+ # Count every agent the call spawned, measured or not. An async spawn
65
+ # settles with no result row at all, so the progress list is the only place
66
+ # it appears — including when it has a measured sibling, which must not mask
67
+ # it. A sync child that failed before its first request has a row but no
68
+ # usage, and is counted by the same subtraction.
69
+ | ([.results[]?] | length) as $result_rows
70
+ | ([.progress[]?] | length) as $progress_rows
71
+ | (if $progress_rows > $result_rows then $progress_rows else $result_rows end) as $spawned
72
+ | (if $spawned > $measured then ($spawned - $measured) else 0 end) as $gaps
73
+ | ([.results[]? | (.extractedToolData.task // [])[]? | level]) as $nested
74
+ | {
75
+ input: (($usage | usum(.input)) + ($nested | usum(.input))),
76
+ output: (($usage | usum(.output)) + ($nested | usum(.output))),
77
+ cacheWrite: (($usage | usum(.cacheWrite)) + ($nested | usum(.cacheWrite))),
78
+ cacheRead: (($usage | usum(.cacheRead)) + ($nested | usum(.cacheRead))),
79
+ cost: (($usage | usum(.cost.total)) + ($nested | usum(.cost))),
80
+ measured: ($measured + ($nested | usum(.measured))),
81
+ unmeasured: ($gaps + ($nested | usum(.unmeasured)))
82
+ };
83
+ ([.messages[]? | select(.role == "assistant")]) as $parent
84
+ | ([.messages[]? | select(.role == "toolResult" and .toolName == "task") | .details | level]) as $subs
85
+ | [
86
+ (($parent | usum(.usage.input)) + ($subs | usum(.input))),
87
+ (($parent | usum(.usage.output)) + ($subs | usum(.output))),
88
+ (($parent | usum(.usage.cacheWrite)) + ($subs | usum(.cacheWrite))),
89
+ (($parent | usum(.usage.cacheRead)) + ($subs | usum(.cacheRead))),
90
+ (($parent | usum(.usage.cost.total)) + ($subs | usum(.cost))),
91
+ ($parent | usum(.duration)),
92
+ ($subs | usum(.cost)),
93
+ ($subs | usum(.measured)),
94
+ ($subs | usum(.unmeasured)),
95
+ ([$parent[] | ((.provider // "") + "/" + (.model // ""))] | last // "")
96
+ ] | @tsv
97
+ '
98
+
24
99
  capture_omp() {
25
100
  local agent="$1"; shift
26
101
  [ "${1:-}" = "--" ] && shift
@@ -51,23 +126,32 @@ capture_omp() {
51
126
 
52
127
  # Extract the final agent_end event (last line that begins with
53
128
  # {"type":"agent_end"). Its usage is per-message (one API call each), not
54
- # pre-aggregated like Claude's single "result" event, so sum it ourselves.
129
+ # pre-aggregated like Claude's single "result" event, and subagent usage sits
130
+ # in the `task` tool results, so aggregate the tree ourselves
131
+ # (see _OMP_USAGE_JQ above).
55
132
  local result_line
56
133
  result_line=$(grep '^{"type":"agent_end"' "$stream_file" | tail -1 || true)
57
134
 
58
- local model input output cache_c cache_r cost duration
59
- if [ -n "$result_line" ]; then
60
- model=$(echo "$result_line" | jq -r '[.messages[]? | select(.role=="assistant") | ((.provider // "") + "/" + (.model // ""))] | last // ""')
61
- input=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.input // 0] | add // 0')
62
- output=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.output // 0] | add // 0')
63
- cache_c=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cacheWrite // 0] | add // 0')
64
- cache_r=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cacheRead // 0] | add // 0')
65
- cost=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cost.total // 0] | add // 0')
66
- duration=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .duration // 0] | add // 0')
135
+ local model input output cache_c cache_r cost duration sub_cost measured unmeasured
136
+ local row=""
137
+ [ -n "$result_line" ] && row=$(echo "$result_line" | jq -r "$_OMP_USAGE_JQ" 2>/dev/null || true)
138
+ if [ -n "$row" ]; then
139
+ IFS=$'\t' read -r input output cache_c cache_r cost duration sub_cost measured unmeasured model \
140
+ <<< "$row"
67
141
  else
142
+ # No parsable agent_end (crash, killed run): the parent's own wall time is
143
+ # the only thing known, and nothing can be claimed about subagents.
68
144
  model=""; input=0; output=0; cache_c=0; cache_r=0; cost=0; duration=$wall_ms
145
+ sub_cost=0; measured=0; unmeasured=0
69
146
  fi
70
147
  [ -n "${PIPE_CAPTURE_MODEL:-}" ] && [ -z "$model" ] && model="$PIPE_CAPTURE_MODEL"
148
+ # Partial the moment any spawned agent's usage was not exposed: the totals
149
+ # above then cover less than the OpenRouter account was charged for. A run
150
+ # with no parsable agent_end is the extreme case — not even the parent's own
151
+ # tokens are known — so its zeros must not read as a settled bill either.
152
+ local complete=true
153
+ [ -z "$row" ] && complete=false
154
+ [ "$unmeasured" -gt 0 ] && complete=false
71
155
 
72
156
  # Recover the user-facing text (concatenate all assistant text blocks) so
73
157
  # downstream consumers that read `omp -p`'s stdout still work.
@@ -75,10 +159,11 @@ capture_omp() {
75
159
  | jq -r 'select(.message.role=="assistant") | .message.content[]? | select(.type=="text") | .text' \
76
160
  || true
77
161
 
78
- # Write metrics record. Same field names as capture_claude's record; only
79
- # the harness-specific exit-code key differs (omp_exit_code vs
80
- # claude_exit_code), by the same convention the two capture_* files
81
- # already use for their own harness-labelled fields.
162
+ # Write metrics record. Same field names as capture_claude's record; the
163
+ # harness-specific exit-code key differs (omp_exit_code vs
164
+ # claude_exit_code), by the same convention the two capture_* files already
165
+ # use for their own harness-labelled fields, and the usage_* /subagent_*
166
+ # keys describe a scope Claude Code's single `result` event does not have.
82
167
  local record_stamp record_file
83
168
  record_stamp=$(date +%s%N)
84
169
  record_file="$metrics_dir/${job_name}-${job_id}-${record_stamp}.json"
@@ -100,6 +185,10 @@ capture_omp() {
100
185
  --argjson duration "$duration" \
101
186
  --argjson wall_ms "$wall_ms" \
102
187
  --argjson exit_code "$exit_code" \
188
+ --argjson sub_cost "$sub_cost" \
189
+ --argjson measured "$measured" \
190
+ --argjson unmeasured "$unmeasured" \
191
+ --argjson complete "$complete" \
103
192
  '{
104
193
  ts: $ts, agent: $agent, event: $event, model: $model,
105
194
  pipeline_id: $pipeline_id, job_id: $job_id, job_name: $job_name,
@@ -107,7 +196,10 @@ capture_omp() {
107
196
  input_tokens: $input, output_tokens: $output,
108
197
  cache_creation_tokens: $cache_c, cache_read_tokens: $cache_r,
109
198
  total_cost_usd: $cost, duration_ms: $duration, wall_ms: $wall_ms,
110
- omp_exit_code: $exit_code
199
+ omp_exit_code: $exit_code,
200
+ usage_scope: "tree", usage_complete: $complete,
201
+ subagents_measured: $measured, subagents_unmeasured: $unmeasured,
202
+ subagent_cost_usd: $sub_cost
111
203
  }' > "$record_file"
112
204
 
113
205
  # If the agent failed, forward stderr so debugging still works, and keep a