@cxi-lmai/ci-agent-platform 3.2.2 → 3.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cxi-lmai/ci-agent-platform",
3
- "version": "3.2.2",
3
+ "version": "3.2.3",
4
4
  "description": "Autonomous dev pipeline on plain GitLab CI or GitHub Actions, driven by Claude Code. A labeled issue goes in, an open merge request comes out.",
5
5
  "keywords": [
6
6
  "claude",
@@ -1,18 +1,37 @@
1
1
  # shellcheck shell=bash
2
2
  # Source-only library. Wraps an `omp --mode json` invocation, sums usage/cost
3
- # from the agent_end event's message transcript, writes a metrics record, and
4
- # surfaces the assistant text on stdout — the omp equivalent of
3
+ # across the whole spawn tree the invocation paid for, writes a metrics record,
4
+ # and surfaces the assistant text on stdout — the omp equivalent of
5
5
  # usage-capture.sh's capture_claude. Same function contract, same metrics
6
- # record shape (bar the harness-specific exit-code key), different event
7
- # parsing because omp's schema differs from Claude's stream-json.
6
+ # record shape (bar the harness-specific exit-code and completeness keys),
7
+ # different event parsing because omp's schema differs from Claude's
8
+ # stream-json.
8
9
  #
9
- # Schema verified live against a real omp install (18.0.6): the final
10
- # "agent_end" line carries {"type":"agent_end","messages":[...],
10
+ # Schema verified against omp 18.1.21 (minimum supported; the nested
11
+ # `extractedToolData.task` carrier below does not exist in older lines): the
12
+ # final "agent_end" line carries {"type":"agent_end","messages":[...],
11
13
  # "isTerminal":true}. Each assistant message in `messages[]` carries its own
12
14
  # `usage` object (per-message, not session-cumulative) and top-level
13
15
  # `model`/`provider`/`duration` fields. See
14
16
  # docs/superpowers/specs/2026-08-26-omp-openrouter-harness-design.md section 3
15
- # for the full verification record.
17
+ # for the original verification record and docs/metrics.md for the aggregation
18
+ # semantics.
19
+ #
20
+ # Subagents run as child sessions, so their model calls are never parent
21
+ # assistant messages. omp exposes them one level at a time instead:
22
+ #
23
+ # - a blocking `task` call's tool result carries `details.usage`, the merge
24
+ # of `details.results[].usage`, each of which counts only that child's own
25
+ # assistant messages;
26
+ # - that child's own `task` calls arrive under
27
+ # `details.results[].extractedToolData.task[]` as the same shape again
28
+ # (omp registers a subprocess handler for the tool), so the tree recurses
29
+ # with every level counted exactly once;
30
+ # - an async spawn exposes no usage at all: the delivery message carries
31
+ # jobs, not tokens. Such children are counted as unmeasured and the record
32
+ # marks itself incomplete rather than passing a parent-only figure off as
33
+ # the total. The CI config overlay in pipeline-common.sh keeps task
34
+ # execution blocking precisely so this stays the exceptional case.
16
35
  #
17
36
  # Usage: capture_omp <agent_name> -- <args to pass to `omp`...>
18
37
  # Stdin: forwarded to `omp` stdin
@@ -21,6 +40,62 @@
21
40
 
22
41
  set -u
23
42
 
43
+ # Aggregates one `agent_end` object into a tab-separated row:
44
+ # input, output, cacheWrite, cacheRead, cost, parent duration, subagent cost,
45
+ # measured subagents, unmeasured subagents, model. The model is deliberately
46
+ # last: it is the one field that can be empty (a run killed before its first
47
+ # assistant turn), and tab is an IFS *whitespace* character, so `read` would
48
+ # collapse an empty leading column and shift every number one place left.
49
+ # Kept to jq 1.6 features (Debian stable ships 1.6; see metrics-snapshot.sh),
50
+ # and assigned from a plain quoted string rather than a `cat` heredoc: sourcing
51
+ # this library must not need a single external binary, because callers run it
52
+ # with a deliberately minimal PATH.
53
+ _OMP_USAGE_JQ='
54
+ def usum(f): map(f // 0) | add // 0;
55
+ # One `task` tool result, recursing into the nested calls its children made.
56
+ def level:
57
+ ([.results[]? | select(.usage != null) | .usage]) as $own
58
+ | (if ($own | length) > 0 then $own
59
+ elif (.usage != null) then [.usage]
60
+ else [] end) as $usage
61
+ | (if ($own | length) > 0 then ($own | length)
62
+ elif (.usage != null) then 1
63
+ else 0 end) as $measured
64
+ # Count every agent the call spawned, measured or not. An async spawn
65
+ # settles with no result row at all, so the progress list is the only place
66
+ # it appears — including when it has a measured sibling, which must not mask
67
+ # it. A sync child that failed before its first request has a row but no
68
+ # usage, and is counted by the same subtraction.
69
+ | ([.results[]?] | length) as $result_rows
70
+ | ([.progress[]?] | length) as $progress_rows
71
+ | (if $progress_rows > $result_rows then $progress_rows else $result_rows end) as $spawned
72
+ | (if $spawned > $measured then ($spawned - $measured) else 0 end) as $gaps
73
+ | ([.results[]? | (.extractedToolData.task // [])[]? | level]) as $nested
74
+ | {
75
+ input: (($usage | usum(.input)) + ($nested | usum(.input))),
76
+ output: (($usage | usum(.output)) + ($nested | usum(.output))),
77
+ cacheWrite: (($usage | usum(.cacheWrite)) + ($nested | usum(.cacheWrite))),
78
+ cacheRead: (($usage | usum(.cacheRead)) + ($nested | usum(.cacheRead))),
79
+ cost: (($usage | usum(.cost.total)) + ($nested | usum(.cost))),
80
+ measured: ($measured + ($nested | usum(.measured))),
81
+ unmeasured: ($gaps + ($nested | usum(.unmeasured)))
82
+ };
83
+ ([.messages[]? | select(.role == "assistant")]) as $parent
84
+ | ([.messages[]? | select(.role == "toolResult" and .toolName == "task") | .details | level]) as $subs
85
+ | [
86
+ (($parent | usum(.usage.input)) + ($subs | usum(.input))),
87
+ (($parent | usum(.usage.output)) + ($subs | usum(.output))),
88
+ (($parent | usum(.usage.cacheWrite)) + ($subs | usum(.cacheWrite))),
89
+ (($parent | usum(.usage.cacheRead)) + ($subs | usum(.cacheRead))),
90
+ (($parent | usum(.usage.cost.total)) + ($subs | usum(.cost))),
91
+ ($parent | usum(.duration)),
92
+ ($subs | usum(.cost)),
93
+ ($subs | usum(.measured)),
94
+ ($subs | usum(.unmeasured)),
95
+ ([$parent[] | ((.provider // "") + "/" + (.model // ""))] | last // "")
96
+ ] | @tsv
97
+ '
98
+
24
99
  capture_omp() {
25
100
  local agent="$1"; shift
26
101
  [ "${1:-}" = "--" ] && shift
@@ -51,23 +126,32 @@ capture_omp() {
51
126
 
52
127
  # Extract the final agent_end event (last line that begins with
53
128
  # {"type":"agent_end"). Its usage is per-message (one API call each), not
54
- # pre-aggregated like Claude's single "result" event, so sum it ourselves.
129
+ # pre-aggregated like Claude's single "result" event, and subagent usage sits
130
+ # in the `task` tool results, so aggregate the tree ourselves
131
+ # (see _OMP_USAGE_JQ above).
55
132
  local result_line
56
133
  result_line=$(grep '^{"type":"agent_end"' "$stream_file" | tail -1 || true)
57
134
 
58
- local model input output cache_c cache_r cost duration
59
- if [ -n "$result_line" ]; then
60
- model=$(echo "$result_line" | jq -r '[.messages[]? | select(.role=="assistant") | ((.provider // "") + "/" + (.model // ""))] | last // ""')
61
- input=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.input // 0] | add // 0')
62
- output=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.output // 0] | add // 0')
63
- cache_c=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cacheWrite // 0] | add // 0')
64
- cache_r=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cacheRead // 0] | add // 0')
65
- cost=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cost.total // 0] | add // 0')
66
- duration=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .duration // 0] | add // 0')
135
+ local model input output cache_c cache_r cost duration sub_cost measured unmeasured
136
+ local row=""
137
+ [ -n "$result_line" ] && row=$(echo "$result_line" | jq -r "$_OMP_USAGE_JQ" 2>/dev/null || true)
138
+ if [ -n "$row" ]; then
139
+ IFS=$'\t' read -r input output cache_c cache_r cost duration sub_cost measured unmeasured model \
140
+ <<< "$row"
67
141
  else
142
+ # No parsable agent_end (crash, killed run): the parent's own wall time is
143
+ # the only thing known, and nothing can be claimed about subagents.
68
144
  model=""; input=0; output=0; cache_c=0; cache_r=0; cost=0; duration=$wall_ms
145
+ sub_cost=0; measured=0; unmeasured=0
69
146
  fi
70
147
  [ -n "${PIPE_CAPTURE_MODEL:-}" ] && [ -z "$model" ] && model="$PIPE_CAPTURE_MODEL"
148
+ # Partial the moment any spawned agent's usage was not exposed: the totals
149
+ # above then cover less than the OpenRouter account was charged for. A run
150
+ # with no parsable agent_end is the extreme case — not even the parent's own
151
+ # tokens are known — so its zeros must not read as a settled bill either.
152
+ local complete=true
153
+ [ -z "$row" ] && complete=false
154
+ [ "$unmeasured" -gt 0 ] && complete=false
71
155
 
72
156
  # Recover the user-facing text (concatenate all assistant text blocks) so
73
157
  # downstream consumers that read `omp -p`'s stdout still work.
@@ -75,10 +159,11 @@ capture_omp() {
75
159
  | jq -r 'select(.message.role=="assistant") | .message.content[]? | select(.type=="text") | .text' \
76
160
  || true
77
161
 
78
- # Write metrics record. Same field names as capture_claude's record; only
79
- # the harness-specific exit-code key differs (omp_exit_code vs
80
- # claude_exit_code), by the same convention the two capture_* files
81
- # already use for their own harness-labelled fields.
162
+ # Write metrics record. Same field names as capture_claude's record; the
163
+ # harness-specific exit-code key differs (omp_exit_code vs
164
+ # claude_exit_code), by the same convention the two capture_* files already
165
+ # use for their own harness-labelled fields, and the usage_* /subagent_*
166
+ # keys describe a scope Claude Code's single `result` event does not have.
82
167
  local record_stamp record_file
83
168
  record_stamp=$(date +%s%N)
84
169
  record_file="$metrics_dir/${job_name}-${job_id}-${record_stamp}.json"
@@ -100,6 +185,10 @@ capture_omp() {
100
185
  --argjson duration "$duration" \
101
186
  --argjson wall_ms "$wall_ms" \
102
187
  --argjson exit_code "$exit_code" \
188
+ --argjson sub_cost "$sub_cost" \
189
+ --argjson measured "$measured" \
190
+ --argjson unmeasured "$unmeasured" \
191
+ --argjson complete "$complete" \
103
192
  '{
104
193
  ts: $ts, agent: $agent, event: $event, model: $model,
105
194
  pipeline_id: $pipeline_id, job_id: $job_id, job_name: $job_name,
@@ -107,7 +196,10 @@ capture_omp() {
107
196
  input_tokens: $input, output_tokens: $output,
108
197
  cache_creation_tokens: $cache_c, cache_read_tokens: $cache_r,
109
198
  total_cost_usd: $cost, duration_ms: $duration, wall_ms: $wall_ms,
110
- omp_exit_code: $exit_code
199
+ omp_exit_code: $exit_code,
200
+ usage_scope: "tree", usage_complete: $complete,
201
+ subagents_measured: $measured, subagents_unmeasured: $unmeasured,
202
+ subagent_cost_usd: $sub_cost
111
203
  }' > "$record_file"
112
204
 
113
205
  # If the agent failed, forward stderr so debugging still works, and keep a