@cxi-lmai/ci-agent-platform 3.2.2 → 3.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cxi-lmai/ci-agent-platform",
|
|
3
|
-
"version": "3.2.
|
|
3
|
+
"version": "3.2.3",
|
|
4
4
|
"description": "Autonomous dev pipeline on plain GitLab CI or GitHub Actions, driven by Claude Code. A labeled issue goes in, an open merge request comes out.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude",
|
|
@@ -1,18 +1,37 @@
|
|
|
1
1
|
# shellcheck shell=bash
|
|
2
2
|
# Source-only library. Wraps an `omp --mode json` invocation, sums usage/cost
|
|
3
|
-
#
|
|
4
|
-
# surfaces the assistant text on stdout — the omp equivalent of
|
|
3
|
+
# across the whole spawn tree the invocation paid for, writes a metrics record,
|
|
4
|
+
# and surfaces the assistant text on stdout — the omp equivalent of
|
|
5
5
|
# usage-capture.sh's capture_claude. Same function contract, same metrics
|
|
6
|
-
# record shape (bar the harness-specific exit-code
|
|
7
|
-
# parsing because omp's schema differs from Claude's
|
|
6
|
+
# record shape (bar the harness-specific exit-code and completeness keys),
|
|
7
|
+
# different event parsing because omp's schema differs from Claude's
|
|
8
|
+
# stream-json.
|
|
8
9
|
#
|
|
9
|
-
# Schema verified
|
|
10
|
-
#
|
|
10
|
+
# Schema verified against omp 18.1.21 (minimum supported; the nested
|
|
11
|
+
# `extractedToolData.task` carrier below does not exist in older lines): the
|
|
12
|
+
# final "agent_end" line carries {"type":"agent_end","messages":[...],
|
|
11
13
|
# "isTerminal":true}. Each assistant message in `messages[]` carries its own
|
|
12
14
|
# `usage` object (per-message, not session-cumulative) and top-level
|
|
13
15
|
# `model`/`provider`/`duration` fields. See
|
|
14
16
|
# docs/superpowers/specs/2026-08-26-omp-openrouter-harness-design.md section 3
|
|
15
|
-
# for the
|
|
17
|
+
# for the original verification record and docs/metrics.md for the aggregation
|
|
18
|
+
# semantics.
|
|
19
|
+
#
|
|
20
|
+
# Subagents run as child sessions, so their model calls are never parent
|
|
21
|
+
# assistant messages. omp exposes them one level at a time instead:
|
|
22
|
+
#
|
|
23
|
+
# - a blocking `task` call's tool result carries `details.usage`, the merge
|
|
24
|
+
# of `details.results[].usage`, each of which counts only that child's own
|
|
25
|
+
# assistant messages;
|
|
26
|
+
# - that child's own `task` calls arrive under
|
|
27
|
+
# `details.results[].extractedToolData.task[]` as the same shape again
|
|
28
|
+
# (omp registers a subprocess handler for the tool), so the tree recurses
|
|
29
|
+
# with every level counted exactly once;
|
|
30
|
+
# - an async spawn exposes no usage at all: the delivery message carries
|
|
31
|
+
# jobs, not tokens. Such children are counted as unmeasured and the record
|
|
32
|
+
# marks itself incomplete rather than passing a parent-only figure off as
|
|
33
|
+
# the total. The CI config overlay in pipeline-common.sh keeps task
|
|
34
|
+
# execution blocking precisely so this stays the exceptional case.
|
|
16
35
|
#
|
|
17
36
|
# Usage: capture_omp <agent_name> -- <args to pass to `omp`...>
|
|
18
37
|
# Stdin: forwarded to `omp` stdin
|
|
@@ -21,6 +40,62 @@
|
|
|
21
40
|
|
|
22
41
|
set -u
|
|
23
42
|
|
|
43
|
+
# Aggregates one `agent_end` object into a tab-separated row:
|
|
44
|
+
# input, output, cacheWrite, cacheRead, cost, parent duration, subagent cost,
|
|
45
|
+
# measured subagents, unmeasured subagents, model. The model is deliberately
|
|
46
|
+
# last: it is the one field that can be empty (a run killed before its first
|
|
47
|
+
# assistant turn), and tab is an IFS *whitespace* character, so `read` would
|
|
48
|
+
# collapse an empty leading column and shift every number one place left.
|
|
49
|
+
# Kept to jq 1.6 features (Debian stable ships 1.6; see metrics-snapshot.sh),
|
|
50
|
+
# and assigned from a plain quoted string rather than a `cat` heredoc: sourcing
|
|
51
|
+
# this library must not need a single external binary, because callers run it
|
|
52
|
+
# with a deliberately minimal PATH.
|
|
53
|
+
_OMP_USAGE_JQ='
|
|
54
|
+
def usum(f): map(f // 0) | add // 0;
|
|
55
|
+
# One `task` tool result, recursing into the nested calls its children made.
|
|
56
|
+
def level:
|
|
57
|
+
([.results[]? | select(.usage != null) | .usage]) as $own
|
|
58
|
+
| (if ($own | length) > 0 then $own
|
|
59
|
+
elif (.usage != null) then [.usage]
|
|
60
|
+
else [] end) as $usage
|
|
61
|
+
| (if ($own | length) > 0 then ($own | length)
|
|
62
|
+
elif (.usage != null) then 1
|
|
63
|
+
else 0 end) as $measured
|
|
64
|
+
# Count every agent the call spawned, measured or not. An async spawn
|
|
65
|
+
# settles with no result row at all, so the progress list is the only place
|
|
66
|
+
# it appears — including when it has a measured sibling, which must not mask
|
|
67
|
+
# it. A sync child that failed before its first request has a row but no
|
|
68
|
+
# usage, and is counted by the same subtraction.
|
|
69
|
+
| ([.results[]?] | length) as $result_rows
|
|
70
|
+
| ([.progress[]?] | length) as $progress_rows
|
|
71
|
+
| (if $progress_rows > $result_rows then $progress_rows else $result_rows end) as $spawned
|
|
72
|
+
| (if $spawned > $measured then ($spawned - $measured) else 0 end) as $gaps
|
|
73
|
+
| ([.results[]? | (.extractedToolData.task // [])[]? | level]) as $nested
|
|
74
|
+
| {
|
|
75
|
+
input: (($usage | usum(.input)) + ($nested | usum(.input))),
|
|
76
|
+
output: (($usage | usum(.output)) + ($nested | usum(.output))),
|
|
77
|
+
cacheWrite: (($usage | usum(.cacheWrite)) + ($nested | usum(.cacheWrite))),
|
|
78
|
+
cacheRead: (($usage | usum(.cacheRead)) + ($nested | usum(.cacheRead))),
|
|
79
|
+
cost: (($usage | usum(.cost.total)) + ($nested | usum(.cost))),
|
|
80
|
+
measured: ($measured + ($nested | usum(.measured))),
|
|
81
|
+
unmeasured: ($gaps + ($nested | usum(.unmeasured)))
|
|
82
|
+
};
|
|
83
|
+
([.messages[]? | select(.role == "assistant")]) as $parent
|
|
84
|
+
| ([.messages[]? | select(.role == "toolResult" and .toolName == "task") | .details | level]) as $subs
|
|
85
|
+
| [
|
|
86
|
+
(($parent | usum(.usage.input)) + ($subs | usum(.input))),
|
|
87
|
+
(($parent | usum(.usage.output)) + ($subs | usum(.output))),
|
|
88
|
+
(($parent | usum(.usage.cacheWrite)) + ($subs | usum(.cacheWrite))),
|
|
89
|
+
(($parent | usum(.usage.cacheRead)) + ($subs | usum(.cacheRead))),
|
|
90
|
+
(($parent | usum(.usage.cost.total)) + ($subs | usum(.cost))),
|
|
91
|
+
($parent | usum(.duration)),
|
|
92
|
+
($subs | usum(.cost)),
|
|
93
|
+
($subs | usum(.measured)),
|
|
94
|
+
($subs | usum(.unmeasured)),
|
|
95
|
+
([$parent[] | ((.provider // "") + "/" + (.model // ""))] | last // "")
|
|
96
|
+
] | @tsv
|
|
97
|
+
'
|
|
98
|
+
|
|
24
99
|
capture_omp() {
|
|
25
100
|
local agent="$1"; shift
|
|
26
101
|
[ "${1:-}" = "--" ] && shift
|
|
@@ -51,23 +126,32 @@ capture_omp() {
|
|
|
51
126
|
|
|
52
127
|
# Extract the final agent_end event (last line that begins with
|
|
53
128
|
# {"type":"agent_end"). Its usage is per-message (one API call each), not
|
|
54
|
-
# pre-aggregated like Claude's single "result" event,
|
|
129
|
+
# pre-aggregated like Claude's single "result" event, and subagent usage sits
|
|
130
|
+
# in the `task` tool results, so aggregate the tree ourselves
|
|
131
|
+
# (see _OMP_USAGE_JQ above).
|
|
55
132
|
local result_line
|
|
56
133
|
result_line=$(grep '^{"type":"agent_end"' "$stream_file" | tail -1 || true)
|
|
57
134
|
|
|
58
|
-
local model input output cache_c cache_r cost duration
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
cache_r=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cacheRead // 0] | add // 0')
|
|
65
|
-
cost=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .usage.cost.total // 0] | add // 0')
|
|
66
|
-
duration=$(echo "$result_line" | jq '[.messages[]? | select(.role=="assistant") | .duration // 0] | add // 0')
|
|
135
|
+
local model input output cache_c cache_r cost duration sub_cost measured unmeasured
|
|
136
|
+
local row=""
|
|
137
|
+
[ -n "$result_line" ] && row=$(echo "$result_line" | jq -r "$_OMP_USAGE_JQ" 2>/dev/null || true)
|
|
138
|
+
if [ -n "$row" ]; then
|
|
139
|
+
IFS=$'\t' read -r input output cache_c cache_r cost duration sub_cost measured unmeasured model \
|
|
140
|
+
<<< "$row"
|
|
67
141
|
else
|
|
142
|
+
# No parsable agent_end (crash, killed run): the parent's own wall time is
|
|
143
|
+
# the only thing known, and nothing can be claimed about subagents.
|
|
68
144
|
model=""; input=0; output=0; cache_c=0; cache_r=0; cost=0; duration=$wall_ms
|
|
145
|
+
sub_cost=0; measured=0; unmeasured=0
|
|
69
146
|
fi
|
|
70
147
|
[ -n "${PIPE_CAPTURE_MODEL:-}" ] && [ -z "$model" ] && model="$PIPE_CAPTURE_MODEL"
|
|
148
|
+
# Partial the moment any spawned agent's usage was not exposed: the totals
|
|
149
|
+
# above then cover less than the OpenRouter account was charged for. A run
|
|
150
|
+
# with no parsable agent_end is the extreme case — not even the parent's own
|
|
151
|
+
# tokens are known — so its zeros must not read as a settled bill either.
|
|
152
|
+
local complete=true
|
|
153
|
+
[ -z "$row" ] && complete=false
|
|
154
|
+
[ "$unmeasured" -gt 0 ] && complete=false
|
|
71
155
|
|
|
72
156
|
# Recover the user-facing text (concatenate all assistant text blocks) so
|
|
73
157
|
# downstream consumers that read `omp -p`'s stdout still work.
|
|
@@ -75,10 +159,11 @@ capture_omp() {
|
|
|
75
159
|
| jq -r 'select(.message.role=="assistant") | .message.content[]? | select(.type=="text") | .text' \
|
|
76
160
|
|| true
|
|
77
161
|
|
|
78
|
-
# Write metrics record. Same field names as capture_claude's record;
|
|
79
|
-
#
|
|
80
|
-
# claude_exit_code), by the same convention the two capture_* files
|
|
81
|
-
#
|
|
162
|
+
# Write metrics record. Same field names as capture_claude's record; the
|
|
163
|
+
# harness-specific exit-code key differs (omp_exit_code vs
|
|
164
|
+
# claude_exit_code), by the same convention the two capture_* files already
|
|
165
|
+
# use for their own harness-labelled fields, and the usage_* /subagent_*
|
|
166
|
+
# keys describe a scope Claude Code's single `result` event does not have.
|
|
82
167
|
local record_stamp record_file
|
|
83
168
|
record_stamp=$(date +%s%N)
|
|
84
169
|
record_file="$metrics_dir/${job_name}-${job_id}-${record_stamp}.json"
|
|
@@ -100,6 +185,10 @@ capture_omp() {
|
|
|
100
185
|
--argjson duration "$duration" \
|
|
101
186
|
--argjson wall_ms "$wall_ms" \
|
|
102
187
|
--argjson exit_code "$exit_code" \
|
|
188
|
+
--argjson sub_cost "$sub_cost" \
|
|
189
|
+
--argjson measured "$measured" \
|
|
190
|
+
--argjson unmeasured "$unmeasured" \
|
|
191
|
+
--argjson complete "$complete" \
|
|
103
192
|
'{
|
|
104
193
|
ts: $ts, agent: $agent, event: $event, model: $model,
|
|
105
194
|
pipeline_id: $pipeline_id, job_id: $job_id, job_name: $job_name,
|
|
@@ -107,7 +196,10 @@ capture_omp() {
|
|
|
107
196
|
input_tokens: $input, output_tokens: $output,
|
|
108
197
|
cache_creation_tokens: $cache_c, cache_read_tokens: $cache_r,
|
|
109
198
|
total_cost_usd: $cost, duration_ms: $duration, wall_ms: $wall_ms,
|
|
110
|
-
omp_exit_code: $exit_code
|
|
199
|
+
omp_exit_code: $exit_code,
|
|
200
|
+
usage_scope: "tree", usage_complete: $complete,
|
|
201
|
+
subagents_measured: $measured, subagents_unmeasured: $unmeasured,
|
|
202
|
+
subagent_cost_usd: $sub_cost
|
|
111
203
|
}' > "$record_file"
|
|
112
204
|
|
|
113
205
|
# If the agent failed, forward stderr so debugging still works, and keep a
|