oh-my-customcode 1.1.48 → 1.1.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +7 -6
  2. package/dist/cli/index.js +1 -1
  3. package/dist/index.js +1 -1
  4. package/package.json +1 -1
  5. package/templates/.claude/agents/agora-runner.md +114 -0
  6. package/templates/.claude/hooks/scripts/agent-teams-advisor.sh +6 -1
  7. package/templates/.claude/hooks/scripts/r007-r008-drift-advisor.sh +41 -3
  8. package/templates/.claude/hooks/scripts/session-env-check.sh +29 -3
  9. package/templates/.claude/rules/MAY-optimization.md +4 -0
  10. package/templates/.claude/rules/MUST-agent-design.md +2 -0
  11. package/templates/.claude/rules/MUST-completion-verification.md +25 -0
  12. package/templates/.claude/rules/MUST-enforcement-policy.md +3 -3
  13. package/templates/.claude/rules/MUST-orchestrator-coordination.md +39 -0
  14. package/templates/.claude/rules/MUST-parallel-execution.md +56 -7
  15. package/templates/.claude/rules/MUST-sync-verification.md +16 -0
  16. package/templates/.claude/rules/MUST-tool-identification.md +22 -4
  17. package/templates/.claude/skills/agora/SKILL.md +325 -0
  18. package/templates/.claude/skills/agora/scripts/agora.sh +761 -0
  19. package/templates/.claude/skills/agora/scripts/anonymize.sh +492 -0
  20. package/templates/.claude/skills/agora/scripts/judge.sh +427 -0
  21. package/templates/.claude/skills/agora/scripts/response-schema.json +26 -0
  22. package/templates/.claude/skills/agora/scripts/reviewers.sh +312 -0
  23. package/templates/.claude/skills/agora/scripts/verdict-schema.json +34 -0
  24. package/templates/.claude/skills/hada-scout/SKILL.md +1 -1
  25. package/templates/.claude/skills/help/SKILL.md +1 -1
  26. package/templates/.claude/skills/pipeline/workflows/auto-dev.yaml +21 -3
  27. package/templates/.claude/skills/sauron-watch/SKILL.md +1 -1
  28. package/templates/.claude/skills/status/SKILL.md +3 -3
  29. package/templates/.claude/skills/token-efficiency-audit/SKILL.md +1 -1
  30. package/templates/CLAUDE.md +3 -3
  31. package/templates/CLAUDE.md.en +3 -3
  32. package/templates/CLAUDE.md.ko +3 -3
  33. package/templates/README.md +5 -5
  34. package/templates/guides/agent-eval/README.md +1 -1
  35. package/templates/manifest.json +3 -3
  36. package/templates/workflows/auto-dev.yaml +21 -3
@@ -0,0 +1,427 @@
1
+ #!/usr/bin/env bash
2
+ # judge.sh — separate-process judge with per-round model rotation.
3
+ # Spec: docs/superpowers/plans/2026-08-15-agora-anonymous-consensus-design.md §3 §8 §11
4
+ #
5
+ # This script receives ONLY the anonymous bundle path. It accepts no path to
6
+ # any other session material, neither as an argument nor through the
7
+ # environment — the judge process has no route back to who said what.
8
+ set -uo pipefail
9
+
10
+ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
11
+ # AGORA_VERDICT_SCHEMA points this script at a substitute schema path. Unset —
12
+ # which is every non-test invocation — it resolves to the file shipped next to
13
+ # this script, byte-for-byte the previous behaviour. Same override convention
14
+ # as AGORA_CLAUDE_BIN / AGORA_AGY_BIN / AGORA_TIMEOUT_SECS /
15
+ # AGORA_FORCE_TIMEOUT_FALLBACK below: a test that needs to exercise the
16
+ # "schema unreadable" path points this at a path that does not exist, instead
17
+ # of moving the real tracked file aside and hoping its `finally` runs.
18
+ VERDICT_SCHEMA="${AGORA_VERDICT_SCHEMA:-$SCRIPT_DIR/verdict-schema.json}"
19
+
20
+ AGORA_CLAUDE_BIN="${AGORA_CLAUDE_BIN:-claude}"
21
+ AGORA_AGY_BIN="${AGORA_AGY_BIN:-agy}"
22
+ AGORA_TIMEOUT_SECS="${AGORA_TIMEOUT_SECS:-300}"
23
+
24
+ # ---------------------------------------------------------------------------
25
+ # _child_env_strip — the `env -u NAME` list that removes every AGORA_*
26
+ # variable from the judge CLI's environment.
27
+ #
28
+ # The header above claims this script accepts no route back to who said what
29
+ # "neither as an argument nor through the environment". Argv was guarded; the
30
+ # environment was not. What agora.sh does not pass is beside the point — the
31
+ # OPERATOR's shell exports reach this process anyway. Measured: with
32
+ # AGORA_OUTPUT_ROOT set in the invoking shell, that value is inherited
33
+ # straight through into the judge CLI, and the session tree it names holds
34
+ # the sealed label-to-vendor record one glob below. That is the whole
35
+ # anonymity property, handed over by inheritance.
36
+ #
37
+ # Prefix rule rather than a hardcoded name list, deliberately: a name list
38
+ # leaks silently the day someone adds AGORA_SOMETHING_NEW, whereas the prefix
39
+ # fails CLOSED — a new variable is stripped unless someone argues it back in.
40
+ # Nothing outside the prefix is touched, so PATH, HOME, the judge CLI's own
41
+ # auth tokens and proxy settings all survive; `env -i` would leave it unable
42
+ # to authenticate at all.
43
+ #
44
+ # Read time and pass time are separate: the config reads above have already
45
+ # happened, so this script goes on using AGORA_CLAUDE_BIN, AGORA_TIMEOUT_SECS
46
+ # and AGORA_VERDICT_SCHEMA as ordinary shell variables. Only the CHILD's copy
47
+ # is removed, which is why test-injected configuration still works.
48
+ #
49
+ # Built from the exported names actually present (compgen -e), so a prefixed
50
+ # variable this script has never heard of is covered too.
51
+ # ---------------------------------------------------------------------------
52
+ _child_env_strip=()
53
+ while IFS= read -r _agora_env_name; do
54
+ [ -n "$_agora_env_name" ] || continue
55
+ _child_env_strip+=(-u "$_agora_env_name")
56
+ done < <(compgen -e 2>/dev/null | grep '^AGORA_' || true)
57
+ unset _agora_env_name
58
+
59
+ # spec REQ-3 rotation roster: three slots, disjoint from the reviewer
60
+ # roster by model (reviewers.sh uses claude-opus-4-8 and gemini-3.1-pro-high).
61
+ JUDGE_ROTATION=(
62
+ 'claude:claude-opus-5'
63
+ 'agy:claude-opus-4-6-thinking'
64
+ 'agy:gpt-oss-120b-medium'
65
+ )
66
+
67
+ # ---------------------------------------------------------------------------
68
+ # judge_model_for_round <round> [offset] — R1/R2/R3 fixed, R4+ cycles; an
69
+ # offset advances to the next rotation slot (used for judge failover).
70
+ # ---------------------------------------------------------------------------
71
+ judge_model_for_round() {
72
+ local round="$1" offset="${2:-0}"
73
+ local n=${#JUDGE_ROTATION[@]}
74
+ local idx=$(( (round - 1 + offset) % n ))
75
+ printf '%s\n' "${JUDGE_ROTATION[$idx]}"
76
+ }
77
+
78
+ # ---------------------------------------------------------------------------
79
+ # judge_prompt <anon_file> <model_id> — assembles the judge's instructions.
80
+ # Only the anonymous bundle's own content is ever embedded here — no path or
81
+ # reference to any other session material.
82
+ # ---------------------------------------------------------------------------
83
+ judge_prompt() {
84
+ local anon_file="$1" model_id="$2"
85
+ cat <<PROMPT
86
+ 당신은 익명 리뷰 합의 절차의 심판입니다.
87
+
88
+ 입력은 아래 익명 번들 JSON 뿐입니다. 리뷰어는 A/B/C 라벨로만 식별되며,
89
+ 라벨이 어떤 도구·모델에 대응하는지는 알 수 없고 추측해서도 안 됩니다.
90
+ 참여 인원은 reviewers 배열의 길이와 같습니다 — 배열에 없는 라벨의 침묵을
91
+ 유의미한 신호로 해석하지 마십시오.
92
+
93
+ 수행할 일은 세 가지입니다.
94
+ 1. 리뷰어 의견을 평가하여 consensus 와 verdict 를 판정합니다.
95
+ 2. 다음 라운드 의제(agenda)를 설정합니다.
96
+ 3. 재작성된 통합 초안(draft)을 제시합니다.
97
+
98
+ 출력은 아래 JSON 스키마를 정확히 따르는 JSON 객체 하나여야 하며,
99
+ "judge" 필드에는 "$model_id" 를 그대로 넣습니다. 다른 텍스트를 덧붙이지 마십시오.
100
+
101
+ $(cat "$VERDICT_SCHEMA")
102
+
103
+ --- 익명 번들 ---
104
+ $(cat "$anon_file")
105
+ PROMPT
106
+ }
107
+
108
+ # ---------------------------------------------------------------------------
109
+ # run_with_timeout <seconds> <cmd>... — R005: macOS ships no GNU timeout.
110
+ # Kept self-contained here (not shared with reviewers.sh) matching this
111
+ # directory's existing convention of scripts that do not import one another.
112
+ # Same construction and same measured fix as reviewers.sh's run_with_timeout:
113
+ # on timeout, kills the WHOLE PROCESS GROUP of the backgrounded command, not
114
+ # just its direct PID — a judge CLI that forks a grandchild is not left
115
+ # running after this function reports it as timed out. Getting a process
116
+ # group at all requires job control (`set -m`); scoped strictly to this
117
+ # function so the rest of the script is unaffected. AGORA_FORCE_TIMEOUT_FALLBACK=1
118
+ # is a test-only switch that forces the wait-based fallback branch even when
119
+ # gtimeout is installed, so both branches can be exercised deterministically.
120
+ # Returns 124 on timeout, otherwise the command's own exit code.
121
+ # ---------------------------------------------------------------------------
122
+ run_with_timeout() {
123
+ local secs="$1"; shift
124
+ if [ "${AGORA_FORCE_TIMEOUT_FALLBACK:-0}" != "1" ] && command -v gtimeout >/dev/null 2>&1; then
125
+ gtimeout "$secs" "$@"
126
+ return $?
127
+ fi
128
+
129
+ local restore_job_control=0
130
+ case $- in
131
+ *m*) ;;
132
+ *) set -m; restore_job_control=1 ;;
133
+ esac
134
+
135
+ "$@" &
136
+ local pid=$!
137
+ # Negative PID targets the whole process group (job-control-assigned pgid
138
+ # == the group leader's pid), so grandchildren the command forks are
139
+ # reaped too, not just the direct child.
140
+ ( sleep "$secs"; kill -TERM -- "-$pid" 2>/dev/null ) &
141
+ local watcher=$!
142
+
143
+ local rc=0
144
+ # 2>/dev/null: suppresses this shell's own job-control "Terminated"
145
+ # notification, a side effect of `wait` reaping a signal-killed background
146
+ # job under `set -m` — noise, not a diagnostic this script itself emits.
147
+ wait "$pid" 2>/dev/null || rc=$?
148
+ # Same measured fix as reviewers.sh: tear the watchdog down by PROCESS
149
+ # GROUP, not by bare pid. `kill -TERM "$watcher"` reaches only the subshell
150
+ # and orphans its `sleep`, which then lingers for the full
151
+ # AGORA_TIMEOUT_SECS under init. It leaks on the SUCCESS path (on timeout
152
+ # the sleep has already elapsed by definition), so every judge attempt that
153
+ # returned normally left one behind. Safe under `set -m` (guaranteed on in
154
+ # this branch): measured pgid(watcher) == pid(watcher) != pgid(this
155
+ # script), and $watcher is still unreaped here, so its pid/group cannot
156
+ # have been recycled onto anything else before the `wait` below.
157
+ kill -TERM -- "-$watcher" 2>/dev/null || true
158
+ wait "$watcher" 2>/dev/null || true
159
+
160
+ [ "$restore_job_control" -eq 1 ] && set +m
161
+
162
+ # 143 = SIGTERM from the watchdog; normalize to the GNU timeout convention.
163
+ [ "$rc" -eq 143 ] && rc=124
164
+ return "$rc"
165
+ }
166
+
167
+ # ---------------------------------------------------------------------------
168
+ # run_sanitized <seconds> <cmd>... — run_with_timeout with the AGORA_*
169
+ # variables removed from the command's environment (see _child_env_strip).
170
+ # Wrapping the command in `env` rather than unsetting the variables in this
171
+ # shell is what keeps the read/pass split above intact. `env` execs the
172
+ # command in place, so the pid and process group that run_with_timeout's
173
+ # watchdog targets are the command's own, exactly as before.
174
+ # ---------------------------------------------------------------------------
175
+ run_sanitized() {
176
+ local secs="$1"; shift
177
+ run_with_timeout "$secs" env ${_child_env_strip[@]+"${_child_env_strip[@]}"} "$@"
178
+ }
179
+
180
+ # ---------------------------------------------------------------------------
181
+ # invoke_judge <model_id> <prompt> — dispatches to the vendor-specific CLI
182
+ # argument shape, applying run_sanitized uniformly so the judge runs under
183
+ # both the timeout and the env strip. argv carries only the prompt string
184
+ # (itself built from nothing but the anonymous bundle) — no session path, no
185
+ # material beyond the bundle itself.
186
+ #
187
+ # Deliberate asymmetry with reviewers.sh: neither branch below passes the
188
+ # permission-bypass flag reviewers.sh's invoke_vendor() adds for the SAME
189
+ # CLI (claude's --enable-auto-mode, agy's --dangerously-skip-permissions —
190
+ # Ruling 12, reviewers.sh:164-183). This is not a missed flag; it is the
191
+ # boundary. The judge's only legitimate input is the anonymous bundle
192
+ # (--anon-file, read by run_judge and turned into argv above) — a
193
+ # permission-bypass flag would let the judge CLI touch the filesystem
194
+ # without prompting, and that access is exactly what this skill's sealing
195
+ # exists to deny: the sealed mapping directory holds the label-to-vendor
196
+ # record the judge must never read. reviewers.sh's reviewers legitimately
197
+ # need filesystem access for their own work (reading the code under
198
+ # review); the judge only judges the anonymous bundle, so it needs none.
199
+ #
200
+ # Named in prose, never as a path literal, and not by accident: a source
201
+ # guard in tests/unit/skills/agora-scripts.test.ts asserts this file
202
+ # contains no spelling of that directory at all. The guard is the cheap
203
+ # deterministic proof of the paragraph above — code that never names the
204
+ # location cannot be assembling a route to it — so describing the
205
+ # directory is fine while writing its path is a test failure. Do not
206
+ # "clarify" this comment by substituting the concrete path back in.
207
+ #
208
+ # Operational consequence, so a future session does not "fix" this by
209
+ # adding the flag: in an environment where the judge CLI would otherwise
210
+ # block on a permission prompt, that attempt times out under
211
+ # AGORA_TIMEOUT_SECS (rc=124, handled in run_judge below) and the 3-slot
212
+ # rotation (spec REQ-3, judge_model_for_round) advances to the next model;
213
+ # exhausting all three slots is exit 4 (run_judge, "every rotation model
214
+ # failed"). If the judge keeps failing for this reason, the fix is to check
215
+ # the CLI's own permission state (e.g. run it once interactively to grant
216
+ # whatever it needs) — never to add the bypass flag here.
217
+ # ---------------------------------------------------------------------------
218
+ invoke_judge() {
219
+ local model_id="$1" prompt="$2"
220
+ local cli="${model_id%%:*}" model="${model_id#*:}"
221
+ case "$cli" in
222
+ claude)
223
+ run_sanitized "$AGORA_TIMEOUT_SECS" "$AGORA_CLAUDE_BIN" \
224
+ -p --model "$model" "$prompt"
225
+ ;;
226
+ agy)
227
+ run_sanitized "$AGORA_TIMEOUT_SECS" "$AGORA_AGY_BIN" \
228
+ -p --model "$model" --output-format json \
229
+ --json-schema "$VERDICT_SCHEMA" "$prompt"
230
+ ;;
231
+ *) return 65 ;;
232
+ esac
233
+ }
234
+
235
+ # ---------------------------------------------------------------------------
236
+ # _emit_judge_stderr_tail <model_id> <err_file> — mirrors reviewers.sh's
237
+ # _emit_vendor_stderr_tail: fold the last lines of a failed judge's own
238
+ # stderr into this script's diagnostic output instead of discarding it, so
239
+ # the actual cause (auth, rate-limit, model deprecation, timeout) is
240
+ # debuggable.
241
+ # ---------------------------------------------------------------------------
242
+ _emit_judge_stderr_tail() {
243
+ local model_id="$1" err_file="$2"
244
+ if [ -s "$err_file" ]; then
245
+ printf '[agora] judge %s stderr (last 20 lines):\n' "$model_id" >&2
246
+ tail -n 20 "$err_file" >&2
247
+ fi
248
+ }
249
+
250
+ # ---------------------------------------------------------------------------
251
+ # validate_verdict <file> — spec §8: `jq -e .` (the caller's syntactic-JSON
252
+ # check) only proves the response IS json; it does not prove the response
253
+ # HAS the right shape. A response can be syntactically valid and still be
254
+ # missing a required field, or carry an enum typo (verdict: "MERGE") that
255
+ # would flow straight into Task 1's decide_stop and silently compare false
256
+ # forever. The required-field list, the declared TYPES and the enum values
257
+ # are all read FROM verdict-schema.json itself — never hardcoded here — so
258
+ # this cannot drift from the schema file the same way anonymize.sh's
259
+ # fingerprint guard reads its own banned-pattern constant directly instead
260
+ # of duplicating it.
261
+ #
262
+ # F1 (review round 3, measured): presence + enums alone let a WRONG-TYPED
263
+ # field through. A judge returning `"agenda": "1. 단일 의제"` (string where
264
+ # the schema declares an array) passed every check here and was written to
265
+ # verdict/round-N.json. agora.sh's run_round then reads it back next round
266
+ # and runs `jq '. + $extra'`, which dies with `string and array cannot be
267
+ # added`; the rc is discarded by the `agenda=$(...)` assignment, so agenda
268
+ # collapses to empty, build_reviewer_prompt renders a BLANK agenda section,
269
+ # and all three vendors are invoked and billed before anonymize.sh's later
270
+ # `--agenda ''` finally surfaces the problem. That is precisely the failure
271
+ # agora.sh's own --extra-agenda pre-check exists to prevent — but that guard
272
+ # only covers USER input into that slot, and the judge's own output lands in
273
+ # the same slot with no equivalent guard. This closes it at the source.
274
+ #
275
+ # Prints every rejection reason to stderr; returns 0 only when every
276
+ # required field is present (and non-null), every field's value matches the
277
+ # type the schema declares for it, and every enum value it carries is one of
278
+ # the schema's allowed values.
279
+ # ---------------------------------------------------------------------------
280
+ validate_verdict() {
281
+ local file="$1"
282
+ local reasons=() key val want got enum_field enum_values r
283
+
284
+ while IFS= read -r key; do
285
+ [ -n "$key" ] || continue
286
+ if ! jq -e --arg k "$key" 'has($k) and (.[$k] != null)' "$file" >/dev/null 2>&1; then
287
+ reasons+=("missing required field: $key")
288
+ fi
289
+ done < <(jq -r '.required[]' "$VERDICT_SCHEMA")
290
+
291
+ # Type check, driven entirely by the schema's own .properties[].type.
292
+ # jq's `type` yields exactly the JSON Schema primitive names (object,
293
+ # array, string, number, boolean, null), so the comparison is direct.
294
+ # Absent and null values are skipped here: every property in this schema
295
+ # is also in .required, so the loop above already reports them, and
296
+ # reporting the same defect twice would only add noise. Properties whose
297
+ # `type` is not a plain string (e.g. a ["string","null"] union) are
298
+ # skipped rather than guessed at, so a future schema edit degrades to
299
+ # "not type-checked" instead of to a false rejection.
300
+ while IFS=$'\t' read -r key want; do
301
+ [ -n "$key" ] && [ -n "$want" ] || continue
302
+ got=$(jq -r --arg k "$key" 'if has($k) then (.[$k] | type) else "absent" end' "$file" 2>/dev/null)
303
+ case "$got" in absent | null) continue ;; esac
304
+ if [ "$got" != "$want" ]; then
305
+ reasons+=("field $key: expected $want, got $got")
306
+ fi
307
+ done < <(jq -r '
308
+ .properties | to_entries[]
309
+ | select(.value.type | type == "string")
310
+ | "\(.key)\t\(.value.type)"' "$VERDICT_SCHEMA")
311
+
312
+ for enum_field in consensus verdict; do
313
+ val=$(jq -r --arg f "$enum_field" '.[$f] // empty' "$file")
314
+ [ -n "$val" ] || continue
315
+ enum_values=$(jq -r --arg f "$enum_field" '.properties[$f].enum[]' "$VERDICT_SCHEMA")
316
+ if ! printf '%s\n' "$enum_values" | grep -qxF -- "$val"; then
317
+ reasons+=("invalid $enum_field: $val")
318
+ fi
319
+ done
320
+
321
+ if [ "${#reasons[@]}" -gt 0 ]; then
322
+ for r in "${reasons[@]}"; do
323
+ printf '[agora] verdict schema violation: %s\n' "$r" >&2
324
+ done
325
+ return 1
326
+ fi
327
+ return 0
328
+ }
329
+
330
+ # ---------------------------------------------------------------------------
331
+ # run_judge --anon-file <path> --out-file <path> --round <N>
332
+ # Tries the three rotation slots in order (spec §11 failover). Writes
333
+ # <out-file> only after the response is confirmed valid JSON AND schema-valid
334
+ # (validate_verdict), then stamps .judge/.round onto it. Each attempt's
335
+ # stderr is captured next to the verdict output itself (never any other
336
+ # session path) and removed on success; a failing attempt's log is left on
337
+ # disk for audit and its tail is folded into this script's own diagnostic
338
+ # stderr.
339
+ # ---------------------------------------------------------------------------
340
+ run_judge() {
341
+ local anon_file='' out_file='' round=''
342
+ while [ "$#" -gt 0 ]; do
343
+ case "$1" in
344
+ --anon-file) anon_file="$2"; shift 2 ;;
345
+ --out-file) out_file="$2"; shift 2 ;;
346
+ --round) round="$2"; shift 2 ;;
347
+ *) printf 'judge.sh: unknown option %s\n' "$1" >&2; return 64 ;;
348
+ esac
349
+ done
350
+ [ -n "$anon_file" ] && [ -n "$out_file" ] && [ -n "$round" ] || {
351
+ printf 'judge.sh: --anon-file, --out-file and --round are required\n' >&2
352
+ return 64
353
+ }
354
+ [ -f "$anon_file" ] || { printf 'judge.sh: %s not found\n' "$anon_file" >&2; return 66; }
355
+
356
+ # F1 (review round): fail explicitly and up front if the schema this
357
+ # script validates against cannot even be read/parsed — required-field and
358
+ # enum checks below are read FROM this file, so a silently-missing schema
359
+ # would silently disable validation instead of loudly failing.
360
+ jq -e . "$VERDICT_SCHEMA" >/dev/null 2>&1 || {
361
+ printf 'judge.sh: cannot read or parse verdict schema at %s\n' "$VERDICT_SCHEMA" >&2
362
+ return 68
363
+ }
364
+
365
+ local out_dir; out_dir="$(dirname "$out_file")"
366
+ mkdir -p "$out_dir"
367
+
368
+ local tmp; tmp=$(mktemp)
369
+ local offset model_id prompt rc err_file
370
+ for offset in 0 1 2; do
371
+ model_id=$(judge_model_for_round "$round" "$offset")
372
+ prompt=$(judge_prompt "$anon_file" "$model_id")
373
+ # F2 (review round): the round is part of the filename so a later round
374
+ # retried in the same out_dir never overwrites an earlier round's audit
375
+ # log — offset alone collided across rounds.
376
+ err_file="$out_dir/.judge-round-$round-attempt-$offset.stderr.log"
377
+
378
+ rc=0
379
+ invoke_judge "$model_id" "$prompt" > "$tmp" 2> "$err_file" || rc=$?
380
+
381
+ if [ "$rc" -eq 124 ]; then
382
+ printf '[agora] judge %s timed out for round %s\n' "$model_id" "$round" >&2
383
+ _emit_judge_stderr_tail "$model_id" "$err_file"
384
+ elif [ "$rc" -ne 0 ]; then
385
+ printf '[agora] judge %s failed (rc=%s) for round %s\n' "$model_id" "$rc" "$round" >&2
386
+ _emit_judge_stderr_tail "$model_id" "$err_file"
387
+ elif ! jq -e . "$tmp" >/dev/null 2>&1; then
388
+ printf '[agora] judge %s returned unparsable output for round %s\n' "$model_id" "$round" >&2
389
+ _emit_judge_stderr_tail "$model_id" "$err_file"
390
+ rc=65
391
+ elif ! validate_verdict "$tmp"; then
392
+ printf '[agora] judge %s returned a schema-violating verdict for round %s\n' "$model_id" "$round" >&2
393
+ rc=65
394
+ else
395
+ # Success is written to the final path ONLY after the response has
396
+ # been confirmed valid JSON — never before the check is settled.
397
+ rm -f "$err_file"
398
+ jq -c --arg j "$model_id" --argjson r "$round" '.judge = $j | .round = $r' "$tmp" > "$out_file"
399
+ printf '[agora] round %s judged by rotation slot %s (%s)\n' "$round" "$(( offset + 1 ))" "$model_id" >&2
400
+ rm -f "$tmp"
401
+ return 0
402
+ fi
403
+
404
+ printf '[agora] advancing judge rotation for round %s\n' "$round" >&2
405
+ done
406
+
407
+ rm -f "$tmp"
408
+ printf '[agora] every rotation model failed for round %s\n' "$round" >&2
409
+ return 4
410
+ }
411
+
412
+ main() {
413
+ case "${1:---help}" in
414
+ --model-for-round) shift; judge_model_for_round "$@" ;;
415
+ --run) shift; run_judge "$@" ;;
416
+ --help | -h)
417
+ cat <<'USAGE'
418
+ Usage:
419
+ judge.sh --model-for-round <round> [offset]
420
+ judge.sh --run --anon-file <path> --out-file <path> --round <N>
421
+ USAGE
422
+ ;;
423
+ *) printf 'judge.sh: unknown option %s\n' "$1" >&2; return 64 ;;
424
+ esac
425
+ }
426
+
427
+ main "$@"
@@ -0,0 +1,26 @@
1
+ {
2
+ "type": "object",
3
+ "required": ["findings", "overall", "rationale"],
4
+ "additionalProperties": false,
5
+ "properties": {
6
+ "findings": {
7
+ "type": "array",
8
+ "items": {
9
+ "type": "object",
10
+ "required": ["id", "severity", "claim", "evidence", "impact", "counter", "verdict"],
11
+ "additionalProperties": false,
12
+ "properties": {
13
+ "id": { "type": "string", "minLength": 1 },
14
+ "severity": { "type": "string", "enum": ["CRITICAL", "HIGH", "MEDIUM", "LOW"] },
15
+ "claim": { "type": "string", "minLength": 1 },
16
+ "evidence": { "type": "string", "minLength": 1 },
17
+ "impact": { "type": "string", "minLength": 1 },
18
+ "counter": { "type": "string", "minLength": 1 },
19
+ "verdict": { "type": "string", "enum": ["KEEP", "MODIFY", "REJECT"] }
20
+ }
21
+ }
22
+ },
23
+ "overall": { "type": "string", "enum": ["BUILD", "BUILD_WITH_CHANGES", "REDESIGN", "ABANDON"] },
24
+ "rationale": { "type": "string", "minLength": 1 }
25
+ }
26
+ }