code-coordinator 0.5.46__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- code_coordinator-0.5.46.dist-info/METADATA +625 -0
- code_coordinator-0.5.46.dist-info/RECORD +295 -0
- code_coordinator-0.5.46.dist-info/WHEEL +5 -0
- code_coordinator-0.5.46.dist-info/entry_points.txt +2 -0
- code_coordinator-0.5.46.dist-info/licenses/LICENSE +110 -0
- code_coordinator-0.5.46.dist-info/top_level.txt +1 -0
- coord/__init__.py +176 -0
- coord/_board_mapping.py +229 -0
- coord/acceptance.py +468 -0
- coord/acceptance_drivers.py +632 -0
- coord/agent.py +7517 -0
- coord/agent_app.py +1555 -0
- coord/agent_update.py +417 -0
- coord/agents/opencode/.gitignore +13 -0
- coord/agents/opencode/agents/work.md +129 -0
- coord/agents/opencode/routing.jsonc +49 -0
- coord/audit.py +301 -0
- coord/auto_loop.py +1440 -0
- coord/board_bool_guard.py +72 -0
- coord/board_service.py +141 -0
- coord/board_wire.py +309 -0
- coord/brain.py +581 -0
- coord/branch_model.py +214 -0
- coord/cargo_cache.py +258 -0
- coord/ci_github.py +386 -0
- coord/ci_store.py +560 -0
- coord/claim.py +353 -0
- coord/cli.py +454 -0
- coord/client.py +610 -0
- coord/commands/__init__.py +1 -0
- coord/commands/_common.py +329 -0
- coord/commands/acceptance.py +916 -0
- coord/commands/agent_ops.py +1339 -0
- coord/commands/audit.py +131 -0
- coord/commands/chat.py +320 -0
- coord/commands/dispatch.py +1780 -0
- coord/commands/dispatch_workers.py +4894 -0
- coord/commands/drive.py +616 -0
- coord/commands/drive_queue.py +1203 -0
- coord/commands/gate_a.py +217 -0
- coord/commands/gates.py +89 -0
- coord/commands/issues.py +681 -0
- coord/commands/lifecycle.py +513 -0
- coord/commands/merge.py +1900 -0
- coord/commands/milestone.py +2081 -0
- coord/commands/plan_followup.py +1243 -0
- coord/commands/plans.py +156 -0
- coord/commands/release.py +2232 -0
- coord/commands/report.py +341 -0
- coord/commands/review.py +1523 -0
- coord/commands/scorecard.py +252 -0
- coord/commands/sessions.py +1930 -0
- coord/commands/setup.py +576 -0
- coord/commands/status.py +2089 -0
- coord/commands/terminal.py +385 -0
- coord/commands/test_gate.py +775 -0
- coord/commands/tui.py +288 -0
- coord/comments.py +718 -0
- coord/config.py +3032 -0
- coord/conflict_fix.py +633 -0
- coord/dao.py +483 -0
- coord/dashboard/__init__.py +0 -0
- coord/dashboard/fixture.py +376 -0
- coord/dashboard/index.html +658 -0
- coord/dashboard/server.py +1894 -0
- coord/dashboard/terminal.py +382 -0
- coord/dashboard/webapp/.gitignore +9 -0
- coord/dashboard/webapp/components.json +17 -0
- coord/dashboard/webapp/dist/assets/Gallery-da3qNiIw.js +71 -0
- coord/dashboard/webapp/dist/assets/Terminal-9CEnUXvW.css +32 -0
- coord/dashboard/webapp/dist/assets/Terminal-skVFCxPU.js +63 -0
- coord/dashboard/webapp/dist/assets/index-DltfZR5f.js +184 -0
- coord/dashboard/webapp/dist/assets/index-Dq4kwTdw.css +1 -0
- coord/dashboard/webapp/dist/assets/workbox-window.prod.es5-BqEJf4Xk.js +2 -0
- coord/dashboard/webapp/dist/icons/icon-192.png +0 -0
- coord/dashboard/webapp/dist/icons/icon-512.png +0 -0
- coord/dashboard/webapp/dist/icons/icon.svg +5 -0
- coord/dashboard/webapp/dist/index.html +38 -0
- coord/dashboard/webapp/dist/manifest.webmanifest +1 -0
- coord/dashboard/webapp/dist/sw.js +1 -0
- coord/dashboard/webapp/dist/workbox-e4022e15.js +1 -0
- coord/dashboard/webapp/e2e/available-gates-terminal.spec.ts +75 -0
- coord/dashboard/webapp/e2e/deep-link.spec.ts +172 -0
- coord/dashboard/webapp/e2e/fixtureServer.ts +155 -0
- coord/dashboard/webapp/e2e/live-update-fixture.spec.ts +113 -0
- coord/dashboard/webapp/e2e/realtime.spec.ts +238 -0
- coord/dashboard/webapp/e2e/shell.spec.ts +309 -0
- coord/dashboard/webapp/e2e/smoke.spec.ts +191 -0
- coord/dashboard/webapp/e2e/terminal.spec.ts +420 -0
- coord/dashboard/webapp/e2e/theme.spec.ts +138 -0
- coord/dashboard/webapp/eslint.config.js +20 -0
- coord/dashboard/webapp/index.html +37 -0
- coord/dashboard/webapp/node_modules/flatted/python/flatted.py +144 -0
- coord/dashboard/webapp/package-lock.json +10584 -0
- coord/dashboard/webapp/package.json +63 -0
- coord/dashboard/webapp/playwright.acceptance.config.ts +166 -0
- coord/dashboard/webapp/playwright.config.ts +93 -0
- coord/dashboard/webapp/postcss.config.js +6 -0
- coord/dashboard/webapp/public/icons/icon-192.png +0 -0
- coord/dashboard/webapp/public/icons/icon-512.png +0 -0
- coord/dashboard/webapp/public/icons/icon.svg +5 -0
- coord/dashboard/webapp/src/App.tsx +140 -0
- coord/dashboard/webapp/src/api/client.ts +199 -0
- coord/dashboard/webapp/src/api/generated.ts +176 -0
- coord/dashboard/webapp/src/components/ConnectionBadge.tsx +52 -0
- coord/dashboard/webapp/src/components/Detail.tsx +800 -0
- coord/dashboard/webapp/src/components/Gallery.tsx +341 -0
- coord/dashboard/webapp/src/components/Home.tsx +435 -0
- coord/dashboard/webapp/src/components/MobileKeyBar.tsx +280 -0
- coord/dashboard/webapp/src/components/PanelHeader.tsx +59 -0
- coord/dashboard/webapp/src/components/PipelineCard.tsx +168 -0
- coord/dashboard/webapp/src/components/SessionCard.tsx +99 -0
- coord/dashboard/webapp/src/components/SessionDetail.tsx +140 -0
- coord/dashboard/webapp/src/components/SessionsList.tsx +81 -0
- coord/dashboard/webapp/src/components/Terminal.tsx +376 -0
- coord/dashboard/webapp/src/components/__tests__/ConnectionBadge.test.tsx +81 -0
- coord/dashboard/webapp/src/components/__tests__/Detail.test.tsx +680 -0
- coord/dashboard/webapp/src/components/__tests__/Gallery.test.tsx +83 -0
- coord/dashboard/webapp/src/components/__tests__/Home.test.tsx +271 -0
- coord/dashboard/webapp/src/components/__tests__/MobileKeyBar.test.tsx +197 -0
- coord/dashboard/webapp/src/components/__tests__/PipelineCard.test.tsx +143 -0
- coord/dashboard/webapp/src/components/__tests__/SessionCard.test.tsx +106 -0
- coord/dashboard/webapp/src/components/__tests__/Terminal.test.tsx +504 -0
- coord/dashboard/webapp/src/components/ui/badge.tsx +41 -0
- coord/dashboard/webapp/src/components/ui/button.tsx +54 -0
- coord/dashboard/webapp/src/components/ui/card.tsx +55 -0
- coord/dashboard/webapp/src/components/ui/dialog.tsx +99 -0
- coord/dashboard/webapp/src/components/ui/dropdown-menu.tsx +189 -0
- coord/dashboard/webapp/src/components/ui/empty-state.tsx +35 -0
- coord/dashboard/webapp/src/components/ui/sheet.tsx +123 -0
- coord/dashboard/webapp/src/components/ui/skeleton.tsx +9 -0
- coord/dashboard/webapp/src/components/ui/tabs.tsx +55 -0
- coord/dashboard/webapp/src/components/ui/theme-provider.tsx +78 -0
- coord/dashboard/webapp/src/components/ui/theme-toggle.tsx +20 -0
- coord/dashboard/webapp/src/components/ui/toast.tsx +123 -0
- coord/dashboard/webapp/src/components/ui/toaster.tsx +30 -0
- coord/dashboard/webapp/src/components/ui/tooltip.tsx +26 -0
- coord/dashboard/webapp/src/components/ui/use-toast.ts +134 -0
- coord/dashboard/webapp/src/index.css +210 -0
- coord/dashboard/webapp/src/lib/pipeline.ts +29 -0
- coord/dashboard/webapp/src/lib/utils.ts +6 -0
- coord/dashboard/webapp/src/main.tsx +46 -0
- coord/dashboard/webapp/src/realtime/RealtimeProvider.tsx +112 -0
- coord/dashboard/webapp/src/realtime/__tests__/RealtimeProvider.test.tsx +189 -0
- coord/dashboard/webapp/src/realtime/__tests__/connection.test.ts +255 -0
- coord/dashboard/webapp/src/realtime/connection.ts +227 -0
- coord/dashboard/webapp/src/realtime/events.ts +100 -0
- coord/dashboard/webapp/src/routes/__tests__/paths.test.ts +92 -0
- coord/dashboard/webapp/src/routes/paths.ts +92 -0
- coord/dashboard/webapp/src/shell/ActivityRail.tsx +335 -0
- coord/dashboard/webapp/src/shell/AppShell.tsx +276 -0
- coord/dashboard/webapp/src/shell/ComingSoon.tsx +33 -0
- coord/dashboard/webapp/src/shell/EmptyDetail.tsx +26 -0
- coord/dashboard/webapp/src/shell/RouteNotFound.tsx +33 -0
- coord/dashboard/webapp/src/shell/ShellLayout.tsx +147 -0
- coord/dashboard/webapp/src/shell/StatusBar.tsx +46 -0
- coord/dashboard/webapp/src/shell/__tests__/ShellLayout.test.tsx +520 -0
- coord/dashboard/webapp/src/shell/__tests__/shellState.test.ts +95 -0
- coord/dashboard/webapp/src/shell/__tests__/stubViewport.ts +40 -0
- coord/dashboard/webapp/src/shell/breakpoints.ts +87 -0
- coord/dashboard/webapp/src/shell/railItems.ts +105 -0
- coord/dashboard/webapp/src/shell/shellState.ts +174 -0
- coord/dashboard/webapp/src/shell/useRegionFocus.ts +95 -0
- coord/dashboard/webapp/src/test-setup.ts +41 -0
- coord/dashboard/webapp/src/vite-env.d.ts +2 -0
- coord/dashboard/webapp/tailwind.config.js +140 -0
- coord/dashboard/webapp/tsconfig.json +25 -0
- coord/dashboard/webapp/tsconfig.node.json +11 -0
- coord/dashboard/webapp/vite.config.ts +71 -0
- coord/db.py +1076 -0
- coord/dead_end.py +332 -0
- coord/deploy/README.md +33 -0
- coord/deploy/coord-agent.service +89 -0
- coord/deploy/coord-db-backup.service +60 -0
- coord/deploy/coord-db-backup.sh +74 -0
- coord/deploy/coord-db-backup.timer +18 -0
- coord/deploy/coord-drive-queue.service +117 -0
- coord/deploy/coord-drive-queue.timer +39 -0
- coord/deploy/coord-notify.service +48 -0
- coord/deploy/coord-notify.timer +24 -0
- coord/deploy/coord-release-propagate.service +83 -0
- coord/deploy/coord-release-propagate.timer +38 -0
- coord/deploy/coord-release-window.service +119 -0
- coord/deploy/coord-release-window.timer +36 -0
- coord/deploy/coord-serve.service +82 -0
- coord/deploy/coord-web-dist-build.service +43 -0
- coord/deploy/coord-web-dist-build.timer +36 -0
- coord/deploy/coord-web.service +125 -0
- coord/deploy_manifest.py +80 -0
- coord/deploy_units.py +384 -0
- coord/deps.py +115 -0
- coord/diagnose.py +1623 -0
- coord/dispatch.py +1009 -0
- coord/dist_name.py +123 -0
- coord/drive.py +3101 -0
- coord/drive_queue.py +2298 -0
- coord/drive_state.py +870 -0
- coord/events.py +381 -0
- coord/failure_class.py +914 -0
- coord/filelock.py +168 -0
- coord/fleet_config_health.py +300 -0
- coord/freshness.py +206 -0
- coord/gate_a.py +469 -0
- coord/gate_b.py +411 -0
- coord/gate_snapshot.py +385 -0
- coord/gates.py +582 -0
- coord/github_ops.py +1954 -0
- coord/goal.py +125 -0
- coord/graph_health.py +348 -0
- coord/health/__init__.py +69 -0
- coord/health/aggregate.py +129 -0
- coord/health/checks/__init__.py +13 -0
- coord/health/checks/agent_install.py +280 -0
- coord/health/checks/cargo_targets.py +171 -0
- coord/health/checks/claude_binary.py +65 -0
- coord/health/checks/deploy_lane_facts.py +458 -0
- coord/health/checks/disk.py +99 -0
- coord/health/checks/fleet_board.py +89 -0
- coord/health/checks/fleet_deploy_lanes.py +469 -0
- coord/health/checks/fleet_phantom.py +69 -0
- coord/health/checks/fleet_unit_drift.py +151 -0
- coord/health/checks/graph.py +192 -0
- coord/health/checks/plan_usage.py +88 -0
- coord/health/checks/repo_state.py +161 -0
- coord/health/checks/spawned_coord.py +465 -0
- coord/health/checks/timer_active.py +254 -0
- coord/health/checks/toolchain.py +547 -0
- coord/health/checks/unit_drift.py +648 -0
- coord/health/checks/unit_enablement.py +171 -0
- coord/health/checks/worktrees.py +96 -0
- coord/health/cli.py +121 -0
- coord/health/context.py +106 -0
- coord/health/fleet_snapshot.py +477 -0
- coord/health/models.py +250 -0
- coord/health/pypi.py +231 -0
- coord/health/registry.py +240 -0
- coord/health/render.py +82 -0
- coord/health/units.py +60 -0
- coord/hooks.py +106 -0
- coord/housekeeping.py +204 -0
- coord/interactive.py +4286 -0
- coord/issue_store.py +1496 -0
- coord/liveness_auditor.py +293 -0
- coord/machine_pause.py +755 -0
- coord/merge_queue.py +4681 -0
- coord/milestone_chat.py +600 -0
- coord/milestone_dispatch.py +943 -0
- coord/milestone_gate.py +709 -0
- coord/milestone_order.py +840 -0
- coord/mock_author.py +334 -0
- coord/models.py +891 -0
- coord/network.py +269 -0
- coord/new_issue_chat.py +229 -0
- coord/notify.py +3226 -0
- coord/openapi.py +404 -0
- coord/overlap_fence.py +133 -0
- coord/parentage.py +200 -0
- coord/parentage_github.py +58 -0
- coord/pipeline.py +481 -0
- coord/plan_parser.py +266 -0
- coord/plans.py +543 -0
- coord/platform_paths.py +43 -0
- coord/pr_body_lint.py +67 -0
- coord/prereqs.py +533 -0
- coord/progress.py +425 -0
- coord/providers/__init__.py +683 -0
- coord/providers/base.py +218 -0
- coord/providers/claude.py +284 -0
- coord/providers/claude_pty.py +610 -0
- coord/providers/opencode.py +896 -0
- coord/reconcile.py +2233 -0
- coord/refine_chat.py +485 -0
- coord/release_cordon.py +525 -0
- coord/release_propagate.py +1176 -0
- coord/release_verify.py +777 -0
- coord/release_window.py +322 -0
- coord/reports.py +1643 -0
- coord/revalidate.py +1101 -0
- coord/review.py +3317 -0
- coord/scorecard.py +484 -0
- coord/serve_app.py +7192 -0
- coord/skills/update-issue/SKILL.md +93 -0
- coord/smoke.py +1030 -0
- coord/split_work.py +210 -0
- coord/stage_projection.py +650 -0
- coord/state.py +5720 -0
- coord/test_author.py +1064 -0
- coord/test_chat.py +352 -0
- coord/test_orchestrator.py +494 -0
- coord/test_report.py +178 -0
- coord/tui_release.py +271 -0
- coord/usage.py +753 -0
- coord/usage_limits.py +358 -0
- coord/usage_rollup.py +709 -0
- coord/worker_events.py +954 -0
|
@@ -0,0 +1,293 @@
|
|
|
1
|
+
"""Cheap, independent, per-turn liveness auditor (#2048).
|
|
2
|
+
|
|
3
|
+
Fills the gap between three existing stall signals:
|
|
4
|
+
|
|
5
|
+
1. Worker self-report (``STATUS:``/``STUCK:`` lines, ``coord/worker_events.py``)
|
|
6
|
+
— free, but a worker that doesn't know it's stuck never says so.
|
|
7
|
+
2. ``EVENT_NEEDS_ATTENTION`` (``coord/notify.py``'s ``attention_signal``) —
|
|
8
|
+
independent, but it's a clock, not a judge: it can say "90 minutes
|
|
9
|
+
elapsed", not "this has been circling for six turns".
|
|
10
|
+
3. Adversarial review / Test agent / sealed oracle — independent judgment,
|
|
11
|
+
but metered and only runs once per stage boundary.
|
|
12
|
+
|
|
13
|
+
This module is tier 2.5: independent judgment, per turn, at clock prices.
|
|
14
|
+
A small model (default Claude Haiku) sees ONLY the assignment's objective
|
|
15
|
+
and the raw output of its single most recent turn — never the transcript,
|
|
16
|
+
never the worker's own ``STATUS:``/``STUCK:`` self-report (feeding those in
|
|
17
|
+
would just re-introduce the tier-1 problem this exists to route around) —
|
|
18
|
+
and rules ``continue`` / ``done`` / ``blocked``. The audit's context is
|
|
19
|
+
fixed-size (~1k tokens in, ~30 out) regardless of how long the session has
|
|
20
|
+
run, which is what keeps it cheap on turn 300 as well as turn 5.
|
|
21
|
+
|
|
22
|
+
**This module gates nothing.** It has no access to (and must never be
|
|
23
|
+
given) anything that could change board state — no ``Assignment.status``,
|
|
24
|
+
no ``review_state``, no ``test_state``. Its only output is a verdict and a
|
|
25
|
+
strike count; ``coord.notify.detect_liveness_stall`` is the only caller,
|
|
26
|
+
and it only ever posts a diagnostic comment + records a durable audit
|
|
27
|
+
trail. See that function's docstring for the "never gates" contract.
|
|
28
|
+
|
|
29
|
+
Runs the check via a one-shot ``claude -p`` subprocess (design decision
|
|
30
|
+
(a) in #2048 — no new API key, no new SDK dependency, matches "No API key
|
|
31
|
+
needed" / "No Anthropic SDK" in CLAUDE.md). The ~1-2s spawn cost is off the
|
|
32
|
+
critical path (the worker never waits on this) and bounded by the debounce
|
|
33
|
+
interval, not by turn count.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import json
|
|
39
|
+
import re
|
|
40
|
+
import subprocess
|
|
41
|
+
from dataclasses import dataclass
|
|
42
|
+
|
|
43
|
+
# ── Verdicts ─────────────────────────────────────────────────────────────
|
|
44
|
+
|
|
45
|
+
CONTINUE = "continue"
|
|
46
|
+
DONE = "done"
|
|
47
|
+
BLOCKED = "blocked"
|
|
48
|
+
KNOWN_VERDICTS = frozenset({CONTINUE, DONE, BLOCKED})
|
|
49
|
+
|
|
50
|
+
DEFAULT_MODEL = "claude-haiku-4-5"
|
|
51
|
+
DEFAULT_TIMEOUT_SECONDS = 30.0
|
|
52
|
+
DEFAULT_STRIKES = 3
|
|
53
|
+
DEFAULT_DEBOUNCE_SECONDS = 60.0
|
|
54
|
+
|
|
55
|
+
_VERDICT_RE = re.compile(r"\b(continue|done|blocked)\b", re.IGNORECASE)
|
|
56
|
+
|
|
57
|
+
# Lines the worker wrote about its OWN state — the tier-1 self-report this
|
|
58
|
+
# auditor exists to be independent of. Stripped from the turn text before
|
|
59
|
+
# it ever reaches the model, so a worker that falsely claims "STATUS:
|
|
60
|
+
# making great progress" can't talk the auditor out of a stall verdict, and
|
|
61
|
+
# a worker's own "STUCK:" line can't be reused as second-hand evidence
|
|
62
|
+
# either.
|
|
63
|
+
_SELF_REPORT_LINE_RE = re.compile(r"^\s*(STATUS|STUCK):.*$", re.MULTILINE | re.IGNORECASE)
|
|
64
|
+
|
|
65
|
+
# The audit's whole cost model rests on a FIXED context size — see the
|
|
66
|
+
# module docstring's "~1,000 in / ~30 out" breakdown, and the issue's own
|
|
67
|
+
# "objective / briefing excerpt — ~300 tokens" line (note: *excerpt*, not
|
|
68
|
+
# the whole document). Nothing upstream of this module enforces that: the
|
|
69
|
+
# caller (``coord.notify.detect_liveness_stall``) passes an assignment's
|
|
70
|
+
# raw ``briefing`` straight through, and a full GitHub-issue briefing
|
|
71
|
+
# regularly runs 5,000-15,000+ tokens (see ``AgentAssignment.to_status_dict``'s
|
|
72
|
+
# docstring in ``coord/agent.py`` — "a full briefing can be tens of KB").
|
|
73
|
+
# Truncate defensively here, in the one place every caller funnels
|
|
74
|
+
# through, so the excerpt the cost model assumes is what the model
|
|
75
|
+
# actually sees regardless of what a caller was handed.
|
|
76
|
+
_OBJECTIVE_EXCERPT_CHARS = 1_200 # ~300 tokens @ ~4 chars/token
|
|
77
|
+
_TURN_EXCERPT_CHARS = 2_000 # ~500 tokens @ ~4 chars/token
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _excerpt(text: str, limit: int) -> str:
|
|
81
|
+
"""Truncate *text* to at most *limit* characters, marking truncation
|
|
82
|
+
so a shortened excerpt is never silently indistinguishable from the
|
|
83
|
+
full text."""
|
|
84
|
+
if len(text) <= limit:
|
|
85
|
+
return text
|
|
86
|
+
return text[:limit].rstrip() + " …[truncated]"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
_AUDIT_SYSTEM_PROMPT = (
|
|
90
|
+
"You are a cheap, independent liveness auditor watching a coding agent "
|
|
91
|
+
"one turn at a time. You do NOT see the conversation history and you "
|
|
92
|
+
"do NOT see anything the worker says about its own status — only the "
|
|
93
|
+
"objective it was given and the raw output of its single MOST RECENT "
|
|
94
|
+
"turn. Judge whether that one turn shows the worker making real "
|
|
95
|
+
"progress toward the objective, having already finished it, or stuck.\n\n"
|
|
96
|
+
"Reply with EXACTLY ONE WORD and nothing else:\n"
|
|
97
|
+
" continue - the turn shows real, concrete progress (a file changed, "
|
|
98
|
+
"a command ran and moved things forward, new information was gathered)\n"
|
|
99
|
+
" done - the turn indicates the objective is complete\n"
|
|
100
|
+
" blocked - the turn shows no real progress: repeating an earlier "
|
|
101
|
+
"action, going in circles, confused, or stuck\n\n"
|
|
102
|
+
"One word only: continue, done, or blocked."
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def strip_self_report_lines(text: str) -> str:
|
|
107
|
+
"""Remove any ``STATUS:``/``STUCK:`` line from *text*.
|
|
108
|
+
|
|
109
|
+
Applied to a turn's raw text before it is ever sent to the auditor —
|
|
110
|
+
see the module docstring's "context isolation" note. Pure string
|
|
111
|
+
transform, safe to unit test without a subprocess.
|
|
112
|
+
"""
|
|
113
|
+
if not text:
|
|
114
|
+
return text
|
|
115
|
+
return _SELF_REPORT_LINE_RE.sub("", text).strip()
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def build_audit_user_message(objective: str, turn_text: str) -> str:
|
|
119
|
+
"""Render the ``(objective, latest turn)`` pair as the auditor's one
|
|
120
|
+
and only user message. No transcript, no history — this string IS the
|
|
121
|
+
entire context the model sees, by construction.
|
|
122
|
+
|
|
123
|
+
Both inputs are excerpted (see ``_OBJECTIVE_EXCERPT_CHARS``/
|
|
124
|
+
``_TURN_EXCERPT_CHARS`` above) so a caller handing this a raw,
|
|
125
|
+
multi-KB assignment briefing or an unusually large turn still gets
|
|
126
|
+
the fixed-size context the auditor's cost model is built on."""
|
|
127
|
+
objective = (objective or "").strip() or "(no objective provided)"
|
|
128
|
+
objective = _excerpt(objective, _OBJECTIVE_EXCERPT_CHARS)
|
|
129
|
+
turn_text = (turn_text or "").strip() or "(no output on this turn)"
|
|
130
|
+
turn_text = _excerpt(turn_text, _TURN_EXCERPT_CHARS)
|
|
131
|
+
return f"OBJECTIVE:\n{objective}\n\nLATEST TURN OUTPUT:\n{turn_text}"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def parse_verdict(text: str) -> str | None:
|
|
135
|
+
"""Pull the verdict word out of the model's reply, or ``None`` if it
|
|
136
|
+
said something unparseable. A ``None`` verdict is treated as "the audit
|
|
137
|
+
itself failed" by :func:`apply_verdict` — it never counts as evidence
|
|
138
|
+
either way, so a flaky/garbled response can't accidentally contribute
|
|
139
|
+
to (or reset) a stall streak."""
|
|
140
|
+
if not text:
|
|
141
|
+
return None
|
|
142
|
+
m = _VERDICT_RE.search(text)
|
|
143
|
+
return m.group(1).lower() if m else None
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
@dataclass
|
|
147
|
+
class AuditOutcome:
|
|
148
|
+
"""Result of one :func:`run_audit` call."""
|
|
149
|
+
|
|
150
|
+
verdict: str | None # one of KNOWN_VERDICTS, or None on parse/subprocess failure
|
|
151
|
+
raw_output: str
|
|
152
|
+
cost_usd: float | None = None
|
|
153
|
+
error: str | None = None
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _build_audit_command(*, model: str, claude_bin: str | None) -> list[str]:
|
|
157
|
+
from coord.providers.claude import ClaudeProvider # noqa: PLC0415
|
|
158
|
+
|
|
159
|
+
provider = ClaudeProvider(binary=claude_bin)
|
|
160
|
+
cmd = provider.oneshot_command(system_prompt=_AUDIT_SYSTEM_PROMPT, output_format="json")
|
|
161
|
+
cmd.extend(["--model", model])
|
|
162
|
+
return cmd
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _parse_json_envelope(stdout: str) -> tuple[str, float | None]:
|
|
166
|
+
"""Pull ``(result_text, total_cost_usd)`` out of a ``claude -p
|
|
167
|
+
--output-format json`` envelope. Falls back to the raw stdout (and no
|
|
168
|
+
cost) for a malformed/unexpected shape — best-effort, matches
|
|
169
|
+
``coord.brain.call_claude``'s own fallback."""
|
|
170
|
+
try:
|
|
171
|
+
outer = json.loads(stdout)
|
|
172
|
+
except (json.JSONDecodeError, ValueError):
|
|
173
|
+
return stdout.strip(), None
|
|
174
|
+
if not isinstance(outer, dict):
|
|
175
|
+
return stdout.strip(), None
|
|
176
|
+
text = outer.get("result")
|
|
177
|
+
if not isinstance(text, str):
|
|
178
|
+
text = stdout.strip()
|
|
179
|
+
cost = outer.get("total_cost_usd")
|
|
180
|
+
if not isinstance(cost, (int, float)):
|
|
181
|
+
cost = None
|
|
182
|
+
else:
|
|
183
|
+
cost = float(cost)
|
|
184
|
+
return text, cost
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def run_audit(
|
|
188
|
+
objective: str,
|
|
189
|
+
turn_text: str,
|
|
190
|
+
*,
|
|
191
|
+
model: str = DEFAULT_MODEL,
|
|
192
|
+
claude_bin: str | None = None,
|
|
193
|
+
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
|
194
|
+
) -> AuditOutcome:
|
|
195
|
+
"""Run one liveness audit: a one-shot ``claude -p`` subprocess call.
|
|
196
|
+
|
|
197
|
+
*objective* and *turn_text* are the ONLY context passed — no transcript,
|
|
198
|
+
no session id, no ``--resume``. Each call is a fresh, independent
|
|
199
|
+
process with no memory of any prior audit, by construction.
|
|
200
|
+
|
|
201
|
+
Never raises: subprocess failures, timeouts, and unparseable replies
|
|
202
|
+
all come back as ``AuditOutcome(verdict=None, ...)`` so a flaky audit
|
|
203
|
+
can never crash (or gate) the caller.
|
|
204
|
+
"""
|
|
205
|
+
cmd = _build_audit_command(model=model, claude_bin=claude_bin)
|
|
206
|
+
user_message = build_audit_user_message(objective, turn_text)
|
|
207
|
+
try:
|
|
208
|
+
result = subprocess.run( # noqa: S603 — fixed argv, no shell
|
|
209
|
+
cmd,
|
|
210
|
+
input=user_message,
|
|
211
|
+
capture_output=True,
|
|
212
|
+
text=True,
|
|
213
|
+
timeout=timeout,
|
|
214
|
+
)
|
|
215
|
+
except (OSError, subprocess.TimeoutExpired) as exc:
|
|
216
|
+
return AuditOutcome(verdict=None, raw_output="", error=str(exc))
|
|
217
|
+
|
|
218
|
+
if result.returncode != 0:
|
|
219
|
+
return AuditOutcome(
|
|
220
|
+
verdict=None,
|
|
221
|
+
raw_output=result.stdout or "",
|
|
222
|
+
error=f"exit {result.returncode}: {(result.stderr or '').strip()[:200]}",
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
text, cost = _parse_json_envelope(result.stdout)
|
|
226
|
+
return AuditOutcome(verdict=parse_verdict(text), raw_output=text, cost_usd=cost)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
# ── Debounce + strike tracking (pure) ───────────────────────────────────────
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
@dataclass
|
|
233
|
+
class AuditState:
|
|
234
|
+
"""Per-assignment liveness-audit tracking state.
|
|
235
|
+
|
|
236
|
+
Persisted via ``coord.state.load_liveness_audit_state`` /
|
|
237
|
+
``save_liveness_audit_state`` (backed by the ``liveness_audits`` table)
|
|
238
|
+
so the strike streak survives across separate ``coord notify``
|
|
239
|
+
invocations — this is polled state, not held in a long-lived process.
|
|
240
|
+
"""
|
|
241
|
+
|
|
242
|
+
consecutive_blocked: int = 0
|
|
243
|
+
last_audit_at: float | None = None
|
|
244
|
+
last_verdict: str | None = None
|
|
245
|
+
# True once this assignment's streak has already reached the strike
|
|
246
|
+
# threshold and the one-shot event has been raised for it. Once set,
|
|
247
|
+
# detect_liveness_stall stops auditing this assignment entirely — a
|
|
248
|
+
# stall that's already been reported doesn't need re-reporting every
|
|
249
|
+
# poll while it stays blocked.
|
|
250
|
+
raised: bool = False
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def should_audit(
|
|
254
|
+
*, last_audit_at: float | None, now: float, debounce_seconds: float
|
|
255
|
+
) -> bool:
|
|
256
|
+
"""Debounce gate: at most one audit per *debounce_seconds*.
|
|
257
|
+
|
|
258
|
+
A stall is a multi-minute phenomenon — auditing every turn buys no
|
|
259
|
+
extra signal and pays for a process spawn each time. ``last_audit_at
|
|
260
|
+
is None`` (never audited) always returns True.
|
|
261
|
+
"""
|
|
262
|
+
if last_audit_at is None:
|
|
263
|
+
return True
|
|
264
|
+
return (now - last_audit_at) >= debounce_seconds
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def apply_verdict(
|
|
268
|
+
state: AuditState, verdict: str | None, *, now: float, strikes: int
|
|
269
|
+
) -> tuple[AuditState, bool]:
|
|
270
|
+
"""Fold one audit verdict into *state*.
|
|
271
|
+
|
|
272
|
+
Returns ``(new_state, just_raised)``. ``just_raised`` is True exactly
|
|
273
|
+
on the call whose BLOCKED verdict brings ``consecutive_blocked`` to
|
|
274
|
+
*strikes* for the FIRST time — edge-triggered, so a streak that stays
|
|
275
|
+
at or above the threshold doesn't re-raise on every subsequent poll.
|
|
276
|
+
A ``None`` verdict (audit failed / unparseable) leaves the streak
|
|
277
|
+
untouched: an audit we can't read must never count as evidence either
|
|
278
|
+
way, in either direction.
|
|
279
|
+
"""
|
|
280
|
+
new_count = state.consecutive_blocked
|
|
281
|
+
if verdict == BLOCKED:
|
|
282
|
+
new_count += 1
|
|
283
|
+
elif verdict in (CONTINUE, DONE):
|
|
284
|
+
new_count = 0
|
|
285
|
+
|
|
286
|
+
just_raised = (not state.raised) and new_count >= strikes
|
|
287
|
+
new_state = AuditState(
|
|
288
|
+
consecutive_blocked=new_count,
|
|
289
|
+
last_audit_at=now,
|
|
290
|
+
last_verdict=verdict if verdict is not None else state.last_verdict,
|
|
291
|
+
raised=state.raised or just_raised,
|
|
292
|
+
)
|
|
293
|
+
return new_state, just_raised
|