code-coordinator 0.5.46__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- code_coordinator-0.5.46.dist-info/METADATA +625 -0
- code_coordinator-0.5.46.dist-info/RECORD +295 -0
- code_coordinator-0.5.46.dist-info/WHEEL +5 -0
- code_coordinator-0.5.46.dist-info/entry_points.txt +2 -0
- code_coordinator-0.5.46.dist-info/licenses/LICENSE +110 -0
- code_coordinator-0.5.46.dist-info/top_level.txt +1 -0
- coord/__init__.py +176 -0
- coord/_board_mapping.py +229 -0
- coord/acceptance.py +468 -0
- coord/acceptance_drivers.py +632 -0
- coord/agent.py +7517 -0
- coord/agent_app.py +1555 -0
- coord/agent_update.py +417 -0
- coord/agents/opencode/.gitignore +13 -0
- coord/agents/opencode/agents/work.md +129 -0
- coord/agents/opencode/routing.jsonc +49 -0
- coord/audit.py +301 -0
- coord/auto_loop.py +1440 -0
- coord/board_bool_guard.py +72 -0
- coord/board_service.py +141 -0
- coord/board_wire.py +309 -0
- coord/brain.py +581 -0
- coord/branch_model.py +214 -0
- coord/cargo_cache.py +258 -0
- coord/ci_github.py +386 -0
- coord/ci_store.py +560 -0
- coord/claim.py +353 -0
- coord/cli.py +454 -0
- coord/client.py +610 -0
- coord/commands/__init__.py +1 -0
- coord/commands/_common.py +329 -0
- coord/commands/acceptance.py +916 -0
- coord/commands/agent_ops.py +1339 -0
- coord/commands/audit.py +131 -0
- coord/commands/chat.py +320 -0
- coord/commands/dispatch.py +1780 -0
- coord/commands/dispatch_workers.py +4894 -0
- coord/commands/drive.py +616 -0
- coord/commands/drive_queue.py +1203 -0
- coord/commands/gate_a.py +217 -0
- coord/commands/gates.py +89 -0
- coord/commands/issues.py +681 -0
- coord/commands/lifecycle.py +513 -0
- coord/commands/merge.py +1900 -0
- coord/commands/milestone.py +2081 -0
- coord/commands/plan_followup.py +1243 -0
- coord/commands/plans.py +156 -0
- coord/commands/release.py +2232 -0
- coord/commands/report.py +341 -0
- coord/commands/review.py +1523 -0
- coord/commands/scorecard.py +252 -0
- coord/commands/sessions.py +1930 -0
- coord/commands/setup.py +576 -0
- coord/commands/status.py +2089 -0
- coord/commands/terminal.py +385 -0
- coord/commands/test_gate.py +775 -0
- coord/commands/tui.py +288 -0
- coord/comments.py +718 -0
- coord/config.py +3032 -0
- coord/conflict_fix.py +633 -0
- coord/dao.py +483 -0
- coord/dashboard/__init__.py +0 -0
- coord/dashboard/fixture.py +376 -0
- coord/dashboard/index.html +658 -0
- coord/dashboard/server.py +1894 -0
- coord/dashboard/terminal.py +382 -0
- coord/dashboard/webapp/.gitignore +9 -0
- coord/dashboard/webapp/components.json +17 -0
- coord/dashboard/webapp/dist/assets/Gallery-da3qNiIw.js +71 -0
- coord/dashboard/webapp/dist/assets/Terminal-9CEnUXvW.css +32 -0
- coord/dashboard/webapp/dist/assets/Terminal-skVFCxPU.js +63 -0
- coord/dashboard/webapp/dist/assets/index-DltfZR5f.js +184 -0
- coord/dashboard/webapp/dist/assets/index-Dq4kwTdw.css +1 -0
- coord/dashboard/webapp/dist/assets/workbox-window.prod.es5-BqEJf4Xk.js +2 -0
- coord/dashboard/webapp/dist/icons/icon-192.png +0 -0
- coord/dashboard/webapp/dist/icons/icon-512.png +0 -0
- coord/dashboard/webapp/dist/icons/icon.svg +5 -0
- coord/dashboard/webapp/dist/index.html +38 -0
- coord/dashboard/webapp/dist/manifest.webmanifest +1 -0
- coord/dashboard/webapp/dist/sw.js +1 -0
- coord/dashboard/webapp/dist/workbox-e4022e15.js +1 -0
- coord/dashboard/webapp/e2e/available-gates-terminal.spec.ts +75 -0
- coord/dashboard/webapp/e2e/deep-link.spec.ts +172 -0
- coord/dashboard/webapp/e2e/fixtureServer.ts +155 -0
- coord/dashboard/webapp/e2e/live-update-fixture.spec.ts +113 -0
- coord/dashboard/webapp/e2e/realtime.spec.ts +238 -0
- coord/dashboard/webapp/e2e/shell.spec.ts +309 -0
- coord/dashboard/webapp/e2e/smoke.spec.ts +191 -0
- coord/dashboard/webapp/e2e/terminal.spec.ts +420 -0
- coord/dashboard/webapp/e2e/theme.spec.ts +138 -0
- coord/dashboard/webapp/eslint.config.js +20 -0
- coord/dashboard/webapp/index.html +37 -0
- coord/dashboard/webapp/node_modules/flatted/python/flatted.py +144 -0
- coord/dashboard/webapp/package-lock.json +10584 -0
- coord/dashboard/webapp/package.json +63 -0
- coord/dashboard/webapp/playwright.acceptance.config.ts +166 -0
- coord/dashboard/webapp/playwright.config.ts +93 -0
- coord/dashboard/webapp/postcss.config.js +6 -0
- coord/dashboard/webapp/public/icons/icon-192.png +0 -0
- coord/dashboard/webapp/public/icons/icon-512.png +0 -0
- coord/dashboard/webapp/public/icons/icon.svg +5 -0
- coord/dashboard/webapp/src/App.tsx +140 -0
- coord/dashboard/webapp/src/api/client.ts +199 -0
- coord/dashboard/webapp/src/api/generated.ts +176 -0
- coord/dashboard/webapp/src/components/ConnectionBadge.tsx +52 -0
- coord/dashboard/webapp/src/components/Detail.tsx +800 -0
- coord/dashboard/webapp/src/components/Gallery.tsx +341 -0
- coord/dashboard/webapp/src/components/Home.tsx +435 -0
- coord/dashboard/webapp/src/components/MobileKeyBar.tsx +280 -0
- coord/dashboard/webapp/src/components/PanelHeader.tsx +59 -0
- coord/dashboard/webapp/src/components/PipelineCard.tsx +168 -0
- coord/dashboard/webapp/src/components/SessionCard.tsx +99 -0
- coord/dashboard/webapp/src/components/SessionDetail.tsx +140 -0
- coord/dashboard/webapp/src/components/SessionsList.tsx +81 -0
- coord/dashboard/webapp/src/components/Terminal.tsx +376 -0
- coord/dashboard/webapp/src/components/__tests__/ConnectionBadge.test.tsx +81 -0
- coord/dashboard/webapp/src/components/__tests__/Detail.test.tsx +680 -0
- coord/dashboard/webapp/src/components/__tests__/Gallery.test.tsx +83 -0
- coord/dashboard/webapp/src/components/__tests__/Home.test.tsx +271 -0
- coord/dashboard/webapp/src/components/__tests__/MobileKeyBar.test.tsx +197 -0
- coord/dashboard/webapp/src/components/__tests__/PipelineCard.test.tsx +143 -0
- coord/dashboard/webapp/src/components/__tests__/SessionCard.test.tsx +106 -0
- coord/dashboard/webapp/src/components/__tests__/Terminal.test.tsx +504 -0
- coord/dashboard/webapp/src/components/ui/badge.tsx +41 -0
- coord/dashboard/webapp/src/components/ui/button.tsx +54 -0
- coord/dashboard/webapp/src/components/ui/card.tsx +55 -0
- coord/dashboard/webapp/src/components/ui/dialog.tsx +99 -0
- coord/dashboard/webapp/src/components/ui/dropdown-menu.tsx +189 -0
- coord/dashboard/webapp/src/components/ui/empty-state.tsx +35 -0
- coord/dashboard/webapp/src/components/ui/sheet.tsx +123 -0
- coord/dashboard/webapp/src/components/ui/skeleton.tsx +9 -0
- coord/dashboard/webapp/src/components/ui/tabs.tsx +55 -0
- coord/dashboard/webapp/src/components/ui/theme-provider.tsx +78 -0
- coord/dashboard/webapp/src/components/ui/theme-toggle.tsx +20 -0
- coord/dashboard/webapp/src/components/ui/toast.tsx +123 -0
- coord/dashboard/webapp/src/components/ui/toaster.tsx +30 -0
- coord/dashboard/webapp/src/components/ui/tooltip.tsx +26 -0
- coord/dashboard/webapp/src/components/ui/use-toast.ts +134 -0
- coord/dashboard/webapp/src/index.css +210 -0
- coord/dashboard/webapp/src/lib/pipeline.ts +29 -0
- coord/dashboard/webapp/src/lib/utils.ts +6 -0
- coord/dashboard/webapp/src/main.tsx +46 -0
- coord/dashboard/webapp/src/realtime/RealtimeProvider.tsx +112 -0
- coord/dashboard/webapp/src/realtime/__tests__/RealtimeProvider.test.tsx +189 -0
- coord/dashboard/webapp/src/realtime/__tests__/connection.test.ts +255 -0
- coord/dashboard/webapp/src/realtime/connection.ts +227 -0
- coord/dashboard/webapp/src/realtime/events.ts +100 -0
- coord/dashboard/webapp/src/routes/__tests__/paths.test.ts +92 -0
- coord/dashboard/webapp/src/routes/paths.ts +92 -0
- coord/dashboard/webapp/src/shell/ActivityRail.tsx +335 -0
- coord/dashboard/webapp/src/shell/AppShell.tsx +276 -0
- coord/dashboard/webapp/src/shell/ComingSoon.tsx +33 -0
- coord/dashboard/webapp/src/shell/EmptyDetail.tsx +26 -0
- coord/dashboard/webapp/src/shell/RouteNotFound.tsx +33 -0
- coord/dashboard/webapp/src/shell/ShellLayout.tsx +147 -0
- coord/dashboard/webapp/src/shell/StatusBar.tsx +46 -0
- coord/dashboard/webapp/src/shell/__tests__/ShellLayout.test.tsx +520 -0
- coord/dashboard/webapp/src/shell/__tests__/shellState.test.ts +95 -0
- coord/dashboard/webapp/src/shell/__tests__/stubViewport.ts +40 -0
- coord/dashboard/webapp/src/shell/breakpoints.ts +87 -0
- coord/dashboard/webapp/src/shell/railItems.ts +105 -0
- coord/dashboard/webapp/src/shell/shellState.ts +174 -0
- coord/dashboard/webapp/src/shell/useRegionFocus.ts +95 -0
- coord/dashboard/webapp/src/test-setup.ts +41 -0
- coord/dashboard/webapp/src/vite-env.d.ts +2 -0
- coord/dashboard/webapp/tailwind.config.js +140 -0
- coord/dashboard/webapp/tsconfig.json +25 -0
- coord/dashboard/webapp/tsconfig.node.json +11 -0
- coord/dashboard/webapp/vite.config.ts +71 -0
- coord/db.py +1076 -0
- coord/dead_end.py +332 -0
- coord/deploy/README.md +33 -0
- coord/deploy/coord-agent.service +89 -0
- coord/deploy/coord-db-backup.service +60 -0
- coord/deploy/coord-db-backup.sh +74 -0
- coord/deploy/coord-db-backup.timer +18 -0
- coord/deploy/coord-drive-queue.service +117 -0
- coord/deploy/coord-drive-queue.timer +39 -0
- coord/deploy/coord-notify.service +48 -0
- coord/deploy/coord-notify.timer +24 -0
- coord/deploy/coord-release-propagate.service +83 -0
- coord/deploy/coord-release-propagate.timer +38 -0
- coord/deploy/coord-release-window.service +119 -0
- coord/deploy/coord-release-window.timer +36 -0
- coord/deploy/coord-serve.service +82 -0
- coord/deploy/coord-web-dist-build.service +43 -0
- coord/deploy/coord-web-dist-build.timer +36 -0
- coord/deploy/coord-web.service +125 -0
- coord/deploy_manifest.py +80 -0
- coord/deploy_units.py +384 -0
- coord/deps.py +115 -0
- coord/diagnose.py +1623 -0
- coord/dispatch.py +1009 -0
- coord/dist_name.py +123 -0
- coord/drive.py +3101 -0
- coord/drive_queue.py +2298 -0
- coord/drive_state.py +870 -0
- coord/events.py +381 -0
- coord/failure_class.py +914 -0
- coord/filelock.py +168 -0
- coord/fleet_config_health.py +300 -0
- coord/freshness.py +206 -0
- coord/gate_a.py +469 -0
- coord/gate_b.py +411 -0
- coord/gate_snapshot.py +385 -0
- coord/gates.py +582 -0
- coord/github_ops.py +1954 -0
- coord/goal.py +125 -0
- coord/graph_health.py +348 -0
- coord/health/__init__.py +69 -0
- coord/health/aggregate.py +129 -0
- coord/health/checks/__init__.py +13 -0
- coord/health/checks/agent_install.py +280 -0
- coord/health/checks/cargo_targets.py +171 -0
- coord/health/checks/claude_binary.py +65 -0
- coord/health/checks/deploy_lane_facts.py +458 -0
- coord/health/checks/disk.py +99 -0
- coord/health/checks/fleet_board.py +89 -0
- coord/health/checks/fleet_deploy_lanes.py +469 -0
- coord/health/checks/fleet_phantom.py +69 -0
- coord/health/checks/fleet_unit_drift.py +151 -0
- coord/health/checks/graph.py +192 -0
- coord/health/checks/plan_usage.py +88 -0
- coord/health/checks/repo_state.py +161 -0
- coord/health/checks/spawned_coord.py +465 -0
- coord/health/checks/timer_active.py +254 -0
- coord/health/checks/toolchain.py +547 -0
- coord/health/checks/unit_drift.py +648 -0
- coord/health/checks/unit_enablement.py +171 -0
- coord/health/checks/worktrees.py +96 -0
- coord/health/cli.py +121 -0
- coord/health/context.py +106 -0
- coord/health/fleet_snapshot.py +477 -0
- coord/health/models.py +250 -0
- coord/health/pypi.py +231 -0
- coord/health/registry.py +240 -0
- coord/health/render.py +82 -0
- coord/health/units.py +60 -0
- coord/hooks.py +106 -0
- coord/housekeeping.py +204 -0
- coord/interactive.py +4286 -0
- coord/issue_store.py +1496 -0
- coord/liveness_auditor.py +293 -0
- coord/machine_pause.py +755 -0
- coord/merge_queue.py +4681 -0
- coord/milestone_chat.py +600 -0
- coord/milestone_dispatch.py +943 -0
- coord/milestone_gate.py +709 -0
- coord/milestone_order.py +840 -0
- coord/mock_author.py +334 -0
- coord/models.py +891 -0
- coord/network.py +269 -0
- coord/new_issue_chat.py +229 -0
- coord/notify.py +3226 -0
- coord/openapi.py +404 -0
- coord/overlap_fence.py +133 -0
- coord/parentage.py +200 -0
- coord/parentage_github.py +58 -0
- coord/pipeline.py +481 -0
- coord/plan_parser.py +266 -0
- coord/plans.py +543 -0
- coord/platform_paths.py +43 -0
- coord/pr_body_lint.py +67 -0
- coord/prereqs.py +533 -0
- coord/progress.py +425 -0
- coord/providers/__init__.py +683 -0
- coord/providers/base.py +218 -0
- coord/providers/claude.py +284 -0
- coord/providers/claude_pty.py +610 -0
- coord/providers/opencode.py +896 -0
- coord/reconcile.py +2233 -0
- coord/refine_chat.py +485 -0
- coord/release_cordon.py +525 -0
- coord/release_propagate.py +1176 -0
- coord/release_verify.py +777 -0
- coord/release_window.py +322 -0
- coord/reports.py +1643 -0
- coord/revalidate.py +1101 -0
- coord/review.py +3317 -0
- coord/scorecard.py +484 -0
- coord/serve_app.py +7192 -0
- coord/skills/update-issue/SKILL.md +93 -0
- coord/smoke.py +1030 -0
- coord/split_work.py +210 -0
- coord/stage_projection.py +650 -0
- coord/state.py +5720 -0
- coord/test_author.py +1064 -0
- coord/test_chat.py +352 -0
- coord/test_orchestrator.py +494 -0
- coord/test_report.py +178 -0
- coord/tui_release.py +271 -0
- coord/usage.py +753 -0
- coord/usage_limits.py +358 -0
- coord/usage_rollup.py +709 -0
- coord/worker_events.py +954 -0
coord/scorecard.py
ADDED
|
@@ -0,0 +1,484 @@
|
|
|
1
|
+
"""Dogfood scorecard (#1559): turn a milestone into evidence, not vibes.
|
|
2
|
+
|
|
3
|
+
``docs/WEB_CONTROL_CENTER.md`` names five metrics the M-W0/M-W1 dogfood
|
|
4
|
+
program is supposed to answer with numbers rather than "it felt like it
|
|
5
|
+
worked": first-pass acceptance rate, human interventions (count + kind),
|
|
6
|
+
cost + wall-clock per issue, escaped defects by stage, and process bugs
|
|
7
|
+
surfaced. This module is the pure aggregator that turns three already-
|
|
8
|
+
fetched inputs — a milestone's GitHub issues (with labels), the repo's board
|
|
9
|
+
assignment rows, and (optionally) the audit trail — into per-issue and
|
|
10
|
+
per-milestone numbers.
|
|
11
|
+
|
|
12
|
+
Split the same way :mod:`coord.usage_rollup` is split from :mod:`coord.usage`:
|
|
13
|
+
this module takes plain data in (``dict``s matching the daemon ``/board``
|
|
14
|
+
wire shape, the GitHub CLI's issue-list JSON shape, and the audit log's
|
|
15
|
+
entry shape) and returns plain data out. No I/O, no ``gh``, no daemon calls.
|
|
16
|
+
The *fetch* side lives in :mod:`coord.commands.scorecard`, kept separate so
|
|
17
|
+
this module is unit-testable against fixtures with no GitHub/daemon
|
|
18
|
+
reachable — the point of #1559's "validate against history" acceptance
|
|
19
|
+
criterion.
|
|
20
|
+
|
|
21
|
+
## The two label conventions
|
|
22
|
+
|
|
23
|
+
Three of the five metrics (first-pass acceptance, interventions, cost/time)
|
|
24
|
+
read straight off data coord already records — see the per-function
|
|
25
|
+
docstrings below for exactly which board columns. The other two have no
|
|
26
|
+
automatic signal and need a labelling convention cheap enough to survive
|
|
27
|
+
contact with a real program; both apply via the existing ``coord issue
|
|
28
|
+
label <repo> <issue> --add <label>`` command, in one call, at the moment of
|
|
29
|
+
discovery — no new CLI surface needed.
|
|
30
|
+
|
|
31
|
+
**Escaped defects** — apply ``escaped:<stage>`` where ``<stage>`` is one of
|
|
32
|
+
:data:`ESCAPED_DEFECT_STAGES` (``review``, ``gate-b``, ``post-merge``,
|
|
33
|
+
``live``), naming the stage that *should* have caught the defect. An issue
|
|
34
|
+
with none of these labels reports as "unlabeled", not "zero" — see
|
|
35
|
+
:func:`build_milestone_scorecard`'s ``escaped_defects.issues_unlabeled``.
|
|
36
|
+
|
|
37
|
+
**Process bugs** — apply ``process-bug`` to a bug filed against coord itself
|
|
38
|
+
under the *same* milestone as the program that surfaced it (so this module
|
|
39
|
+
never needs a second GitHub call to find them). Once the fix lands, apply
|
|
40
|
+
either ``regression-test:landed`` or ``regression-test:missing``. A
|
|
41
|
+
``process-bug`` issue carrying neither sub-label reports ``regression_test
|
|
42
|
+
= "unknown"`` — still triaging, not "no test".
|
|
43
|
+
|
|
44
|
+
## First-pass acceptance and intervention kind — the board convention
|
|
45
|
+
|
|
46
|
+
An issue's *root* work assignment is a ``type in ("work", "test-author")``
|
|
47
|
+
row with no ``review_of_assignment_id`` and ``review_iteration`` 0 (see
|
|
48
|
+
:func:`is_root_work`). Every follow-up dispatched against that root —
|
|
49
|
+
whether ``coord assign --interactive --fix-of``/``--rework-of`` (human) or
|
|
50
|
+
``auto_loop``'s headless review→fix bounce (automated) — keeps
|
|
51
|
+
``type="work"`` and sets ``review_of_assignment_id`` to the assignment id of
|
|
52
|
+
whatever it's a follow-up *to*, which is the ROOT's id only for the first
|
|
53
|
+
fix/rework iteration dispatched directly off the root's review. A review
|
|
54
|
+
dispatched against a completed FIX (itself ``type="work"``, eligible for
|
|
55
|
+
review the same as the root — see ``review.py:dispatch_review``'s
|
|
56
|
+
``review_of_assignment_id=completed.assignment_id``) sets its own
|
|
57
|
+
``review_of_assignment_id`` to that fix's id, not the root's; a second
|
|
58
|
+
fix/rework chained off *that* review therefore points at the immediately
|
|
59
|
+
preceding iteration, not the root (see
|
|
60
|
+
``dispatch_workers.py:_dispatch_fix_of``/``_dispatch_rework_of`` — both
|
|
61
|
+
resolve ``work`` from the review's own ``review_of_assignment_id``, which
|
|
62
|
+
may itself be a fix — and ``auto_loop.py``'s equivalent bounce-fix dispatch,
|
|
63
|
+
same "whatever the review points at" resolution). This doesn't confuse
|
|
64
|
+
:func:`is_root_work` or :func:`classify_intervention` below — both only
|
|
65
|
+
check whether ``review_of_assignment_id`` is *set*, never its target, so a
|
|
66
|
+
row anywhere in a multi-iteration chain is correctly never mistaken for a
|
|
67
|
+
root and always correctly counted as a follow-up. Every follow-up in the
|
|
68
|
+
chain, human or automated, does prefix ``issue_title`` with ``"[fix-N] "``
|
|
69
|
+
(``auto_loop.py`` uses that exact tag too, so title alone can't tell them
|
|
70
|
+
apart). What DOES distinguish a human follow-up is
|
|
71
|
+
``provider_name == "claude-pty"`` — the same signal
|
|
72
|
+
``coord.reconcile.is_interactive_merge_session`` and
|
|
73
|
+
``coord.config.Config.attention_threshold_for`` already key off of for
|
|
74
|
+
"is this an interactive session", set only by the interactive dispatch
|
|
75
|
+
front doors, never by ``auto_loop``'s HTTP ``/assign`` POST to the headless
|
|
76
|
+
agent host. ``"[rework-N] "`` (only ever set by the human ``--rework-of``
|
|
77
|
+
front door) further splits "fix" from "rescue" — see
|
|
78
|
+
:func:`classify_intervention`.
|
|
79
|
+
|
|
80
|
+
"First-pass acceptance" (:func:`build_milestone_scorecard`'s
|
|
81
|
+
``first_pass``) is therefore: **exactly one root work assignment, reaching
|
|
82
|
+
``status="merged"``, with zero human fix/rework descendants.** An issue
|
|
83
|
+
with no board data at all (never dispatched through coord, or the data
|
|
84
|
+
didn't survive) reports ``"unknown"`` — never lumped in with ``"no"``.
|
|
85
|
+
|
|
86
|
+
A second root (e.g. ``coord retry`` dispatching a fresh, unlinked
|
|
87
|
+
``type="work"`` row with no ``review_of_assignment_id`` after a failure,
|
|
88
|
+
possibly on a different machine) also fails ``first_pass``, but doesn't fit
|
|
89
|
+
any of ``interventions.by_kind`` — it's not a fix/rescue/nudge/abandon, it's
|
|
90
|
+
a second unrelated attempt. That case is its own signal,
|
|
91
|
+
:attr:`IssueScorecard.multiple_roots` / the totals'
|
|
92
|
+
``interventions.issues_with_multiple_roots``, so ``first_pass="no"`` with
|
|
93
|
+
zero interventions counted doesn't read as a mystery.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
from __future__ import annotations
|
|
97
|
+
|
|
98
|
+
import re
|
|
99
|
+
from dataclasses import asdict, dataclass, field
|
|
100
|
+
from typing import Any
|
|
101
|
+
|
|
102
|
+
__all__ = [
|
|
103
|
+
"ESCAPED_DEFECT_STAGES",
|
|
104
|
+
"INTERVENTION_KINDS",
|
|
105
|
+
"PROCESS_BUG_LABEL",
|
|
106
|
+
"REGRESSION_TEST_LANDED_LABEL",
|
|
107
|
+
"REGRESSION_TEST_MISSING_LABEL",
|
|
108
|
+
"IssueScorecard",
|
|
109
|
+
"MilestoneScorecard",
|
|
110
|
+
"classify_intervention",
|
|
111
|
+
"is_root_work",
|
|
112
|
+
"build_milestone_scorecard",
|
|
113
|
+
"scorecard_to_dict",
|
|
114
|
+
]
|
|
115
|
+
|
|
116
|
+
# ── Label conventions ───────────────────────────────────────────────────────
|
|
117
|
+
|
|
118
|
+
ESCAPED_DEFECT_STAGES = ("review", "gate-b", "post-merge", "live")
|
|
119
|
+
_ESCAPED_LABEL_RE = re.compile(
|
|
120
|
+
r"^escaped:(" + "|".join(re.escape(s) for s in ESCAPED_DEFECT_STAGES) + r")$"
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
PROCESS_BUG_LABEL = "process-bug"
|
|
124
|
+
REGRESSION_TEST_LANDED_LABEL = "regression-test:landed"
|
|
125
|
+
REGRESSION_TEST_MISSING_LABEL = "regression-test:missing"
|
|
126
|
+
|
|
127
|
+
# ── Intervention kinds ───────────────────────────────────────────────────────
|
|
128
|
+
|
|
129
|
+
INTERVENTION_KINDS = ("fix", "rescue", "nudge", "abandon")
|
|
130
|
+
|
|
131
|
+
# Assignment types a human --fix-of/--rework-of dispatch always keeps (see
|
|
132
|
+
# dispatch_workers.py:_dispatch_fix_of/_dispatch_rework_of, and auto_loop.py's
|
|
133
|
+
# fix_type="test-author" carve-out for a test-author slice fix).
|
|
134
|
+
_ROOT_WORK_TYPES = frozenset({"work", "test-author"})
|
|
135
|
+
|
|
136
|
+
# Human-attended, per-issue conversational sessions (coord.config.
|
|
137
|
+
# INTERACTIVE_SESSION_TYPES minus the milestone/pre-creation-scoped members
|
|
138
|
+
# that don't attach to one issue's execution: "audit", "milestone-chat",
|
|
139
|
+
# "new-issue-chat").
|
|
140
|
+
_NUDGE_TYPES = frozenset({"chat", "troubleshoot", "test-chat", "refinement"})
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def is_root_work(row: dict) -> bool:
|
|
144
|
+
"""Whether *row* is an issue's original (non-fix, non-rework) dispatch.
|
|
145
|
+
|
|
146
|
+
``type`` in :data:`_ROOT_WORK_TYPES`, no ``review_of_assignment_id``, and
|
|
147
|
+
``review_iteration`` 0 (missing/``None`` counts as 0 — predates #1176).
|
|
148
|
+
"""
|
|
149
|
+
if str(row.get("type") or "work") not in _ROOT_WORK_TYPES:
|
|
150
|
+
return False
|
|
151
|
+
if row.get("review_of_assignment_id"):
|
|
152
|
+
return False
|
|
153
|
+
try:
|
|
154
|
+
iteration = int(row.get("review_iteration") or 0)
|
|
155
|
+
except (TypeError, ValueError):
|
|
156
|
+
iteration = 0
|
|
157
|
+
return iteration == 0
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def classify_intervention(row: dict) -> str | None:
|
|
161
|
+
"""Classify one board assignment row as a human-intervention kind.
|
|
162
|
+
|
|
163
|
+
Returns ``None`` for anything that isn't one — including ``auto_loop``'s
|
|
164
|
+
headless bounce-fix dispatch, which shares ``review_of_assignment_id``/
|
|
165
|
+
``review_iteration``/``"[fix-N] "``-title shape with a human fix but
|
|
166
|
+
never sets ``provider_name="claude-pty"`` (only the interactive dispatch
|
|
167
|
+
front doors in ``dispatch_workers.py`` do — see module docstring), and a
|
|
168
|
+
mechanical ``type="conflict-fix"`` automated merge repair
|
|
169
|
+
(``conflict_fix.py`` — the coordinator's own automatic
|
|
170
|
+
mechanical-conflict resolution, not a human touching anything, and it
|
|
171
|
+
never sets ``provider_name="claude-pty"`` either).
|
|
172
|
+
|
|
173
|
+
``"abandon"`` is NOT classified here — it's a per-issue fact (closed,
|
|
174
|
+
board data exists, nothing ever merged), not a single assignment's
|
|
175
|
+
shape; see :func:`build_milestone_scorecard`.
|
|
176
|
+
"""
|
|
177
|
+
rtype = str(row.get("type") or "work")
|
|
178
|
+
provider = str(row.get("provider_name") or "")
|
|
179
|
+
title = str(row.get("issue_title") or "")
|
|
180
|
+
if (
|
|
181
|
+
rtype in _ROOT_WORK_TYPES
|
|
182
|
+
and row.get("review_of_assignment_id")
|
|
183
|
+
and provider == "claude-pty"
|
|
184
|
+
):
|
|
185
|
+
if title.startswith("[rework-"):
|
|
186
|
+
return "rescue"
|
|
187
|
+
# "[fix-N] " (the interactive --fix-of front door) and any other
|
|
188
|
+
# human follow-up shape without a recognized title tag both land
|
|
189
|
+
# here — "fix" is the more common of the two human flavours.
|
|
190
|
+
return "fix"
|
|
191
|
+
if rtype in _NUDGE_TYPES:
|
|
192
|
+
return "nudge"
|
|
193
|
+
return None
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _labels_of(issue: dict) -> set[str]:
|
|
197
|
+
out: set[str] = set()
|
|
198
|
+
for entry in issue.get("labels") or []:
|
|
199
|
+
name = entry.get("name") if isinstance(entry, dict) else entry
|
|
200
|
+
if name:
|
|
201
|
+
out.add(str(name))
|
|
202
|
+
return out
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _escaped_defect_stage(labels: set[str]) -> str | None:
|
|
206
|
+
for label in labels:
|
|
207
|
+
m = _ESCAPED_LABEL_RE.match(label)
|
|
208
|
+
if m:
|
|
209
|
+
return m.group(1)
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _process_bug_fields(labels: set[str]) -> tuple[bool, str | None]:
|
|
214
|
+
if PROCESS_BUG_LABEL not in labels:
|
|
215
|
+
return False, None
|
|
216
|
+
if REGRESSION_TEST_LANDED_LABEL in labels:
|
|
217
|
+
return True, "landed"
|
|
218
|
+
if REGRESSION_TEST_MISSING_LABEL in labels:
|
|
219
|
+
return True, "missing"
|
|
220
|
+
return True, "unknown"
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# ── Per-issue / per-milestone result shapes ─────────────────────────────────
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
@dataclass
|
|
227
|
+
class IssueScorecard:
|
|
228
|
+
"""The five metrics resolved for one issue."""
|
|
229
|
+
|
|
230
|
+
number: int
|
|
231
|
+
title: str
|
|
232
|
+
state: str # "OPEN" | "CLOSED" | "" (unknown)
|
|
233
|
+
|
|
234
|
+
# First-pass acceptance: "yes" | "no" | "unknown" (no board data at all).
|
|
235
|
+
first_pass: str
|
|
236
|
+
|
|
237
|
+
# Count per INTERVENTION_KINDS, always all four keys present (0, never
|
|
238
|
+
# omitted) once has_assignment_data is True.
|
|
239
|
+
interventions: dict[str, int]
|
|
240
|
+
has_assignment_data: bool
|
|
241
|
+
|
|
242
|
+
# True when more than one root work assignment exists for this issue
|
|
243
|
+
# (e.g. `coord retry` dispatching a fresh unlinked `type="work"` row
|
|
244
|
+
# after a failure, possibly on a different machine) — see
|
|
245
|
+
# `build_milestone_scorecard`'s docstring. This is why `first_pass` can
|
|
246
|
+
# be "no" with all four `interventions` counts at 0: none of
|
|
247
|
+
# fix/rescue/nudge/abandon fit "a second, unrelated root appeared."
|
|
248
|
+
multiple_roots: bool
|
|
249
|
+
|
|
250
|
+
# Cost + wall-clock, from coord.usage_rollup.aggregate(by="issue").
|
|
251
|
+
cost_captured: float
|
|
252
|
+
cost_est: float
|
|
253
|
+
cost_total: float
|
|
254
|
+
duration_secs: float
|
|
255
|
+
legs: int
|
|
256
|
+
has_cost_data: bool
|
|
257
|
+
|
|
258
|
+
# Escaped defects / process bugs — label-sourced, see module docstring.
|
|
259
|
+
escaped_defect_stage: str | None
|
|
260
|
+
process_bug: bool
|
|
261
|
+
regression_test: str | None # "landed" | "missing" | "unknown" | None
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
@dataclass
|
|
265
|
+
class MilestoneScorecard:
|
|
266
|
+
"""The full report for one milestone: per-issue rows + the aggregate."""
|
|
267
|
+
|
|
268
|
+
milestone: Any
|
|
269
|
+
milestone_title: str
|
|
270
|
+
repo_name: str
|
|
271
|
+
issues: list[IssueScorecard] = field(default_factory=list)
|
|
272
|
+
totals: dict = field(default_factory=dict)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _aggregate_totals(cards: list[IssueScorecard]) -> dict:
|
|
276
|
+
fp_yes = sum(1 for c in cards if c.first_pass == "yes")
|
|
277
|
+
fp_no = sum(1 for c in cards if c.first_pass == "no")
|
|
278
|
+
fp_unknown = sum(1 for c in cards if c.first_pass == "unknown")
|
|
279
|
+
fp_decided = fp_yes + fp_no
|
|
280
|
+
fp_rate = (fp_yes / fp_decided) if fp_decided else None
|
|
281
|
+
|
|
282
|
+
by_kind = {
|
|
283
|
+
k: sum(c.interventions.get(k, 0) for c in cards) for k in INTERVENTION_KINDS
|
|
284
|
+
}
|
|
285
|
+
issues_with_unknown_data = sum(1 for c in cards if not c.has_assignment_data)
|
|
286
|
+
issues_with_multiple_roots = sum(1 for c in cards if c.multiple_roots)
|
|
287
|
+
|
|
288
|
+
cost_total = sum(c.cost_total for c in cards)
|
|
289
|
+
cost_captured = sum(c.cost_captured for c in cards)
|
|
290
|
+
cost_est = sum(c.cost_est for c in cards)
|
|
291
|
+
duration_total = sum(c.duration_secs for c in cards)
|
|
292
|
+
issues_with_cost_data = sum(1 for c in cards if c.has_cost_data)
|
|
293
|
+
|
|
294
|
+
escaped_by_stage = {
|
|
295
|
+
s: sum(1 for c in cards if c.escaped_defect_stage == s)
|
|
296
|
+
for s in ESCAPED_DEFECT_STAGES
|
|
297
|
+
}
|
|
298
|
+
escaped_unlabeled = sum(1 for c in cards if c.escaped_defect_stage is None)
|
|
299
|
+
|
|
300
|
+
process_bugs = [c for c in cards if c.process_bug]
|
|
301
|
+
|
|
302
|
+
return {
|
|
303
|
+
"issue_count": len(cards),
|
|
304
|
+
"first_pass": {
|
|
305
|
+
"yes": fp_yes,
|
|
306
|
+
"no": fp_no,
|
|
307
|
+
"unknown": fp_unknown,
|
|
308
|
+
"rate": fp_rate,
|
|
309
|
+
},
|
|
310
|
+
"interventions": {
|
|
311
|
+
"by_kind": by_kind,
|
|
312
|
+
"total": sum(by_kind.values()),
|
|
313
|
+
"issues_with_unknown_data": issues_with_unknown_data,
|
|
314
|
+
# first_pass="no" with all four by_kind counts at 0 means one of
|
|
315
|
+
# these — a second, unrelated root dispatch, not a fix/rescue/
|
|
316
|
+
# nudge/abandon. See IssueScorecard.multiple_roots.
|
|
317
|
+
"issues_with_multiple_roots": issues_with_multiple_roots,
|
|
318
|
+
},
|
|
319
|
+
"cost": {
|
|
320
|
+
"total_usd": cost_total,
|
|
321
|
+
"captured_usd": cost_captured,
|
|
322
|
+
"estimated_usd": cost_est,
|
|
323
|
+
"duration_secs": duration_total,
|
|
324
|
+
"issues_with_data": issues_with_cost_data,
|
|
325
|
+
"issues_without_data": len(cards) - issues_with_cost_data,
|
|
326
|
+
},
|
|
327
|
+
"escaped_defects": {
|
|
328
|
+
"by_stage": escaped_by_stage,
|
|
329
|
+
"issues_unlabeled": escaped_unlabeled,
|
|
330
|
+
},
|
|
331
|
+
"process_bugs": {
|
|
332
|
+
"count": len(process_bugs),
|
|
333
|
+
"regression_test": {
|
|
334
|
+
"landed": sum(1 for c in process_bugs if c.regression_test == "landed"),
|
|
335
|
+
"missing": sum(1 for c in process_bugs if c.regression_test == "missing"),
|
|
336
|
+
"unknown": sum(1 for c in process_bugs if c.regression_test == "unknown"),
|
|
337
|
+
},
|
|
338
|
+
},
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _score_one_issue(
|
|
343
|
+
*, issue: dict, rows: list[dict], cost_group: dict | None, audited_merged: bool
|
|
344
|
+
) -> IssueScorecard:
|
|
345
|
+
labels = _labels_of(issue)
|
|
346
|
+
escaped_stage = _escaped_defect_stage(labels)
|
|
347
|
+
process_bug, regression_test = _process_bug_fields(labels)
|
|
348
|
+
|
|
349
|
+
roots = [r for r in rows if is_root_work(r)]
|
|
350
|
+
counts = {k: 0 for k in INTERVENTION_KINDS}
|
|
351
|
+
for row in rows:
|
|
352
|
+
kind = classify_intervention(row)
|
|
353
|
+
if kind:
|
|
354
|
+
counts[kind] += 1
|
|
355
|
+
|
|
356
|
+
has_data = bool(rows)
|
|
357
|
+
merged = audited_merged or any(str(r.get("status")) == "merged" for r in roots)
|
|
358
|
+
human_followups = counts["fix"] + counts["rescue"] + counts["nudge"]
|
|
359
|
+
multiple_roots = len(roots) > 1
|
|
360
|
+
|
|
361
|
+
if not has_data:
|
|
362
|
+
first_pass = "unknown"
|
|
363
|
+
elif len(roots) == 1 and merged and human_followups == 0:
|
|
364
|
+
first_pass = "yes"
|
|
365
|
+
else:
|
|
366
|
+
first_pass = "no"
|
|
367
|
+
|
|
368
|
+
# Abandon: closed, we have board history, but nothing ever merged — the
|
|
369
|
+
# automated pipeline didn't land this; someone finished (or dropped) it
|
|
370
|
+
# outside coord's tracked chain.
|
|
371
|
+
state = str(issue.get("state") or "").upper()
|
|
372
|
+
if has_data and state == "CLOSED" and not merged:
|
|
373
|
+
counts["abandon"] = 1
|
|
374
|
+
|
|
375
|
+
if cost_group is not None:
|
|
376
|
+
cost_captured = float(cost_group.get("cost_captured") or 0.0)
|
|
377
|
+
cost_est = float(cost_group.get("cost_est") or 0.0)
|
|
378
|
+
cost_total = float(cost_group.get("cost_total") or (cost_captured + cost_est))
|
|
379
|
+
duration_secs = float(cost_group.get("duration_secs") or 0.0)
|
|
380
|
+
legs = int(cost_group.get("legs") or 0)
|
|
381
|
+
has_cost_data = legs > 0
|
|
382
|
+
else:
|
|
383
|
+
cost_captured = cost_est = cost_total = duration_secs = 0.0
|
|
384
|
+
legs = 0
|
|
385
|
+
has_cost_data = False
|
|
386
|
+
|
|
387
|
+
return IssueScorecard(
|
|
388
|
+
number=int(issue["number"]),
|
|
389
|
+
title=str(issue.get("title") or ""),
|
|
390
|
+
state=state,
|
|
391
|
+
first_pass=first_pass,
|
|
392
|
+
interventions=counts,
|
|
393
|
+
has_assignment_data=has_data,
|
|
394
|
+
multiple_roots=multiple_roots,
|
|
395
|
+
cost_captured=cost_captured,
|
|
396
|
+
cost_est=cost_est,
|
|
397
|
+
cost_total=cost_total,
|
|
398
|
+
duration_secs=duration_secs,
|
|
399
|
+
legs=legs,
|
|
400
|
+
has_cost_data=has_cost_data,
|
|
401
|
+
escaped_defect_stage=escaped_stage,
|
|
402
|
+
process_bug=process_bug,
|
|
403
|
+
regression_test=regression_test,
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def build_milestone_scorecard(
|
|
408
|
+
*,
|
|
409
|
+
milestone: Any,
|
|
410
|
+
milestone_title: str,
|
|
411
|
+
repo_name: str,
|
|
412
|
+
issues: list[dict],
|
|
413
|
+
assignment_rows: list[dict],
|
|
414
|
+
audit_entries: list[dict] | None = None,
|
|
415
|
+
pricing: dict | None = None,
|
|
416
|
+
) -> MilestoneScorecard:
|
|
417
|
+
"""Build the full scorecard for one milestone.
|
|
418
|
+
|
|
419
|
+
Parameters mirror the three fetch seams :mod:`coord.commands.scorecard`
|
|
420
|
+
composes: *issues* is ``coord.github_ops.get_milestone_issues``'s shape
|
|
421
|
+
(``{"number", "title", "state", "labels"}``, open+closed);
|
|
422
|
+
*assignment_rows* is ``coord.usage.fetch_usage_rows()``'s shape (the
|
|
423
|
+
daemon ``/board`` wire format, ALL repos — this function filters to
|
|
424
|
+
*repo_name* itself so callers don't have to); *audit_entries* is
|
|
425
|
+
optional (``coord.state.list_audit_log`` entries) and used only as a
|
|
426
|
+
durability cross-check for "merged" — a board row that never got its
|
|
427
|
+
``status`` flipped (a reconcile-sweep gap) still counts as merged if the
|
|
428
|
+
audit trail recorded the merge event, so a real board-sync hiccup
|
|
429
|
+
doesn't misreport an accepted issue as abandoned. *pricing* is
|
|
430
|
+
``coord.usage.pricing_dict_from_config(cfg.pricing)``'s shape; ``None``
|
|
431
|
+
uses the built-in default rates (matches ``aggregate()``'s own default).
|
|
432
|
+
"""
|
|
433
|
+
from coord.usage_rollup import Window, aggregate, row_issue_number # noqa: PLC0415
|
|
434
|
+
|
|
435
|
+
repo_rows = [r for r in assignment_rows if str(r.get("repo_name") or "") == repo_name]
|
|
436
|
+
|
|
437
|
+
cost_result = aggregate(
|
|
438
|
+
repo_rows, by="issue", window=Window(), pricing=pricing or {}
|
|
439
|
+
)
|
|
440
|
+
cost_by_issue = {g["key"]: g for g in cost_result["groups"]}
|
|
441
|
+
|
|
442
|
+
audited_merged: set[int] = set()
|
|
443
|
+
for entry in audit_entries or []:
|
|
444
|
+
if entry.get("event_type") != "merged":
|
|
445
|
+
continue
|
|
446
|
+
if str(entry.get("repo") or "") != repo_name:
|
|
447
|
+
continue
|
|
448
|
+
issue_no = entry.get("issue")
|
|
449
|
+
if issue_no is not None:
|
|
450
|
+
audited_merged.add(int(issue_no))
|
|
451
|
+
|
|
452
|
+
by_issue: dict[int, list[dict]] = {}
|
|
453
|
+
for row in repo_rows:
|
|
454
|
+
by_issue.setdefault(row_issue_number(row), []).append(row)
|
|
455
|
+
|
|
456
|
+
cards = [
|
|
457
|
+
_score_one_issue(
|
|
458
|
+
issue=issue,
|
|
459
|
+
rows=by_issue.get(int(issue["number"]), []),
|
|
460
|
+
cost_group=cost_by_issue.get(int(issue["number"])),
|
|
461
|
+
audited_merged=int(issue["number"]) in audited_merged,
|
|
462
|
+
)
|
|
463
|
+
for issue in issues
|
|
464
|
+
]
|
|
465
|
+
|
|
466
|
+
return MilestoneScorecard(
|
|
467
|
+
milestone=milestone,
|
|
468
|
+
milestone_title=milestone_title,
|
|
469
|
+
repo_name=repo_name,
|
|
470
|
+
issues=cards,
|
|
471
|
+
totals=_aggregate_totals(cards),
|
|
472
|
+
)
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def scorecard_to_dict(card: MilestoneScorecard) -> dict:
|
|
476
|
+
"""Plain-dict rendering for JSON output — ``dataclasses.asdict`` with no
|
|
477
|
+
surprises since every field is already a JSON-safe primitive/dict/list."""
|
|
478
|
+
return {
|
|
479
|
+
"milestone": card.milestone,
|
|
480
|
+
"milestone_title": card.milestone_title,
|
|
481
|
+
"repo_name": card.repo_name,
|
|
482
|
+
"issues": [asdict(c) for c in card.issues],
|
|
483
|
+
"totals": card.totals,
|
|
484
|
+
}
|