pocketshell 0.4.37__tar.gz → 0.4.38__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pocketshell-0.4.37 → pocketshell-0.4.38}/PKG-INFO +2 -2
- {pocketshell-0.4.37 → pocketshell-0.4.38}/pyproject.toml +11 -4
- pocketshell-0.4.38/tests/data/quse-0.0.11-usage.json +141 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_usage.py +71 -2
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_watch_ci.py +17 -4
- pocketshell-0.4.38/tests/test_watch_ci_cancelled.py +558 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/uv.lock +16 -3
- {pocketshell-0.4.37 → pocketshell-0.4.38}/.gitignore +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/README.md +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/scheduler/README.md +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/scheduler/pocketshell-usage-capture.service +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/scheduler/pocketshell-usage-capture.timer +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/__init__.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/__main__.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/agent_card_push.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/agent_log.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/agents.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/agents_kind.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/cards.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/cgroup_agents.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/cli.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/daemon.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/env.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/github.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/hooks.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/jobs.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/logs.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/profiles.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/prune_attachments.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/push.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/qr_share.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/repos.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/resume.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/sessions.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/tree.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/usage.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/usage_capture.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/src/pocketshell/usage_reset.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/__init__.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/data/quse-0.0.9-usage.json +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_agent_card_push.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_agent_log.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_agents.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_agents_kind.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_cards.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_cards_push_notify.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_cgroup_agents.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_cli.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_daemon.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_env.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_github.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_hooks.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_jobs.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_logs.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_profiles.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_prune_attachments.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_push.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_qr_share.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_repos.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_resume.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_sessions.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_tree.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_usage_capture.py +0 -0
- {pocketshell-0.4.37 → pocketshell-0.4.38}/tests/test_usage_reset.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pocketshell
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.38
|
|
4
4
|
Summary: Unified server-side Python utility for the PocketShell Android client.
|
|
5
5
|
Project-URL: Homepage, https://github.com/alexeygrigorev/pocketshell
|
|
6
6
|
Project-URL: Issues, https://github.com/alexeygrigorev/pocketshell/issues
|
|
@@ -21,7 +21,7 @@ Requires-Python: >=3.11
|
|
|
21
21
|
Requires-Dist: click>=8.2.0
|
|
22
22
|
Requires-Dist: google-auth>=2.0.0
|
|
23
23
|
Requires-Dist: pyyaml>=6.0
|
|
24
|
-
Requires-Dist: quse==0.0.
|
|
24
|
+
Requires-Dist: quse==0.0.11
|
|
25
25
|
Requires-Dist: tmuxctl>=0.3.3
|
|
26
26
|
Provides-Extra: dev
|
|
27
27
|
Requires-Dist: pytest>=8.4.0; extra == 'dev'
|
|
@@ -8,7 +8,7 @@ name = "pocketshell"
|
|
|
8
8
|
# scripts/check-pypi-version.sh enforces this; .github/workflows/build.yml
|
|
9
9
|
# runs that check before publishing to PyPI. See
|
|
10
10
|
# tools/pocketshell/README.md ("Release flow") for the bump procedure.
|
|
11
|
-
version = "0.4.
|
|
11
|
+
version = "0.4.38"
|
|
12
12
|
description = "Unified server-side Python utility for the PocketShell Android client."
|
|
13
13
|
readme = "README.md"
|
|
14
14
|
requires-python = ">=3.11"
|
|
@@ -35,11 +35,18 @@ classifiers = [
|
|
|
35
35
|
dependencies = [
|
|
36
36
|
"click>=8.2.0",
|
|
37
37
|
# Usage backend + single source of truth for the unified provider schema
|
|
38
|
-
# (issue #1318). Pinned exactly: `pocketshell usage --json` expects quse
|
|
39
|
-
#
|
|
38
|
+
# (issue #1318). Pinned exactly: `pocketshell usage --json` expects quse's
|
|
39
|
+
# provider-keyed `--json` document (per provider: status +
|
|
40
40
|
# short_term/long_term {percent_remaining, reset_at, window} + error) and
|
|
41
41
|
# fails loudly on any schema drift, so the version is frozen, not a range.
|
|
42
|
-
|
|
42
|
+
# 0.0.11 (#1564): quse labels each Codex window from its actual
|
|
43
|
+
# `limit_window_seconds` instead of assuming primary==5h / secondary==7d,
|
|
44
|
+
# and omits a window Codex drops (`present: false`) as a null placeholder
|
|
45
|
+
# rather than a phantom "0% / unavailable" ghost row. Codex temporarily
|
|
46
|
+
# removed the 5h window, so its `primary_window` now carries the WEEKLY
|
|
47
|
+
# (604800s) span — the old positional labels mislabeled weekly data as a
|
|
48
|
+
# "5h window" with a 5-day reset and left a "7d" ghost.
|
|
49
|
+
"quse==0.0.11",
|
|
43
50
|
# FCM HTTP v1 push delivery (#690): service-account OAuth2 bearer minting
|
|
44
51
|
# for `pocketshell push` / the `usage --capture` reset-push send. Imported
|
|
45
52
|
# lazily and fail-soft — a host without it (or without a Firebase
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
{
|
|
2
|
+
"claude": {
|
|
3
|
+
"details": {
|
|
4
|
+
"limit_reached": true,
|
|
5
|
+
"subscription": null,
|
|
6
|
+
"windows": {
|
|
7
|
+
"five_hour": {
|
|
8
|
+
"reset_at": "2026-07-15T11:39:59Z",
|
|
9
|
+
"used_percent": 18.0
|
|
10
|
+
},
|
|
11
|
+
"seven_day": {
|
|
12
|
+
"reset_at": "2026-07-16T14:59:59Z",
|
|
13
|
+
"used_percent": 95.0
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
},
|
|
17
|
+
"error": null,
|
|
18
|
+
"long_term": {
|
|
19
|
+
"percent_remaining": 5.0,
|
|
20
|
+
"reset_at": "2026-07-16T14:59:59Z",
|
|
21
|
+
"window": "7d"
|
|
22
|
+
},
|
|
23
|
+
"short_term": {
|
|
24
|
+
"percent_remaining": 82.0,
|
|
25
|
+
"reset_at": "2026-07-15T11:39:59Z",
|
|
26
|
+
"window": "5h"
|
|
27
|
+
},
|
|
28
|
+
"status": "ok"
|
|
29
|
+
},
|
|
30
|
+
"codex": {
|
|
31
|
+
"details": {
|
|
32
|
+
"limit_reached": false,
|
|
33
|
+
"reset_credits": [
|
|
34
|
+
{
|
|
35
|
+
"expires_at": "2026-07-31T19:09:12Z",
|
|
36
|
+
"status": "available",
|
|
37
|
+
"title": "Full reset"
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"expires_at": "2026-08-11T21:09:47Z",
|
|
41
|
+
"status": "available",
|
|
42
|
+
"title": "Full reset"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"expires_at": "2026-08-12T18:09:45Z",
|
|
46
|
+
"status": "available",
|
|
47
|
+
"title": "Full reset"
|
|
48
|
+
}
|
|
49
|
+
],
|
|
50
|
+
"reset_credits_available": 3,
|
|
51
|
+
"reset_credits_error": null,
|
|
52
|
+
"windows": {
|
|
53
|
+
"primary_window": {
|
|
54
|
+
"limit_window_seconds": 604800,
|
|
55
|
+
"present": true,
|
|
56
|
+
"reset_at": "2026-07-21T20:37:32Z",
|
|
57
|
+
"used_percent": 31.0
|
|
58
|
+
},
|
|
59
|
+
"secondary_window": {
|
|
60
|
+
"limit_window_seconds": null,
|
|
61
|
+
"present": false,
|
|
62
|
+
"reset_at": null,
|
|
63
|
+
"used_percent": 0.0
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
"error": null,
|
|
68
|
+
"long_term": {
|
|
69
|
+
"percent_remaining": null,
|
|
70
|
+
"reset_at": null,
|
|
71
|
+
"window": null
|
|
72
|
+
},
|
|
73
|
+
"short_term": {
|
|
74
|
+
"percent_remaining": 69.0,
|
|
75
|
+
"reset_at": "2026-07-21T20:37:32Z",
|
|
76
|
+
"window": "7d"
|
|
77
|
+
},
|
|
78
|
+
"status": "ok"
|
|
79
|
+
},
|
|
80
|
+
"copilot": {
|
|
81
|
+
"details": {
|
|
82
|
+
"limit_reached": false,
|
|
83
|
+
"premium_entitlement": 1500,
|
|
84
|
+
"premium_percent_remaining": 97.1,
|
|
85
|
+
"premium_remaining": 1457
|
|
86
|
+
},
|
|
87
|
+
"error": null,
|
|
88
|
+
"long_term": {
|
|
89
|
+
"percent_remaining": 97.1,
|
|
90
|
+
"reset_at": "2026-08-01T00:00:00Z",
|
|
91
|
+
"window": "monthly"
|
|
92
|
+
},
|
|
93
|
+
"short_term": {
|
|
94
|
+
"percent_remaining": 100.0,
|
|
95
|
+
"reset_at": null,
|
|
96
|
+
"window": null
|
|
97
|
+
},
|
|
98
|
+
"status": "ok"
|
|
99
|
+
},
|
|
100
|
+
"zai": {
|
|
101
|
+
"details": {
|
|
102
|
+
"limit_reached": false,
|
|
103
|
+
"max_used_percent": 25.0,
|
|
104
|
+
"windows": {
|
|
105
|
+
"five_hour": {
|
|
106
|
+
"limit": null,
|
|
107
|
+
"remaining": null,
|
|
108
|
+
"reset_at": null,
|
|
109
|
+
"used_percent": 1.0,
|
|
110
|
+
"window_hours": 5
|
|
111
|
+
},
|
|
112
|
+
"monthly_web_search": {
|
|
113
|
+
"limit": 4000,
|
|
114
|
+
"remaining": 3962,
|
|
115
|
+
"reset_at": "2026-07-27T14:04:58Z",
|
|
116
|
+
"used_percent": 1.0,
|
|
117
|
+
"window_hours": 5
|
|
118
|
+
},
|
|
119
|
+
"weekly": {
|
|
120
|
+
"limit": null,
|
|
121
|
+
"remaining": null,
|
|
122
|
+
"reset_at": "2026-07-18T14:04:58Z",
|
|
123
|
+
"used_percent": 25.0,
|
|
124
|
+
"window_hours": null
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
},
|
|
128
|
+
"error": null,
|
|
129
|
+
"long_term": {
|
|
130
|
+
"percent_remaining": 75.0,
|
|
131
|
+
"reset_at": "2026-07-18T14:04:58Z",
|
|
132
|
+
"window": "weekly"
|
|
133
|
+
},
|
|
134
|
+
"short_term": {
|
|
135
|
+
"percent_remaining": 99.0,
|
|
136
|
+
"reset_at": null,
|
|
137
|
+
"window": "5h"
|
|
138
|
+
},
|
|
139
|
+
"status": "ok"
|
|
140
|
+
}
|
|
141
|
+
}
|
|
@@ -36,6 +36,16 @@ from pocketshell.usage import (
|
|
|
36
36
|
|
|
37
37
|
_PYPROJECT = Path(__file__).resolve().parent.parent / "pyproject.toml"
|
|
38
38
|
_FIXTURE = Path(__file__).resolve().parent / "data" / "quse-0.0.9-usage.json"
|
|
39
|
+
# Captured LIVE from the pinned quse 0.0.11 (`quse --json`) on 2026-07-15 —
|
|
40
|
+
# the exact Codex "no 5h window" shape (#1564): Codex temporarily removed the
|
|
41
|
+
# 5h window, so its `primary_window` now carries the WEEKLY (604800s) span and
|
|
42
|
+
# `secondary_window` is `present: false`. quse 0.0.11 labels the primary from
|
|
43
|
+
# its actual `limit_window_seconds` (→ `short_term.window == "7d"`, NOT the old
|
|
44
|
+
# positional "5h") and emits the dropped window as a null placeholder
|
|
45
|
+
# (`long_term: {percent_remaining: null, ...}`) instead of a phantom
|
|
46
|
+
# "0% / unavailable" ghost row. Claude / Copilot / zai in the same document
|
|
47
|
+
# keep their correct 5h / 7d / monthly / weekly labels (class coverage).
|
|
48
|
+
_FIXTURE_0011 = Path(__file__).resolve().parent / "data" / "quse-0.0.11-usage.json"
|
|
39
49
|
|
|
40
50
|
|
|
41
51
|
def _fake_completed(
|
|
@@ -62,9 +72,11 @@ def _quse_keyed_json() -> str:
|
|
|
62
72
|
|
|
63
73
|
|
|
64
74
|
def test_pyproject_pins_quse_exactly() -> None:
|
|
65
|
-
# AC: pocketshell pins quse==0.0.
|
|
75
|
+
# AC: pocketshell pins quse==0.0.11 as a hard dependency (frozen contract).
|
|
76
|
+
# 0.0.11 (#1564) labels Codex windows from the actual `limit_window_seconds`
|
|
77
|
+
# and omits a dropped window rather than emitting a phantom "5h"/ghost row.
|
|
66
78
|
text = _PYPROJECT.read_text()
|
|
67
|
-
assert '"quse==0.0.
|
|
79
|
+
assert '"quse==0.0.11"' in text, "pyproject must pin quse==0.0.11 in dependencies"
|
|
68
80
|
|
|
69
81
|
|
|
70
82
|
def test_resolve_quse_binary_uses_pinned_env_next_to_interpreter(tmp_path: Path) -> None:
|
|
@@ -189,6 +201,63 @@ def test_flatten_passes_unified_fields_through_unchanged() -> None:
|
|
|
189
201
|
}
|
|
190
202
|
|
|
191
203
|
|
|
204
|
+
def test_flatten_codex_0011_no_5h_window_weekly_only_no_ghost() -> None:
|
|
205
|
+
"""#1564: quse 0.0.11 fixes Codex's mislabeled/ghost windows at the source.
|
|
206
|
+
|
|
207
|
+
Feeds the REAL captured quse 0.0.11 Codex "no 5h window" shape through the
|
|
208
|
+
flatten and asserts the corrected wire shape the app consumes:
|
|
209
|
+
|
|
210
|
+
- `short_term` is the WEEKLY window labeled **"7d"** (from Codex's real
|
|
211
|
+
`limit_window_seconds=604800`) with its real reset — NOT the old phantom
|
|
212
|
+
"5h" window that showed weekly data under a 5h label with a 5-day reset.
|
|
213
|
+
- `long_term` is a null placeholder (`percent_remaining: null`) for the
|
|
214
|
+
window Codex DROPPED — the app parser treats a null-percent window as
|
|
215
|
+
absent, so there is NO "0% / unavailable" ghost row.
|
|
216
|
+
|
|
217
|
+
The flatten passes quse's unified fields through verbatim (D22 / #1318):
|
|
218
|
+
pocketshell does NOT relabel — the correct labels come from quse 0.0.11.
|
|
219
|
+
"""
|
|
220
|
+
out = normalize_usage_stdout(_FIXTURE_0011.read_text())
|
|
221
|
+
by_provider = {r["provider"]: r for r in (json.loads(ln) for ln in out.splitlines())}
|
|
222
|
+
|
|
223
|
+
codex = by_provider["codex"]
|
|
224
|
+
# The single real Codex window is the WEEKLY one, labeled "7d" — no phantom 5h.
|
|
225
|
+
assert codex["short_term"]["window"] == "7d"
|
|
226
|
+
assert codex["short_term"]["window"] != "5h"
|
|
227
|
+
assert codex["short_term"]["reset_at"] == "2026-07-21T20:37:32Z"
|
|
228
|
+
assert codex["short_term"]["percent_remaining"] == 69.0
|
|
229
|
+
# The dropped window is a null placeholder → the app parser omits it (no
|
|
230
|
+
# "0% / unavailable" ghost row). percent_remaining MUST be null here.
|
|
231
|
+
assert codex["long_term"]["percent_remaining"] is None
|
|
232
|
+
assert codex["long_term"]["window"] is None
|
|
233
|
+
assert codex["long_term"]["reset_at"] is None
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def test_flatten_codex_0011_leaves_other_providers_unchanged() -> None:
|
|
237
|
+
"""#1564 class coverage: the quse Codex fix does NOT regress other cards.
|
|
238
|
+
|
|
239
|
+
Claude keeps its 5h + 7d windows, Copilot keeps its monthly window, and zai
|
|
240
|
+
keeps its 5h + weekly windows — all labeled correctly in the same quse
|
|
241
|
+
0.0.11 document. Only Codex's window shape changed.
|
|
242
|
+
"""
|
|
243
|
+
out = normalize_usage_stdout(_FIXTURE_0011.read_text())
|
|
244
|
+
by_provider = {r["provider"]: r for r in (json.loads(ln) for ln in out.splitlines())}
|
|
245
|
+
|
|
246
|
+
claude = by_provider["claude"]
|
|
247
|
+
assert claude["short_term"]["window"] == "5h"
|
|
248
|
+
assert claude["short_term"]["reset_at"] == "2026-07-15T11:39:59Z"
|
|
249
|
+
assert claude["long_term"]["window"] == "7d"
|
|
250
|
+
assert claude["long_term"]["reset_at"] == "2026-07-16T14:59:59Z"
|
|
251
|
+
|
|
252
|
+
copilot = by_provider["copilot"]
|
|
253
|
+
assert copilot["long_term"]["window"] == "monthly"
|
|
254
|
+
assert copilot["long_term"]["percent_remaining"] == 97.1
|
|
255
|
+
|
|
256
|
+
zai = by_provider["zai"]
|
|
257
|
+
assert zai["short_term"]["window"] == "5h"
|
|
258
|
+
assert zai["long_term"]["window"] == "weekly"
|
|
259
|
+
|
|
260
|
+
|
|
192
261
|
def test_flatten_raises_on_non_json() -> None:
|
|
193
262
|
try:
|
|
194
263
|
normalize_usage_stdout("this is not json")
|
|
@@ -9,9 +9,11 @@ without a real `gh` binary, network, or wall-clock wait.
|
|
|
9
9
|
Exit-code contract under test:
|
|
10
10
|
|
|
11
11
|
0 green — all required checks passed
|
|
12
|
-
1 failed — a required check failed
|
|
12
|
+
1 failed — a required check GENUINELY failed
|
|
13
13
|
2 hang — no progress past --no-progress-timeout OR wall-clock cap
|
|
14
14
|
3 unresolved — run / inputs couldn't be resolved (or gh stayed broken)
|
|
15
|
+
4 superseded — a newer run replaced this one (routine concurrency-cancel)
|
|
16
|
+
5 no_verdict — cancelled, so no verdict was reached (issue #1650)
|
|
15
17
|
|
|
16
18
|
The load-bearing case is the HANG one: a run whose jobs NEVER change state must
|
|
17
19
|
exit 2, not loop forever. The test drives a virtual clock so a stalled run is
|
|
@@ -198,15 +200,26 @@ def test_required_failure_exits_1_and_names_job():
|
|
|
198
200
|
assert outcome.likely_infra is False
|
|
199
201
|
|
|
200
202
|
|
|
201
|
-
def
|
|
203
|
+
def test_cancelled_run_is_not_a_failure():
|
|
204
|
+
"""Issue #1650: `cancelled` is NOT a failure — it is a NON-verdict.
|
|
205
|
+
|
|
206
|
+
This test previously asserted exit 1 and so ENCODED the bug: `main`'s push
|
|
207
|
+
concurrency group cancels superseded runs by design, which made every
|
|
208
|
+
superseded-run watch a guaranteed false FAILED. Hard-cut per D22 — the old
|
|
209
|
+
expectation is deleted, not kept alongside. Full class coverage (superseded
|
|
210
|
+
vs user-cancelled vs genuine-failure-then-cancelled) lives in
|
|
211
|
+
test_watch_ci_cancelled.py.
|
|
212
|
+
"""
|
|
202
213
|
gh = FakeGh()
|
|
203
214
|
jobs = _all_required_jobs()
|
|
204
215
|
jobs[1] = _job("Python utility tests (pocketshell)", "completed", "cancelled", db_id=7)
|
|
205
216
|
gh.queue_run_state(status="completed", conclusion="cancelled", jobs=jobs)
|
|
217
|
+
gh.run_list_json = [] # no newer run → not superseded, just no verdict
|
|
206
218
|
watcher, _ = _make_watcher(gh)
|
|
207
219
|
outcome = watcher.watch(run_id="123")
|
|
208
|
-
assert outcome.result == wci.
|
|
209
|
-
assert outcome.exit_code ==
|
|
220
|
+
assert outcome.result == wci.RESULT_NO_VERDICT
|
|
221
|
+
assert outcome.exit_code == 5
|
|
222
|
+
assert outcome.signature is None
|
|
210
223
|
|
|
211
224
|
|
|
212
225
|
# ── 2 hang: no-progress (the load-bearing case) ──────────────────────────────
|
|
@@ -0,0 +1,558 @@
|
|
|
1
|
+
"""Regression tests for the watch-ci.py fabricated-infra-failure defects (#1650).
|
|
2
|
+
|
|
3
|
+
An on-call ran `scripts/watch-ci.py` against `main` run 29520839502 and got a
|
|
4
|
+
confident, specific, and ENTIRELY FICTIONAL report:
|
|
5
|
+
|
|
6
|
+
FAILED — a required check failed
|
|
7
|
+
signature: echo "::error title=sdkmanager still broken after repair::... (issue #771).
|
|
8
|
+
Treat as EMULATOR INFRA UNAVAILABLE, not a test failure."
|
|
9
|
+
|
|
10
|
+
Neither part happened. The run was merely SUPERSEDED — `#1648` landed and
|
|
11
|
+
`main`'s push concurrency group cancelled the older in-flight run (the
|
|
12
|
+
documented, intended design; see process.md "the `main` push concurrency group
|
|
13
|
+
cancels older in-flight runs when newer merges land"). `Unit tests`, `Python
|
|
14
|
+
utility tests (pocketshell)` and `Integration tests (Docker)` were all green.
|
|
15
|
+
|
|
16
|
+
Two independent defects, reproduced separately below:
|
|
17
|
+
|
|
18
|
+
D1. `cancelled` was in FAILING_CONCLUSIONS, so a routine concurrency-cancel
|
|
19
|
+
became a guaranteed false `FAILED` on the exact path the on-call is told
|
|
20
|
+
to watch (process.md "Never babysit CI").
|
|
21
|
+
D2. The signature grep matched a CONDITIONAL `echo` in the workflow's script
|
|
22
|
+
body — captured by GitHub's command echo (the `ESC[36;1m` prefix) — that
|
|
23
|
+
NEVER FIRED. It could not tell "this error occurred" from "this string
|
|
24
|
+
exists in the script text", so it attached a wrong diagnosis AND a wrong
|
|
25
|
+
issue number.
|
|
26
|
+
|
|
27
|
+
The load-bearing NEGATIVE cases (G6) matter more than the false alarms: a fix
|
|
28
|
+
that makes the watcher blind to real failures is far worse than the bug, because
|
|
29
|
+
the on-call would stop noticing red CI. `test_genuine_*` below assert that a
|
|
30
|
+
real infra failure is still detected WITH its real signature, and a real test
|
|
31
|
+
failure is still reported as a real failure — never softened to
|
|
32
|
+
cancelled/superseded/unknown.
|
|
33
|
+
|
|
34
|
+
Class coverage (G2) — not just the one reported instance:
|
|
35
|
+
|
|
36
|
+
| case | expected result |
|
|
37
|
+
|-----------------------------------|-----------------|
|
|
38
|
+
| superseded by concurrency | superseded (4) |
|
|
39
|
+
| user/API cancelled, no newer run | no_verdict (5) |
|
|
40
|
+
| genuine infra failure | failed (1) + real signature, likely_infra |
|
|
41
|
+
| genuine test failure | failed (1) + real signature |
|
|
42
|
+
| genuine failure THEN cancelled | failed (1) |
|
|
43
|
+
| timeout / hang | hang (2) |
|
|
44
|
+
| no verdict available (probe down) | no_verdict (5) |
|
|
45
|
+
|
|
46
|
+
Fixtures are captured from the REAL run/log (`gh run view --log`), not
|
|
47
|
+
hand-idealised, per the #847 happy-fixture-masks-reality lesson.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
from __future__ import annotations
|
|
51
|
+
|
|
52
|
+
import pytest
|
|
53
|
+
|
|
54
|
+
from tests.test_watch_ci import ( # the shared offline harness
|
|
55
|
+
REQUIRED,
|
|
56
|
+
FakeGh,
|
|
57
|
+
_job,
|
|
58
|
+
_make_watcher,
|
|
59
|
+
wci,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
# ── Real captured fixtures (run 29520839502, main @ ab3e0caf) ────────────────
|
|
63
|
+
|
|
64
|
+
# The EXACT job set from `gh run view 29520839502 --json status,conclusion,jobs`.
|
|
65
|
+
# The three cheap required checks passed; only the emulator shards + the
|
|
66
|
+
# aggregate were cancelled when #1648 (136fe702) superseded the run at 18:50:18.
|
|
67
|
+
def _superseded_main_jobs() -> list[dict]:
|
|
68
|
+
return [
|
|
69
|
+
_job("Python utility tests (pocketshell)", "completed", "success", 87697273282),
|
|
70
|
+
_job("Integration tests (Docker)", "completed", "success", 87697273298),
|
|
71
|
+
_job("Unit tests", "completed", "success", 87697273300),
|
|
72
|
+
_job(
|
|
73
|
+
"Emulator journey subset (load-bearing, Docker agents) (1)",
|
|
74
|
+
"completed",
|
|
75
|
+
"cancelled",
|
|
76
|
+
87701705006,
|
|
77
|
+
),
|
|
78
|
+
_job(
|
|
79
|
+
"Emulator journey subset (load-bearing, Docker agents) (0)",
|
|
80
|
+
"completed",
|
|
81
|
+
"cancelled",
|
|
82
|
+
87701705021,
|
|
83
|
+
),
|
|
84
|
+
_job(
|
|
85
|
+
"Emulator journey subset (load-bearing, Docker agents) (2)",
|
|
86
|
+
"completed",
|
|
87
|
+
"cancelled",
|
|
88
|
+
87701705035,
|
|
89
|
+
),
|
|
90
|
+
_job(
|
|
91
|
+
"Emulator journey aggregate verdict (#1458)",
|
|
92
|
+
"completed",
|
|
93
|
+
"cancelled",
|
|
94
|
+
87713160666,
|
|
95
|
+
),
|
|
96
|
+
]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# The EXACT line `gh run view --job 87701705006 --log` emitted at line 529 that
|
|
100
|
+
# the watcher turned into its fabricated signature. The `ESC[36;1m ... ESC[0m`
|
|
101
|
+
# wrapper is GitHub's COMMAND ECHO: it is the `run:` step's script body being
|
|
102
|
+
# printed, NOT output from an executed command. The `if` guarding this echo
|
|
103
|
+
# never fired — the sdkmanager was fine.
|
|
104
|
+
_REAL_ECHOED_SDKMANAGER_LINE = (
|
|
105
|
+
"Emulator journey subset (load-bearing, Docker agents) (1)\t"
|
|
106
|
+
"Repair Android cmdline-tools + accept licenses (issue\t"
|
|
107
|
+
"2026-07-16T18:03:10.7886203Z \x1b[36;1m "
|
|
108
|
+
'echo "::error title=sdkmanager still broken after repair::The '
|
|
109
|
+
"freshly-installed cmdline-tools sdkmanager still failed to run (issue "
|
|
110
|
+
'#771). Treat as EMULATOR INFRA UNAVAILABLE, not a test failure."\x1b[0m'
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# Real surrounding context from the same captured log.
|
|
114
|
+
_REAL_CANCELLED_JOB_LOG = "\n".join(
|
|
115
|
+
[
|
|
116
|
+
"Emulator journey subset (load-bearing, Docker agents) (1)\tSet up job\t"
|
|
117
|
+
"2026-07-16T18:02:55.1000000Z Job is about to start running on the runner",
|
|
118
|
+
_REAL_ECHOED_SDKMANAGER_LINE,
|
|
119
|
+
"Emulator journey subset (load-bearing, Docker agents) (1)\t"
|
|
120
|
+
"Repair Android cmdline-tools + accept licenses (issue\t"
|
|
121
|
+
"2026-07-16T18:03:12.0000000Z sdkmanager ok",
|
|
122
|
+
"Emulator journey subset (load-bearing, Docker agents) (1)\t"
|
|
123
|
+
"Retry journey subset on a fresh cold-booted emulator\t"
|
|
124
|
+
"2026-07-16T18:50:19.0573036Z ##[error]The operation was canceled.",
|
|
125
|
+
]
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _newer_run_on_main(newer: bool = True) -> list[dict]:
|
|
130
|
+
"""`gh run list --branch main` output; newest-first, as gh returns it."""
|
|
131
|
+
runs = []
|
|
132
|
+
if newer:
|
|
133
|
+
runs.append(
|
|
134
|
+
{
|
|
135
|
+
"databaseId": 29521111111,
|
|
136
|
+
"workflowName": "Tests",
|
|
137
|
+
"headBranch": "main",
|
|
138
|
+
"status": "in_progress",
|
|
139
|
+
"createdAt": "2026-07-16T18:50:18Z",
|
|
140
|
+
}
|
|
141
|
+
)
|
|
142
|
+
runs.append(
|
|
143
|
+
{
|
|
144
|
+
"databaseId": 29520839502,
|
|
145
|
+
"workflowName": "Tests",
|
|
146
|
+
"headBranch": "main",
|
|
147
|
+
"status": "completed",
|
|
148
|
+
"createdAt": "2026-07-16T18:02:40Z",
|
|
149
|
+
}
|
|
150
|
+
)
|
|
151
|
+
return runs
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _superseded_run_state() -> dict:
|
|
155
|
+
return {
|
|
156
|
+
"status": "completed",
|
|
157
|
+
"conclusion": "cancelled",
|
|
158
|
+
"databaseId": 29520839502,
|
|
159
|
+
"headBranch": "main",
|
|
160
|
+
"workflowName": "Tests",
|
|
161
|
+
"createdAt": "2026-07-16T18:02:40Z",
|
|
162
|
+
"jobs": _superseded_main_jobs(),
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
# ── D1: superseded-by-concurrency is NOT a failure ───────────────────────────
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def test_superseded_by_concurrency_is_not_reported_as_failed():
|
|
170
|
+
"""THE reported instance. Run 29520839502 was superseded, not broken.
|
|
171
|
+
|
|
172
|
+
Before the fix this returned RESULT_FAILED/exit 1 — a standing false-alarm
|
|
173
|
+
generator, because `main`'s concurrency-cancel is routine and by design.
|
|
174
|
+
"""
|
|
175
|
+
gh = FakeGh()
|
|
176
|
+
gh.queue_run_state(**_superseded_run_state())
|
|
177
|
+
gh.run_list_json = _newer_run_on_main(newer=True)
|
|
178
|
+
gh.log_text = _REAL_CANCELLED_JOB_LOG
|
|
179
|
+
watcher, _ = _make_watcher(gh)
|
|
180
|
+
|
|
181
|
+
outcome = watcher.watch(run_id="29520839502")
|
|
182
|
+
|
|
183
|
+
assert outcome.result != wci.RESULT_FAILED, (
|
|
184
|
+
"a concurrency-superseded run must NOT be reported as a failure "
|
|
185
|
+
f"(got {outcome.result}: {outcome.reason})"
|
|
186
|
+
)
|
|
187
|
+
assert outcome.result == wci.RESULT_SUPERSEDED
|
|
188
|
+
assert outcome.exit_code == 4
|
|
189
|
+
assert "supersed" in outcome.reason.lower()
|
|
190
|
+
# The on-call's action is to re-watch the newest head — surface it.
|
|
191
|
+
assert "29521111111" in outcome.reason
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def test_superseded_run_never_fabricates_a_signature():
|
|
195
|
+
"""D1+D2 together: the fictional story is signature-free.
|
|
196
|
+
|
|
197
|
+
A superseded run has no failure, so there is nothing to diagnose. Attaching
|
|
198
|
+
ANY signature here is fabrication (G5: an infra claim needs a captured
|
|
199
|
+
signature AND a clean re-run).
|
|
200
|
+
"""
|
|
201
|
+
gh = FakeGh()
|
|
202
|
+
gh.queue_run_state(**_superseded_run_state())
|
|
203
|
+
gh.run_list_json = _newer_run_on_main(newer=True)
|
|
204
|
+
gh.log_text = _REAL_CANCELLED_JOB_LOG
|
|
205
|
+
watcher, _ = _make_watcher(gh)
|
|
206
|
+
|
|
207
|
+
outcome = watcher.watch(run_id="29520839502")
|
|
208
|
+
|
|
209
|
+
assert outcome.signature is None
|
|
210
|
+
assert outcome.likely_infra is False
|
|
211
|
+
assert "771" not in wci.render_human_summary(outcome), (
|
|
212
|
+
"must never surface a guessed issue number for a run that did not fail"
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def test_cancelled_without_a_newer_run_is_no_verdict_not_failed():
|
|
217
|
+
"""User/API cancelled: no newer run on the branch, so it was not superseded.
|
|
218
|
+
|
|
219
|
+
We STILL cannot claim a failure — a cancelled run produced no verdict. The
|
|
220
|
+
watcher must say exactly that rather than invent one.
|
|
221
|
+
"""
|
|
222
|
+
gh = FakeGh()
|
|
223
|
+
gh.queue_run_state(**_superseded_run_state())
|
|
224
|
+
gh.run_list_json = _newer_run_on_main(newer=False) # only ourselves
|
|
225
|
+
gh.log_text = _REAL_CANCELLED_JOB_LOG
|
|
226
|
+
watcher, _ = _make_watcher(gh)
|
|
227
|
+
|
|
228
|
+
outcome = watcher.watch(run_id="29520839502")
|
|
229
|
+
|
|
230
|
+
assert outcome.result == wci.RESULT_NO_VERDICT
|
|
231
|
+
assert outcome.exit_code == 5
|
|
232
|
+
assert outcome.signature is None
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def test_cancelled_with_unreachable_probe_is_no_verdict_not_failed():
|
|
236
|
+
"""The supersede probe itself failing must not resurrect the false FAILED.
|
|
237
|
+
|
|
238
|
+
If we cannot tell WHY it was cancelled, "no verdict" is the honest answer.
|
|
239
|
+
"""
|
|
240
|
+
gh = FakeGh()
|
|
241
|
+
gh.queue_run_state(**_superseded_run_state())
|
|
242
|
+
gh.run_list_json = None # `gh run list` errors out
|
|
243
|
+
gh.log_text = _REAL_CANCELLED_JOB_LOG
|
|
244
|
+
watcher, _ = _make_watcher(gh)
|
|
245
|
+
|
|
246
|
+
outcome = watcher.watch(run_id="29520839502")
|
|
247
|
+
|
|
248
|
+
assert outcome.result == wci.RESULT_NO_VERDICT
|
|
249
|
+
assert outcome.exit_code == 5
|
|
250
|
+
assert outcome.signature is None
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# ── D2: a signature must come from output that ACTUALLY EXECUTED ─────────────
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def test_echoed_conditional_script_text_is_not_a_signature():
|
|
257
|
+
"""The exact fabrication vector, at the unit level.
|
|
258
|
+
|
|
259
|
+
`extract_signature` used to return this echoed `run:` body verbatim, because
|
|
260
|
+
the loose pattern `(Install|installing).*failed` happened to match
|
|
261
|
+
"freshly-installed ... still failed to run" inside the ECHOED script text.
|
|
262
|
+
A conditional echo that never fired is not evidence of anything.
|
|
263
|
+
"""
|
|
264
|
+
sig = wci.extract_signature(_REAL_CANCELLED_JOB_LOG)
|
|
265
|
+
if sig is not None:
|
|
266
|
+
assert "sdkmanager" not in sig, (
|
|
267
|
+
"matched a conditional echo that never executed — the #1650 "
|
|
268
|
+
f"fabrication: {sig!r}"
|
|
269
|
+
)
|
|
270
|
+
assert "#771" not in sig
|
|
271
|
+
assert "echo" not in sig
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def test_command_echo_lines_are_ignored_even_when_they_look_like_errors():
|
|
275
|
+
"""Class coverage: ANY echoed script body, not just the sdkmanager one."""
|
|
276
|
+
log = "\n".join(
|
|
277
|
+
[
|
|
278
|
+
"j\tstep\t2026-07-16T18:03:10.0000000Z \x1b[36;1mif ! foo; then\x1b[0m",
|
|
279
|
+
"j\tstep\t2026-07-16T18:03:10.0000000Z \x1b[36;1m "
|
|
280
|
+
'echo "error: everything is on fire"\x1b[0m',
|
|
281
|
+
"j\tstep\t2026-07-16T18:03:10.0000000Z \x1b[36;1mfi\x1b[0m",
|
|
282
|
+
"j\tstep\t2026-07-16T18:03:11.0000000Z all good",
|
|
283
|
+
]
|
|
284
|
+
)
|
|
285
|
+
assert wci.extract_signature(log) is None
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def test_echoed_annotation_is_ignored_even_without_the_ansi_marker():
|
|
289
|
+
"""Defence in depth: the vector must stay closed if ANSI is ever stripped.
|
|
290
|
+
|
|
291
|
+
A real annotation is emitted by the runner as `##[error]...`; it is never the
|
|
292
|
+
literal text `echo "::error`. So an echoed annotation is script text even
|
|
293
|
+
when the `ESC[36;1m` command-echo marker is absent.
|
|
294
|
+
"""
|
|
295
|
+
log = (
|
|
296
|
+
"j\tRepair Android cmdline-tools\t2026-07-16T18:03:10.0000000Z "
|
|
297
|
+
'echo "::error title=sdkmanager still broken after repair::The '
|
|
298
|
+
"freshly-installed cmdline-tools sdkmanager still failed to run (issue "
|
|
299
|
+
'#771). Treat as EMULATOR INFRA UNAVAILABLE, not a test failure."'
|
|
300
|
+
)
|
|
301
|
+
sig = wci.extract_signature(log)
|
|
302
|
+
assert sig is None or "sdkmanager" not in sig
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def test_usage_help_text_mentioning_an_error_is_not_a_signature():
|
|
306
|
+
"""A `--help`/usage line that merely NAMES an error is not an occurrence."""
|
|
307
|
+
log = "\n".join(
|
|
308
|
+
[
|
|
309
|
+
"j\tstep\t2026-07-16T18:03:10.0000000Z \x1b[36;1m"
|
|
310
|
+
'echo "usage: gradlew [--fail-fast] # BUILD FAILED means a test broke"\x1b[0m',
|
|
311
|
+
"j\tstep\t2026-07-16T18:03:11.0000000Z done",
|
|
312
|
+
]
|
|
313
|
+
)
|
|
314
|
+
assert wci.extract_signature(log) is None
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
# ── G6 NEGATIVE CASES: real failures MUST still be caught ────────────────────
|
|
318
|
+
# These are the load-bearing assertions. Going blind to real red CI would be far
|
|
319
|
+
# worse than the false alarms this issue is about.
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def test_genuine_test_failure_is_still_reported_failed_with_its_signature():
|
|
323
|
+
gh = FakeGh()
|
|
324
|
+
jobs = [_job(n, "completed", "success", i) for i, n in enumerate(REQUIRED)]
|
|
325
|
+
jobs[0] = _job("Unit tests", "completed", "failure", 99)
|
|
326
|
+
gh.queue_run_state(
|
|
327
|
+
status="completed",
|
|
328
|
+
conclusion="failure",
|
|
329
|
+
databaseId=1,
|
|
330
|
+
headBranch="main",
|
|
331
|
+
workflowName="Tests",
|
|
332
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
333
|
+
jobs=jobs,
|
|
334
|
+
)
|
|
335
|
+
gh.log_text = (
|
|
336
|
+
"Unit tests\tRun tests\t2026-07-16T10:00:05Z FooBarTest > checks FAILED\n"
|
|
337
|
+
"Unit tests\tRun tests\t2026-07-16T10:00:06Z BUILD FAILED in 12s\n"
|
|
338
|
+
)
|
|
339
|
+
watcher, _ = _make_watcher(gh)
|
|
340
|
+
|
|
341
|
+
outcome = watcher.watch(run_id="1")
|
|
342
|
+
|
|
343
|
+
assert outcome.result == wci.RESULT_FAILED
|
|
344
|
+
assert outcome.exit_code == 1
|
|
345
|
+
assert "Unit tests" in outcome.failing_jobs
|
|
346
|
+
assert outcome.signature is not None and "FAILED" in outcome.signature
|
|
347
|
+
assert outcome.likely_infra is False
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def test_genuine_infra_failure_is_still_detected_with_its_real_signature():
|
|
351
|
+
"""A real, EXECUTED infra error keeps its signature + likely_infra tag (G5)."""
|
|
352
|
+
gh = FakeGh()
|
|
353
|
+
jobs = [_job(n, "completed", "success", i) for i, n in enumerate(REQUIRED)]
|
|
354
|
+
jobs[3] = _job(REQUIRED[3], "completed", "failure", 42)
|
|
355
|
+
gh.queue_run_state(
|
|
356
|
+
status="completed",
|
|
357
|
+
conclusion="failure",
|
|
358
|
+
databaseId=1,
|
|
359
|
+
headBranch="main",
|
|
360
|
+
workflowName="Tests",
|
|
361
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
362
|
+
jobs=jobs,
|
|
363
|
+
)
|
|
364
|
+
# Note: a REAL runner-emitted annotation, not an echoed script body.
|
|
365
|
+
gh.log_text = (
|
|
366
|
+
"emu\tBoot emulator\t2026-07-16T18:10:00.0000000Z \x1b[36;1m"
|
|
367
|
+
'echo "starting emulator"\x1b[0m\n'
|
|
368
|
+
"emu\tBoot emulator\t2026-07-16T18:40:00.0000000Z "
|
|
369
|
+
"##[error]Timed out waiting for emulator to boot after 1800s\n"
|
|
370
|
+
)
|
|
371
|
+
watcher, _ = _make_watcher(gh)
|
|
372
|
+
|
|
373
|
+
outcome = watcher.watch(run_id="1")
|
|
374
|
+
|
|
375
|
+
assert outcome.result == wci.RESULT_FAILED
|
|
376
|
+
assert outcome.exit_code == 1
|
|
377
|
+
assert outcome.signature is not None
|
|
378
|
+
assert "Timed out waiting for emulator" in outcome.signature
|
|
379
|
+
assert outcome.likely_infra is True, "a real infra signature must still tag infra"
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def test_genuine_failure_that_is_later_cancelled_is_still_failed():
|
|
383
|
+
"""Precedence: a real failure outranks the cancel that followed it.
|
|
384
|
+
|
|
385
|
+
The concurrency-cancel fix must not become a laundry service for real red.
|
|
386
|
+
"""
|
|
387
|
+
gh = FakeGh()
|
|
388
|
+
jobs = _superseded_main_jobs()
|
|
389
|
+
jobs[2] = _job("Unit tests", "completed", "failure", 87697273300)
|
|
390
|
+
gh.queue_run_state(
|
|
391
|
+
status="completed",
|
|
392
|
+
conclusion="cancelled", # run cancelled AFTER Unit tests genuinely failed
|
|
393
|
+
databaseId=29520839502,
|
|
394
|
+
headBranch="main",
|
|
395
|
+
workflowName="Tests",
|
|
396
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
397
|
+
jobs=jobs,
|
|
398
|
+
)
|
|
399
|
+
gh.run_list_json = _newer_run_on_main(newer=True) # a newer run DOES exist
|
|
400
|
+
gh.log_text = (
|
|
401
|
+
"Unit tests\tRun tests\t2026-07-16T18:30:00Z BUILD FAILED in 12s\n"
|
|
402
|
+
)
|
|
403
|
+
watcher, _ = _make_watcher(gh)
|
|
404
|
+
|
|
405
|
+
outcome = watcher.watch(run_id="29520839502")
|
|
406
|
+
|
|
407
|
+
assert outcome.result == wci.RESULT_FAILED, (
|
|
408
|
+
"a genuinely failed required check must stay FAILED even though the run "
|
|
409
|
+
"was later cancelled/superseded — otherwise real red CI goes unnoticed"
|
|
410
|
+
)
|
|
411
|
+
assert outcome.exit_code == 1
|
|
412
|
+
assert "Unit tests" in outcome.failing_jobs
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def test_timed_out_conclusion_is_still_a_real_failure():
|
|
416
|
+
"""`timed_out` is NOT cancelled — it stays a real failure."""
|
|
417
|
+
gh = FakeGh()
|
|
418
|
+
jobs = [_job(n, "completed", "success", i) for i, n in enumerate(REQUIRED)]
|
|
419
|
+
jobs[0] = _job("Unit tests", "completed", "timed_out", 5)
|
|
420
|
+
gh.queue_run_state(
|
|
421
|
+
status="completed",
|
|
422
|
+
conclusion="failure",
|
|
423
|
+
databaseId=1,
|
|
424
|
+
headBranch="main",
|
|
425
|
+
workflowName="Tests",
|
|
426
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
427
|
+
jobs=jobs,
|
|
428
|
+
)
|
|
429
|
+
gh.log_text = "Unit tests\tRun\t2026-07-16T10:00:06Z ##[error]The job running has exceeded the maximum execution time\n"
|
|
430
|
+
watcher, _ = _make_watcher(gh)
|
|
431
|
+
|
|
432
|
+
outcome = watcher.watch(run_id="1")
|
|
433
|
+
|
|
434
|
+
assert outcome.result == wci.RESULT_FAILED
|
|
435
|
+
assert outcome.exit_code == 1
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def test_unclassified_failure_reports_honestly_without_guessing():
|
|
439
|
+
"""No confident signature ⇒ say 'unclassified', never guess (G5).
|
|
440
|
+
|
|
441
|
+
A fabricated signature is worse than no signature: it sends the on-call to
|
|
442
|
+
the wrong issue with false confidence.
|
|
443
|
+
"""
|
|
444
|
+
gh = FakeGh()
|
|
445
|
+
jobs = [_job(n, "completed", "success", i) for i, n in enumerate(REQUIRED)]
|
|
446
|
+
jobs[0] = _job("Unit tests", "completed", "failure", 99)
|
|
447
|
+
gh.queue_run_state(
|
|
448
|
+
status="completed",
|
|
449
|
+
conclusion="failure",
|
|
450
|
+
databaseId=1,
|
|
451
|
+
headBranch="main",
|
|
452
|
+
workflowName="Tests",
|
|
453
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
454
|
+
jobs=jobs,
|
|
455
|
+
)
|
|
456
|
+
gh.log_text = "Unit tests\tRun\t2026-07-16T10:00:00Z nothing conclusive here\n"
|
|
457
|
+
watcher, _ = _make_watcher(gh)
|
|
458
|
+
|
|
459
|
+
outcome = watcher.watch(run_id="1")
|
|
460
|
+
|
|
461
|
+
assert outcome.result == wci.RESULT_FAILED
|
|
462
|
+
assert outcome.signature is None
|
|
463
|
+
assert outcome.likely_infra is False
|
|
464
|
+
summary = wci.render_human_summary(outcome)
|
|
465
|
+
assert "unclassified" in summary.lower()
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
# ── D3: a healthy long emulator shard must not trip the no-progress guard ────
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def test_default_no_progress_timeout_exceeds_the_emulator_job_cap():
|
|
472
|
+
"""The emulator shards run 20-40 min (cap 95) with NO job-state change.
|
|
473
|
+
|
|
474
|
+
The old 1200s (20 min) default meant every `main` emulator watch reported a
|
|
475
|
+
bogus HANG while the shards were healthy and running.
|
|
476
|
+
"""
|
|
477
|
+
job_cap_s = 95 * 60 # .github/workflows/tests.yml: timeout-minutes: 95
|
|
478
|
+
assert wci.DEFAULT_NO_PROGRESS_TIMEOUT_S > job_cap_s, (
|
|
479
|
+
"a healthy emulator shard would trip the no-progress guard"
|
|
480
|
+
)
|
|
481
|
+
# The wall-clock cap must leave room for the no-progress guard to be the
|
|
482
|
+
# thing that fires, otherwise raising it changes nothing.
|
|
483
|
+
assert wci.DEFAULT_MAX_WALL_CLOCK_S > wci.DEFAULT_NO_PROGRESS_TIMEOUT_S
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def test_healthy_long_running_emulator_shard_is_not_a_hang():
|
|
487
|
+
"""A shard in_progress for 40 min with no state change is HEALTHY, not hung."""
|
|
488
|
+
gh = FakeGh()
|
|
489
|
+
jobs = [_job(n, "completed", "success", i) for i, n in enumerate(REQUIRED)]
|
|
490
|
+
# The emulator shard grinds away, unchanged, for 40 minutes...
|
|
491
|
+
for _ in range(200):
|
|
492
|
+
shard = list(jobs)
|
|
493
|
+
shard[3] = _job(REQUIRED[3], "in_progress", None, 42)
|
|
494
|
+
gh.queue_run_state(
|
|
495
|
+
status="in_progress",
|
|
496
|
+
conclusion=None,
|
|
497
|
+
databaseId=1,
|
|
498
|
+
headBranch="main",
|
|
499
|
+
workflowName="Tests",
|
|
500
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
501
|
+
jobs=shard,
|
|
502
|
+
)
|
|
503
|
+
# ...then finishes green.
|
|
504
|
+
gh.queue_run_state(
|
|
505
|
+
status="completed",
|
|
506
|
+
conclusion="success",
|
|
507
|
+
databaseId=1,
|
|
508
|
+
headBranch="main",
|
|
509
|
+
workflowName="Tests",
|
|
510
|
+
createdAt="2026-07-16T18:02:40Z",
|
|
511
|
+
jobs=jobs,
|
|
512
|
+
)
|
|
513
|
+
gh.repeat_last_run_state = True
|
|
514
|
+
watcher, _ = _make_watcher(
|
|
515
|
+
gh,
|
|
516
|
+
interval_s=12.0, # 200 polls * 12s = 40 min of no job-state change
|
|
517
|
+
no_progress_timeout_s=wci.DEFAULT_NO_PROGRESS_TIMEOUT_S,
|
|
518
|
+
max_wall_clock_s=wci.DEFAULT_MAX_WALL_CLOCK_S,
|
|
519
|
+
)
|
|
520
|
+
|
|
521
|
+
outcome = watcher.watch(run_id="1")
|
|
522
|
+
|
|
523
|
+
assert outcome.result == wci.RESULT_GREEN, (
|
|
524
|
+
f"healthy long shard misreported as {outcome.result}: {outcome.reason}"
|
|
525
|
+
)
|
|
526
|
+
assert outcome.exit_code == 0
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
# ── The exit-code contract stays a single source of truth ────────────────────
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def test_exit_code_contract_is_complete_and_distinct():
|
|
533
|
+
codes = wci.EXIT_CODE
|
|
534
|
+
assert codes[wci.RESULT_GREEN] == 0
|
|
535
|
+
assert codes[wci.RESULT_FAILED] == 1
|
|
536
|
+
assert codes[wci.RESULT_HANG] == 2
|
|
537
|
+
assert codes[wci.RESULT_UNRESOLVED] == 3
|
|
538
|
+
assert codes[wci.RESULT_SUPERSEDED] == 4
|
|
539
|
+
assert codes[wci.RESULT_NO_VERDICT] == 5
|
|
540
|
+
# Every result must render a headline (no KeyError on a new verdict).
|
|
541
|
+
for result in codes:
|
|
542
|
+
outcome = wci.WatchOutcome(
|
|
543
|
+
result=result,
|
|
544
|
+
exit_code=codes[result],
|
|
545
|
+
required={},
|
|
546
|
+
failing_jobs=[],
|
|
547
|
+
signature=None,
|
|
548
|
+
likely_infra=False,
|
|
549
|
+
reason="x",
|
|
550
|
+
run_id="1",
|
|
551
|
+
polls=1,
|
|
552
|
+
elapsed_s=1.0,
|
|
553
|
+
)
|
|
554
|
+
assert wci.render_human_summary(outcome)
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
if __name__ == "__main__": # pragma: no cover
|
|
558
|
+
raise SystemExit(pytest.main([__file__, "-v"]))
|
|
@@ -3,8 +3,7 @@ revision = 3
|
|
|
3
3
|
requires-python = ">=3.11"
|
|
4
4
|
|
|
5
5
|
[options]
|
|
6
|
-
exclude-newer = "2026-
|
|
7
|
-
exclude-newer-span = "P7D"
|
|
6
|
+
exclude-newer = "2026-07-16T22:00:00Z"
|
|
8
7
|
|
|
9
8
|
[[package]]
|
|
10
9
|
name = "annotated-doc"
|
|
@@ -315,12 +314,13 @@ wheels = [
|
|
|
315
314
|
|
|
316
315
|
[[package]]
|
|
317
316
|
name = "pocketshell"
|
|
318
|
-
version = "0.4.
|
|
317
|
+
version = "0.4.37"
|
|
319
318
|
source = { editable = "." }
|
|
320
319
|
dependencies = [
|
|
321
320
|
{ name = "click" },
|
|
322
321
|
{ name = "google-auth" },
|
|
323
322
|
{ name = "pyyaml" },
|
|
323
|
+
{ name = "quse" },
|
|
324
324
|
{ name = "tmuxctl" },
|
|
325
325
|
]
|
|
326
326
|
|
|
@@ -346,6 +346,7 @@ requires-dist = [
|
|
|
346
346
|
{ name = "pytest", marker = "extra == 'dev'", specifier = ">=8.4.0" },
|
|
347
347
|
{ name = "pyyaml", specifier = ">=6.0" },
|
|
348
348
|
{ name = "qrcode", extras = ["pil"], marker = "extra == 'qr'", specifier = ">=7.4" },
|
|
349
|
+
{ name = "quse", specifier = "==0.0.11" },
|
|
349
350
|
{ name = "ruff", marker = "extra == 'dev'", specifier = ">=0.15.0" },
|
|
350
351
|
{ name = "tmuxctl", specifier = ">=0.3.3" },
|
|
351
352
|
]
|
|
@@ -484,6 +485,18 @@ pil = [
|
|
|
484
485
|
{ name = "pillow" },
|
|
485
486
|
]
|
|
486
487
|
|
|
488
|
+
[[package]]
|
|
489
|
+
name = "quse"
|
|
490
|
+
version = "0.0.11"
|
|
491
|
+
source = { registry = "https://pypi.org/simple" }
|
|
492
|
+
dependencies = [
|
|
493
|
+
{ name = "click" },
|
|
494
|
+
]
|
|
495
|
+
sdist = { url = "https://files.pythonhosted.org/packages/fd/5d/c659757cf4532e8f347476881caf998e4339f1e9dc7c71ee7887270f2069/quse-0.0.11.tar.gz", hash = "sha256:3728b89b5011638a48c68465870ec9be824844ce3b1e2a8f0a9a4d3dc888124b", size = 52456, upload-time = "2026-07-15T08:26:18.888Z" }
|
|
496
|
+
wheels = [
|
|
497
|
+
{ url = "https://files.pythonhosted.org/packages/d0/c1/53e5a3ab471833b0fec1dcaaa81affd07e2bbbefa059e793ca7b333934f2/quse-0.0.11-py3-none-any.whl", hash = "sha256:2fae4365f7c9c78c544e7afff5ce9aa2e5543892c3736402ff06bb1e3d16230a", size = 19051, upload-time = "2026-07-15T08:26:18Z" },
|
|
498
|
+
]
|
|
499
|
+
|
|
487
500
|
[[package]]
|
|
488
501
|
name = "rich"
|
|
489
502
|
version = "15.0.0"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|