@hasna/hooks 0.5.0 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -8
- package/bin/index.js +1756 -382
- package/bin/serve.js +5258 -0
- package/dist/cf/provision.d.ts +24 -0
- package/dist/config.d.ts +18 -0
- package/dist/db/legacy-import.d.ts +1 -1
- package/dist/db/migrations/004_hooks_table.d.ts +9 -0
- package/dist/db/pg-migrations.d.ts +1 -1
- package/dist/db/storage-sync.d.ts +26 -6
- package/dist/index.d.ts +18 -2
- package/dist/index.js +5351 -286
- package/dist/lib/custom-install.d.ts +19 -0
- package/dist/lib/manifest.d.ts +70 -0
- package/dist/lib/resolve.d.ts +19 -0
- package/dist/lib/run.d.ts +37 -0
- package/dist/lib/store.d.ts +69 -0
- package/dist/lib/sync.d.ts +35 -0
- package/dist/serve.d.ts +36 -0
- package/dist/storage.d.ts +2 -2
- package/dist/storage.js +133 -42
- package/hooks/codewith-native-common.test.ts +15 -2
- package/hooks/hook-scanoutput/README.md +151 -0
- package/hooks/hook-scanoutput/package.json +12 -0
- package/hooks/hook-scanoutput/src/hook.test.ts +217 -0
- package/hooks/hook-scanoutput/src/hook.ts +319 -0
- package/hooks/hook-workspace-repos-guard/README.md +63 -0
- package/hooks/hook-workspace-repos-guard/package.json +12 -0
- package/hooks/hook-workspace-repos-guard/src/hook.test.ts +312 -0
- package/hooks/hook-workspace-repos-guard/src/hook.ts +352 -0
- package/hooks/hook-workspace-repos-guard/tsconfig.json +21 -0
- package/hooks/mention-context/README.md +109 -0
- package/hooks/mention-context/package.json +9 -0
- package/hooks/mention-context/src/hasna-mention-context.py +1218 -0
- package/hooks/mention-context/src/hasna-mention-warm.py +521 -0
- package/hooks/mention-context/src/hook.test.ts +68 -0
- package/hooks/mention-context/src/test_run_capture.py +345 -0
- package/package.json +9 -5
|
@@ -0,0 +1,521 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
hasna-mention-warm — out-of-band cache warmer for hasna-mention-context.
|
|
4
|
+
|
|
5
|
+
WHY THIS EXISTS, MEASURED RATHER THAN ASSUMED (station01, 2026-08-05):
|
|
6
|
+
|
|
7
|
+
gh api graphql (branch+head+3 commits+5 PRs) 1.02 / 1.14 / 1.18 s
|
|
8
|
+
gh --version, no network at all 0.15 - 0.24 s (43 MB Go binary)
|
|
9
|
+
raw TLS+HTTP to api.github.com 0.33 - 0.55 s
|
|
10
|
+
todos projects --json 7.64 s, 1.35 MB
|
|
11
|
+
todos list --project <id> --json 5.5 - 7.1 s
|
|
12
|
+
hasna-mention-context total budget 0.900 s
|
|
13
|
+
|
|
14
|
+
The GitHub call needs more wall time than the hook is allowed to EXIST for, so
|
|
15
|
+
it was killed before it could ever write its cache — which left the cache cold,
|
|
16
|
+
which guaranteed the next prompt also timed out. A self-perpetuating cold cache,
|
|
17
|
+
visible as `degraded github (timeout)` on every single run and as a missing
|
|
18
|
+
`hasna__todos__gh.json`. The todos calls are 6-8x the whole budget and were
|
|
19
|
+
never even attempted inline.
|
|
20
|
+
|
|
21
|
+
Raising the hook's timeout is the wrong fix: it would breach the 900 ms deadline
|
|
22
|
+
on EVERY prompt that mentions a repo, to buy data that is near-static between
|
|
23
|
+
prompts. So the slow work moves here, where nothing is waiting on it, and the
|
|
24
|
+
hook becomes a pure cache reader.
|
|
25
|
+
|
|
26
|
+
This writes exactly the cache files the hook already reads, in exactly the shape
|
|
27
|
+
the hook's own `shape_github` produces — it imports the hook module rather than
|
|
28
|
+
restating its schema, so the two cannot drift.
|
|
29
|
+
|
|
30
|
+
Sources warmed, each on its own age threshold so one 5-minute cron self-paces:
|
|
31
|
+
|
|
32
|
+
gh branch, origin HEAD, last 3 commits, open PRs > 240 s
|
|
33
|
+
npm registry version + description > 540 s
|
|
34
|
+
slow todos project resolution + last 3 BUG rows > 6 h
|
|
35
|
+
|
|
36
|
+
Usage:
|
|
37
|
+
hasna-mention-warm.py # tokens from the hook's own log + seeds
|
|
38
|
+
hasna-mention-warm.py hasna/todos ... # explicit tokens
|
|
39
|
+
hasna-mention-warm.py --force # ignore age thresholds
|
|
40
|
+
hasna-mention-warm.py --verbose # per-source timing to stdout
|
|
41
|
+
|
|
42
|
+
Exit status is 0 unless a token was explicitly requested and every source for it
|
|
43
|
+
failed, so a cron wrapper can tell "nothing to do" from "everything broke".
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
import argparse
|
|
47
|
+
import importlib.util
|
|
48
|
+
import json
|
|
49
|
+
import os
|
|
50
|
+
import re
|
|
51
|
+
import subprocess
|
|
52
|
+
import sys
|
|
53
|
+
import tempfile
|
|
54
|
+
import time
|
|
55
|
+
|
|
56
|
+
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
57
|
+
HOOK_PATH = os.path.join(HERE, "hasna-mention-context.py")
|
|
58
|
+
|
|
59
|
+
# --------------------------------------------------------------------------
|
|
60
|
+
# Hosted-store credentials for the `todos` child process
|
|
61
|
+
#
|
|
62
|
+
# WHY, MEASURED RATHER THAN ASSUMED (station01, 2026-08-06, todos 17b532d3):
|
|
63
|
+
# cron does not source the shell profile and does not inherit BASH_ENV, so the
|
|
64
|
+
# HASNA_TODOS_* names are absent from this process, run_capture() spawns `todos`
|
|
65
|
+
# with subprocess.run(cmd) and NO env= argument, so the child inherits that same
|
|
66
|
+
# empty configuration and falls back SILENTLY to a stale on-box store at rc=0.
|
|
67
|
+
# Under `env -i HOME=... PATH=<the crontab PATH>`, same binary in both arms
|
|
68
|
+
# (/home/hasna/.bun/bin/todos and /home/hasna/.local/bin/todos are one inode):
|
|
69
|
+
#
|
|
70
|
+
# no loader HASNA_TODOS_* unset `todos projects --json` = 2353 rows
|
|
71
|
+
# `todos show 17b532d3` rc=1 "Could not resolve task ID"
|
|
72
|
+
# todos.env HASNA_TODOS_* set `todos projects --json` = 2793 rows
|
|
73
|
+
# `todos show 17b532d3` rc=0 ; NEGCTL `todos show zz000000` rc=1
|
|
74
|
+
#
|
|
75
|
+
# 2793 is the count a login shell sees (byte-identical payload, 1359664 B). The
|
|
76
|
+
# 440-row gap is only the visible half: the stale store cannot see rows created
|
|
77
|
+
# today at all, so warm_slow() was resolving projects and BUG rows against a
|
|
78
|
+
# store that does not contain recent work — and writing that into the cache the
|
|
79
|
+
# hook serves.
|
|
80
|
+
#
|
|
81
|
+
# setdefault, not overwrite: under cron nothing is set and the file supplies
|
|
82
|
+
# everything, while an interactive run that already carries a deliberate
|
|
83
|
+
# configuration keeps it. The shell equivalent (`set -a; . file`) overwrites;
|
|
84
|
+
# this is the additive form of the same idiom and cannot regress a working case.
|
|
85
|
+
# Failure here is never fatal — a warmer that cannot read the env file should
|
|
86
|
+
# still warm gh and npm.
|
|
87
|
+
CLOUD_ENV_PATH = os.path.join(os.path.expanduser("~"), ".hasna", "cloud", "todos.env")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def load_cloud_env(path=CLOUD_ENV_PATH):
|
|
91
|
+
"""Load KEY=VALUE lines from a cloud env file into os.environ WITHOUT
|
|
92
|
+
overriding anything already set. Returns the list of names applied, so a
|
|
93
|
+
caller can assert coverage; values are never returned, logged or printed."""
|
|
94
|
+
applied = []
|
|
95
|
+
try:
|
|
96
|
+
with open(path, "r", errors="replace") as f:
|
|
97
|
+
for line in f:
|
|
98
|
+
line = line.strip()
|
|
99
|
+
if not line or line.startswith("#"):
|
|
100
|
+
continue
|
|
101
|
+
if line.startswith("export "):
|
|
102
|
+
line = line[len("export "):].strip()
|
|
103
|
+
if "=" not in line:
|
|
104
|
+
continue
|
|
105
|
+
k, v = line.split("=", 1)
|
|
106
|
+
k = k.strip()
|
|
107
|
+
v = v.strip()
|
|
108
|
+
if len(v) >= 2 and v[0] == v[-1] and v[0] in ("'", '"'):
|
|
109
|
+
v = v[1:-1]
|
|
110
|
+
if not k:
|
|
111
|
+
continue
|
|
112
|
+
if os.environ.get(k):
|
|
113
|
+
continue
|
|
114
|
+
os.environ[k] = v
|
|
115
|
+
applied.append(k)
|
|
116
|
+
except Exception:
|
|
117
|
+
return applied
|
|
118
|
+
return applied
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
CLOUD_ENV_APPLIED = load_cloud_env()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def load_hook():
|
|
125
|
+
"""Import the hook module so cache paths, shapes and the sanitizer have ONE
|
|
126
|
+
definition. The hook only runs main() under __main__, so importing it is
|
|
127
|
+
side-effect free apart from reading /proc for its own start time."""
|
|
128
|
+
spec = importlib.util.spec_from_file_location("hasna_mention_context", HOOK_PATH)
|
|
129
|
+
mod = importlib.util.module_from_spec(spec)
|
|
130
|
+
spec.loader.exec_module(mod)
|
|
131
|
+
return mod
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
H = load_hook()
|
|
135
|
+
|
|
136
|
+
# Age above which each source is refetched. Chosen so a 5-minute cron keeps the
|
|
137
|
+
# gh entry comfortably inside the hook's serve-clean window with headroom for a
|
|
138
|
+
# missed tick, without refetching on every single pass.
|
|
139
|
+
AGE_GH = 240
|
|
140
|
+
AGE_NPM = 240 # cheap (~520 ms); refresh on essentially every pass
|
|
141
|
+
AGE_SLOW = 6 * 3600
|
|
142
|
+
|
|
143
|
+
LOCK_PATH = os.path.join(H.CACHE_DIR, ".warm.lock")
|
|
144
|
+
LOCK_STALE = 900 # a lock older than this is from a killed run
|
|
145
|
+
LOG_PATH = os.path.join(H.CACHE_DIR, "warm.jsonl")
|
|
146
|
+
|
|
147
|
+
# Tokens always kept warm regardless of whether anyone mentioned them recently.
|
|
148
|
+
# Deliberately small: every seed costs a GraphQL call per pass.
|
|
149
|
+
SEED_TOKENS = [("hasna", "todos")]
|
|
150
|
+
|
|
151
|
+
LOG_LOOKBACK_DAYS = 14
|
|
152
|
+
MAX_TOKENS_PER_RUN = 24
|
|
153
|
+
MAX_SLOW_PER_RUN = 2 # stagger the 7-42 s todos tier across passes
|
|
154
|
+
NET_TIMEOUT = 25.0 # generous: nothing is waiting on this process
|
|
155
|
+
TODOS_TIMEOUT = 90.0
|
|
156
|
+
|
|
157
|
+
VERBOSE = False
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def say(*a):
|
|
161
|
+
if VERBOSE:
|
|
162
|
+
print(*a, flush=True)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# --------------------------------------------------------------------------
|
|
166
|
+
# Lock — one warmer at a time, self-healing if a previous run was killed
|
|
167
|
+
# --------------------------------------------------------------------------
|
|
168
|
+
|
|
169
|
+
def acquire_lock():
|
|
170
|
+
try:
|
|
171
|
+
os.makedirs(H.CACHE_DIR, exist_ok=True)
|
|
172
|
+
try:
|
|
173
|
+
age = time.time() - os.path.getmtime(LOCK_PATH)
|
|
174
|
+
if age > LOCK_STALE:
|
|
175
|
+
os.unlink(LOCK_PATH)
|
|
176
|
+
else:
|
|
177
|
+
return False
|
|
178
|
+
except FileNotFoundError:
|
|
179
|
+
pass
|
|
180
|
+
fd = os.open(LOCK_PATH, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
|
181
|
+
os.write(fd, str(os.getpid()).encode())
|
|
182
|
+
os.close(fd)
|
|
183
|
+
return True
|
|
184
|
+
except FileExistsError:
|
|
185
|
+
return False
|
|
186
|
+
except Exception:
|
|
187
|
+
return True # never let lock trouble stop the warm
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def release_lock():
|
|
191
|
+
try:
|
|
192
|
+
os.unlink(LOCK_PATH)
|
|
193
|
+
except Exception:
|
|
194
|
+
pass
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
# --------------------------------------------------------------------------
|
|
198
|
+
# Token discovery — warm what people actually mention
|
|
199
|
+
# --------------------------------------------------------------------------
|
|
200
|
+
|
|
201
|
+
def tokens_from_log():
|
|
202
|
+
"""Read the hook's own log for repos mentioned recently. Self-tuning: the
|
|
203
|
+
repos you talk about are the repos that stay warm."""
|
|
204
|
+
import calendar
|
|
205
|
+
out, cutoff = [], time.time() - LOG_LOOKBACK_DAYS * 86400
|
|
206
|
+
try:
|
|
207
|
+
with open(H.LOG_PATH) as f:
|
|
208
|
+
lines = f.readlines()
|
|
209
|
+
except Exception:
|
|
210
|
+
return out
|
|
211
|
+
for line in lines[-4000:]:
|
|
212
|
+
try:
|
|
213
|
+
rec = json.loads(line)
|
|
214
|
+
# timegm, not mktime: the log stamps are UTC and mktime would read
|
|
215
|
+
# them as local, shifting the cutoff by the box's offset.
|
|
216
|
+
t = calendar.timegm(time.strptime(rec.get("at", ""), "%Y-%m-%dT%H:%M:%SZ"))
|
|
217
|
+
if t < cutoff:
|
|
218
|
+
continue
|
|
219
|
+
for tok in rec.get("tokens") or []:
|
|
220
|
+
if "/" in tok:
|
|
221
|
+
org, word = tok.split("/", 1)
|
|
222
|
+
if org in ("hasna", "hasnaxyz") and re.fullmatch(r"[a-z0-9][a-z0-9-]{0,38}", word):
|
|
223
|
+
out.append((org, word))
|
|
224
|
+
except Exception:
|
|
225
|
+
continue
|
|
226
|
+
return out
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def select_tokens(explicit):
|
|
230
|
+
"""Explicit tokens mean ONLY those tokens.
|
|
231
|
+
|
|
232
|
+
Merging them with the log set made `warm.py hasna/todos --force` fan out to
|
|
233
|
+
every repo mentioned in the last fortnight and run for over two minutes —
|
|
234
|
+
the operator asked about one repo and got the whole corpus.
|
|
235
|
+
"""
|
|
236
|
+
if explicit:
|
|
237
|
+
seen, ordered = set(), []
|
|
238
|
+
for tok in explicit:
|
|
239
|
+
if tok not in seen:
|
|
240
|
+
seen.add(tok)
|
|
241
|
+
ordered.append(tok)
|
|
242
|
+
return ordered
|
|
243
|
+
seen, ordered = set(), []
|
|
244
|
+
for tok in SEED_TOKENS + list(reversed(tokens_from_log())):
|
|
245
|
+
if tok not in seen:
|
|
246
|
+
seen.add(tok)
|
|
247
|
+
ordered.append(tok)
|
|
248
|
+
return ordered[:MAX_TOKENS_PER_RUN]
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def needs(org, word, source, max_age, force):
|
|
252
|
+
if force:
|
|
253
|
+
return True
|
|
254
|
+
_, age = H.cache_read(org, word, source)
|
|
255
|
+
return age is None or age > max_age
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
# --------------------------------------------------------------------------
|
|
259
|
+
# gh + npm — the hook's own probes, run without a deadline
|
|
260
|
+
# --------------------------------------------------------------------------
|
|
261
|
+
|
|
262
|
+
def warm_gh(org, word, tmpdir):
|
|
263
|
+
rc, out = H.run_capture(
|
|
264
|
+
[H.GH_BIN, "api", "graphql", "-f", "query=" + H.GQL,
|
|
265
|
+
"-f", "o=" + org, "-f", "n=" + word],
|
|
266
|
+
timeout=NET_TIMEOUT, tmpdir=tmpdir, tag=f"warm-gh-{org}-{word}",
|
|
267
|
+
)
|
|
268
|
+
if rc is None:
|
|
269
|
+
return "timeout"
|
|
270
|
+
try:
|
|
271
|
+
d = json.loads(out)
|
|
272
|
+
except Exception:
|
|
273
|
+
return "unparseable"
|
|
274
|
+
repo = (d.get("data") or {}).get("repository")
|
|
275
|
+
if repo is None:
|
|
276
|
+
errs = d.get("errors") or []
|
|
277
|
+
if any((e or {}).get("type") == "NOT_FOUND" for e in errs):
|
|
278
|
+
return "notfound"
|
|
279
|
+
return "error"
|
|
280
|
+
H.cache_write(org, word, "gh", H.shape_github(repo))
|
|
281
|
+
return "ok"
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def warm_npm(org, word, tmpdir):
|
|
285
|
+
body = os.path.join(tmpdir, f"warm-npm-{org}-{word}.body")
|
|
286
|
+
rc, code = H.run_capture(
|
|
287
|
+
[H.CURL_BIN, "-sS", "--max-time", str(NET_TIMEOUT),
|
|
288
|
+
"-o", body, "-w", "%{http_code}",
|
|
289
|
+
f"https://registry.npmjs.org/@{org}%2F{word}/latest"],
|
|
290
|
+
timeout=NET_TIMEOUT + 5, tmpdir=tmpdir, tag=f"warm-npmw-{org}-{word}",
|
|
291
|
+
)
|
|
292
|
+
if rc is None:
|
|
293
|
+
return "timeout"
|
|
294
|
+
if code.strip()[-3:] != "200":
|
|
295
|
+
return "http" + code.strip()[-3:]
|
|
296
|
+
try:
|
|
297
|
+
with open(body) as f:
|
|
298
|
+
d = json.load(f)
|
|
299
|
+
except Exception:
|
|
300
|
+
return "unparseable"
|
|
301
|
+
H.cache_write(org, word, "npm",
|
|
302
|
+
{"version": d.get("version"), "description": d.get("description")})
|
|
303
|
+
return "ok"
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
# --------------------------------------------------------------------------
|
|
307
|
+
# slow — todos project resolution and BUG rows
|
|
308
|
+
# --------------------------------------------------------------------------
|
|
309
|
+
|
|
310
|
+
def todos_json(args, tmpdir, tag):
|
|
311
|
+
"""Run a todos command and parse it, branching EXPLICITLY on the error
|
|
312
|
+
object. `todos` writes {"error": ...} to STDOUT, and the near-universal
|
|
313
|
+
consumer shape degrades that to an empty list — so a missing project and an
|
|
314
|
+
empty project print the same number. Returns (rows, error_string)."""
|
|
315
|
+
rc, out = H.run_capture(["todos"] + args, timeout=TODOS_TIMEOUT,
|
|
316
|
+
tmpdir=tmpdir, tag=tag)
|
|
317
|
+
if rc is None:
|
|
318
|
+
return None, "timeout"
|
|
319
|
+
try:
|
|
320
|
+
d = json.loads(out)
|
|
321
|
+
except Exception:
|
|
322
|
+
return None, "unparseable"
|
|
323
|
+
if isinstance(d, dict):
|
|
324
|
+
if d.get("error"):
|
|
325
|
+
return None, str(d["error"])[:120]
|
|
326
|
+
d = d.get("entries") or d.get("tasks") or d.get("projects") or d.get("data") or []
|
|
327
|
+
if not isinstance(d, list):
|
|
328
|
+
return None, "unexpected-shape"
|
|
329
|
+
return d, None
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def candidate_names(org, word):
|
|
333
|
+
"""The same filesystem convention probe_checkout uses, so project matching
|
|
334
|
+
and checkout matching cannot disagree. No hardcoded project ids."""
|
|
335
|
+
if org == "hasna":
|
|
336
|
+
return ["open-" + word, word]
|
|
337
|
+
return [word, "iapp-" + word, "platform-" + word, "open-" + word]
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def warm_slow(org, word, tmpdir, projects_cache):
|
|
341
|
+
"""Resolve the todos project for a repo and collect its last 3 BUG rows.
|
|
342
|
+
|
|
343
|
+
Duplicate project rows are the norm, not the exception: `open-todos` exists
|
|
344
|
+
FOUR times, with 3 / 101 / 58 / 150 task counters on three different
|
|
345
|
+
machine_ids, and the BUG rows are spread across all four (2+7+14+34 = 57).
|
|
346
|
+
Picking by name alone would silently return a 3-task fossil. So the primary
|
|
347
|
+
is chosen by MOST RECENT TASK ACTIVITY, which is derived from the data
|
|
348
|
+
rather than hardcoded, and the duplicate count is reported because an agent
|
|
349
|
+
filing a bug into the wrong row is the hazard this field exists to prevent.
|
|
350
|
+
Bugs are merged across every matching row, so a bug filed into a secondary
|
|
351
|
+
is still seen.
|
|
352
|
+
"""
|
|
353
|
+
if projects_cache.get("rows") is None:
|
|
354
|
+
rows, err = todos_json(["projects", "--json"], tmpdir, "warm-projects")
|
|
355
|
+
projects_cache["rows"] = rows if rows is not None else []
|
|
356
|
+
projects_cache["err"] = err
|
|
357
|
+
say(f" projects: {len(projects_cache['rows'])} rows err={err}")
|
|
358
|
+
rows = projects_cache["rows"]
|
|
359
|
+
if not rows:
|
|
360
|
+
return "no-projects"
|
|
361
|
+
|
|
362
|
+
names = candidate_names(org, word)
|
|
363
|
+
matches = [r for r in rows if str(r.get("name") or "") in names]
|
|
364
|
+
if not matches:
|
|
365
|
+
return "no-match"
|
|
366
|
+
|
|
367
|
+
per_project, bugs = [], []
|
|
368
|
+
for m in matches[:6]: # bound the fan-out
|
|
369
|
+
pid = m.get("id")
|
|
370
|
+
if not pid:
|
|
371
|
+
continue
|
|
372
|
+
trows, err = todos_json(
|
|
373
|
+
["list", "--project", str(pid), "--json", "--limit", "3000"],
|
|
374
|
+
tmpdir, f"warm-tasks-{str(pid)[:8]}")
|
|
375
|
+
if trows is None:
|
|
376
|
+
say(f" tasks {str(pid)[:8]}: err={err}")
|
|
377
|
+
continue
|
|
378
|
+
newest = ""
|
|
379
|
+
for r in trows:
|
|
380
|
+
created = str(r.get("created_at") or "")
|
|
381
|
+
newest = max(newest, created)
|
|
382
|
+
title = str(r.get("title") or "")
|
|
383
|
+
if title.strip().upper().startswith("BUG"):
|
|
384
|
+
bugs.append({
|
|
385
|
+
"date": created[:10] or "?",
|
|
386
|
+
"status": H.sanitize(r.get("status"), 12) or "?",
|
|
387
|
+
# Untrusted: written by other agents. Sanitized here AND
|
|
388
|
+
# again at render time.
|
|
389
|
+
"title": H.sanitize(title, 96),
|
|
390
|
+
"project": str(pid)[:8],
|
|
391
|
+
})
|
|
392
|
+
per_project.append({
|
|
393
|
+
"id": str(pid),
|
|
394
|
+
"name": H.sanitize(m.get("name"), 40),
|
|
395
|
+
"path": H.sanitize(m.get("path"), 120),
|
|
396
|
+
"tasks": len(trows),
|
|
397
|
+
"newest": newest,
|
|
398
|
+
})
|
|
399
|
+
|
|
400
|
+
if not per_project:
|
|
401
|
+
return "no-tasks-readable"
|
|
402
|
+
|
|
403
|
+
per_project.sort(key=lambda p: (p["newest"], p["tasks"]), reverse=True)
|
|
404
|
+
primary = per_project[0]
|
|
405
|
+
bugs.sort(key=lambda b: b["date"], reverse=True)
|
|
406
|
+
|
|
407
|
+
H.cache_write(org, word, "slow", {
|
|
408
|
+
"todos_key": primary["name"],
|
|
409
|
+
"project": primary,
|
|
410
|
+
"duplicates": len(per_project),
|
|
411
|
+
"bug_total": len(bugs),
|
|
412
|
+
"bugs": bugs[:3],
|
|
413
|
+
})
|
|
414
|
+
say(f" slow: primary={primary['id'][:8]} tasks={primary['tasks']} "
|
|
415
|
+
f"dups={len(per_project)} bugs={len(bugs)}")
|
|
416
|
+
return "ok"
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
# --------------------------------------------------------------------------
|
|
420
|
+
|
|
421
|
+
def main():
|
|
422
|
+
global VERBOSE
|
|
423
|
+
ap = argparse.ArgumentParser()
|
|
424
|
+
ap.add_argument("tokens", nargs="*", help="org/word, e.g. hasna/todos")
|
|
425
|
+
ap.add_argument("--force", action="store_true", help="ignore age thresholds")
|
|
426
|
+
ap.add_argument("--verbose", action="store_true")
|
|
427
|
+
ap.add_argument("--fast", action="store_true", help="skip the slow todos tier")
|
|
428
|
+
a = ap.parse_args()
|
|
429
|
+
VERBOSE = a.verbose
|
|
430
|
+
|
|
431
|
+
explicit = []
|
|
432
|
+
for t in a.tokens:
|
|
433
|
+
if "/" in t:
|
|
434
|
+
org, word = t.split("/", 1)
|
|
435
|
+
if org in ("hasna", "hasnaxyz"):
|
|
436
|
+
explicit.append((org, word))
|
|
437
|
+
|
|
438
|
+
if not acquire_lock():
|
|
439
|
+
say("another warmer holds the lock; exiting")
|
|
440
|
+
return 0
|
|
441
|
+
|
|
442
|
+
# A killed run (SIGTERM from a cron wrapper, or an operator timeout) does not
|
|
443
|
+
# run `finally`, so it would strand the lock and mute the warmer until the
|
|
444
|
+
# stale-lock window elapsed. Release on signal too.
|
|
445
|
+
import signal
|
|
446
|
+
|
|
447
|
+
def _bail(signum, frame):
|
|
448
|
+
release_lock()
|
|
449
|
+
os._exit(143)
|
|
450
|
+
|
|
451
|
+
for _sig in (signal.SIGTERM, signal.SIGINT, signal.SIGHUP):
|
|
452
|
+
try:
|
|
453
|
+
signal.signal(_sig, _bail)
|
|
454
|
+
except Exception:
|
|
455
|
+
pass
|
|
456
|
+
|
|
457
|
+
started = time.time()
|
|
458
|
+
result = {}
|
|
459
|
+
tmpdir = tempfile.mkdtemp(prefix="hasna-warm-")
|
|
460
|
+
projects_cache = {"rows": None}
|
|
461
|
+
slow_budget = MAX_SLOW_PER_RUN
|
|
462
|
+
try:
|
|
463
|
+
for org, word in select_tokens(explicit):
|
|
464
|
+
key = f"{org}/{word}"
|
|
465
|
+
r = {}
|
|
466
|
+
say(f" {key}")
|
|
467
|
+
if needs(org, word, "gh", AGE_GH, a.force):
|
|
468
|
+
s = time.time()
|
|
469
|
+
r["gh"] = warm_gh(org, word, tmpdir)
|
|
470
|
+
say(f" gh: {r['gh']} in {int((time.time()-s)*1000)}ms")
|
|
471
|
+
if needs(org, word, "npm", AGE_NPM, a.force):
|
|
472
|
+
s = time.time()
|
|
473
|
+
r["npm"] = warm_npm(org, word, tmpdir)
|
|
474
|
+
say(f" npm: {r['npm']} in {int((time.time()-s)*1000)}ms")
|
|
475
|
+
# The slow tier costs 7-42 s per token because each todos call is
|
|
476
|
+
# 5-7 s and a repo can match several project rows. Staggering it
|
|
477
|
+
# keeps any single pass short; the 6 h threshold means a token comes
|
|
478
|
+
# round again long before its entry expires.
|
|
479
|
+
if not a.fast and slow_budget > 0 and \
|
|
480
|
+
needs(org, word, "slow", AGE_SLOW, a.force):
|
|
481
|
+
s = time.time()
|
|
482
|
+
r["slow"] = warm_slow(org, word, tmpdir, projects_cache)
|
|
483
|
+
slow_budget -= 1
|
|
484
|
+
say(f" slow: {r['slow']} in {int((time.time()-s)*1000)}ms")
|
|
485
|
+
if r:
|
|
486
|
+
result[key] = r
|
|
487
|
+
finally:
|
|
488
|
+
import shutil
|
|
489
|
+
shutil.rmtree(tmpdir, ignore_errors=True)
|
|
490
|
+
release_lock()
|
|
491
|
+
|
|
492
|
+
try:
|
|
493
|
+
H.cache_prune()
|
|
494
|
+
with open(LOG_PATH, "a") as f:
|
|
495
|
+
f.write(json.dumps({
|
|
496
|
+
"at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
497
|
+
"elapsed_ms": int((time.time() - started) * 1000),
|
|
498
|
+
"result": result,
|
|
499
|
+
}) + "\n")
|
|
500
|
+
except Exception:
|
|
501
|
+
pass
|
|
502
|
+
|
|
503
|
+
say(f"done in {int((time.time()-started)*1000)}ms: {result}")
|
|
504
|
+
if explicit and result:
|
|
505
|
+
for org, word in explicit:
|
|
506
|
+
if any(v == "ok" for v in (result.get(f"{org}/{word}") or {}).values()):
|
|
507
|
+
return 0
|
|
508
|
+
return 1
|
|
509
|
+
return 0
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
if __name__ == "__main__":
|
|
513
|
+
try:
|
|
514
|
+
sys.exit(main())
|
|
515
|
+
except KeyboardInterrupt:
|
|
516
|
+
release_lock()
|
|
517
|
+
sys.exit(1)
|
|
518
|
+
except Exception as e:
|
|
519
|
+
release_lock()
|
|
520
|
+
print("warm failed:", type(e).__name__, str(e)[:200], file=sys.stderr)
|
|
521
|
+
sys.exit(1)
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { existsSync } from "node:fs";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* The mention-context hook is Python, and its regression suite is Python. This
|
|
7
|
+
* wrapper exists so that suite runs under the repository's existing `bun test`
|
|
8
|
+
* step, with no CI workflow change and no second command for a contributor to
|
|
9
|
+
* remember.
|
|
10
|
+
*
|
|
11
|
+
* It deliberately FAILS rather than skips when python3 is missing. A skipped
|
|
12
|
+
* test and a passing one are indistinguishable in the summary line, and this
|
|
13
|
+
* suite gates a defect that already reached production once: repositories were
|
|
14
|
+
* reported carrying each other's git HEAD, because concurrent probes shared one
|
|
15
|
+
* temp file.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
const SUITE = join(import.meta.dir, "test_run_capture.py");
|
|
19
|
+
|
|
20
|
+
describe("mention-context python regression suite", () => {
|
|
21
|
+
test("python3 is available to run it", () => {
|
|
22
|
+
const found = Bun.which("python3");
|
|
23
|
+
expect(
|
|
24
|
+
found,
|
|
25
|
+
"python3 is not on PATH; the mention-context regression suite cannot run, " +
|
|
26
|
+
"and a silent skip here would read as a pass",
|
|
27
|
+
).toBeTruthy();
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
test("the suite file is present", () => {
|
|
31
|
+
expect(existsSync(SUITE), `missing suite at ${SUITE}`).toBe(true);
|
|
32
|
+
});
|
|
33
|
+
|
|
34
|
+
test(
|
|
35
|
+
"concurrent captures do not cross temp files",
|
|
36
|
+
async () => {
|
|
37
|
+
const python = Bun.which("python3");
|
|
38
|
+
if (!python) throw new Error("python3 not on PATH");
|
|
39
|
+
|
|
40
|
+
const proc = Bun.spawn([python, SUITE, "-v"], {
|
|
41
|
+
stdout: "pipe",
|
|
42
|
+
stderr: "pipe",
|
|
43
|
+
});
|
|
44
|
+
const [stdout, stderr, exitCode] = await Promise.all([
|
|
45
|
+
new Response(proc.stdout).text(),
|
|
46
|
+
new Response(proc.stderr).text(),
|
|
47
|
+
proc.exited,
|
|
48
|
+
]);
|
|
49
|
+
|
|
50
|
+
// unittest writes its report to stderr; surface both so a failure here is
|
|
51
|
+
// diagnosable from the bun test output alone rather than needing a re-run.
|
|
52
|
+
if (exitCode !== 0) {
|
|
53
|
+
throw new Error(
|
|
54
|
+
`python regression suite failed (exit ${exitCode})\n` +
|
|
55
|
+
`--- stderr ---\n${stderr}\n--- stdout ---\n${stdout}`,
|
|
56
|
+
);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// Assert the suite actually ran tests. `unittest` exits 0 when it collects
|
|
60
|
+
// nothing, so a zero exit alone cannot distinguish "all passed" from
|
|
61
|
+
// "nothing ran" — which is the same class of vacuous check this suite was
|
|
62
|
+
// written to close.
|
|
63
|
+
expect(stderr).toMatch(/Ran [1-9]\d* tests?/);
|
|
64
|
+
expect(stderr).toMatch(/\bOK\b/);
|
|
65
|
+
},
|
|
66
|
+
120_000, // the suite forces timed interleavings; a loaded machine needs room
|
|
67
|
+
);
|
|
68
|
+
});
|