pi-crew 0.9.48 → 0.9.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/AGENTS.md +18 -0
  2. package/CHANGELOG.md +105 -0
  3. package/dist/build-meta.json +22 -12
  4. package/dist/index.mjs +430 -388
  5. package/dist/index.mjs.map +3 -3
  6. package/docs/decisions/2026-07-24-oidc-trusted-publishing.md +112 -0
  7. package/package.json +2 -2
  8. package/skills/.gitkeep +0 -0
  9. package/skills/distill-persona/BUILD-NOTES.md +55 -0
  10. package/skills/distill-persona/SKILL.md +612 -0
  11. package/skills/distill-persona/UPGRADE-LOG-RESEARCH-SKILLS.md +100 -0
  12. package/skills/distill-persona/references/coverage-manifest.md +65 -0
  13. package/skills/distill-persona/references/distillation-field-synthesis-pass2.md +59 -0
  14. package/skills/distill-persona/references/distillation-field-synthesis.md +108 -0
  15. package/skills/distill-persona/references/handoff.md +42 -0
  16. package/skills/distill-persona/references/research/lesson-memory-shortcut.md +33 -0
  17. package/skills/distill-persona/references/research/r1-a-examples.md +23 -0
  18. package/skills/distill-persona/references/research/r1-b-scripts.md +26 -0
  19. package/skills/distill-persona/references/research/r1-c-human-readme.md +31 -0
  20. package/skills/distill-persona/references/research/r1-d-tests.md +28 -0
  21. package/skills/distill-persona/references/research/r1-verification.md +36 -0
  22. package/skills/distill-persona/references/research/r2-low-yield.md +26 -0
  23. package/skills/distill-persona/scripts/fidelity_eval.py +244 -0
  24. package/skills/distill-persona/scripts/validate-skill-structure.mjs +177 -0
  25. package/skills/distill-software/BUILD-NOTES.md +56 -0
  26. package/skills/distill-software/SKILL.md +302 -0
  27. package/skills/distill-software/references/handoff.md +47 -0
  28. package/skills/distill-software/scripts/code_dna.py +290 -0
  29. package/skills/research/DISTILLATION-PROCESS-CHECKLIST.md +120 -0
  30. package/skills/research/EXCAVATION-CHECKLIST.md +142 -0
  31. package/skills/research/FIDELITY.md +180 -0
  32. package/skills/research/SKILL.md +432 -0
  33. package/skills/research/references/anti-patterns.md +184 -0
  34. package/skills/research/references/fidelity.md +241 -0
  35. package/skills/research/references/handoff.md +48 -0
  36. package/skills/research/references/research-protocol.md +162 -0
  37. package/skills/research/references/source-inventory.md +135 -0
  38. package/skills/research/references/verified-models.md +163 -0
  39. package/skills/research/scripts/__pycache__/safe_io.cpython-312.pyc +0 -0
  40. package/skills/research/scripts/code_dna.py +233 -0
  41. package/skills/research/scripts/emit_run_summary.py +142 -0
  42. package/skills/research/scripts/safe_io.py +314 -0
  43. package/skills/research/scripts/source_evaluator.py +234 -0
  44. package/skills/research/scripts/validate-skill-structure.mjs +177 -0
  45. package/skills/research/scripts/verify_citations.py +225 -0
  46. package/skills/security-priority.json +28 -0
  47. package/src/config/config.ts +1 -0
  48. package/src/config/role-tools.ts +6 -3
  49. package/src/config/types.ts +8 -0
  50. package/src/runtime/background-runner.ts +11 -16
  51. package/src/runtime/heartbeat-watcher.ts +28 -1
  52. package/src/runtime/task-runner.ts +165 -119
  53. package/src/schema/config-schema.ts +1 -0
  54. package/src/utils/gh-protocol.ts +9 -8
  55. package/workflows/distill.workflow.md +198 -0
@@ -0,0 +1,177 @@
1
+ #!/usr/bin/env node
2
+ // validate-skill-structure.mjs — structural invariant checker for generated distill skills.
3
+ // Implements F10 (awesome-persona-distill-skills finding): hard-fail if any structural assertion fails.
4
+ // Run AFTER Phase 3 build, BEFORE Phase 4 behavioral fidelity. Self-contained (stdlib only).
5
+ //
6
+ // Usage:
7
+ // node validate-skill-structure.mjs <path-to-SKILL.md>
8
+ // node validate-skill-structure.mjs <skill-dir>
9
+ //
10
+ // Exit codes: 0 = all assertions pass (all-green → may ship); 1 = one or more failed (iterate).
11
+
12
+ import { readFileSync, readdirSync, statSync, existsSync } from 'node:fs';
13
+ import { join, dirname, basename } from 'node:path';
14
+
15
+ const target = process.argv.find((a) => !a.startsWith('-') && a !== process.argv[0] && a !== process.argv[1]);
16
+ if (!target) {
17
+ console.error('Usage: validate-skill-structure.mjs <path-to-SKILL.md | skill-dir> [--engine]');
18
+ process.exit(2);
19
+ }
20
+
21
+ // Resolve to a SKILL.md path
22
+ let skillPath = target;
23
+ if (statSync(target).isDirectory()) {
24
+ skillPath = join(target, 'SKILL.md');
25
+ }
26
+ if (!existsSync(skillPath)) {
27
+ console.error(`✗ Not found: ${skillPath}`);
28
+ process.exit(2);
29
+ }
30
+
31
+ const src = readFileSync(skillPath, 'utf8');
32
+ const dir = dirname(skillPath);
33
+ const name = basename(dir);
34
+ const isEngine = process.argv.includes('--engine');
35
+ const PLACEHOLDERS = [/<person>/i, /<target>/i, /<topic>/i, /<field>/i, /TODO/i, /TBD/i, /XXX/i, /YYYY-MM-DD/i, /\.\.\.\s*<\/?/i];
36
+
37
+ const failures = [];
38
+ const passes = [];
39
+ const check = (label, ok, detail = '') => {
40
+ (ok ? passes : failures).push(ok ? ` ✓ ${label}` : ` ✗ ${label}${detail ? ' — ' + detail : ''}`);
41
+ };
42
+
43
+ // --- Frontmatter ---
44
+ const fm = src.match(/^---\n([\s\S]*?)\n---/);
45
+ check('frontmatter block present', !!fm, 'no --- block at top');
46
+ const fmText = fm ? fm[1] : '';
47
+ const fmField = (key) => {
48
+ const m = fmText.match(new RegExp(`^${key}:\\s*(.+)$`, 'm'));
49
+ return m ? m[1].trim() : null;
50
+ };
51
+ const hasFm = (key) => new RegExp(`^${key}:`, 'm').test(fmText);
52
+
53
+ check('frontmatter: name', hasFm('name'));
54
+ check('frontmatter: description', hasFm('description'));
55
+ const desc = fmField('description');
56
+ check('frontmatter: description ≤1 sentence (one terminal punct or one clause)',
57
+ desc ? desc.split(/[.。!!??]/).filter(Boolean).length <= 2 : false,
58
+ desc ? `got: "${desc.slice(0, 60)}…"` : 'missing');
59
+ check('frontmatter: triggers', hasFm('triggers') || hasFm('trigger'), 'no triggers field');
60
+ if (!isEngine) check('frontmatter: distilled (staleness date)', hasFm('distilled') || hasFm('调研时间'), 'no distilled/调研时间 staleness anchor');
61
+ const distilled = fmField('distilled') || fmField('调研时间');
62
+ check('frontmatter: distilled is valid date (YYYY-MM-DD)',
63
+ !distilled || /^\d{4}-\d{2}-\d{2}/.test(distilled),
64
+ distilled ? `got "${distilled}"` : '');
65
+ if (!isEngine) check('frontmatter: target (person|topic|software)', hasFm('target'), 'no target field');
66
+
67
+ // --- Body sections ---
68
+ check('Agentic Protocol section present', /回答工作流|Agentic Protocol/i.test(src));
69
+ check('Agentic Protocol Step 1', /Step 1|第一步|步骤 1|### 1\b/i.test(src));
70
+ check('Agentic Protocol Step 2', /Step 2|第二步|步骤 2|### 2\b/i.test(src));
71
+ check('Agentic Protocol Step 3', /Step 3|第三步|步骤 3|### 3\b/i.test(src));
72
+
73
+ // honest boundaries: count items (numbered list or bullets) under an honest-boundaries heading
74
+ function countListItems(haystack, headingRe, windowChars = 2000) {
75
+ const m = haystack.match(headingRe);
76
+ if (!m) return { found: false, count: 0, block: '' };
77
+ const after = haystack.slice(m.index);
78
+ const section = after.match(/([\s\S]*?)\n##(?=[^#])/);
79
+ const block = section ? section[1] : after.slice(0, windowChars);
80
+ // count BOTH list items AND table data rows (rows with | content |, excluding separator rows)
81
+ const listItems = (block.match(/^\s*(?:\d+[.)]|[-*])\s+\S/gm) || []).length;
82
+ const tableRows = (block.match(/^\s*\|(?![\s:|-]+\|?\s*$).+\|/gm) || []).length;
83
+ return { found: true, count: listItems + tableRows, block };
84
+ }
85
+ if (!isEngine) {
86
+ const boundary = countListItems(src, /诚实边界|honest boundar(?:y|ies)/i);
87
+ check('honest boundaries (M11) ≥3', boundary.count >= 3, `found ${boundary.count} items`);
88
+ }
89
+
90
+ // --- Anti-drift tables (M9a 内在张力, M9b 反例黑名单, M12 fallback tree) — Darwin gap #2: validator previously skipped these
91
+ if (!isEngine) {
92
+ const tension = countListItems(src, /M9a|内在张力|inner tension|internal tension/i);
93
+ check('内在张力 (M9a) ≥3 tension pairs', tension.count >= 3, tension.found ? `found ${tension.count} items` : 'section missing');
94
+
95
+ const blacklist = countListItems(src, /M9b|反例黑名单|anti.?pattern blacklist/i);
96
+ check('反例黑名单 (M9b) ≥7 rows', blacklist.count >= 7, blacklist.found ? `found ${blacklist.count} rows` : 'section missing');
97
+
98
+ const fallback = countListItems(src, /M12|失败模式.*[Ff]allback|[Ff]allback\s*树/i);
99
+ check('失败模式Fallback树 (M12) ≥8 rows', fallback.count >= 8, fallback.found ? `found ${fallback.count} rows` : 'section missing');
100
+ }
101
+
102
+ // --- FIDELITY.md companion artifact (Darwin gap #3: ship-gate requires it but nothing checked)
103
+ const fidelityPath = join(dir, 'FIDELITY.md');
104
+ if (existsSync(fidelityPath)) {
105
+ const fm2 = readFileSync(fidelityPath, 'utf8');
106
+ const totalMatch = fm2.match(/(?:总分|total)[::\s*]*\*{0,2}([0-9]+)\s*\*?\s*(?:\/|/|\sout\sof\s)\s*\*?\s*100/i);
107
+ check('FIDELITY.md total score present', !!totalMatch, totalMatch ? `=${totalMatch[1]}/100` : 'no /100 score found');
108
+ const qCount = (fm2.match(/(?:Q[1-5]|问题[1-5]|question\s*[1-5])/gi) || []).length;
109
+ check('FIDELITY.md ≥5 test questions', qCount >= 5, `found ${qCount} question refs`);
110
+ check('FIDELITY.md flags single-agent/self-score caveat', /单\s*agent|single.?agent|self.?score|upper.?bound|independent/i.test(fm2), 'add single-agent upper-bound caveat');
111
+ } else if (!isEngine) {
112
+ check('FIDELITY.md companion present', false, 'no FIDELITY.md in skill dir');
113
+ }
114
+
115
+ // --- EXCAVATION-CHECKLIST.md (Phase 1 protocol: track + verify each part was really read)
116
+ const checklistPath = join(dir, 'EXCAVATION-CHECKLIST.md');
117
+ if (existsSync(checklistPath)) {
118
+ const cl = readFileSync(checklistPath, 'utf8');
119
+ // dangling rows: status ⬜ not-started or ⏳ reading left at ship time = silently skipped/forgotten
120
+ const dangling = (cl.match(/\|\s*[⬜⏳][^|]*\|/gu) || []).length;
121
+ check('checklist: no dangling ⬜/⏳ rows (every part resolved)', dangling === 0, dangling ? `${dangling} row(s) not-started/reading — resolve to ✅📄/⏭/🧠` : '');
122
+ // memory ratio: parse "memory-ratio: NN%" if declared
123
+ const ratioMatch = cl.match(/memory-ratio[:\s]*([0-9]+)\s*%/i);
124
+ if (ratioMatch) {
125
+ const ratio = parseInt(ratioMatch[1], 10);
126
+ check('checklist: 🧠 memory-ratio ≤30%', ratio <= 30, `declared ${ratio}% (>30% = recap, not distillation)`);
127
+ } else {
128
+ check('checklist: declares memory-ratio', false, 'add "🧠 memory-ratio: NN% (X/Y findings)" header line');
129
+ }
130
+ // proof-of-read present: at least one verbatim-quote-with-location cell (heuristic — a ✅ row should carry a quote)
131
+ const proofCells = (cl.match(/"[^"]{8,}"[^|]*/g) || []).length;
132
+ check('checklist: ≥1 proof-of-read (verbatim quote in a ✅ row)', proofCells >= 1, `${proofCells} quote cell(s) found`);
133
+ } else if (!isEngine) {
134
+ check('EXCAVATION-CHECKLIST.md present', false, 'no excavation checklist — Phase 1 protocol requires one');
135
+ }
136
+
137
+ // --- DISTILLATION-PROCESS-CHECKLIST.md (whole-pipeline tracking + the 3-empty-rounds deep-dive gate)
138
+ const procPath = join(dir, 'DISTILLATION-PROCESS-CHECKLIST.md');
139
+ if (existsSync(procPath)) {
140
+ const pc = readFileSync(procPath, 'utf8');
141
+ // dangling phases: ⬜/⏳ left in the phase-progress table at ship = a phase skipped
142
+ const danglingPhases = (pc.match(/\|\s*[⬜⏳][^|]*\|/gu) || []).length;
143
+ check('process: no dangling ⬜/⏳ phases (every phase completed)', danglingPhases === 0, danglingPhases ? `${danglingPhases} phase(s) not done` : '');
144
+ // deep-dive round log present
145
+ check('process: deep-dive round log present', /round log|\| *Round.*New findings/i.test(pc), 'add the round-log table');
146
+ // 3-empty-rounds gate recorded (≥3 consecutive zero-new rounds before leaving a phase)
147
+ const gateFired = /gate fires|3 consecutive (empty|zero)|GATE FIRES/i.test(pc);
148
+ check('process: 3-empty-rounds gate recorded (≥3 zero-new rounds)', gateFired, 'record a round-log row marked gate-fired before declaring any research phase done');
149
+ } else if (!isEngine) {
150
+ check('DISTILLATION-PROCESS-CHECKLIST.md present', false, 'no process checklist — every phase + the 3-round deep-dive gate must be tracked');
151
+ }
152
+
153
+ // --- Placeholders ---
154
+ const foundPlaceholders = PLACEHOLDERS.filter((re) => re.test(src));
155
+ if (!isEngine) check('no unresolved placeholder text', foundPlaceholders.length === 0,
156
+ foundPlaceholders.map((re) => src.match(re)?.[0]).filter(Boolean).join(', '));
157
+
158
+ // --- Self-containment (F9): references/sources should not dangle outside the dir ---
159
+ const refsDir = join(dir, 'references');
160
+ if (!isEngine) check('self-contained: no external repo paths in body (e.g. 07-调研/)',
161
+ !/[\w/-]*调研与分析?\/|src\/|source\//i.test(src.replace(/```[\s\S]*?```/g, '')),
162
+ 'body references paths outside skill dir');
163
+
164
+ // --- Software-specific assertions (if target: software) ---
165
+ if (/target:\s*software/i.test(fmText) || /code-dna|代码表达DNA|code expression/i.test(src)) {
166
+ check('[software] Code Expression-DNA section present', /代码表达DNA|Code Expression-DNA|code.?dna/i.test(src));
167
+ check('[software] toolchain matrix present', /toolchain|eslint|oxlint|biome|tsconfig/i.test(src));
168
+ check('[software] distilled_against commit anchor', hasFm('distilled_against') || /distilled_against/i.test(src));
169
+ }
170
+
171
+ // --- Report ---
172
+ console.log(`\nvalidate-skill-structure: ${name}`);
173
+ console.log(`path: ${skillPath}\n`);
174
+ passes.forEach((l) => console.log(l));
175
+ failures.forEach((l) => console.log(l));
176
+ console.log(`\n${passes.length} pass, ${failures.length} fail — ${failures.length === 0 ? '✅ ALL-GREEN (may ship)' : '🔴 NOT READY (iterate Phase 2→3)'}${isEngine ? ' [engine mode — generated-skill checks skipped]' : ''}\n`);
177
+ process.exit(failures.length === 0 ? 0 : 1);
@@ -0,0 +1,225 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ verify_citations.py — borrowed from Geek-skills-deep-research, ported for the
4
+ research skill (F13: wired INTO the Agentic Protocol Step 2, never orphaned).
5
+
6
+ Structural citation-integrity checker: resolves every citation marker `[n]` in
7
+ the report against the LOCAL source pool (sources.json); flags unresolved
8
+ citations, dangling references, non-sequential numbering, and source
9
+ concentration (> 25% from any single source).
10
+
11
+ ⚠️ This is a STRUCTURAL check only — it does NOT make network requests.
12
+ It does NOT verify HTTP 404s, redirect chains, or content-drift. A URL that
13
+ has gone dead or been hijacked will still pass if it matches the local source
14
+ pool. For live-URL verification, use a separate network checker (e.g.
15
+ WebFetch HEAD) before relying on citation-liveness.
16
+
17
+ Stdlib only. Python 3.9+.
18
+
19
+ Usage:
20
+ python3 verify_citations.py <report.md> <sources.json> [--output results.json]
21
+ python3 verify_citations.py --self-test # run a quick self-test
22
+
23
+ Exit codes:
24
+ 0 no unresolved / no dangling / no concentration issues
25
+ 1 one or more issues found (see --output report)
26
+ 2 usage error
27
+ """
28
+ import json
29
+ import re
30
+ import sys
31
+ from pathlib import Path
32
+ from urllib.parse import urlparse
33
+
34
+
35
+ SOURCE_CONCENTRATION_LIMIT = 0.25 # No single source > 25% of citations
36
+
37
+
38
+ def normalize_url(url: str) -> str:
39
+ """Lowercase + strip trailing slash + strip common tracking params."""
40
+ url = url.strip().rstrip("/")
41
+ parsed = urlparse(url.lower())
42
+ drop_params = {"utm_source", "utm_medium", "utm_campaign", "ref", "fbclid", "gclid"}
43
+ # Best-effort: keep query but strip params in drop_params
44
+ if parsed.query:
45
+ parts = [p for p in parsed.query.split("&") if p.split("=")[0] not in drop_params]
46
+ query = "&".join(parts)
47
+ else:
48
+ query = ""
49
+ return f"{parsed.scheme}://{parsed.netloc}{parsed.path}{('?' + query) if query else ''}".rstrip("/")
50
+
51
+
52
+ def url_signature(url: str):
53
+ """Robust (domain, path_segments[:3]) for matching."""
54
+ parsed = urlparse(url.lower())
55
+ host = parsed.hostname or ""
56
+ if host.startswith("www."):
57
+ host = host[4:]
58
+ segments = [s for s in parsed.path.split("/") if s][:3]
59
+ return (host, tuple(segments))
60
+
61
+
62
+ def extract_citations(text: str):
63
+ """Find [n] markers in body. Returns list of int indices."""
64
+ return [int(m) for m in re.findall(r"\[(\d+)\]", text)]
65
+
66
+
67
+ def extract_references(text: str):
68
+ """Find reference list. Returns dict {int_idx: (label, url)}."""
69
+ refs = {}
70
+ # Match common patterns: [1] Label — URL or [1] Label - http...
71
+ for m in re.finditer(r"^\s*\[(\d+)\]\s+([^—\-]+?)\s*[—\-]\s*(\S+)", text, re.MULTILINE):
72
+ refs[int(m.group(1))] = (m.group(2).strip(), m.group(3).strip())
73
+ return refs
74
+
75
+
76
+ def main(argv):
77
+ if not argv or argv[0] in ("-h", "--help"):
78
+ print(__doc__)
79
+ return 0
80
+ if argv[0] == "--self-test":
81
+ return self_test()
82
+ if len(argv) < 2:
83
+ print("Usage: verify_citations.py <report.md> <sources.json> [--output results.json]", file=sys.stderr)
84
+ return 2
85
+
86
+ report_path = Path(argv[0])
87
+ sources_path = Path(argv[1])
88
+ output_path = None
89
+ if len(argv) >= 4 and argv[2] == "--output":
90
+ output_path = Path(argv[3])
91
+
92
+ report = report_path.read_text(encoding="utf-8")
93
+ sources_data = json.loads(sources_path.read_text(encoding="utf-8"))
94
+
95
+ sources = sources_data.get("sources", [])
96
+ sources_normalized = {normalize_url(s["url"]): s for s in sources}
97
+
98
+ citations = extract_citations(report)
99
+ refs = extract_references(report)
100
+
101
+ issues = []
102
+ stats = {
103
+ "report_citations": len(citations),
104
+ "unique_citations": len(set(citations)),
105
+ "references_declared": len(refs),
106
+ "sources_in_pool": len(sources),
107
+ "concentration": {},
108
+ }
109
+
110
+ # 1. Every [n] has a matching reference
111
+ for c in set(citations):
112
+ if c not in refs:
113
+ issues.append({
114
+ "severity": "fatal",
115
+ "type": "unresolved_citation",
116
+ "detail": f"Citation [{c}] has no matching reference entry",
117
+ })
118
+
119
+ # 2. Every reference URL exists in source pool
120
+ for idx, (label, url) in refs.items():
121
+ if normalize_url(url) not in sources_normalized:
122
+ # Try signature match
123
+ sig = url_signature(url)
124
+ match = None
125
+ for s_url, s in sources_normalized.items():
126
+ if url_signature(s_url) == sig:
127
+ match = s
128
+ break
129
+ if not match:
130
+ issues.append({
131
+ "severity": "fatal",
132
+ "type": "url_not_in_source_pool",
133
+ "detail": f"Reference [{idx}] URL '{url}' not found in source pool",
134
+ })
135
+
136
+ # 3. Dangling references (in list but never cited)
137
+ for idx in refs:
138
+ if idx not in set(citations):
139
+ issues.append({
140
+ "severity": "warn",
141
+ "type": "dangling_reference",
142
+ "detail": f"Reference [{idx}] declared but never cited in body",
143
+ })
144
+
145
+ # 4. Sequential numbering, no gaps
146
+ if refs:
147
+ max_idx = max(refs.keys())
148
+ missing = [i for i in range(1, max_idx + 1) if i not in refs]
149
+ if missing:
150
+ issues.append({
151
+ "severity": "warn",
152
+ "type": "non_sequential_numbering",
153
+ "detail": f"Reference numbering gaps: {missing}",
154
+ })
155
+
156
+ # 5. Source concentration
157
+ cite_counts = {}
158
+ for c in citations:
159
+ if c in refs:
160
+ label = refs[c][0]
161
+ cite_counts[label] = cite_counts.get(label, 0) + 1
162
+ total = max(len(citations), 1)
163
+ for label, count in cite_counts.items():
164
+ ratio = count / total
165
+ stats["concentration"][label] = round(ratio, 3)
166
+ if ratio > SOURCE_CONCENTRATION_LIMIT:
167
+ issues.append({
168
+ "severity": "fatal",
169
+ "type": "source_concentration",
170
+ "detail": f"Source '{label}' has {count}/{total} citations ({ratio*100:.1f}%) > {SOURCE_CONCENTRATION_LIMIT*100:.0f}% limit",
171
+ })
172
+
173
+ result = {
174
+ "ok": all(i["severity"] != "fatal" for i in issues),
175
+ "stats": stats,
176
+ "issues": issues,
177
+ }
178
+
179
+ if output_path:
180
+ output_path.write_text(json.dumps(result, indent=2), encoding="utf-8")
181
+ else:
182
+ print(json.dumps(result, indent=2))
183
+
184
+ return 0 if result["ok"] else 1
185
+
186
+
187
+ def self_test():
188
+ """Run a quick self-test on a fabricated report."""
189
+ report = """# Test
190
+
191
+ Claim one [1]. Claim two [2]. Claim three [3]. Claim four [4]. Claim five [5].
192
+
193
+ ## References
194
+
195
+ [1] Author A — https://example.com/a
196
+ [2] Author B — https://example.org/b
197
+ [3] Author C — https://example.net/c
198
+ [4] Author D — https://example.io/d
199
+ [5] Author E — https://example.dev/e
200
+ """
201
+ sources = {
202
+ "sources": [
203
+ {"url": "https://example.com/a", "title": "Author A's paper"},
204
+ {"url": "https://example.org/b", "title": "Author B's notes"},
205
+ {"url": "https://example.net/c", "title": "Author C's x"},
206
+ {"url": "https://example.io/d", "title": "Author D's y"},
207
+ {"url": "https://example.dev/e", "title": "Author E's z"},
208
+ ]
209
+ }
210
+ import tempfile
211
+ with tempfile.NamedTemporaryFile("w", suffix=".md", delete=False) as f:
212
+ f.write(report)
213
+ rp = f.name
214
+ with tempfile.NamedTemporaryFile("w", suffix=".json", delete=False) as f:
215
+ json.dump(sources, f)
216
+ sp = f.name
217
+ code = main([rp, sp])
218
+ Path(rp).unlink()
219
+ Path(sp).unlink()
220
+ print(f"\nself-test: exit={code} (expected 0 — 1 citation per source, no concentration)")
221
+ return code
222
+
223
+
224
+ if __name__ == "__main__":
225
+ sys.exit(main(sys.argv[1:]))
@@ -0,0 +1,28 @@
1
+ {
2
+ "version": "1.0.0",
3
+ "generated": "2026-05-28T06:10:00Z",
4
+ "source": "source/Anthropic-Cybersecurity-Skills/",
5
+ "description": "Prioritized security skills for pi-crew security-reviewer role",
6
+ "priority_skills": [
7
+ { "id": "detecting-ai-model-prompt-injection-attacks", "priority": "critical", "atlas": ["AML.T0051"], "category": "agent-security" },
8
+ { "id": "detecting-supply-chain-attacks-in-ci-cd", "priority": "critical", "atlas": ["AML.T0010", "AML.T0104"], "category": "supply-chain" },
9
+ { "id": "detecting-anomalous-authentication-patterns", "priority": "high", "atlas": ["AML.T0043", "AML.T0018"], "category": "auth" },
10
+ { "id": "detecting-typosquatting-packages-in-npm-pypi", "priority": "high", "atlas": [], "category": "supply-chain" },
11
+ { "id": "detecting-path-traversal", "priority": "high", "atlas": [], "category": "code-security" },
12
+ { "id": "detecting-command-injection", "priority": "high", "atlas": [], "category": "code-security" },
13
+ { "id": "detecting-sensitive-data-exposure", "priority": "high", "atlas": ["AML.T0067"], "category": "secrets" },
14
+ { "id": "detecting-context-poisoning-in-agent-loops", "priority": "high", "atlas": ["AML.T0051"], "category": "agent-security" },
15
+ { "id": "detecting-tool-invocation-abuse", "priority": "medium", "atlas": ["AML.T0051", "AML.T0054"], "category": "agent-security" },
16
+ { "id": "detecting-malicious-skill-loading", "priority": "medium", "atlas": ["AML.T0062"], "category": "agent-security" },
17
+ { "id": "detecting-credential-leakage-in-logs", "priority": "medium", "atlas": [], "category": "secrets" },
18
+ { "id": "detecting-session-fixation", "priority": "medium", "atlas": ["AML.T0018"], "category": "auth" },
19
+ { "id": "detecting-data-exfiltration-indicators", "priority": "medium", "atlas": ["AML.T0067"], "category": "data-security" },
20
+ { "id": "detecting-serverless-function-injection", "priority": "medium", "atlas": [], "category": "code-security" },
21
+ { "id": "detecting-race-condition-vulnerabilities", "priority": "medium", "atlas": ["AML.T0054"], "category": "code-security" },
22
+ { "id": "detecting-agent-privilege-escalation", "priority": "medium", "atlas": ["AML.T0054"], "category": "agent-security" },
23
+ { "id": "detecting-malicious-npm-packages", "priority": "low", "atlas": [], "category": "supply-chain" },
24
+ { "id": "detecting-dependency-confusion-attacks", "priority": "low", "atlas": [], "category": "supply-chain" },
25
+ { "id": "detecting-token-hijacking", "priority": "low", "atlas": ["AML.T0018"], "category": "auth" },
26
+ { "id": "detecting-race-condition-in-file-operations", "priority": "low", "atlas": ["AML.T0054"], "category": "code-security" }
27
+ ]
28
+ }
@@ -707,6 +707,7 @@ function parseRuntimeConfig(value: unknown): CrewRuntimeConfig | undefined {
707
707
  allowChildProcessFallback: parseWithSchema(Type.Boolean(), obj.allowChildProcessFallback),
708
708
  maxTurns: parsePositiveInteger(obj.maxTurns, LIMIT_CEILINGS.runtimeMaxTurns),
709
709
  graceTurns: parsePositiveInteger(obj.graceTurns, LIMIT_CEILINGS.runtimeGraceTurns),
710
+ taskTimeoutMs: parsePositiveInteger(obj.taskTimeoutMs, LIMIT_CEILINGS.runtimeMaxTurns),
710
711
  inheritContext: parseWithSchema(Type.Boolean(), obj.inheritContext) ?? true,
711
712
  promptMode: parseWithSchema(Type.Union([Type.Literal("replace"), Type.Literal("append")]), obj.promptMode),
712
713
  groupJoin: parseWithSchema(Type.Union([Type.Literal("off"), Type.Literal("group"), Type.Literal("smart")]), obj.groupJoin),
@@ -11,10 +11,13 @@ export interface RoleToolConfig {
11
11
  }
12
12
 
13
13
  export const ROLE_TOOL_CONFIGS: Record<string, RoleToolConfig> = {
14
- // Explorer - Read-only, no write or execute
14
+ // Explorer - Read-only exploration; bash is included for git log/show
15
+ // (decisions stream needs commit-history mining) but edit/write stay
16
+ // excluded. State-mutation safety is enforced separately by
17
+ // READ_ONLY_ROLES in role-permission.ts.
15
18
  explorer: {
16
- tools: ["read", "grep", "find", "ls", "glob"],
17
- excludeTools: ["edit", "write", "bash", "web"],
19
+ tools: ["read", "grep", "find", "ls", "glob", "bash"],
20
+ excludeTools: ["edit", "write", "web"],
18
21
  },
19
22
 
20
23
  // Analyst - Read and analyze, limited execution
@@ -44,6 +44,14 @@ export interface CrewRuntimeConfig {
44
44
  allowChildProcessFallback?: boolean;
45
45
  maxTurns?: number;
46
46
  graceTurns?: number;
47
+ /**
48
+ * W2 fix — wall-clock timeout per task in milliseconds. When the task
49
+ * exceeds this limit, input.signal is aborted which triggers the existing
50
+ * SIGTERM → SIGKILL escalation in child-pi.ts. Default 0 (no timeout).
51
+ * Prevents runaway agent loops (e.g. 11_build in oh-my-pi distill run that
52
+ * re-verified completed files 14+ times).
53
+ */
54
+ taskTimeoutMs?: number;
47
55
  inheritContext?: boolean;
48
56
  promptMode?: "replace" | "append";
49
57
  groupJoin?: "off" | "group" | "smart";
@@ -8,6 +8,7 @@ import { withRunLockSync } from "../state/locks.ts";
8
8
  import { createRunPaths, loadRunManifestById, saveRunManifestAsync, updateRunStatus } from "../state/state-store.ts";
9
9
  import type { TeamRunManifest, TeamTaskState } from "../state/types.ts";
10
10
  import { allTeams, discoverTeams } from "../teams/discover-teams.ts";
11
+ import { errorMessage } from "../utils/guards.ts";
11
12
  import { projectCrewRoot } from "../utils/paths.ts";
12
13
  import { allWorkflows, discoverWorkflows } from "../workflows/discover-workflows.ts";
13
14
  // Heavy runtime — lazy-loaded to avoid pulling team-runner into background-runner
@@ -175,7 +176,7 @@ function setupUnhandledRejectionGuard(
175
176
  setExitFlag: () => void,
176
177
  ): void {
177
178
  process.on("unhandledRejection", (reason, promise) => {
178
- const message = reason instanceof Error ? reason.message : String(reason);
179
+ const message = errorMessage(reason);
179
180
  console.error("[background-runner] UNHANDLED REJECTION:", reason);
180
181
  console.error("[background-runner] Stack:", reason instanceof Error ? reason.stack : "N/A");
181
182
  try {
@@ -234,9 +235,7 @@ function runCleanup(
234
235
  try {
235
236
  killed = terminateActiveChildPiProcesses();
236
237
  } catch (error) {
237
- console.log(
238
- `[background-runner] runCleanup: terminateActiveChildPiProcesses error: ${error instanceof Error ? error.message : String(error)}`,
239
- );
238
+ console.log(`[background-runner] runCleanup: terminateActiveChildPiProcesses error: ${errorMessage(error)}`);
240
239
  }
241
240
  console.log(`[background-runner] runCleanup: killed ${killed} child processes`);
242
241
  // FIX Issue #5: Unregister this worker from the orphan registry on exit.
@@ -245,13 +244,13 @@ function runCleanup(
245
244
  try {
246
245
  unregisterWorker(process.pid);
247
246
  } catch (error) {
248
- console.log(`[background-runner] runCleanup: unregisterWorker error: ${error instanceof Error ? error.message : String(error)}`);
247
+ console.log(`[background-runner] runCleanup: unregisterWorker error: ${errorMessage(error)}`);
249
248
  if (eventsPath) {
250
249
  try {
251
250
  appendEvent(eventsPath, {
252
251
  type: "background.unregister_worker_failed",
253
252
  runId: argValue("--run-id") ?? "unknown",
254
- message: `unregisterWorker failed: ${error instanceof Error ? error.message : String(error)}`,
253
+ message: `unregisterWorker failed: ${errorMessage(error)}`,
255
254
  data: { pid: process.pid },
256
255
  });
257
256
  } catch {
@@ -391,7 +390,7 @@ async function main(): Promise<void> {
391
390
  staleMs: 30_000,
392
391
  });
393
392
  } catch (lockErr) {
394
- throw new Error(`Failed to acquire lock for run '${runId}': ${lockErr instanceof Error ? lockErr.message : String(lockErr)}`);
393
+ throw new Error(`Failed to acquire lock for run '${runId}': ${errorMessage(lockErr)}`);
395
394
  }
396
395
  if (!loaded) throw new Error(`Run '${runId}' not found.`);
397
396
  let { manifest, tasks } = loaded;
@@ -750,9 +749,7 @@ async function main(): Promise<void> {
750
749
  }
751
750
  console.log(`[background-runner] executeTeamRun returned, status=${result.manifest.status}`);
752
751
  } catch (execError) {
753
- console.log(
754
- `[background-runner] executeTeamRun THREW: ${execError instanceof Error ? execError.message : String(execError)}`,
755
- );
752
+ console.log(`[background-runner] executeTeamRun THREW: ${errorMessage(execError)}`);
756
753
  console.log(`[background-runner] stack: ${execError instanceof Error ? execError.stack : "N/A"}`);
757
754
  throw execError;
758
755
  }
@@ -782,7 +779,7 @@ async function main(): Promise<void> {
782
779
  } catch {
783
780
  /* best-effort */
784
781
  }
785
- const message = error instanceof Error ? error.message : String(error);
782
+ const message = errorMessage(error);
786
783
  manifest = updateRunStatus(manifest, "failed", message);
787
784
  appendEvent(manifest.eventsPath, {
788
785
  type: "async.failed",
@@ -790,7 +787,7 @@ async function main(): Promise<void> {
790
787
  message,
791
788
  });
792
789
  process.exitCode = 1;
793
- console.log(`[background-runner] catch block, error=${error instanceof Error ? error.message : String(error)}`);
790
+ console.log(`[background-runner] catch block, error=${errorMessage(error)}`);
794
791
  } finally {
795
792
  // FIX Issue #4: Use shared runCleanup() function for consistent cleanup
796
793
  // across all exit paths (normal, unhandled rejection, main() exception).
@@ -807,9 +804,7 @@ async function main(): Promise<void> {
807
804
  manifest.eventsPath,
808
805
  );
809
806
  } catch (cleanupError) {
810
- console.error(
811
- `[background-runner] runCleanup threw: ${cleanupError instanceof Error ? cleanupError.message : String(cleanupError)}`,
812
- );
807
+ console.error(`[background-runner] runCleanup threw: ${errorMessage(cleanupError)}`);
813
808
  }
814
809
  // NOTE: If exitDueToRejection was set, runCleanup() already called process.exit(1)
815
810
  // so this finally block never continues past that point.
@@ -828,7 +823,7 @@ async function main(): Promise<void> {
828
823
  try {
829
824
  await main();
830
825
  } catch (err) {
831
- console.error(`[background-runner] DEBUG: main() uncaught: ${err instanceof Error ? err.message : String(err)}`);
826
+ console.error(`[background-runner] DEBUG: main() uncaught: ${errorMessage(err)}`);
832
827
  // FIX Issue #1: Set the flag so the finally block's runCleanup() call
833
828
  // will trigger process.exit(1) after cleanup completes. Previously this
834
829
  // called process.exit(1) directly, bypassing the finally block and leaving
@@ -1,4 +1,5 @@
1
1
  import * as fs from "node:fs";
2
+ import * as path from "node:path";
2
3
  import type { NotificationDescriptor } from "../extension/notification-router.ts";
3
4
  import type { MetricRegistry } from "../observability/metric-registry.ts";
4
5
  import { appendEvent } from "../state/event-log.ts";
@@ -142,6 +143,28 @@ export class HeartbeatWatcher {
142
143
  if (level === "dead" && isProcessAlive) {
143
144
  level = "stale";
144
145
  }
146
+ // W8 fix: completion-artifact check — prevents false-positive "dead"
147
+ // during the exit-before-manifest-update race. When a worker process
148
+ // exits normally after completing its task, the result artifact is
149
+ // already on disk, but the manifest status update may lag by a few
150
+ // seconds (status + finishedAt are set atomically in task-runner.ts).
151
+ // If the result file exists, the task completed — downgrade to
152
+ // "stale" so the watcher doesn't fire a misleading "dead" notification
153
+ // for a task that already produced its output.
154
+ // W8-fix-v2 — path-traversal defense-in-depth. task.id is
155
+ // generated internally (e.g. "ts1", "ts2") but we still
156
+ // resolve the candidate path and verify it's strictly
157
+ // contained within <artifactsRoot>/results/. If task.id
158
+ // contained "../" or absolute path segments, the containment
159
+ // check fails and we skip the W8 check (fail-closed: don't
160
+ // accidentally treat a malicious task ID as "completed").
161
+ if (level === "dead" && !isProcessAlive && loaded.manifest.artifactsRoot) {
162
+ const resultsDir = path.resolve(loaded.manifest.artifactsRoot, "results");
163
+ const candidate = path.resolve(resultsDir, `${task.id}.txt`);
164
+ if (candidate.startsWith(resultsDir + path.sep) && fs.existsSync(candidate)) {
165
+ level = "stale";
166
+ }
167
+ }
145
168
  this.opts.registry
146
169
  .gauge("crew.heartbeat.staleness_ms", "Heartbeat elapsed since last seen, milliseconds")
147
170
  .set({ runId: run.runId, taskId: task.id }, Number.isFinite(elapsed) ? elapsed : thresholds.deadMs);
@@ -161,12 +184,16 @@ export class HeartbeatWatcher {
161
184
  elapsedMs: Number.isFinite(elapsed) ? elapsed : undefined,
162
185
  },
163
186
  });
187
+ // W9 fix — prefix title with short run label (first 8 chars of runId)
188
+ // so ambient notifications are scannable when multiple runs are
189
+ // in flight. Full runId remains in the notification object.
190
+ const runLabel = run.runId.slice(0, 8);
164
191
  this.opts.router.enqueue({
165
192
  id: `dead_${run.runId}_${task.id}`,
166
193
  severity: "warning",
167
194
  source: "heartbeat-watcher",
168
195
  runId: run.runId,
169
- title: `Task ${task.id} heartbeat dead`,
196
+ title: `[${runLabel}] Task ${task.id} heartbeat dead`,
170
197
  body: "Background watcher detected a stuck worker.",
171
198
  });
172
199
  this.opts.onDead?.(run.runId, task.id, Number.isFinite(elapsed) ? elapsed : thresholds.deadMs);