pi-crew 0.9.48 → 0.9.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +18 -0
- package/CHANGELOG.md +105 -0
- package/dist/build-meta.json +22 -12
- package/dist/index.mjs +430 -388
- package/dist/index.mjs.map +3 -3
- package/docs/decisions/2026-07-24-oidc-trusted-publishing.md +112 -0
- package/package.json +2 -2
- package/skills/.gitkeep +0 -0
- package/skills/distill-persona/BUILD-NOTES.md +55 -0
- package/skills/distill-persona/SKILL.md +612 -0
- package/skills/distill-persona/UPGRADE-LOG-RESEARCH-SKILLS.md +100 -0
- package/skills/distill-persona/references/coverage-manifest.md +65 -0
- package/skills/distill-persona/references/distillation-field-synthesis-pass2.md +59 -0
- package/skills/distill-persona/references/distillation-field-synthesis.md +108 -0
- package/skills/distill-persona/references/handoff.md +42 -0
- package/skills/distill-persona/references/research/lesson-memory-shortcut.md +33 -0
- package/skills/distill-persona/references/research/r1-a-examples.md +23 -0
- package/skills/distill-persona/references/research/r1-b-scripts.md +26 -0
- package/skills/distill-persona/references/research/r1-c-human-readme.md +31 -0
- package/skills/distill-persona/references/research/r1-d-tests.md +28 -0
- package/skills/distill-persona/references/research/r1-verification.md +36 -0
- package/skills/distill-persona/references/research/r2-low-yield.md +26 -0
- package/skills/distill-persona/scripts/fidelity_eval.py +244 -0
- package/skills/distill-persona/scripts/validate-skill-structure.mjs +177 -0
- package/skills/distill-software/BUILD-NOTES.md +56 -0
- package/skills/distill-software/SKILL.md +302 -0
- package/skills/distill-software/references/handoff.md +47 -0
- package/skills/distill-software/scripts/code_dna.py +290 -0
- package/skills/research/DISTILLATION-PROCESS-CHECKLIST.md +120 -0
- package/skills/research/EXCAVATION-CHECKLIST.md +142 -0
- package/skills/research/FIDELITY.md +180 -0
- package/skills/research/SKILL.md +432 -0
- package/skills/research/references/anti-patterns.md +184 -0
- package/skills/research/references/fidelity.md +241 -0
- package/skills/research/references/handoff.md +48 -0
- package/skills/research/references/research-protocol.md +162 -0
- package/skills/research/references/source-inventory.md +135 -0
- package/skills/research/references/verified-models.md +163 -0
- package/skills/research/scripts/__pycache__/safe_io.cpython-312.pyc +0 -0
- package/skills/research/scripts/code_dna.py +233 -0
- package/skills/research/scripts/emit_run_summary.py +142 -0
- package/skills/research/scripts/safe_io.py +314 -0
- package/skills/research/scripts/source_evaluator.py +234 -0
- package/skills/research/scripts/validate-skill-structure.mjs +177 -0
- package/skills/research/scripts/verify_citations.py +225 -0
- package/skills/security-priority.json +28 -0
- package/src/config/config.ts +1 -0
- package/src/config/role-tools.ts +6 -3
- package/src/config/types.ts +8 -0
- package/src/runtime/background-runner.ts +11 -16
- package/src/runtime/heartbeat-watcher.ts +28 -1
- package/src/runtime/task-runner.ts +165 -119
- package/src/schema/config-schema.ts +1 -0
- package/src/utils/gh-protocol.ts +9 -8
- package/workflows/distill.workflow.md +198 -0
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// validate-skill-structure.mjs — structural invariant checker for generated distill skills.
|
|
3
|
+
// Implements F10 (awesome-persona-distill-skills finding): hard-fail if any structural assertion fails.
|
|
4
|
+
// Run AFTER Phase 3 build, BEFORE Phase 4 behavioral fidelity. Self-contained (stdlib only).
|
|
5
|
+
//
|
|
6
|
+
// Usage:
|
|
7
|
+
// node validate-skill-structure.mjs <path-to-SKILL.md>
|
|
8
|
+
// node validate-skill-structure.mjs <skill-dir>
|
|
9
|
+
//
|
|
10
|
+
// Exit codes: 0 = all assertions pass (all-green → may ship); 1 = one or more failed (iterate).
|
|
11
|
+
|
|
12
|
+
import { readFileSync, readdirSync, statSync, existsSync } from 'node:fs';
|
|
13
|
+
import { join, dirname, basename } from 'node:path';
|
|
14
|
+
|
|
15
|
+
const target = process.argv.find((a) => !a.startsWith('-') && a !== process.argv[0] && a !== process.argv[1]);
|
|
16
|
+
if (!target) {
|
|
17
|
+
console.error('Usage: validate-skill-structure.mjs <path-to-SKILL.md | skill-dir> [--engine]');
|
|
18
|
+
process.exit(2);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
// Resolve to a SKILL.md path
|
|
22
|
+
let skillPath = target;
|
|
23
|
+
if (statSync(target).isDirectory()) {
|
|
24
|
+
skillPath = join(target, 'SKILL.md');
|
|
25
|
+
}
|
|
26
|
+
if (!existsSync(skillPath)) {
|
|
27
|
+
console.error(`✗ Not found: ${skillPath}`);
|
|
28
|
+
process.exit(2);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
const src = readFileSync(skillPath, 'utf8');
|
|
32
|
+
const dir = dirname(skillPath);
|
|
33
|
+
const name = basename(dir);
|
|
34
|
+
const isEngine = process.argv.includes('--engine');
|
|
35
|
+
const PLACEHOLDERS = [/<person>/i, /<target>/i, /<topic>/i, /<field>/i, /TODO/i, /TBD/i, /XXX/i, /YYYY-MM-DD/i, /\.\.\.\s*<\/?/i];
|
|
36
|
+
|
|
37
|
+
const failures = [];
|
|
38
|
+
const passes = [];
|
|
39
|
+
const check = (label, ok, detail = '') => {
|
|
40
|
+
(ok ? passes : failures).push(ok ? ` ✓ ${label}` : ` ✗ ${label}${detail ? ' — ' + detail : ''}`);
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
// --- Frontmatter ---
|
|
44
|
+
const fm = src.match(/^---\n([\s\S]*?)\n---/);
|
|
45
|
+
check('frontmatter block present', !!fm, 'no --- block at top');
|
|
46
|
+
const fmText = fm ? fm[1] : '';
|
|
47
|
+
const fmField = (key) => {
|
|
48
|
+
const m = fmText.match(new RegExp(`^${key}:\\s*(.+)$`, 'm'));
|
|
49
|
+
return m ? m[1].trim() : null;
|
|
50
|
+
};
|
|
51
|
+
const hasFm = (key) => new RegExp(`^${key}:`, 'm').test(fmText);
|
|
52
|
+
|
|
53
|
+
check('frontmatter: name', hasFm('name'));
|
|
54
|
+
check('frontmatter: description', hasFm('description'));
|
|
55
|
+
const desc = fmField('description');
|
|
56
|
+
check('frontmatter: description ≤1 sentence (one terminal punct or one clause)',
|
|
57
|
+
desc ? desc.split(/[.。!!??]/).filter(Boolean).length <= 2 : false,
|
|
58
|
+
desc ? `got: "${desc.slice(0, 60)}…"` : 'missing');
|
|
59
|
+
check('frontmatter: triggers', hasFm('triggers') || hasFm('trigger'), 'no triggers field');
|
|
60
|
+
if (!isEngine) check('frontmatter: distilled (staleness date)', hasFm('distilled') || hasFm('调研时间'), 'no distilled/调研时间 staleness anchor');
|
|
61
|
+
const distilled = fmField('distilled') || fmField('调研时间');
|
|
62
|
+
check('frontmatter: distilled is valid date (YYYY-MM-DD)',
|
|
63
|
+
!distilled || /^\d{4}-\d{2}-\d{2}/.test(distilled),
|
|
64
|
+
distilled ? `got "${distilled}"` : '');
|
|
65
|
+
if (!isEngine) check('frontmatter: target (person|topic|software)', hasFm('target'), 'no target field');
|
|
66
|
+
|
|
67
|
+
// --- Body sections ---
|
|
68
|
+
check('Agentic Protocol section present', /回答工作流|Agentic Protocol/i.test(src));
|
|
69
|
+
check('Agentic Protocol Step 1', /Step 1|第一步|步骤 1|### 1\b/i.test(src));
|
|
70
|
+
check('Agentic Protocol Step 2', /Step 2|第二步|步骤 2|### 2\b/i.test(src));
|
|
71
|
+
check('Agentic Protocol Step 3', /Step 3|第三步|步骤 3|### 3\b/i.test(src));
|
|
72
|
+
|
|
73
|
+
// honest boundaries: count items (numbered list or bullets) under an honest-boundaries heading
|
|
74
|
+
function countListItems(haystack, headingRe, windowChars = 2000) {
|
|
75
|
+
const m = haystack.match(headingRe);
|
|
76
|
+
if (!m) return { found: false, count: 0, block: '' };
|
|
77
|
+
const after = haystack.slice(m.index);
|
|
78
|
+
const section = after.match(/([\s\S]*?)\n##(?=[^#])/);
|
|
79
|
+
const block = section ? section[1] : after.slice(0, windowChars);
|
|
80
|
+
// count BOTH list items AND table data rows (rows with | content |, excluding separator rows)
|
|
81
|
+
const listItems = (block.match(/^\s*(?:\d+[.)]|[-*])\s+\S/gm) || []).length;
|
|
82
|
+
const tableRows = (block.match(/^\s*\|(?![\s:|-]+\|?\s*$).+\|/gm) || []).length;
|
|
83
|
+
return { found: true, count: listItems + tableRows, block };
|
|
84
|
+
}
|
|
85
|
+
if (!isEngine) {
|
|
86
|
+
const boundary = countListItems(src, /诚实边界|honest boundar(?:y|ies)/i);
|
|
87
|
+
check('honest boundaries (M11) ≥3', boundary.count >= 3, `found ${boundary.count} items`);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// --- Anti-drift tables (M9a 内在张力, M9b 反例黑名单, M12 fallback tree) — Darwin gap #2: validator previously skipped these
|
|
91
|
+
if (!isEngine) {
|
|
92
|
+
const tension = countListItems(src, /M9a|内在张力|inner tension|internal tension/i);
|
|
93
|
+
check('内在张力 (M9a) ≥3 tension pairs', tension.count >= 3, tension.found ? `found ${tension.count} items` : 'section missing');
|
|
94
|
+
|
|
95
|
+
const blacklist = countListItems(src, /M9b|反例黑名单|anti.?pattern blacklist/i);
|
|
96
|
+
check('反例黑名单 (M9b) ≥7 rows', blacklist.count >= 7, blacklist.found ? `found ${blacklist.count} rows` : 'section missing');
|
|
97
|
+
|
|
98
|
+
const fallback = countListItems(src, /M12|失败模式.*[Ff]allback|[Ff]allback\s*树/i);
|
|
99
|
+
check('失败模式Fallback树 (M12) ≥8 rows', fallback.count >= 8, fallback.found ? `found ${fallback.count} rows` : 'section missing');
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
// --- FIDELITY.md companion artifact (Darwin gap #3: ship-gate requires it but nothing checked)
|
|
103
|
+
const fidelityPath = join(dir, 'FIDELITY.md');
|
|
104
|
+
if (existsSync(fidelityPath)) {
|
|
105
|
+
const fm2 = readFileSync(fidelityPath, 'utf8');
|
|
106
|
+
const totalMatch = fm2.match(/(?:总分|total)[::\s*]*\*{0,2}([0-9]+)\s*\*?\s*(?:\/|/|\sout\sof\s)\s*\*?\s*100/i);
|
|
107
|
+
check('FIDELITY.md total score present', !!totalMatch, totalMatch ? `=${totalMatch[1]}/100` : 'no /100 score found');
|
|
108
|
+
const qCount = (fm2.match(/(?:Q[1-5]|问题[1-5]|question\s*[1-5])/gi) || []).length;
|
|
109
|
+
check('FIDELITY.md ≥5 test questions', qCount >= 5, `found ${qCount} question refs`);
|
|
110
|
+
check('FIDELITY.md flags single-agent/self-score caveat', /单\s*agent|single.?agent|self.?score|upper.?bound|independent/i.test(fm2), 'add single-agent upper-bound caveat');
|
|
111
|
+
} else if (!isEngine) {
|
|
112
|
+
check('FIDELITY.md companion present', false, 'no FIDELITY.md in skill dir');
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// --- EXCAVATION-CHECKLIST.md (Phase 1 protocol: track + verify each part was really read)
|
|
116
|
+
const checklistPath = join(dir, 'EXCAVATION-CHECKLIST.md');
|
|
117
|
+
if (existsSync(checklistPath)) {
|
|
118
|
+
const cl = readFileSync(checklistPath, 'utf8');
|
|
119
|
+
// dangling rows: status ⬜ not-started or ⏳ reading left at ship time = silently skipped/forgotten
|
|
120
|
+
const dangling = (cl.match(/\|\s*[⬜⏳][^|]*\|/gu) || []).length;
|
|
121
|
+
check('checklist: no dangling ⬜/⏳ rows (every part resolved)', dangling === 0, dangling ? `${dangling} row(s) not-started/reading — resolve to ✅📄/⏭/🧠` : '');
|
|
122
|
+
// memory ratio: parse "memory-ratio: NN%" if declared
|
|
123
|
+
const ratioMatch = cl.match(/memory-ratio[:\s]*([0-9]+)\s*%/i);
|
|
124
|
+
if (ratioMatch) {
|
|
125
|
+
const ratio = parseInt(ratioMatch[1], 10);
|
|
126
|
+
check('checklist: 🧠 memory-ratio ≤30%', ratio <= 30, `declared ${ratio}% (>30% = recap, not distillation)`);
|
|
127
|
+
} else {
|
|
128
|
+
check('checklist: declares memory-ratio', false, 'add "🧠 memory-ratio: NN% (X/Y findings)" header line');
|
|
129
|
+
}
|
|
130
|
+
// proof-of-read present: at least one verbatim-quote-with-location cell (heuristic — a ✅ row should carry a quote)
|
|
131
|
+
const proofCells = (cl.match(/"[^"]{8,}"[^|]*/g) || []).length;
|
|
132
|
+
check('checklist: ≥1 proof-of-read (verbatim quote in a ✅ row)', proofCells >= 1, `${proofCells} quote cell(s) found`);
|
|
133
|
+
} else if (!isEngine) {
|
|
134
|
+
check('EXCAVATION-CHECKLIST.md present', false, 'no excavation checklist — Phase 1 protocol requires one');
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// --- DISTILLATION-PROCESS-CHECKLIST.md (whole-pipeline tracking + the 3-empty-rounds deep-dive gate)
|
|
138
|
+
const procPath = join(dir, 'DISTILLATION-PROCESS-CHECKLIST.md');
|
|
139
|
+
if (existsSync(procPath)) {
|
|
140
|
+
const pc = readFileSync(procPath, 'utf8');
|
|
141
|
+
// dangling phases: ⬜/⏳ left in the phase-progress table at ship = a phase skipped
|
|
142
|
+
const danglingPhases = (pc.match(/\|\s*[⬜⏳][^|]*\|/gu) || []).length;
|
|
143
|
+
check('process: no dangling ⬜/⏳ phases (every phase completed)', danglingPhases === 0, danglingPhases ? `${danglingPhases} phase(s) not done` : '');
|
|
144
|
+
// deep-dive round log present
|
|
145
|
+
check('process: deep-dive round log present', /round log|\| *Round.*New findings/i.test(pc), 'add the round-log table');
|
|
146
|
+
// 3-empty-rounds gate recorded (≥3 consecutive zero-new rounds before leaving a phase)
|
|
147
|
+
const gateFired = /gate fires|3 consecutive (empty|zero)|GATE FIRES/i.test(pc);
|
|
148
|
+
check('process: 3-empty-rounds gate recorded (≥3 zero-new rounds)', gateFired, 'record a round-log row marked gate-fired before declaring any research phase done');
|
|
149
|
+
} else if (!isEngine) {
|
|
150
|
+
check('DISTILLATION-PROCESS-CHECKLIST.md present', false, 'no process checklist — every phase + the 3-round deep-dive gate must be tracked');
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// --- Placeholders ---
|
|
154
|
+
const foundPlaceholders = PLACEHOLDERS.filter((re) => re.test(src));
|
|
155
|
+
if (!isEngine) check('no unresolved placeholder text', foundPlaceholders.length === 0,
|
|
156
|
+
foundPlaceholders.map((re) => src.match(re)?.[0]).filter(Boolean).join(', '));
|
|
157
|
+
|
|
158
|
+
// --- Self-containment (F9): references/sources should not dangle outside the dir ---
|
|
159
|
+
const refsDir = join(dir, 'references');
|
|
160
|
+
if (!isEngine) check('self-contained: no external repo paths in body (e.g. 07-调研/)',
|
|
161
|
+
!/[\w/-]*调研与分析?\/|src\/|source\//i.test(src.replace(/```[\s\S]*?```/g, '')),
|
|
162
|
+
'body references paths outside skill dir');
|
|
163
|
+
|
|
164
|
+
// --- Software-specific assertions (if target: software) ---
|
|
165
|
+
if (/target:\s*software/i.test(fmText) || /code-dna|代码表达DNA|code expression/i.test(src)) {
|
|
166
|
+
check('[software] Code Expression-DNA section present', /代码表达DNA|Code Expression-DNA|code.?dna/i.test(src));
|
|
167
|
+
check('[software] toolchain matrix present', /toolchain|eslint|oxlint|biome|tsconfig/i.test(src));
|
|
168
|
+
check('[software] distilled_against commit anchor', hasFm('distilled_against') || /distilled_against/i.test(src));
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
// --- Report ---
|
|
172
|
+
console.log(`\nvalidate-skill-structure: ${name}`);
|
|
173
|
+
console.log(`path: ${skillPath}\n`);
|
|
174
|
+
passes.forEach((l) => console.log(l));
|
|
175
|
+
failures.forEach((l) => console.log(l));
|
|
176
|
+
console.log(`\n${passes.length} pass, ${failures.length} fail — ${failures.length === 0 ? '✅ ALL-GREEN (may ship)' : '🔴 NOT READY (iterate Phase 2→3)'}${isEngine ? ' [engine mode — generated-skill checks skipped]' : ''}\n`);
|
|
177
|
+
process.exit(failures.length === 0 ? 0 : 1);
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
verify_citations.py — borrowed from Geek-skills-deep-research, ported for the
|
|
4
|
+
research skill (F13: wired INTO the Agentic Protocol Step 2, never orphaned).
|
|
5
|
+
|
|
6
|
+
Structural citation-integrity checker: resolves every citation marker `[n]` in
|
|
7
|
+
the report against the LOCAL source pool (sources.json); flags unresolved
|
|
8
|
+
citations, dangling references, non-sequential numbering, and source
|
|
9
|
+
concentration (> 25% from any single source).
|
|
10
|
+
|
|
11
|
+
⚠️ This is a STRUCTURAL check only — it does NOT make network requests.
|
|
12
|
+
It does NOT verify HTTP 404s, redirect chains, or content-drift. A URL that
|
|
13
|
+
has gone dead or been hijacked will still pass if it matches the local source
|
|
14
|
+
pool. For live-URL verification, use a separate network checker (e.g.
|
|
15
|
+
WebFetch HEAD) before relying on citation-liveness.
|
|
16
|
+
|
|
17
|
+
Stdlib only. Python 3.9+.
|
|
18
|
+
|
|
19
|
+
Usage:
|
|
20
|
+
python3 verify_citations.py <report.md> <sources.json> [--output results.json]
|
|
21
|
+
python3 verify_citations.py --self-test # run a quick self-test
|
|
22
|
+
|
|
23
|
+
Exit codes:
|
|
24
|
+
0 no unresolved / no dangling / no concentration issues
|
|
25
|
+
1 one or more issues found (see --output report)
|
|
26
|
+
2 usage error
|
|
27
|
+
"""
|
|
28
|
+
import json
|
|
29
|
+
import re
|
|
30
|
+
import sys
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
from urllib.parse import urlparse
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
SOURCE_CONCENTRATION_LIMIT = 0.25 # No single source > 25% of citations
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def normalize_url(url: str) -> str:
|
|
39
|
+
"""Lowercase + strip trailing slash + strip common tracking params."""
|
|
40
|
+
url = url.strip().rstrip("/")
|
|
41
|
+
parsed = urlparse(url.lower())
|
|
42
|
+
drop_params = {"utm_source", "utm_medium", "utm_campaign", "ref", "fbclid", "gclid"}
|
|
43
|
+
# Best-effort: keep query but strip params in drop_params
|
|
44
|
+
if parsed.query:
|
|
45
|
+
parts = [p for p in parsed.query.split("&") if p.split("=")[0] not in drop_params]
|
|
46
|
+
query = "&".join(parts)
|
|
47
|
+
else:
|
|
48
|
+
query = ""
|
|
49
|
+
return f"{parsed.scheme}://{parsed.netloc}{parsed.path}{('?' + query) if query else ''}".rstrip("/")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def url_signature(url: str):
|
|
53
|
+
"""Robust (domain, path_segments[:3]) for matching."""
|
|
54
|
+
parsed = urlparse(url.lower())
|
|
55
|
+
host = parsed.hostname or ""
|
|
56
|
+
if host.startswith("www."):
|
|
57
|
+
host = host[4:]
|
|
58
|
+
segments = [s for s in parsed.path.split("/") if s][:3]
|
|
59
|
+
return (host, tuple(segments))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def extract_citations(text: str):
|
|
63
|
+
"""Find [n] markers in body. Returns list of int indices."""
|
|
64
|
+
return [int(m) for m in re.findall(r"\[(\d+)\]", text)]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def extract_references(text: str):
|
|
68
|
+
"""Find reference list. Returns dict {int_idx: (label, url)}."""
|
|
69
|
+
refs = {}
|
|
70
|
+
# Match common patterns: [1] Label — URL or [1] Label - http...
|
|
71
|
+
for m in re.finditer(r"^\s*\[(\d+)\]\s+([^—\-]+?)\s*[—\-]\s*(\S+)", text, re.MULTILINE):
|
|
72
|
+
refs[int(m.group(1))] = (m.group(2).strip(), m.group(3).strip())
|
|
73
|
+
return refs
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def main(argv):
|
|
77
|
+
if not argv or argv[0] in ("-h", "--help"):
|
|
78
|
+
print(__doc__)
|
|
79
|
+
return 0
|
|
80
|
+
if argv[0] == "--self-test":
|
|
81
|
+
return self_test()
|
|
82
|
+
if len(argv) < 2:
|
|
83
|
+
print("Usage: verify_citations.py <report.md> <sources.json> [--output results.json]", file=sys.stderr)
|
|
84
|
+
return 2
|
|
85
|
+
|
|
86
|
+
report_path = Path(argv[0])
|
|
87
|
+
sources_path = Path(argv[1])
|
|
88
|
+
output_path = None
|
|
89
|
+
if len(argv) >= 4 and argv[2] == "--output":
|
|
90
|
+
output_path = Path(argv[3])
|
|
91
|
+
|
|
92
|
+
report = report_path.read_text(encoding="utf-8")
|
|
93
|
+
sources_data = json.loads(sources_path.read_text(encoding="utf-8"))
|
|
94
|
+
|
|
95
|
+
sources = sources_data.get("sources", [])
|
|
96
|
+
sources_normalized = {normalize_url(s["url"]): s for s in sources}
|
|
97
|
+
|
|
98
|
+
citations = extract_citations(report)
|
|
99
|
+
refs = extract_references(report)
|
|
100
|
+
|
|
101
|
+
issues = []
|
|
102
|
+
stats = {
|
|
103
|
+
"report_citations": len(citations),
|
|
104
|
+
"unique_citations": len(set(citations)),
|
|
105
|
+
"references_declared": len(refs),
|
|
106
|
+
"sources_in_pool": len(sources),
|
|
107
|
+
"concentration": {},
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
# 1. Every [n] has a matching reference
|
|
111
|
+
for c in set(citations):
|
|
112
|
+
if c not in refs:
|
|
113
|
+
issues.append({
|
|
114
|
+
"severity": "fatal",
|
|
115
|
+
"type": "unresolved_citation",
|
|
116
|
+
"detail": f"Citation [{c}] has no matching reference entry",
|
|
117
|
+
})
|
|
118
|
+
|
|
119
|
+
# 2. Every reference URL exists in source pool
|
|
120
|
+
for idx, (label, url) in refs.items():
|
|
121
|
+
if normalize_url(url) not in sources_normalized:
|
|
122
|
+
# Try signature match
|
|
123
|
+
sig = url_signature(url)
|
|
124
|
+
match = None
|
|
125
|
+
for s_url, s in sources_normalized.items():
|
|
126
|
+
if url_signature(s_url) == sig:
|
|
127
|
+
match = s
|
|
128
|
+
break
|
|
129
|
+
if not match:
|
|
130
|
+
issues.append({
|
|
131
|
+
"severity": "fatal",
|
|
132
|
+
"type": "url_not_in_source_pool",
|
|
133
|
+
"detail": f"Reference [{idx}] URL '{url}' not found in source pool",
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
# 3. Dangling references (in list but never cited)
|
|
137
|
+
for idx in refs:
|
|
138
|
+
if idx not in set(citations):
|
|
139
|
+
issues.append({
|
|
140
|
+
"severity": "warn",
|
|
141
|
+
"type": "dangling_reference",
|
|
142
|
+
"detail": f"Reference [{idx}] declared but never cited in body",
|
|
143
|
+
})
|
|
144
|
+
|
|
145
|
+
# 4. Sequential numbering, no gaps
|
|
146
|
+
if refs:
|
|
147
|
+
max_idx = max(refs.keys())
|
|
148
|
+
missing = [i for i in range(1, max_idx + 1) if i not in refs]
|
|
149
|
+
if missing:
|
|
150
|
+
issues.append({
|
|
151
|
+
"severity": "warn",
|
|
152
|
+
"type": "non_sequential_numbering",
|
|
153
|
+
"detail": f"Reference numbering gaps: {missing}",
|
|
154
|
+
})
|
|
155
|
+
|
|
156
|
+
# 5. Source concentration
|
|
157
|
+
cite_counts = {}
|
|
158
|
+
for c in citations:
|
|
159
|
+
if c in refs:
|
|
160
|
+
label = refs[c][0]
|
|
161
|
+
cite_counts[label] = cite_counts.get(label, 0) + 1
|
|
162
|
+
total = max(len(citations), 1)
|
|
163
|
+
for label, count in cite_counts.items():
|
|
164
|
+
ratio = count / total
|
|
165
|
+
stats["concentration"][label] = round(ratio, 3)
|
|
166
|
+
if ratio > SOURCE_CONCENTRATION_LIMIT:
|
|
167
|
+
issues.append({
|
|
168
|
+
"severity": "fatal",
|
|
169
|
+
"type": "source_concentration",
|
|
170
|
+
"detail": f"Source '{label}' has {count}/{total} citations ({ratio*100:.1f}%) > {SOURCE_CONCENTRATION_LIMIT*100:.0f}% limit",
|
|
171
|
+
})
|
|
172
|
+
|
|
173
|
+
result = {
|
|
174
|
+
"ok": all(i["severity"] != "fatal" for i in issues),
|
|
175
|
+
"stats": stats,
|
|
176
|
+
"issues": issues,
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
if output_path:
|
|
180
|
+
output_path.write_text(json.dumps(result, indent=2), encoding="utf-8")
|
|
181
|
+
else:
|
|
182
|
+
print(json.dumps(result, indent=2))
|
|
183
|
+
|
|
184
|
+
return 0 if result["ok"] else 1
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def self_test():
|
|
188
|
+
"""Run a quick self-test on a fabricated report."""
|
|
189
|
+
report = """# Test
|
|
190
|
+
|
|
191
|
+
Claim one [1]. Claim two [2]. Claim three [3]. Claim four [4]. Claim five [5].
|
|
192
|
+
|
|
193
|
+
## References
|
|
194
|
+
|
|
195
|
+
[1] Author A — https://example.com/a
|
|
196
|
+
[2] Author B — https://example.org/b
|
|
197
|
+
[3] Author C — https://example.net/c
|
|
198
|
+
[4] Author D — https://example.io/d
|
|
199
|
+
[5] Author E — https://example.dev/e
|
|
200
|
+
"""
|
|
201
|
+
sources = {
|
|
202
|
+
"sources": [
|
|
203
|
+
{"url": "https://example.com/a", "title": "Author A's paper"},
|
|
204
|
+
{"url": "https://example.org/b", "title": "Author B's notes"},
|
|
205
|
+
{"url": "https://example.net/c", "title": "Author C's x"},
|
|
206
|
+
{"url": "https://example.io/d", "title": "Author D's y"},
|
|
207
|
+
{"url": "https://example.dev/e", "title": "Author E's z"},
|
|
208
|
+
]
|
|
209
|
+
}
|
|
210
|
+
import tempfile
|
|
211
|
+
with tempfile.NamedTemporaryFile("w", suffix=".md", delete=False) as f:
|
|
212
|
+
f.write(report)
|
|
213
|
+
rp = f.name
|
|
214
|
+
with tempfile.NamedTemporaryFile("w", suffix=".json", delete=False) as f:
|
|
215
|
+
json.dump(sources, f)
|
|
216
|
+
sp = f.name
|
|
217
|
+
code = main([rp, sp])
|
|
218
|
+
Path(rp).unlink()
|
|
219
|
+
Path(sp).unlink()
|
|
220
|
+
print(f"\nself-test: exit={code} (expected 0 — 1 citation per source, no concentration)")
|
|
221
|
+
return code
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
if __name__ == "__main__":
|
|
225
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": "1.0.0",
|
|
3
|
+
"generated": "2026-05-28T06:10:00Z",
|
|
4
|
+
"source": "source/Anthropic-Cybersecurity-Skills/",
|
|
5
|
+
"description": "Prioritized security skills for pi-crew security-reviewer role",
|
|
6
|
+
"priority_skills": [
|
|
7
|
+
{ "id": "detecting-ai-model-prompt-injection-attacks", "priority": "critical", "atlas": ["AML.T0051"], "category": "agent-security" },
|
|
8
|
+
{ "id": "detecting-supply-chain-attacks-in-ci-cd", "priority": "critical", "atlas": ["AML.T0010", "AML.T0104"], "category": "supply-chain" },
|
|
9
|
+
{ "id": "detecting-anomalous-authentication-patterns", "priority": "high", "atlas": ["AML.T0043", "AML.T0018"], "category": "auth" },
|
|
10
|
+
{ "id": "detecting-typosquatting-packages-in-npm-pypi", "priority": "high", "atlas": [], "category": "supply-chain" },
|
|
11
|
+
{ "id": "detecting-path-traversal", "priority": "high", "atlas": [], "category": "code-security" },
|
|
12
|
+
{ "id": "detecting-command-injection", "priority": "high", "atlas": [], "category": "code-security" },
|
|
13
|
+
{ "id": "detecting-sensitive-data-exposure", "priority": "high", "atlas": ["AML.T0067"], "category": "secrets" },
|
|
14
|
+
{ "id": "detecting-context-poisoning-in-agent-loops", "priority": "high", "atlas": ["AML.T0051"], "category": "agent-security" },
|
|
15
|
+
{ "id": "detecting-tool-invocation-abuse", "priority": "medium", "atlas": ["AML.T0051", "AML.T0054"], "category": "agent-security" },
|
|
16
|
+
{ "id": "detecting-malicious-skill-loading", "priority": "medium", "atlas": ["AML.T0062"], "category": "agent-security" },
|
|
17
|
+
{ "id": "detecting-credential-leakage-in-logs", "priority": "medium", "atlas": [], "category": "secrets" },
|
|
18
|
+
{ "id": "detecting-session-fixation", "priority": "medium", "atlas": ["AML.T0018"], "category": "auth" },
|
|
19
|
+
{ "id": "detecting-data-exfiltration-indicators", "priority": "medium", "atlas": ["AML.T0067"], "category": "data-security" },
|
|
20
|
+
{ "id": "detecting-serverless-function-injection", "priority": "medium", "atlas": [], "category": "code-security" },
|
|
21
|
+
{ "id": "detecting-race-condition-vulnerabilities", "priority": "medium", "atlas": ["AML.T0054"], "category": "code-security" },
|
|
22
|
+
{ "id": "detecting-agent-privilege-escalation", "priority": "medium", "atlas": ["AML.T0054"], "category": "agent-security" },
|
|
23
|
+
{ "id": "detecting-malicious-npm-packages", "priority": "low", "atlas": [], "category": "supply-chain" },
|
|
24
|
+
{ "id": "detecting-dependency-confusion-attacks", "priority": "low", "atlas": [], "category": "supply-chain" },
|
|
25
|
+
{ "id": "detecting-token-hijacking", "priority": "low", "atlas": ["AML.T0018"], "category": "auth" },
|
|
26
|
+
{ "id": "detecting-race-condition-in-file-operations", "priority": "low", "atlas": ["AML.T0054"], "category": "code-security" }
|
|
27
|
+
]
|
|
28
|
+
}
|
package/src/config/config.ts
CHANGED
|
@@ -707,6 +707,7 @@ function parseRuntimeConfig(value: unknown): CrewRuntimeConfig | undefined {
|
|
|
707
707
|
allowChildProcessFallback: parseWithSchema(Type.Boolean(), obj.allowChildProcessFallback),
|
|
708
708
|
maxTurns: parsePositiveInteger(obj.maxTurns, LIMIT_CEILINGS.runtimeMaxTurns),
|
|
709
709
|
graceTurns: parsePositiveInteger(obj.graceTurns, LIMIT_CEILINGS.runtimeGraceTurns),
|
|
710
|
+
taskTimeoutMs: parsePositiveInteger(obj.taskTimeoutMs, LIMIT_CEILINGS.runtimeMaxTurns),
|
|
710
711
|
inheritContext: parseWithSchema(Type.Boolean(), obj.inheritContext) ?? true,
|
|
711
712
|
promptMode: parseWithSchema(Type.Union([Type.Literal("replace"), Type.Literal("append")]), obj.promptMode),
|
|
712
713
|
groupJoin: parseWithSchema(Type.Union([Type.Literal("off"), Type.Literal("group"), Type.Literal("smart")]), obj.groupJoin),
|
package/src/config/role-tools.ts
CHANGED
|
@@ -11,10 +11,13 @@ export interface RoleToolConfig {
|
|
|
11
11
|
}
|
|
12
12
|
|
|
13
13
|
export const ROLE_TOOL_CONFIGS: Record<string, RoleToolConfig> = {
|
|
14
|
-
// Explorer - Read-only
|
|
14
|
+
// Explorer - Read-only exploration; bash is included for git log/show
|
|
15
|
+
// (decisions stream needs commit-history mining) but edit/write stay
|
|
16
|
+
// excluded. State-mutation safety is enforced separately by
|
|
17
|
+
// READ_ONLY_ROLES in role-permission.ts.
|
|
15
18
|
explorer: {
|
|
16
|
-
tools: ["read", "grep", "find", "ls", "glob"],
|
|
17
|
-
excludeTools: ["edit", "write", "
|
|
19
|
+
tools: ["read", "grep", "find", "ls", "glob", "bash"],
|
|
20
|
+
excludeTools: ["edit", "write", "web"],
|
|
18
21
|
},
|
|
19
22
|
|
|
20
23
|
// Analyst - Read and analyze, limited execution
|
package/src/config/types.ts
CHANGED
|
@@ -44,6 +44,14 @@ export interface CrewRuntimeConfig {
|
|
|
44
44
|
allowChildProcessFallback?: boolean;
|
|
45
45
|
maxTurns?: number;
|
|
46
46
|
graceTurns?: number;
|
|
47
|
+
/**
|
|
48
|
+
* W2 fix — wall-clock timeout per task in milliseconds. When the task
|
|
49
|
+
* exceeds this limit, input.signal is aborted which triggers the existing
|
|
50
|
+
* SIGTERM → SIGKILL escalation in child-pi.ts. Default 0 (no timeout).
|
|
51
|
+
* Prevents runaway agent loops (e.g. 11_build in oh-my-pi distill run that
|
|
52
|
+
* re-verified completed files 14+ times).
|
|
53
|
+
*/
|
|
54
|
+
taskTimeoutMs?: number;
|
|
47
55
|
inheritContext?: boolean;
|
|
48
56
|
promptMode?: "replace" | "append";
|
|
49
57
|
groupJoin?: "off" | "group" | "smart";
|
|
@@ -8,6 +8,7 @@ import { withRunLockSync } from "../state/locks.ts";
|
|
|
8
8
|
import { createRunPaths, loadRunManifestById, saveRunManifestAsync, updateRunStatus } from "../state/state-store.ts";
|
|
9
9
|
import type { TeamRunManifest, TeamTaskState } from "../state/types.ts";
|
|
10
10
|
import { allTeams, discoverTeams } from "../teams/discover-teams.ts";
|
|
11
|
+
import { errorMessage } from "../utils/guards.ts";
|
|
11
12
|
import { projectCrewRoot } from "../utils/paths.ts";
|
|
12
13
|
import { allWorkflows, discoverWorkflows } from "../workflows/discover-workflows.ts";
|
|
13
14
|
// Heavy runtime — lazy-loaded to avoid pulling team-runner into background-runner
|
|
@@ -175,7 +176,7 @@ function setupUnhandledRejectionGuard(
|
|
|
175
176
|
setExitFlag: () => void,
|
|
176
177
|
): void {
|
|
177
178
|
process.on("unhandledRejection", (reason, promise) => {
|
|
178
|
-
const message =
|
|
179
|
+
const message = errorMessage(reason);
|
|
179
180
|
console.error("[background-runner] UNHANDLED REJECTION:", reason);
|
|
180
181
|
console.error("[background-runner] Stack:", reason instanceof Error ? reason.stack : "N/A");
|
|
181
182
|
try {
|
|
@@ -234,9 +235,7 @@ function runCleanup(
|
|
|
234
235
|
try {
|
|
235
236
|
killed = terminateActiveChildPiProcesses();
|
|
236
237
|
} catch (error) {
|
|
237
|
-
console.log(
|
|
238
|
-
`[background-runner] runCleanup: terminateActiveChildPiProcesses error: ${error instanceof Error ? error.message : String(error)}`,
|
|
239
|
-
);
|
|
238
|
+
console.log(`[background-runner] runCleanup: terminateActiveChildPiProcesses error: ${errorMessage(error)}`);
|
|
240
239
|
}
|
|
241
240
|
console.log(`[background-runner] runCleanup: killed ${killed} child processes`);
|
|
242
241
|
// FIX Issue #5: Unregister this worker from the orphan registry on exit.
|
|
@@ -245,13 +244,13 @@ function runCleanup(
|
|
|
245
244
|
try {
|
|
246
245
|
unregisterWorker(process.pid);
|
|
247
246
|
} catch (error) {
|
|
248
|
-
console.log(`[background-runner] runCleanup: unregisterWorker error: ${
|
|
247
|
+
console.log(`[background-runner] runCleanup: unregisterWorker error: ${errorMessage(error)}`);
|
|
249
248
|
if (eventsPath) {
|
|
250
249
|
try {
|
|
251
250
|
appendEvent(eventsPath, {
|
|
252
251
|
type: "background.unregister_worker_failed",
|
|
253
252
|
runId: argValue("--run-id") ?? "unknown",
|
|
254
|
-
message: `unregisterWorker failed: ${
|
|
253
|
+
message: `unregisterWorker failed: ${errorMessage(error)}`,
|
|
255
254
|
data: { pid: process.pid },
|
|
256
255
|
});
|
|
257
256
|
} catch {
|
|
@@ -391,7 +390,7 @@ async function main(): Promise<void> {
|
|
|
391
390
|
staleMs: 30_000,
|
|
392
391
|
});
|
|
393
392
|
} catch (lockErr) {
|
|
394
|
-
throw new Error(`Failed to acquire lock for run '${runId}': ${
|
|
393
|
+
throw new Error(`Failed to acquire lock for run '${runId}': ${errorMessage(lockErr)}`);
|
|
395
394
|
}
|
|
396
395
|
if (!loaded) throw new Error(`Run '${runId}' not found.`);
|
|
397
396
|
let { manifest, tasks } = loaded;
|
|
@@ -750,9 +749,7 @@ async function main(): Promise<void> {
|
|
|
750
749
|
}
|
|
751
750
|
console.log(`[background-runner] executeTeamRun returned, status=${result.manifest.status}`);
|
|
752
751
|
} catch (execError) {
|
|
753
|
-
console.log(
|
|
754
|
-
`[background-runner] executeTeamRun THREW: ${execError instanceof Error ? execError.message : String(execError)}`,
|
|
755
|
-
);
|
|
752
|
+
console.log(`[background-runner] executeTeamRun THREW: ${errorMessage(execError)}`);
|
|
756
753
|
console.log(`[background-runner] stack: ${execError instanceof Error ? execError.stack : "N/A"}`);
|
|
757
754
|
throw execError;
|
|
758
755
|
}
|
|
@@ -782,7 +779,7 @@ async function main(): Promise<void> {
|
|
|
782
779
|
} catch {
|
|
783
780
|
/* best-effort */
|
|
784
781
|
}
|
|
785
|
-
const message =
|
|
782
|
+
const message = errorMessage(error);
|
|
786
783
|
manifest = updateRunStatus(manifest, "failed", message);
|
|
787
784
|
appendEvent(manifest.eventsPath, {
|
|
788
785
|
type: "async.failed",
|
|
@@ -790,7 +787,7 @@ async function main(): Promise<void> {
|
|
|
790
787
|
message,
|
|
791
788
|
});
|
|
792
789
|
process.exitCode = 1;
|
|
793
|
-
console.log(`[background-runner] catch block, error=${
|
|
790
|
+
console.log(`[background-runner] catch block, error=${errorMessage(error)}`);
|
|
794
791
|
} finally {
|
|
795
792
|
// FIX Issue #4: Use shared runCleanup() function for consistent cleanup
|
|
796
793
|
// across all exit paths (normal, unhandled rejection, main() exception).
|
|
@@ -807,9 +804,7 @@ async function main(): Promise<void> {
|
|
|
807
804
|
manifest.eventsPath,
|
|
808
805
|
);
|
|
809
806
|
} catch (cleanupError) {
|
|
810
|
-
console.error(
|
|
811
|
-
`[background-runner] runCleanup threw: ${cleanupError instanceof Error ? cleanupError.message : String(cleanupError)}`,
|
|
812
|
-
);
|
|
807
|
+
console.error(`[background-runner] runCleanup threw: ${errorMessage(cleanupError)}`);
|
|
813
808
|
}
|
|
814
809
|
// NOTE: If exitDueToRejection was set, runCleanup() already called process.exit(1)
|
|
815
810
|
// so this finally block never continues past that point.
|
|
@@ -828,7 +823,7 @@ async function main(): Promise<void> {
|
|
|
828
823
|
try {
|
|
829
824
|
await main();
|
|
830
825
|
} catch (err) {
|
|
831
|
-
console.error(`[background-runner] DEBUG: main() uncaught: ${
|
|
826
|
+
console.error(`[background-runner] DEBUG: main() uncaught: ${errorMessage(err)}`);
|
|
832
827
|
// FIX Issue #1: Set the flag so the finally block's runCleanup() call
|
|
833
828
|
// will trigger process.exit(1) after cleanup completes. Previously this
|
|
834
829
|
// called process.exit(1) directly, bypassing the finally block and leaving
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import * as fs from "node:fs";
|
|
2
|
+
import * as path from "node:path";
|
|
2
3
|
import type { NotificationDescriptor } from "../extension/notification-router.ts";
|
|
3
4
|
import type { MetricRegistry } from "../observability/metric-registry.ts";
|
|
4
5
|
import { appendEvent } from "../state/event-log.ts";
|
|
@@ -142,6 +143,28 @@ export class HeartbeatWatcher {
|
|
|
142
143
|
if (level === "dead" && isProcessAlive) {
|
|
143
144
|
level = "stale";
|
|
144
145
|
}
|
|
146
|
+
// W8 fix: completion-artifact check — prevents false-positive "dead"
|
|
147
|
+
// during the exit-before-manifest-update race. When a worker process
|
|
148
|
+
// exits normally after completing its task, the result artifact is
|
|
149
|
+
// already on disk, but the manifest status update may lag by a few
|
|
150
|
+
// seconds (status + finishedAt are set atomically in task-runner.ts).
|
|
151
|
+
// If the result file exists, the task completed — downgrade to
|
|
152
|
+
// "stale" so the watcher doesn't fire a misleading "dead" notification
|
|
153
|
+
// for a task that already produced its output.
|
|
154
|
+
// W8-fix-v2 — path-traversal defense-in-depth. task.id is
|
|
155
|
+
// generated internally (e.g. "ts1", "ts2") but we still
|
|
156
|
+
// resolve the candidate path and verify it's strictly
|
|
157
|
+
// contained within <artifactsRoot>/results/. If task.id
|
|
158
|
+
// contained "../" or absolute path segments, the containment
|
|
159
|
+
// check fails and we skip the W8 check (fail-closed: don't
|
|
160
|
+
// accidentally treat a malicious task ID as "completed").
|
|
161
|
+
if (level === "dead" && !isProcessAlive && loaded.manifest.artifactsRoot) {
|
|
162
|
+
const resultsDir = path.resolve(loaded.manifest.artifactsRoot, "results");
|
|
163
|
+
const candidate = path.resolve(resultsDir, `${task.id}.txt`);
|
|
164
|
+
if (candidate.startsWith(resultsDir + path.sep) && fs.existsSync(candidate)) {
|
|
165
|
+
level = "stale";
|
|
166
|
+
}
|
|
167
|
+
}
|
|
145
168
|
this.opts.registry
|
|
146
169
|
.gauge("crew.heartbeat.staleness_ms", "Heartbeat elapsed since last seen, milliseconds")
|
|
147
170
|
.set({ runId: run.runId, taskId: task.id }, Number.isFinite(elapsed) ? elapsed : thresholds.deadMs);
|
|
@@ -161,12 +184,16 @@ export class HeartbeatWatcher {
|
|
|
161
184
|
elapsedMs: Number.isFinite(elapsed) ? elapsed : undefined,
|
|
162
185
|
},
|
|
163
186
|
});
|
|
187
|
+
// W9 fix — prefix title with short run label (first 8 chars of runId)
|
|
188
|
+
// so ambient notifications are scannable when multiple runs are
|
|
189
|
+
// in flight. Full runId remains in the notification object.
|
|
190
|
+
const runLabel = run.runId.slice(0, 8);
|
|
164
191
|
this.opts.router.enqueue({
|
|
165
192
|
id: `dead_${run.runId}_${task.id}`,
|
|
166
193
|
severity: "warning",
|
|
167
194
|
source: "heartbeat-watcher",
|
|
168
195
|
runId: run.runId,
|
|
169
|
-
title: `Task ${task.id} heartbeat dead`,
|
|
196
|
+
title: `[${runLabel}] Task ${task.id} heartbeat dead`,
|
|
170
197
|
body: "Background watcher detected a stuck worker.",
|
|
171
198
|
});
|
|
172
199
|
this.opts.onDead?.(run.runId, task.id, Number.isFinite(elapsed) ? elapsed : thresholds.deadMs);
|