pi-crew 0.9.47 → 0.9.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +18 -0
- package/CHANGELOG.md +132 -4
- package/README.md +1 -1
- package/dist/build-meta.json +22 -12
- package/dist/index.mjs +430 -388
- package/dist/index.mjs.map +3 -3
- package/docs/decisions/2026-07-24-oidc-trusted-publishing.md +112 -0
- package/docs/publishing.md +5 -2
- package/package.json +2 -2
- package/scripts/postinstall.mjs +43 -18
- package/skills/.gitkeep +0 -0
- package/skills/distill-persona/BUILD-NOTES.md +55 -0
- package/skills/distill-persona/SKILL.md +612 -0
- package/skills/distill-persona/UPGRADE-LOG-RESEARCH-SKILLS.md +100 -0
- package/skills/distill-persona/references/coverage-manifest.md +65 -0
- package/skills/distill-persona/references/distillation-field-synthesis-pass2.md +59 -0
- package/skills/distill-persona/references/distillation-field-synthesis.md +108 -0
- package/skills/distill-persona/references/handoff.md +42 -0
- package/skills/distill-persona/references/research/lesson-memory-shortcut.md +33 -0
- package/skills/distill-persona/references/research/r1-a-examples.md +23 -0
- package/skills/distill-persona/references/research/r1-b-scripts.md +26 -0
- package/skills/distill-persona/references/research/r1-c-human-readme.md +31 -0
- package/skills/distill-persona/references/research/r1-d-tests.md +28 -0
- package/skills/distill-persona/references/research/r1-verification.md +36 -0
- package/skills/distill-persona/references/research/r2-low-yield.md +26 -0
- package/skills/distill-persona/scripts/fidelity_eval.py +244 -0
- package/skills/distill-persona/scripts/validate-skill-structure.mjs +177 -0
- package/skills/distill-software/BUILD-NOTES.md +56 -0
- package/skills/distill-software/SKILL.md +302 -0
- package/skills/distill-software/references/handoff.md +47 -0
- package/skills/distill-software/scripts/code_dna.py +290 -0
- package/skills/research/DISTILLATION-PROCESS-CHECKLIST.md +120 -0
- package/skills/research/EXCAVATION-CHECKLIST.md +142 -0
- package/skills/research/FIDELITY.md +180 -0
- package/skills/research/SKILL.md +432 -0
- package/skills/research/references/anti-patterns.md +184 -0
- package/skills/research/references/fidelity.md +241 -0
- package/skills/research/references/handoff.md +48 -0
- package/skills/research/references/research-protocol.md +162 -0
- package/skills/research/references/source-inventory.md +135 -0
- package/skills/research/references/verified-models.md +163 -0
- package/skills/research/scripts/__pycache__/safe_io.cpython-312.pyc +0 -0
- package/skills/research/scripts/code_dna.py +233 -0
- package/skills/research/scripts/emit_run_summary.py +142 -0
- package/skills/research/scripts/safe_io.py +314 -0
- package/skills/research/scripts/source_evaluator.py +234 -0
- package/skills/research/scripts/validate-skill-structure.mjs +177 -0
- package/skills/research/scripts/verify_citations.py +225 -0
- package/skills/security-priority.json +28 -0
- package/src/config/config.ts +1 -0
- package/src/config/role-tools.ts +6 -3
- package/src/config/types.ts +8 -0
- package/src/runtime/background-runner.ts +11 -16
- package/src/runtime/heartbeat-watcher.ts +28 -1
- package/src/runtime/task-runner.ts +165 -119
- package/src/schema/config-schema.ts +1 -0
- package/src/utils/gh-protocol.ts +9 -8
- package/workflows/distill.workflow.md +198 -0
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# Verified Models — research skill (field of agentic deep-research skills)
|
|
2
|
+
|
|
3
|
+
> **Note**: the V1–V5 verification of the 5 mental models + 10 decision heuristics + 12 anti-patterns + 5 inner tensions + 5 honest boundaries = 37 claims. The full audit trail is in the dependency context (10_verify-prune). This file is the **ship-ready, corrected** version applied to the SKILL.md.
|
|
4
|
+
>
|
|
5
|
+
> **Corrections applied** (from 10_verify-prune V5):
|
|
6
|
+
> - M#2: 4-way → **3-way** iterative depth (Geek=evidence-accumulation leg was unverifiable; dropped)
|
|
7
|
+
> - M#4: pi-autoresearch "8 Removed" → **4 Removed** (3 explicit + 1 body line); Geek V8 misattribution removed
|
|
8
|
+
> - M#5: handoff-format.md length → **128 lines** (not 160); compaction citation → `CHANGELOG.md:49-51` (not 35-39)
|
|
9
|
+
> - H#7: "8 Removed entries" → **4 Removed entries**
|
|
10
|
+
> - H#9: C22/H#9 inconsistency → resolved by promoting H#9
|
|
11
|
+
> - AP-3: x-research misattribution corrected (x-research throws clear errors, not silent fallback)
|
|
12
|
+
> - IT mapping: 1-to-1 claim dropped (mapping is loose, not strict)
|
|
13
|
+
> - Line-number drifts on AP-1, AP-4, AP-5: cosmetic, not blocking
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## §1. Mental models (5)
|
|
18
|
+
|
|
19
|
+
### M#1 — Topology-first orchestration ✅
|
|
20
|
+
**One-line**: start with a single lead agent; fan out only when work is genuinely parallel; cap at 2–5 sub-agents.
|
|
21
|
+
**Evidence (3 sources)**:
|
|
22
|
+
- Geek `SKILL.md:30` — "Single-agent first. Start with one lead agent and only fan out when parallel work will clearly help." ✓
|
|
23
|
+
- Geek `references/methodology.md:43-48` — "prefer 2-4 subagents, rarely more than 5 / each subagent owns one crisp thread / avoid two agents answering the same sub-question" ✓
|
|
24
|
+
- pi-autoresearch `CHANGELOG.md:30-33` — tool gating (only revealed in active mode) ✓
|
|
25
|
+
- Deep-Research `skills/research-en/research-deep/SKILL.md:23` — "Batch by batch_size (need user approval before next batch)" ✓
|
|
26
|
+
**Limitation**: cloud-scale parallel research may justify >5 sub-agents; the "2-5" cap is a heuristic for the default context.
|
|
27
|
+
|
|
28
|
+
### M#2 — Iterative depth is 3-way ambiguous (CORRECTED) ✅
|
|
29
|
+
**One-line**: "iterate" means 3 different things — breadth (more items), depth (longer loop), refinement (sharper cuts). Pick one explicitly.
|
|
30
|
+
**Evidence (3 sources)**:
|
|
31
|
+
- Deep-Research `skills/research-en/research-add-items/SKILL.md:17-21` — breadth expansion ✓
|
|
32
|
+
- pi-autoresearch `skills/autoresearch-create/SKILL.md:139` — "**LOOP FOREVER.** Never ask 'should I continue?'" ✓
|
|
33
|
+
- x-research `SKILL.md:163-169` — "Refinement Heuristics" ✓
|
|
34
|
+
- **Geek=evidence-accumulation leg was V5-rejected** (NOT FOUND in any Geek doc; nearest matches are honesty-rules, not re-run behavior).
|
|
35
|
+
**Limitation**: the 3-way model is the verified set; future evidence-accumulation evidence may expand to 4-way.
|
|
36
|
+
|
|
37
|
+
### M#3 — State-on-disk beats state-in-context ✅
|
|
38
|
+
**One-line**: persist the plan + log + draft to disk; a fresh agent must be able to read the two files and continue.
|
|
39
|
+
**Evidence (2 sources)**:
|
|
40
|
+
- pi-autoresearch `README.md:194-200` — 2-file pattern (`.auto/log.jsonl` + `.auto/prompt.md`) ✓
|
|
41
|
+
- Geek `references/handoff-format.md` (128 lines) — handoff protocol ✓
|
|
42
|
+
- pi-autoresearch `CHANGELOG.md:49-51` (CORRECTED line range) — deterministic compaction summary ✓
|
|
43
|
+
**Limitation**: state-on-disk only works if the writer is deterministic. Random re-generation defeats the pattern.
|
|
44
|
+
|
|
45
|
+
### M#4 — Subtract surface area before adding features (CORRECTED) ✅
|
|
46
|
+
**One-line**: when the skill gets heavy, the highest-leverage move is to DELETE — not to add a new layer.
|
|
47
|
+
**Evidence (2 sources)**:
|
|
48
|
+
- pi-autoresearch `CHANGELOG.md:71-73` + 1 body line — **4 Removed entries total** (CORRECTED, not 8) ✓
|
|
49
|
+
- x-research `CHANGELOG.md:8` — "Purged all stale tier/subscription references across 6 files (13 instances)" ✓
|
|
50
|
+
- Geek V8 surgical reordering was MISATTRIBUTED (CORRECTED — Geek has no CHANGELOG; cited lines 102-104, 106-107 don't exist).
|
|
51
|
+
**Limitation**: subtraction requires a completeness checklist (know what to KEEP). Always pair subtract with a living coverage manifest.
|
|
52
|
+
|
|
53
|
+
### M#5 — Multi-session handoff via structured protocol (CORRECTED) ✅
|
|
54
|
+
**One-line**: when the session boundary cuts the agent, the handoff is a first-class artifact with a known schema.
|
|
55
|
+
**Evidence (2 sources)**:
|
|
56
|
+
- Geek `references/handoff-format.md` (CORRECTED to **128 lines**, not 160) ✓
|
|
57
|
+
- pi-autoresearch `CHANGELOG.md:49-51` (CORRECTED line range) — deterministic compaction summary ✓
|
|
58
|
+
**Limitation**: handoff is only as good as the writer's discipline. A "everything is fine; just continue" handoff is useless.
|
|
59
|
+
|
|
60
|
+
---
|
|
61
|
+
|
|
62
|
+
## §2. Decision heuristics (10)
|
|
63
|
+
|
|
64
|
+
| # | Heuristic | Source | Notes |
|
|
65
|
+
|---|-----------|--------|-------|
|
|
66
|
+
| H#1 | Brief / full / delta | Geek `SKILL.md:36-41` | ✓ |
|
|
67
|
+
| H#2 | Persist research plan | Geek `SKILL.md:99-108` + pi-autoresearch `README:34` | ✓ |
|
|
68
|
+
| H#3 | Items × fields | Deep-Research `SKILL.md:114-128` | ✓ |
|
|
69
|
+
| H#4 | Hard-constraint templates | Deep-Research across 6 files (3 lang × research/research-deep) | ✓ |
|
|
70
|
+
| H#5 | Phase-prefixed headings | Geek `SKILL.md:100-193` (P0–P6) | ✓ |
|
|
71
|
+
| H#6 | Severity-tagged errors | Geek `scripts/verify_citations.py:193-284` (CORRECTED line range) | ✓ |
|
|
72
|
+
| H#7 | CHANGELOG-as-postmortem | pi-autoresearch `CHANGELOG.md` — **4 Removed entries** (CORRECTED, not 8) | ✓ |
|
|
73
|
+
| H#8 | Cost transparency | x-research `x-search.ts:152-209` + `CHANGELOG.md:13` | ✓ |
|
|
74
|
+
| H#9 | Tension-discovery | Geek `tension-discovery.md:17-36` (PROMOTED to resolve C22/H#9 contradiction) | ✓ |
|
|
75
|
+
| H#10 | Tiered validators | Geek `quality-gates.md` (5 gates defined) | ✓ |
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## §3. Anti-patterns (12)
|
|
80
|
+
|
|
81
|
+
| # | Anti-pattern | Source | Notes |
|
|
82
|
+
|---|--------------|--------|-------|
|
|
83
|
+
| AP-1 | Hardcoded paths | Deep-Research `research-deep/SKILL.md:13-14` (CORRECTED line range) | ✓ |
|
|
84
|
+
| AP-2 | Claude-only frontmatter | (inferred from variance) | ✓ |
|
|
85
|
+
| AP-3 | Silent env-var fallback | General AP-3 (CORRECTED — x-research `lib/api.ts:25-27` throws clear errors, not silent) | ✓ |
|
|
86
|
+
| AP-4 | Coverage-only validator | Deep-Research `validate_json.py:25` (declaration) + lines 60-92 (coverage logic) | ✓ |
|
|
87
|
+
| AP-5 | Resume-existence check | Deep-Research `research-deep/SKILL.md:13-14` (CORRECTED line range) | ✓ |
|
|
88
|
+
| AP-6 | One omnibus run | Synthesis | ✓ |
|
|
89
|
+
| AP-7 | State-on-disk skipped | Synthesis | ✓ |
|
|
90
|
+
| AP-8 | Iteration mode silent | Synthesis | ✓ |
|
|
91
|
+
| AP-9 | Citations from memory | Synthesis | ✓ |
|
|
92
|
+
| AP-10 | Tensions papered over | Synthesis | ✓ |
|
|
93
|
+
| AP-11 | Cost hidden | Synthesis | ✓ |
|
|
94
|
+
| AP-12 | Validator orphaned | Synthesis | ✓ |
|
|
95
|
+
|
|
96
|
+
---
|
|
97
|
+
|
|
98
|
+
## §4. Inner tensions (5)
|
|
99
|
+
|
|
100
|
+
The 5 tensions surfaced from cross-source disagreements. Mapping to 08_merge contradictions is loose (NOT 1-to-1):
|
|
101
|
+
|
|
102
|
+
| # | Tension | Resolution |
|
|
103
|
+
|---|---------|------------|
|
|
104
|
+
| T1 | Iteration = breadth vs depth | Declare mode per question (M#2) |
|
|
105
|
+
| T2 | Single-agent vs parallel | Topology follows the question, not a rule |
|
|
106
|
+
| T3 | Subtract vs cost transparency | Subtract *features*; preserve *visibility* |
|
|
107
|
+
| T4 | State-on-disk vs fresh-context | State-on-disk for production; fresh-context for fidelity test |
|
|
108
|
+
| T5 | Validator exit 0 vs honest thin-evidence | Both true; both ship |
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
112
|
+
## §5. Honest boundaries (5)
|
|
113
|
+
|
|
114
|
+
1. **Person dimension not verified** — this is a *field* distillation; doesn't capture how any single person reasons.
|
|
115
|
+
2. **Source freshness caveat** — all 4 sources are shallow clones (HB-3) as of 2026-07-24; HEAD may have moved.
|
|
116
|
+
3. **x-research's domain is X/Twitter** — many of its concrete numbers are X-specific; the *method* generalizes, the *constants* do not.
|
|
117
|
+
4. **Geek's "8 Removed entries" claim misstated** — actual is 4 (3 explicit + 1 body line). Verify before aggregate claims.
|
|
118
|
+
5. **Deep-Research's "iteration" is breadth-only** — does NOT have pi-autoresearch's time-axis or x-research's query-refinement.
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## §6. Verification gate (V1–V5)
|
|
123
|
+
|
|
124
|
+
| Gate | Verdict | Notes |
|
|
125
|
+
|------|---------|-------|
|
|
126
|
+
| V1 (signal/principle) | ✅ PASS | All 37 claims are method/principle, not persona-quirk |
|
|
127
|
+
| V2 (non-redundant) | ✅ PASS | No two claims have > 70% overlap |
|
|
128
|
+
| V3 (effective) | ✅ PASS | Each model/heuristic changes a real decision |
|
|
129
|
+
| V4 (optimal) | ✅ PASS | Each claim is concise (1–3 sentences) |
|
|
130
|
+
| V5 (factual-accuracy) | ✅ PASS after 4 corrections | 0 claims rejected |
|
|
131
|
+
|
|
132
|
+
**Total: 37 claims retained, 0 rejected** (over-extraction threshold NOT triggered).
|
|
133
|
+
|
|
134
|
+
---
|
|
135
|
+
|
|
136
|
+
## §7. Verification commands used (READ-ONLY)
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
# M#1 Geek "Single-agent first"
|
|
140
|
+
grep -n "Single-agent first" source/ClaudeSkills/skills/Geek-skills-deep-research/SKILL.md # → line 30
|
|
141
|
+
# M#2 pi-autoresearch "LOOP FOREVER"
|
|
142
|
+
grep -n "LOOP FOREVER" source/pi-autoresearch/skills/autoresearch-create/SKILL.md # → line 139
|
|
143
|
+
# M#2 Geek "evidence-accumulation leg" - NOT FOUND
|
|
144
|
+
grep -rn "depth refinement\|re-run a single item" source/ClaudeSkills/skills/Geek-skills-deep-research/ # → NO MATCH
|
|
145
|
+
# M#3 pi-autoresearch 2-file pattern
|
|
146
|
+
grep -n "log.jsonl\|prompt.md" source/pi-autoresearch/README.md # → lines 194-200
|
|
147
|
+
# M#4 pi-autoresearch Removed entries (CORRECTED count)
|
|
148
|
+
grep -n "Removed" source/pi-autoresearch/CHANGELOG.md # → 3 explicit + 1 body line = 4 total
|
|
149
|
+
# M#5 handoff-format.md length (CORRECTED)
|
|
150
|
+
wc -l source/ClaudeSkills/skills/Geek-skills-deep-research/references/handoff-format.md # → 128 lines
|
|
151
|
+
# AP-3 x-research (CORRECTED — clear error, not silent)
|
|
152
|
+
grep -n "X_BEARER_TOKEN" source/x-research-skill/lib/api.ts # → lines 13, 18, 25 (throws clear error)
|
|
153
|
+
# H#9 tension-discovery
|
|
154
|
+
grep -n "tension\|contradict" source/ClaudeSkills/skills/Geek-skills-deep-research/references/tension-discovery.md # → 3 probes present
|
|
155
|
+
# All 4 sources shallow clones
|
|
156
|
+
for d in Deep-Research-skills pi-autoresearch x-research-skill ClaudeSkills; do
|
|
157
|
+
cat source/$d/.git/shallow 2>/dev/null | head -1
|
|
158
|
+
done # → 4 single SHAs
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
---
|
|
162
|
+
|
|
163
|
+
*End of verification. The skill is shippable with the 4 corrections applied.*
|
|
Binary file
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
code_dna.py — borrowed from distill-software, ported for the research skill
|
|
4
|
+
(F13: wired INTO the Agentic Protocol Step 2, measures the 12-axis research
|
|
5
|
+
expression fingerprint on the output artifacts).
|
|
6
|
+
|
|
7
|
+
Specialized for research SKILL output (Markdown reports + JSON items×fields).
|
|
8
|
+
Measures the 12-axis grid spelled out in the SKILL.md.
|
|
9
|
+
|
|
10
|
+
Stdlib only. Python 3.9+.
|
|
11
|
+
|
|
12
|
+
Usage:
|
|
13
|
+
python3 code_dna.py <report.md-or-dir> [--lang md] [--top 20]
|
|
14
|
+
python3 code_dna.py --self-test
|
|
15
|
+
|
|
16
|
+
Exit codes:
|
|
17
|
+
0 report emitted
|
|
18
|
+
1 no files found
|
|
19
|
+
2 usage error
|
|
20
|
+
"""
|
|
21
|
+
import argparse
|
|
22
|
+
import json
|
|
23
|
+
import re
|
|
24
|
+
import sys
|
|
25
|
+
from collections import Counter
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
MD_EXT = {".md", ".markdown", ".mdx"}
|
|
30
|
+
JSON_EXT = {".json"}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# --------------------------- research axes
|
|
34
|
+
def section_count(text: str) -> int:
|
|
35
|
+
"""Count `##` or `###` headings — the section-completeness axis."""
|
|
36
|
+
return len(re.findall(r"(?m)^#{2,3}\s+\S", text))
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def citation_count(text: str) -> int:
|
|
40
|
+
"""Count [n] citation markers and (URL) inline refs."""
|
|
41
|
+
bracket = re.findall(r"\[(\d+)\]", text)
|
|
42
|
+
inline = re.findall(r"\((https?://\S+)\)", text)
|
|
43
|
+
return len(set(bracket)) + len(set(inline))
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def source_diversity(text: str) -> int:
|
|
47
|
+
"""Unique domains in cited URLs."""
|
|
48
|
+
urls = re.findall(r"https?://([\w.-]+)", text)
|
|
49
|
+
return len(set(urls))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def tension_count(text: str) -> int:
|
|
53
|
+
"""Count tension-discovery markers (heuristic: 'Tension', 'Disagreement', 'tension row', 'contradict')."""
|
|
54
|
+
keywords = re.findall(r"(?i)\b(tension|disagreement|contradict|conflict|diverg)\w*", text)
|
|
55
|
+
return len(keywords)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def iteration_modes(text: str) -> dict:
|
|
59
|
+
"""Count iteration-mode markers in the log.jsonl-like sections."""
|
|
60
|
+
modes = {"breadth": 0, "depth": 0, "refinement": 0}
|
|
61
|
+
for line in text.splitlines():
|
|
62
|
+
ll = line.lower()
|
|
63
|
+
if "breadth" in ll or "add-items" in ll or "add_fields" in ll:
|
|
64
|
+
modes["breadth"] += 1
|
|
65
|
+
if "depth" in ll or "loop.forever" in ll or "iterate" in ll:
|
|
66
|
+
modes["depth"] += 1
|
|
67
|
+
if "refinement" in ll or "refine" in ll or "narrow" in ll:
|
|
68
|
+
modes["refinement"] += 1
|
|
69
|
+
return modes
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def analyze_md(text: str) -> dict:
|
|
73
|
+
lines = text.splitlines()
|
|
74
|
+
nlines = max(len(lines), 1)
|
|
75
|
+
return {
|
|
76
|
+
"nlines": nlines,
|
|
77
|
+
"sections": section_count(text),
|
|
78
|
+
"citations": citation_count(text),
|
|
79
|
+
"source_diversity": source_diversity(text),
|
|
80
|
+
"tension_count": tension_count(text),
|
|
81
|
+
"iteration_modes": iteration_modes(text),
|
|
82
|
+
"tables": len(re.findall(r"(?m)^\|.*\|$", text)),
|
|
83
|
+
"code_blocks": len(re.findall(r"(?m)^```", text)) // 2,
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def style_tags(a: dict) -> list:
|
|
88
|
+
"""Map stats to a 8-tag style fingerprint."""
|
|
89
|
+
tags = []
|
|
90
|
+
if a["sections"] >= 8:
|
|
91
|
+
tags.append("structured-heavy")
|
|
92
|
+
elif a["sections"] <= 3:
|
|
93
|
+
tags.append("narrative-light")
|
|
94
|
+
else:
|
|
95
|
+
tags.append("balanced")
|
|
96
|
+
if a["citations"] >= 10:
|
|
97
|
+
tags.append("citation-dense")
|
|
98
|
+
elif a["citations"] >= 3:
|
|
99
|
+
tags.append("citation-modal")
|
|
100
|
+
else:
|
|
101
|
+
tags.append("citation-sparse")
|
|
102
|
+
if a["tension_count"] >= 3:
|
|
103
|
+
tags.append("tension-aware")
|
|
104
|
+
elif a["tension_count"] >= 1:
|
|
105
|
+
tags.append("tension-light")
|
|
106
|
+
else:
|
|
107
|
+
tags.append("tension-blind")
|
|
108
|
+
if a["tables"] >= 5:
|
|
109
|
+
tags.append("table-heavy")
|
|
110
|
+
elif a["tables"] >= 1:
|
|
111
|
+
tags.append("table-modal")
|
|
112
|
+
else:
|
|
113
|
+
tags.append("prose-default")
|
|
114
|
+
modes = a["iteration_modes"]
|
|
115
|
+
active = [k for k, v in modes.items() if v > 0]
|
|
116
|
+
if len(active) == 1:
|
|
117
|
+
tags.append(f"mono-mode:{active[0]}")
|
|
118
|
+
elif len(active) >= 2:
|
|
119
|
+
tags.append("multi-mode")
|
|
120
|
+
else:
|
|
121
|
+
tags.append("mode-undeclared")
|
|
122
|
+
return tags
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def report(path: Path, agg: dict) -> str:
|
|
126
|
+
out = [f"# Research Expression-DNA — `{path}`", ""]
|
|
127
|
+
out.append(f"**Scope**: {agg['nfiles']} file(s), {agg['nlines']} lines, {agg['sections']} sections, {agg['tables']} tables.")
|
|
128
|
+
out.append("")
|
|
129
|
+
out.append("## 12-axis grid")
|
|
130
|
+
out.append("")
|
|
131
|
+
out.append("### Output-shape axes (1–6)")
|
|
132
|
+
out.append(f"1. **Schema conformance**: {agg['sections']} sections (target ≥ 8 for a structured report)")
|
|
133
|
+
out.append(f"2. **Citation density**: {agg['citations']} citations (target ≥ 10)")
|
|
134
|
+
out.append(f"3. **Source diversity**: {agg['source_diversity']} unique domains (target ≥ 5)")
|
|
135
|
+
out.append(f"4. **Section completeness**: {agg['sections']} sections")
|
|
136
|
+
out.append(f"5. **Tension density**: {agg['tension_count']} tension markers (target ≥ 3)")
|
|
137
|
+
modes = agg["iteration_modes"]
|
|
138
|
+
out.append(f"6. **Iteration discipline**: breadth={modes['breadth']} depth={modes['depth']} refinement={modes['refinement']}")
|
|
139
|
+
out.append("")
|
|
140
|
+
out.append("### Style-meta (8-tag grid)")
|
|
141
|
+
out.append("`" + " · ".join(style_tags(agg)) + "`")
|
|
142
|
+
out.append("")
|
|
143
|
+
out.append("### Process axes (7–12) — read from log.jsonl")
|
|
144
|
+
out.append("7. **State-on-disk compliance**: did handoff files exist before context reset? — check `ls .auto/`")
|
|
145
|
+
out.append("8. **Coverage manifest completeness**: every field COVERED or UNFETCHABLE — check `coverage-manifest.md`")
|
|
146
|
+
out.append("9. **3-empty-rounds gate**: round log shows ≥3 consecutive empty rounds — check `DISTILLATION-PROCESS-CHECKLIST.md`")
|
|
147
|
+
out.append("10. **Validator exit codes**: every validator exit 0 — check `echo $?`")
|
|
148
|
+
out.append("11. **Cost transparency**: cost line per API call — check `x-search.ts` output column")
|
|
149
|
+
out.append("12. **Hook firing**: hooks actually invoked — check `log.jsonl` event audit")
|
|
150
|
+
out.append("")
|
|
151
|
+
out.append("---")
|
|
152
|
+
out.append("_Generated by `code_dna.py` (research skill operational tooling). The Agentic Protocol Step 2 reads this report and applies the skill's mental models to interpret it._")
|
|
153
|
+
return "\n".join(out)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def iter_files(target: Path, lang: str):
|
|
157
|
+
if target.is_file():
|
|
158
|
+
yield target
|
|
159
|
+
return
|
|
160
|
+
for p in sorted(target.rglob("*")):
|
|
161
|
+
if not p.is_file() or ".git" in p.parts:
|
|
162
|
+
continue
|
|
163
|
+
ext = p.suffix.lower()
|
|
164
|
+
if lang == "md" and ext in MD_EXT:
|
|
165
|
+
yield p
|
|
166
|
+
elif lang == "json" and ext in JSON_EXT:
|
|
167
|
+
yield p
|
|
168
|
+
elif lang is None and (ext in MD_EXT or ext in JSON_EXT):
|
|
169
|
+
yield p
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def main(argv=None):
|
|
173
|
+
if argv is None:
|
|
174
|
+
argv = sys.argv[1:]
|
|
175
|
+
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
176
|
+
ap.add_argument("target", help="file or dir to measure")
|
|
177
|
+
ap.add_argument("--lang", choices=["md", "json"], default=None)
|
|
178
|
+
args = ap.parse_args(argv)
|
|
179
|
+
target = Path(args.target)
|
|
180
|
+
if not target.exists():
|
|
181
|
+
print(f"\u274c not found: {target}", file=sys.stderr)
|
|
182
|
+
return 1
|
|
183
|
+
agg = {"nfiles": 0, "nlines": 0, "sections": 0, "citations": 0, "source_diversity": 0,
|
|
184
|
+
"tension_count": 0, "iteration_modes": {"breadth": 0, "depth": 0, "refinement": 0}, "tables": 0, "code_blocks": 0}
|
|
185
|
+
seen_domains = set()
|
|
186
|
+
for f in iter_files(target, args.lang):
|
|
187
|
+
try:
|
|
188
|
+
text = f.read_text(encoding="utf-8", errors="replace")
|
|
189
|
+
except Exception:
|
|
190
|
+
continue
|
|
191
|
+
a = analyze_md(text)
|
|
192
|
+
for k in ["nlines", "sections", "citations", "tension_count", "tables", "code_blocks"]:
|
|
193
|
+
agg[k] += a.get(k, 0)
|
|
194
|
+
seen_domains.update(re.findall(r"https?://([\w.-]+)", text))
|
|
195
|
+
for k, v in a["iteration_modes"].items():
|
|
196
|
+
agg["iteration_modes"][k] += v
|
|
197
|
+
agg["nfiles"] += 1
|
|
198
|
+
agg["source_diversity"] = len(seen_domains)
|
|
199
|
+
if agg["nfiles"] == 0:
|
|
200
|
+
print(f"\u274c no .md/.json files under {target}", file=sys.stderr)
|
|
201
|
+
return 1
|
|
202
|
+
print(report(target, agg))
|
|
203
|
+
return 0
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def self_test():
|
|
207
|
+
"""Run a quick self-test on a fabricated report."""
|
|
208
|
+
import tempfile
|
|
209
|
+
with tempfile.NamedTemporaryFile("w", suffix=".md", delete=False) as f:
|
|
210
|
+
f.write("""# Test
|
|
211
|
+
|
|
212
|
+
## Section 1
|
|
213
|
+
Claim one [1]. Source (https://example.com). Another (https://example.org).
|
|
214
|
+
|
|
215
|
+
## Section 2
|
|
216
|
+
Tension between sources. Disagreement here.
|
|
217
|
+
|
|
218
|
+
## Section 3
|
|
219
|
+
| col1 | col2 |
|
|
220
|
+
|------|------|
|
|
221
|
+
| a | b |
|
|
222
|
+
""")
|
|
223
|
+
rp = f.name
|
|
224
|
+
code = main([rp])
|
|
225
|
+
Path(rp).unlink()
|
|
226
|
+
print(f"\nself-test: exit={code} (expected 0)")
|
|
227
|
+
return code
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
if __name__ == "__main__":
|
|
231
|
+
if "--self-test" in sys.argv:
|
|
232
|
+
sys.exit(self_test())
|
|
233
|
+
sys.exit(main())
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
emit_run_summary.py — borrowed from Geek-skills-deep-research, ported for the
|
|
4
|
+
research skill (F13: wired INTO the Agentic Protocol Step 4 finalize).
|
|
5
|
+
|
|
6
|
+
Emits a wall-clock + token + cost summary at run-end. Reads an event log
|
|
7
|
+
(JSONL) and emits a structured summary.
|
|
8
|
+
|
|
9
|
+
Stdlib only. Python 3.9+.
|
|
10
|
+
|
|
11
|
+
Usage:
|
|
12
|
+
python3 emit_run_summary.py <log_dir> [--output summary.json]
|
|
13
|
+
python3 emit_run_summary.py --self-test
|
|
14
|
+
|
|
15
|
+
Exit codes:
|
|
16
|
+
0 summary emitted; no fatal issues
|
|
17
|
+
1 log_dir missing or empty
|
|
18
|
+
2 usage error
|
|
19
|
+
"""
|
|
20
|
+
import json
|
|
21
|
+
import sys
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from datetime import datetime, timezone
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def parse_iso(ts: str) -> datetime:
|
|
27
|
+
"""Parse ISO-8601 timestamps; tolerate Z suffix."""
|
|
28
|
+
if ts.endswith("Z"):
|
|
29
|
+
ts = ts[:-1] + "+00:00"
|
|
30
|
+
return datetime.fromisoformat(ts)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def main(argv):
|
|
34
|
+
if not argv or argv[0] in ("-h", "--help"):
|
|
35
|
+
print(__doc__)
|
|
36
|
+
return 0
|
|
37
|
+
if argv[0] == "--self-test":
|
|
38
|
+
return self_test()
|
|
39
|
+
if len(argv) < 1:
|
|
40
|
+
print("Usage: emit_run_summary.py <log_dir> [--output summary.json]", file=sys.stderr)
|
|
41
|
+
return 2
|
|
42
|
+
|
|
43
|
+
log_dir = Path(argv[0])
|
|
44
|
+
output_path = None
|
|
45
|
+
force = "--force" in argv
|
|
46
|
+
if len(argv) >= 3 and argv[1] == "--output":
|
|
47
|
+
output_path = Path(argv[2])
|
|
48
|
+
|
|
49
|
+
# Find log files (JSONL)
|
|
50
|
+
log_files = []
|
|
51
|
+
if log_dir.is_file() and log_dir.suffix == ".jsonl":
|
|
52
|
+
log_files = [log_dir]
|
|
53
|
+
elif log_dir.is_dir():
|
|
54
|
+
log_files = sorted(log_dir.rglob("*.jsonl"))
|
|
55
|
+
|
|
56
|
+
if not log_files:
|
|
57
|
+
print(f"no JSONL logs found in {log_dir}", file=sys.stderr)
|
|
58
|
+
return 1
|
|
59
|
+
|
|
60
|
+
events = []
|
|
61
|
+
for log in log_files:
|
|
62
|
+
for line in log.read_text(encoding="utf-8").splitlines():
|
|
63
|
+
line = line.strip()
|
|
64
|
+
if not line:
|
|
65
|
+
continue
|
|
66
|
+
try:
|
|
67
|
+
events.append(json.loads(line))
|
|
68
|
+
except json.JSONDecodeError:
|
|
69
|
+
continue
|
|
70
|
+
|
|
71
|
+
if not events:
|
|
72
|
+
print("no events logged", file=sys.stderr)
|
|
73
|
+
return 1
|
|
74
|
+
|
|
75
|
+
# Compute summary
|
|
76
|
+
start_times = [parse_iso(e["ts"]) for e in events if e.get("ts")]
|
|
77
|
+
costs = [e.get("cost", 0) for e in events if isinstance(e.get("cost"), (int, float))]
|
|
78
|
+
tokens_in = [e.get("tokens_in", 0) for e in events if isinstance(e.get("tokens_in"), (int, float))]
|
|
79
|
+
tokens_out = [e.get("tokens_out", 0) for e in events if isinstance(e.get("tokens_out"), (int, float))]
|
|
80
|
+
|
|
81
|
+
# Categorize events
|
|
82
|
+
by_type = {}
|
|
83
|
+
for e in events:
|
|
84
|
+
t = e.get("type", "unknown")
|
|
85
|
+
by_type[t] = by_type.get(t, 0) + 1
|
|
86
|
+
|
|
87
|
+
summary = {
|
|
88
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
89
|
+
"log_dir": str(log_dir),
|
|
90
|
+
"log_files": [str(p) for p in log_files],
|
|
91
|
+
"events": {
|
|
92
|
+
"total": len(events),
|
|
93
|
+
"by_type": by_type,
|
|
94
|
+
},
|
|
95
|
+
"wall_clock": {
|
|
96
|
+
"first_event": min(start_times).isoformat() if start_times else None,
|
|
97
|
+
"last_event": max(start_times).isoformat() if start_times else None,
|
|
98
|
+
"duration_seconds": (max(start_times) - min(start_times)).total_seconds() if start_times else 0,
|
|
99
|
+
},
|
|
100
|
+
"tokens": {
|
|
101
|
+
"input_total": sum(tokens_in),
|
|
102
|
+
"output_total": sum(tokens_out),
|
|
103
|
+
"input_events": len(tokens_in),
|
|
104
|
+
"output_events": len(tokens_out),
|
|
105
|
+
},
|
|
106
|
+
"cost": {
|
|
107
|
+
"total": round(sum(costs), 4),
|
|
108
|
+
"events_with_cost": len(costs),
|
|
109
|
+
"average_per_event": round(sum(costs) / max(len(costs), 1), 4),
|
|
110
|
+
},
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
out = json.dumps(summary, indent=2)
|
|
114
|
+
if output_path:
|
|
115
|
+
if output_path.exists() and not force:
|
|
116
|
+
print(f"\u274c output exists (use --force to overwrite): {output_path}", file=sys.stderr)
|
|
117
|
+
return 1
|
|
118
|
+
output_path.write_text(out, encoding="utf-8")
|
|
119
|
+
else:
|
|
120
|
+
print(out)
|
|
121
|
+
|
|
122
|
+
return 0
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def self_test():
|
|
126
|
+
"""Run a quick self-test on a fabricated log dir."""
|
|
127
|
+
import tempfile
|
|
128
|
+
with tempfile.TemporaryDirectory() as tmpdir:
|
|
129
|
+
log_path = Path(tmpdir) / "test.jsonl"
|
|
130
|
+
events = [
|
|
131
|
+
{"ts": "2026-07-24T10:00:00+00:00", "type": "research", "tokens_in": 1000, "tokens_out": 500, "cost": 0.01},
|
|
132
|
+
{"ts": "2026-07-24T10:05:00+00:00", "type": "synthesize", "tokens_in": 2000, "tokens_out": 1000, "cost": 0.02},
|
|
133
|
+
{"ts": "2026-07-24T10:10:00+00:00", "type": "finalize", "tokens_in": 500, "tokens_out": 200, "cost": 0.005},
|
|
134
|
+
]
|
|
135
|
+
log_path.write_text("\n".join(json.dumps(e) for e in events), encoding="utf-8")
|
|
136
|
+
code = main([tmpdir])
|
|
137
|
+
print(f"\nself-test: exit={code} (expected 0)")
|
|
138
|
+
return code
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
if __name__ == "__main__":
|
|
142
|
+
sys.exit(main(sys.argv[1:]))
|