pi-crew 0.9.48 → 0.9.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/AGENTS.md +18 -0
  2. package/CHANGELOG.md +105 -0
  3. package/dist/build-meta.json +22 -12
  4. package/dist/index.mjs +430 -388
  5. package/dist/index.mjs.map +3 -3
  6. package/docs/decisions/2026-07-24-oidc-trusted-publishing.md +112 -0
  7. package/package.json +2 -2
  8. package/skills/.gitkeep +0 -0
  9. package/skills/distill-persona/BUILD-NOTES.md +55 -0
  10. package/skills/distill-persona/SKILL.md +612 -0
  11. package/skills/distill-persona/UPGRADE-LOG-RESEARCH-SKILLS.md +100 -0
  12. package/skills/distill-persona/references/coverage-manifest.md +65 -0
  13. package/skills/distill-persona/references/distillation-field-synthesis-pass2.md +59 -0
  14. package/skills/distill-persona/references/distillation-field-synthesis.md +108 -0
  15. package/skills/distill-persona/references/handoff.md +42 -0
  16. package/skills/distill-persona/references/research/lesson-memory-shortcut.md +33 -0
  17. package/skills/distill-persona/references/research/r1-a-examples.md +23 -0
  18. package/skills/distill-persona/references/research/r1-b-scripts.md +26 -0
  19. package/skills/distill-persona/references/research/r1-c-human-readme.md +31 -0
  20. package/skills/distill-persona/references/research/r1-d-tests.md +28 -0
  21. package/skills/distill-persona/references/research/r1-verification.md +36 -0
  22. package/skills/distill-persona/references/research/r2-low-yield.md +26 -0
  23. package/skills/distill-persona/scripts/fidelity_eval.py +244 -0
  24. package/skills/distill-persona/scripts/validate-skill-structure.mjs +177 -0
  25. package/skills/distill-software/BUILD-NOTES.md +56 -0
  26. package/skills/distill-software/SKILL.md +302 -0
  27. package/skills/distill-software/references/handoff.md +47 -0
  28. package/skills/distill-software/scripts/code_dna.py +290 -0
  29. package/skills/research/DISTILLATION-PROCESS-CHECKLIST.md +120 -0
  30. package/skills/research/EXCAVATION-CHECKLIST.md +142 -0
  31. package/skills/research/FIDELITY.md +180 -0
  32. package/skills/research/SKILL.md +432 -0
  33. package/skills/research/references/anti-patterns.md +184 -0
  34. package/skills/research/references/fidelity.md +241 -0
  35. package/skills/research/references/handoff.md +48 -0
  36. package/skills/research/references/research-protocol.md +162 -0
  37. package/skills/research/references/source-inventory.md +135 -0
  38. package/skills/research/references/verified-models.md +163 -0
  39. package/skills/research/scripts/__pycache__/safe_io.cpython-312.pyc +0 -0
  40. package/skills/research/scripts/code_dna.py +233 -0
  41. package/skills/research/scripts/emit_run_summary.py +142 -0
  42. package/skills/research/scripts/safe_io.py +314 -0
  43. package/skills/research/scripts/source_evaluator.py +234 -0
  44. package/skills/research/scripts/validate-skill-structure.mjs +177 -0
  45. package/skills/research/scripts/verify_citations.py +225 -0
  46. package/skills/security-priority.json +28 -0
  47. package/src/config/config.ts +1 -0
  48. package/src/config/role-tools.ts +6 -3
  49. package/src/config/types.ts +8 -0
  50. package/src/runtime/background-runner.ts +11 -16
  51. package/src/runtime/heartbeat-watcher.ts +28 -1
  52. package/src/runtime/task-runner.ts +165 -119
  53. package/src/schema/config-schema.ts +1 -0
  54. package/src/utils/gh-protocol.ts +9 -8
  55. package/workflows/distill.workflow.md +198 -0
@@ -485,131 +485,177 @@ export async function runTeamTask(input: TaskRunnerInput): Promise<{ manifest: T
485
485
  data: { role: task.role, model: model ?? "default" },
486
486
  });
487
487
  upsertCrewAgent(manifest, recordFromTask(manifest, task, "child-process"));
488
- const childResult = await runChildPi({
489
- cwd: task.cwd,
490
- task: prompt,
491
- agent: input.agent,
492
- model,
493
- signal: input.signal,
494
- transcriptPath,
495
- maxDepth: input.limits?.maxTaskDepth,
496
- skillPaths,
497
- maxTurns: input.runtimeConfig?.maxTurns,
498
- graceTurns: input.runtimeConfig?.graceTurns,
499
- inheritContext: input.runtimeConfig?.inheritContext,
500
- parentContext: input.parentContext,
501
- excludeContextBash: input.runtimeConfig?.excludeContextBash,
502
- sessionId: manifest.sessionId,
503
- role: task.role,
504
- runId: manifest.runId,
505
- agentId: task.id,
506
- artifactsRoot: manifest.artifactsRoot,
507
- steeringFile: resolveContainedPath(`${manifest.artifactsRoot}/steering`, `${task.id}.jsonl`),
508
- onSpawn: (pid) => {
509
- try {
510
- ({ task, tasks } = checkpointTask(manifest, tasks, task, "child-spawned", pid));
511
- if (task.pendingSteers?.length) {
512
- const steeringDir = `${manifest.artifactsRoot}/steering`;
513
- // Fire-and-forget async write for steering events
514
- void appendSteeringAsync(steeringDir, task.id, task.pendingSteers);
515
- task.pendingSteers = [];
516
- tasks = persistSingleTaskUpdate(manifest, tasks, task);
517
- }
518
- } catch (err) {
519
- logInternalError("task-runner.on-spawn", err as Error, `pid=${pid}, taskId=${task.id}`);
520
- }
521
- },
522
- onLifecycleEvent: (event: ChildPiLifecycleEvent) => {
523
- void appendEventAsync(manifest.eventsPath, {
524
- type: `worker.${event.type}` as const,
525
- runId: manifest.runId,
526
- taskId: task.id,
527
- message: `Worker lifecycle: ${event.type}${event.error ? ` error=${event.error}` : ""}${event.exitCode != null ? ` exit=${event.exitCode}` : ""}`,
528
- data: { ...event },
529
- }).catch((error) =>
530
- logInternalError("task-runner.lifecycle-event", error, `taskId=${task.id}, type=${event.type}`),
531
- );
532
- },
533
- onStdoutLine: (line) => {
534
- appendCrewAgentOutput(manifest, task.id, line);
535
- persistHeartbeat();
536
- // Check for supervisor contact requests from child Pi
537
- const contact = parseSupervisorContactFromLine(line);
538
- if (contact) {
539
- recordSupervisorContact(manifest, {
540
- runId: manifest.runId,
541
- ...contact,
542
- });
488
+ // W2 fix — wall-clock timeout per task. We create our own
489
+ // AbortController, link the caller's signal to it, and abort
490
+ // from a timer. The internal signal is passed to runChildPi so
491
+ // the existing SIGTERM → SIGKILL escalation in child-pi.ts
492
+ // handles cleanup. Prevents runaway agent loops (e.g. 11_build
493
+ // in the oh-my-pi distill run that re-verified completed files
494
+ // 14+ times).
495
+ const taskTimeoutMs = input.runtimeConfig?.taskTimeoutMs ?? 0;
496
+ const timeoutController = new AbortController();
497
+ // W2 fix (memory leak) — store the listener reference so we can
498
+ // removeEventListener() it in the finally block below. { once: true }
499
+ // alone is NOT enough: when the timeout fires first, the listener
500
+ // never fires → { once: true } never auto-removes → listener stays
501
+ // attached to input.signal for the rest of the run (run-level
502
+ // signal = long-lived → leak accumulates per task run).
503
+ let externalAbortListener: (() => void) | undefined;
504
+ if (input.signal) {
505
+ if (input.signal.aborted) {
506
+ timeoutController.abort(input.signal.reason);
507
+ } else {
508
+ externalAbortListener = () => timeoutController.abort(input.signal!.reason);
509
+ input.signal.addEventListener("abort", externalAbortListener, { once: true });
510
+ }
511
+ }
512
+ let timeoutHandle: ReturnType<typeof setTimeout> | undefined;
513
+ if (taskTimeoutMs > 0 && !timeoutController.signal.aborted) {
514
+ timeoutHandle = setTimeout(() => {
515
+ if (!timeoutController.signal.aborted) {
516
+ timeoutController.abort(new Error(`Task exceeded wall-clock timeout of ${taskTimeoutMs}ms`));
543
517
  }
544
- },
545
- onJsonEvent: (event) => {
546
- // Top-level error boundary: prevent any single event from crashing the task.
547
- // Errors are logged but processing continues so subsequent events still update state.
548
- try {
549
- appendCrewAgentEvent(manifest, task.id, event);
550
- if (collectedJsonEvents && event && typeof event === "object" && !Array.isArray(event))
551
- collectedJsonEvents.push(event as Record<string, unknown>);
552
- if (collectedJsonEvents && collectedJsonEvents.length > 1000) {
553
- collectedJsonEvents.splice(0, collectedJsonEvents.length - 1000);
554
- }
555
- // Accumulate lifetime usage via message_end events (survives compaction)
556
- if (event && typeof event === "object" && (event as Record<string, unknown>).type === "message_end") {
557
- const msg = (event as Record<string, unknown>).message as Record<string, unknown> | undefined;
558
- if (msg?.role === "assistant") {
559
- const usage = msg.usage as Record<string, number> | undefined;
560
- if (usage) {
561
- task.lifetimeUsage = {
562
- input: (task.lifetimeUsage?.input ?? 0) + (usage.input ?? 0),
563
- output: (task.lifetimeUsage?.output ?? 0) + (usage.output ?? 0),
564
- cacheWrite: (task.lifetimeUsage?.cacheWrite ?? 0) + (usage.cacheWrite ?? 0),
565
- };
566
- }
518
+ }, taskTimeoutMs);
519
+ timeoutHandle.unref?.();
520
+ }
521
+ let childResult;
522
+ try {
523
+ childResult = await runChildPi({
524
+ cwd: task.cwd,
525
+ task: prompt,
526
+ agent: input.agent,
527
+ model,
528
+ signal: timeoutController.signal,
529
+ transcriptPath,
530
+ maxDepth: input.limits?.maxTaskDepth,
531
+ skillPaths,
532
+ maxTurns: input.runtimeConfig?.maxTurns,
533
+ graceTurns: input.runtimeConfig?.graceTurns,
534
+ inheritContext: input.runtimeConfig?.inheritContext,
535
+ parentContext: input.parentContext,
536
+ excludeContextBash: input.runtimeConfig?.excludeContextBash,
537
+ sessionId: manifest.sessionId,
538
+ role: task.role,
539
+ runId: manifest.runId,
540
+ agentId: task.id,
541
+ artifactsRoot: manifest.artifactsRoot,
542
+ steeringFile: resolveContainedPath(`${manifest.artifactsRoot}/steering`, `${task.id}.jsonl`),
543
+ onSpawn: (pid) => {
544
+ try {
545
+ ({ task, tasks } = checkpointTask(manifest, tasks, task, "child-spawned", pid));
546
+ if (task.pendingSteers?.length) {
547
+ const steeringDir = `${manifest.artifactsRoot}/steering`;
548
+ // Fire-and-forget async write for steering events
549
+ void appendSteeringAsync(steeringDir, task.id, task.pendingSteers);
550
+ task.pendingSteers = [];
551
+ tasks = persistSingleTaskUpdate(manifest, tasks, task);
567
552
  }
553
+ } catch (err) {
554
+ logInternalError("task-runner.on-spawn", err as Error, `pid=${pid}, taskId=${task.id}`);
568
555
  }
569
- persistHeartbeat();
570
- // Bug #3 fix: Write worker JSON events to background.log for debugging when running in background mode.
571
- // This supplements the event log so developers can see what the child Pi worker produced.
572
- if (process.env.PI_CREW_BACKGROUND_MODE === "1" && event) {
573
- const bgLogPath = `${manifest.stateRoot}/background.log`;
574
- const eventLine =
575
- typeof event === "object" && !Array.isArray(event) ? JSON.stringify(event) : String(event);
576
- // Fire-and-forget async write for background log
577
- void appendBackgroundLogAsync(bgLogPath, eventLine);
578
- }
579
- // Always keep in-memory agentProgress fresh (cheap) so the UI/events see
580
- // the latest progress, but THROTTLE the disk persist. Previously this
581
- // did a full locked read-parse-write of tasks.json on EVERY child JSON
582
- // event — a 200-event task produced 200 such cycles (Round 15 P1).
583
- // Final state is force-flushed on task completion (persistHeartbeat(true)).
584
- const nextProgress = applyAgentProgressEvent(
585
- task.agentProgress ?? emptyCrewAgentProgress(),
586
- event,
587
- task.startedAt,
556
+ },
557
+ onLifecycleEvent: (event: ChildPiLifecycleEvent) => {
558
+ void appendEventAsync(manifest.eventsPath, {
559
+ type: `worker.${event.type}` as const,
560
+ runId: manifest.runId,
561
+ taskId: task.id,
562
+ message: `Worker lifecycle: ${event.type}${event.error ? ` error=${event.error}` : ""}${event.exitCode != null ? ` exit=${event.exitCode}` : ""}`,
563
+ data: { ...event },
564
+ }).catch((error) =>
565
+ logInternalError("task-runner.lifecycle-event", error, `taskId=${task.id}, type=${event.type}`),
588
566
  );
589
- task = { ...task, agentProgress: nextProgress };
590
- tasks = updateTask(tasks, task);
591
- const progressNow = Date.now();
592
- if (progressNow - lastTaskProgressPersistedAt >= 500) {
593
- tasks = persistSingleTaskUpdate(manifest, tasks, task);
594
- lastTaskProgressPersistedAt = progressNow;
595
- }
596
- // Bridge event to UI event bus for near-instant updates
597
- const bridgeEvent = bridgeEventFromJsonEvent(manifest.runId, task.id, event);
598
- if (bridgeEvent) streamBridge?.handler(bridgeEvent);
599
- // Feed overflow recovery tracker
600
- if (input.onJsonEvent) {
601
- input.onJsonEvent(task.id, manifest.runId, event);
567
+ },
568
+ onStdoutLine: (line) => {
569
+ appendCrewAgentOutput(manifest, task.id, line);
570
+ persistHeartbeat();
571
+ // Check for supervisor contact requests from child Pi
572
+ const contact = parseSupervisorContactFromLine(line);
573
+ if (contact) {
574
+ recordSupervisorContact(manifest, {
575
+ runId: manifest.runId,
576
+ ...contact,
577
+ });
602
578
  }
603
- if (!finalCheckpointWritten && isFinalChildEvent(event)) {
604
- finalCheckpointWritten = true;
605
- ({ task, tasks } = checkpointTask(manifest, tasks, task, "child-stdout-final"));
579
+ },
580
+ onJsonEvent: (event) => {
581
+ // Top-level error boundary: prevent any single event from crashing the task.
582
+ // Errors are logged but processing continues so subsequent events still update state.
583
+ try {
584
+ appendCrewAgentEvent(manifest, task.id, event);
585
+ if (collectedJsonEvents && event && typeof event === "object" && !Array.isArray(event))
586
+ collectedJsonEvents.push(event as Record<string, unknown>);
587
+ if (collectedJsonEvents && collectedJsonEvents.length > 1000) {
588
+ collectedJsonEvents.splice(0, collectedJsonEvents.length - 1000);
589
+ }
590
+ // Accumulate lifetime usage via message_end events (survives compaction)
591
+ if (event && typeof event === "object" && (event as Record<string, unknown>).type === "message_end") {
592
+ const msg = (event as Record<string, unknown>).message as Record<string, unknown> | undefined;
593
+ if (msg?.role === "assistant") {
594
+ const usage = msg.usage as Record<string, number> | undefined;
595
+ if (usage) {
596
+ task.lifetimeUsage = {
597
+ input: (task.lifetimeUsage?.input ?? 0) + (usage.input ?? 0),
598
+ output: (task.lifetimeUsage?.output ?? 0) + (usage.output ?? 0),
599
+ cacheWrite: (task.lifetimeUsage?.cacheWrite ?? 0) + (usage.cacheWrite ?? 0),
600
+ };
601
+ }
602
+ }
603
+ }
604
+ persistHeartbeat();
605
+ // Bug #3 fix: Write worker JSON events to background.log for debugging when running in background mode.
606
+ // This supplements the event log so developers can see what the child Pi worker produced.
607
+ if (process.env.PI_CREW_BACKGROUND_MODE === "1" && event) {
608
+ const bgLogPath = `${manifest.stateRoot}/background.log`;
609
+ const eventLine =
610
+ typeof event === "object" && !Array.isArray(event) ? JSON.stringify(event) : String(event);
611
+ // Fire-and-forget async write for background log
612
+ void appendBackgroundLogAsync(bgLogPath, eventLine);
613
+ }
614
+ // Always keep in-memory agentProgress fresh (cheap) so the UI/events see
615
+ // the latest progress, but THROTTLE the disk persist. Previously this
616
+ // did a full locked read-parse-write of tasks.json on EVERY child JSON
617
+ // event — a 200-event task produced 200 such cycles (Round 15 P1).
618
+ // Final state is force-flushed on task completion (persistHeartbeat(true)).
619
+ const nextProgress = applyAgentProgressEvent(
620
+ task.agentProgress ?? emptyCrewAgentProgress(),
621
+ event,
622
+ task.startedAt,
623
+ );
624
+ task = { ...task, agentProgress: nextProgress };
625
+ tasks = updateTask(tasks, task);
626
+ const progressNow = Date.now();
627
+ if (progressNow - lastTaskProgressPersistedAt >= 500) {
628
+ tasks = persistSingleTaskUpdate(manifest, tasks, task);
629
+ lastTaskProgressPersistedAt = progressNow;
630
+ }
631
+ // Bridge event to UI event bus for near-instant updates
632
+ const bridgeEvent = bridgeEventFromJsonEvent(manifest.runId, task.id, event);
633
+ if (bridgeEvent) streamBridge?.handler(bridgeEvent);
634
+ // Feed overflow recovery tracker
635
+ if (input.onJsonEvent) {
636
+ input.onJsonEvent(task.id, manifest.runId, event);
637
+ }
638
+ if (!finalCheckpointWritten && isFinalChildEvent(event)) {
639
+ finalCheckpointWritten = true;
640
+ ({ task, tasks } = checkpointTask(manifest, tasks, task, "child-stdout-final"));
641
+ }
642
+ persistChildProgress(event);
643
+ } catch (err) {
644
+ logInternalError("task-runner.on-json-event", err as Error, `taskId=${task.id}`);
606
645
  }
607
- persistChildProgress(event);
608
- } catch (err) {
609
- logInternalError("task-runner.on-json-event", err as Error, `taskId=${task.id}`);
610
- }
611
- },
612
- });
646
+ },
647
+ });
648
+ } finally {
649
+ if (timeoutHandle) clearTimeout(timeoutHandle);
650
+ // W2 fix — release the listener so it doesn't leak. {once:true}
651
+ // only auto-removes when the listener FIRES; if the timeout
652
+ // fires first, the listener never fires and stays attached
653
+ // to input.signal (run-level signal = long-lived → leak per
654
+ // task). Remove explicitly here.
655
+ if (externalAbortListener && input.signal) {
656
+ input.signal.removeEventListener("abort", externalAbortListener);
657
+ }
658
+ }
613
659
  const evidenceStatus = childResult.exitStatus?.cancelled
614
660
  ? "cancelled"
615
661
  : childResult.error || (childResult.exitCode && childResult.exitCode !== 0)
@@ -51,6 +51,7 @@ export const PiTeamsRuntimeConfigSchema = Type.Object(
51
51
  allowChildProcessFallback: Type.Optional(Type.Boolean()),
52
52
  maxTurns: Type.Optional(Type.Integer({ minimum: 1 })),
53
53
  graceTurns: Type.Optional(Type.Integer({ minimum: 1 })),
54
+ taskTimeoutMs: Type.Optional(Type.Integer({ minimum: 1 })),
54
55
  inheritContext: Type.Optional(Type.Boolean()),
55
56
  promptMode: Type.Optional(Type.Union([Type.Literal("replace"), Type.Literal("append")])),
56
57
  groupJoin: Type.Optional(Type.Union([Type.Literal("off"), Type.Literal("group"), Type.Literal("smart")])),
@@ -22,6 +22,7 @@
22
22
  * Repo resolution: git remote get-url origin from cwd.
23
23
  */
24
24
  import { execFileSync } from "node:child_process";
25
+ import { errorMessage } from "./guards.ts";
25
26
 
26
27
  /** Resolve the default repo from `git remote get-url origin` in cwd. */
27
28
  export function resolveDefaultRepo(cwd: string): string {
@@ -44,7 +45,7 @@ export function resolveDefaultRepo(cwd: string): string {
44
45
 
45
46
  throw new Error(`Could not parse git remote URL: ${remoteUrl}`);
46
47
  } catch (err) {
47
- const msg = err instanceof Error ? err.message : String(err);
48
+ const msg = errorMessage(err);
48
49
  throw new Error(`Failed to resolve default repo from git: ${msg}`);
49
50
  }
50
51
  }
@@ -314,7 +315,7 @@ function runGh(cwd: string, args: string[]): string {
314
315
  stdio: ["pipe", "pipe", "pipe"],
315
316
  });
316
317
  } catch (err) {
317
- const msg = err instanceof Error ? err.message : String(err);
318
+ const msg = errorMessage(err);
318
319
  throw new Error(`gh command failed: ${msg}`);
319
320
  }
320
321
  }
@@ -375,7 +376,7 @@ export function resolveGitHubUrl(parsed: Parsed, scheme: "issue" | "pr", cwd: st
375
376
  try {
376
377
  items = ghJson<GitHubListItem[]>(cwd, args);
377
378
  } catch (err) {
378
- const msg = err instanceof Error ? err.message : String(err);
379
+ const msg = errorMessage(err);
379
380
  throw new Error(`${scheme}:// listing failed: ${msg}`);
380
381
  }
381
382
 
@@ -395,7 +396,7 @@ export function resolveGitHubUrl(parsed: Parsed, scheme: "issue" | "pr", cwd: st
395
396
  try {
396
397
  repo = resolveDefaultRepo(cwd);
397
398
  } catch (err) {
398
- const msg = err instanceof Error ? err.message : String(err);
399
+ const msg = errorMessage(err);
399
400
  throw new Error(
400
401
  `${scheme}://${(parsed as ParsedSingle).number} could not resolve a default repo from cwd '${cwd}': ${msg}\nUse ${scheme}://<owner>/<repo>/${(parsed as ParsedSingle).number} instead.`,
401
402
  );
@@ -417,7 +418,7 @@ export function resolveGitHubUrl(parsed: Parsed, scheme: "issue" | "pr", cwd: st
417
418
  notes: [`Full diff for ${scheme}://${repo}/${parsed.number}`],
418
419
  };
419
420
  } catch (err) {
420
- const msg = err instanceof Error ? err.message : String(err);
421
+ const msg = errorMessage(err);
421
422
  throw new Error(`pr://${parsed.number}/diff failed: ${msg}`);
422
423
  }
423
424
  }
@@ -456,7 +457,7 @@ export function resolveGitHubUrl(parsed: Parsed, scheme: "issue" | "pr", cwd: st
456
457
  notes: [`File listing for pr://${repo}/${parsed.number}`],
457
458
  };
458
459
  } catch (err) {
459
- const msg = err instanceof Error ? err.message : String(err);
460
+ const msg = errorMessage(err);
460
461
  throw new Error(`pr://${parsed.number}/diff failed: ${msg}`);
461
462
  }
462
463
  }
@@ -503,7 +504,7 @@ export function resolveGitHubUrl(parsed: Parsed, scheme: "issue" | "pr", cwd: st
503
504
  notes: [`Diff for file ${parsed.index}/${fileLines.length}: ${fileName} in pr://${repo}/${parsed.number}`],
504
505
  };
505
506
  } catch (err) {
506
- const msg = err instanceof Error ? err.message : String(err);
507
+ const msg = errorMessage(err);
507
508
  throw new Error(`pr://${parsed.number}/diff/${parsed.index} failed: ${msg}`);
508
509
  }
509
510
  }
@@ -539,7 +540,7 @@ export function resolveGitHubUrl(parsed: Parsed, scheme: "issue" | "pr", cwd: st
539
540
  notes: [`${scheme}://${repo}/${single.number} via gh`],
540
541
  };
541
542
  } catch (err) {
542
- const msg = err instanceof Error ? err.message : String(err);
543
+ const msg = errorMessage(err);
543
544
  throw new Error(`${scheme}://${repo}/${single.number} failed: ${msg}`);
544
545
  }
545
546
  }
@@ -0,0 +1,198 @@
1
+ ---
2
+ name: distill
3
+ description: Adapter for distill-persona/distill-software — orchestrates the full distillation pipeline as one command. Parallel research → merge → triple-verify → verify-prune(V1-V5) → build → fidelity. **Run with team='implementation'** (needs explorer/analyst/planner/critic/executor/verifier roles; the default team lacks analyst+critic). Pass the target in {goal} (e.g. "distill Karpathy persona" / "distill oh-my-pi codebase conventions" / "distill the field of perf-debugging").
4
+ topology: complex-dag
5
+ ---
6
+
7
+ <!--
8
+ This workflow is the pi-crew ADAPTER for the distill-* skills. It bakes the
9
+ methodology (distill-persona / distill-software) into enforced orchestration:
10
+ parallel research, an INDEPENDENT fresh-context verifier (the F2' independence
11
+ fix), and the V1-V5 extraction gate. Every step loads the relevant distill
12
+ skill for methodology; the workflow enforces structure + independence.
13
+
14
+ Flavor is chosen in {goal}: "persona"/"person" → distill-persona (6 streams);
15
+ "codebase"/"software"/"engineer" → distill-software (6+extra streams + code-DNA);
16
+ "field"/"topic" → distill-persona topic variant. Detect from {goal}; if unclear,
17
+ default to person.
18
+ -->
19
+
20
+ ## phase0-route
21
+ role: planner
22
+ output: distill-plan.md
23
+
24
+ Read the distill skill matching {goal}'s flavor (distill-persona for person/field, distill-software for codebase/engineer). Determine: flavor, target, `language`+`distilled_against` (software) or research-date (person), cost tier, and the output skill dir. Create the skill dir + `references/research/`. Write a one-paragraph run plan (flavor, target, staleness anchors, tier, coverage manifest approach) to distill-plan.md. Do NOT research yet — just route + scaffold.
25
+
26
+ ## research-1
27
+ role: explorer
28
+ parallelGroup: research
29
+ dependsOn: phase0-route
30
+ output: research/01-writings.md
31
+
32
+ Stream 1 — **writings/docs**: for person → books/essays/papers/newsletters; for software → design docs/ADRs/RFCs/READMEs/commit messages. Follow the distill skill's stream-1 spec. Write findings (with sources + credibility) to `<skill-dir>/references/research/01-writings.md`. Research not persisted = not done.
33
+
34
+ **W5 fix — WRITE ACCESS CLARIFICATION**: You CAN and SHOULD write your findings to `<output-dir>/references/research/0N-<stream>.md` (the run artifacts dir, NOT the target project). The READ-ONLY restriction applies to the TARGET codebase (don't edit the project being learned from). If the output dir doesn't exist, `mkdir -p` it first. Do NOT emit your full output as TEXT in your result message — write to the file directly. The previous run's Stream 2 + Stream 4 deferred file writes and emitted TEXT, forcing a downstream worker to re-save — avoid that.
35
+
36
+ ## research-2
37
+ role: explorer
38
+ parallelGroup: research
39
+ dependsOn: phase0-route
40
+ output: research/02-conversations.md
41
+
42
+ Stream 2 — **conversations/discourse**: person → podcasts/AMAs/interviews (stance-change moments, refusals); software → code-review comments/PR threads/incident retros. Write to `references/research/02-conversations.md`.
43
+
44
+
45
+
46
+ **W5 fix — WRITE ACCESS CLARIFICATION**: You CAN and SHOULD write your findings to `<output-dir>/references/research/0N-<stream>.md` (the run artifacts dir, NOT the target project). The READ-ONLY restriction applies to the TARGET codebase (don't edit the project being learned from). If the output dir doesn't exist, `mkdir -p` it first. Do NOT emit your full output as TEXT in your result message — write to the file directly. The previous run's Stream 2 + Stream 4 deferred file writes and emitted TEXT, forcing a downstream worker to re-save — avoid that.
47
+ ## research-3
48
+ role: explorer
49
+ parallelGroup: research
50
+ dependsOn: phase0-route
51
+ output: research/03-expression-dna.md
52
+
53
+ Stream 3 — **expression/DNA**: person → prose expression-DNA (sentence length, analogy density, certainty spectrum, 口癖); software → CODE Expression-DNA (run `code_dna.py` if software flavor; naming/function-length/comment/error-handling/type-strictness). Write to `references/research/03-expression-dna.md`.
54
+
55
+
56
+
57
+ **W5 fix — WRITE ACCESS CLARIFICATION**: You CAN and SHOULD write your findings to `<output-dir>/references/research/0N-<stream>.md` (the run artifacts dir, NOT the target project). The READ-ONLY restriction applies to the TARGET codebase (don't edit the project being learned from). If the output dir doesn't exist, `mkdir -p` it first. Do NOT emit your full output as TEXT in your result message — write to the file directly. The previous run's Stream 2 + Stream 4 deferred file writes and emitted TEXT, forcing a downstream worker to re-save — avoid that.
58
+ ## research-4
59
+ role: explorer
60
+ parallelGroup: research
61
+ dependsOn: phase0-route
62
+ output: research/04-external-views.md
63
+
64
+ Stream 4 — **critics/failures**: person → biography/criticism/peer-contrast; software → postmortems/bug reports/dep CVEs/arch-review. Write to `references/research/04-external-views.md`.
65
+
66
+
67
+
68
+ **W5 fix — WRITE ACCESS CLARIFICATION**: You CAN and SHOULD write your findings to `<output-dir>/references/research/0N-<stream>.md` (the run artifacts dir, NOT the target project). The READ-ONLY restriction applies to the TARGET codebase (don't edit the project being learned from). If the output dir doesn't exist, `mkdir -p` it first. Do NOT emit your full output as TEXT in your result message — write to the file directly. The previous run's Stream 2 + Stream 4 deferred file writes and emitted TEXT, forcing a downstream worker to re-save — avoid that.
69
+ ## research-5
70
+ role: explorer
71
+ parallelGroup: research
72
+ dependsOn: phase0-route
73
+ output: research/05-decisions.md
74
+
75
+ Stream 5 — **decisions** (where mental models live): person → major life/career decisions + say-vs-do gaps; software → ADRs/tradeoff records/"why X over Y" in commits+PRs (mine with `git log --grep`). Write to `references/research/05-decisions.md`.
76
+
77
+
78
+
79
+ **W5 fix — WRITE ACCESS CLARIFICATION**: You CAN and SHOULD write your findings to `<output-dir>/references/research/0N-<stream>.md` (the run artifacts dir, NOT the target project). The READ-ONLY restriction applies to the TARGET codebase (don't edit the project being learned from). If the output dir doesn't exist, `mkdir -p` it first. Do NOT emit your full output as TEXT in your result message — write to the file directly. The previous run's Stream 2 + Stream 4 deferred file writes and emitted TEXT, forcing a downstream worker to re-save — avoid that.
80
+ ## research-6
81
+ role: explorer
82
+ parallelGroup: research
83
+ dependsOn: phase0-route
84
+ output: research/06-timeline.md
85
+
86
+ Stream 6 — **timeline**: person → chronology + last 12 months (anti-staleness); software → `git log` IS the timeline (architecture evolution, what's actively changing). Write to `references/research/06-timeline.md`. For software flavor, ALSO sweep the extra streams (tests-as-invariants, CI/lint, dep manifests, release-pipeline, risk-posture, concurrency, platform-hardening) — fold each into the matching research file or add `07-extra.md`.
87
+
88
+
89
+
90
+ **W5 fix — WRITE ACCESS CLARIFICATION**: You CAN and SHOULD write your findings to `<output-dir>/references/research/0N-<stream>.md` (the run artifacts dir, NOT the target project). The READ-ONLY restriction applies to the TARGET codebase (don't edit the project being learned from). If the output dir doesn't exist, `mkdir -p` it first. Do NOT emit your full output as TEXT in your result message — write to the file directly. The previous run's Stream 2 + Stream 4 deferred file writes and emitted TEXT, forcing a downstream worker to re-save — avoid that.
91
+ ## merge
92
+ role: analyst
93
+ dependsOn: research-1, research-2, research-3, research-4, research-5, research-6
94
+ output: research-coverage.md
95
+
96
+ Read all 6 (or 7) research files. Produce the Phase 1.5 coverage table: streams × source-count × key-findings × contradictions × gaps. **Surface contradictions explicitly — do NOT average them into false consensus** (contradiction-as-signal). Flag any stream with thin coverage. Write to `references/research-coverage.md`.
97
+
98
+ ## synthesize
99
+ role: planner
100
+ dependsOn: merge
101
+ output: models.md
102
+
103
+ Read the research files + coverage. List all candidate claims (usually 15-30). Apply **triple-verification** to each (cross-domain/module recurrence + generative + exclusive): all-3 → mental model (3-7); 1-2 → decision heuristic (5-10); 0 → discard. The exclusivity test is the anti-bloat weapon (kills "use version control" generic advice). Extract expression-DNA, values+anti-patterns, ≥2 preserved inner tensions, intellectual lineage, ≥3 honest boundaries + staleness date. Write the model set to `references/models.md`.
104
+
105
+ **W4 fix — target-equivalent check**: Before proposing any model that implies "create new file/module/helper" in the target, you MUST grep the target codebase for an existing equivalent (e.g. `rg -l "errorMessage|toErrorMessage" <target>/src/` for an error-message helper claim). The oh-my-pi→pi-crew run's H11 said "create `src/runtime/error-message.ts`" but the helper already existed at `src/utils/guards.ts:96` — would have shipped a duplicate. Any "create" claim without a grep-confirmed absence in target is **OVERSTATED** — re-frame as "adopt existing X at target:file:line" or remove.
106
+
107
+ ## verify-prune
108
+ role: critic
109
+ dependsOn: synthesize
110
+
111
+ **Fresh-context extraction verification (V1-V5)** — do NOT trust the synthesis at face value. Apply to every candidate model: V1 signal (method/principle, not persona-content/quirk — flag quirk-vs-principle ⚠️); V2 non-redundant (>70% overlap → merge/drop); V3 effective (changes a real decision?); V4 optimal (simplest form); **V5 factual-accuracy** — every quoted constant/function-name/path/regex/threshold is grep-verified against the actual source (codebase) or the research files (person); cite evidence; reject over-absolute claims. Output the PRUNED model set + reject reasons (audit trail) to `references/verified-models.md`. Over-extraction is worse than under-extraction — default to PRUNE.
112
+
113
+ ## three-filter
114
+ role: critic
115
+ dependsOn: verify-prune
116
+ output: three-filter.md
117
+
118
+ **Phase 2.5 (new — SELF-UPGRADE DIRECTIVE).** Apply the 3-chiều filter to every KEEP candidate from `verified-models.md`, using `target-analysis.md` as ground truth. For each candidate, walk the 3 dimensions:
119
+ 1. **RELEVANCE** — relevant to target's domain/scale/context? NO → SKIP.
120
+ 2. **PRESENCE** — target already has something similar? NO → NECESSITY (does target need it? NO → SKIP, YES → ADOPT). YES → QUALITY COMPARISON: source BETTER → IMPROVE, EQUAL/WORSE → SKIP, COMPLEMENTARY → MERGE.
121
+ 3. **For each SELECTED** → note adaptation needed for target context (license boundary, toolchain differences, existing patterns to preserve).
122
+
123
+ Output: per-candidate verdict (SKIP / ADOPT / IMPROVE / MERGE) with target-evidence citations. Write to `references/three-filter.md`. **This is the gate that prevents applying patterns the target already has better.**
124
+
125
+ ## effectiveness-gate
126
+ role: critic
127
+ dependsOn: three-filter
128
+ output: effectiveness-gate.md
129
+
130
+ **Phase 2.6 (new — MANDATORY pre-apply gate).** For every SELECTED candidate from `three-filter.md`, run the EFFECTIVENESS VERIFICATION — SELECTED ≠ TO-APPLY. Each SELECTED must prove it's effective FOR THIS TARGET:
131
+ a. **CONCRETE DELTA** — exactly what changes in the target? (1-line: "AGENTS.md +rule X" / "lint +rule Y" / "src/runtime/new-file.ts +function Z")
132
+ b. **EFFECTIVENESS PROOF** (≥1 of): GENERATIVE (changes a real decision/answer/behavior? name the case), PROBLEM-EXISTS (target has a problem this solves? grep/test/convention-gap evidence), DELTA-TEST (apply in isolation → measure improvement on 1 target case)
133
+ c. **CONFLICT CHECK** — conflicts with target's existing practice? resolve or downgrade
134
+ d. **VERDICT**: ✅ EFFECTIVENESS-VERIFIED → TO-APPLY, or ❌ REJECTED → log reason (rejection goes to APPLY-LOG)
135
+
136
+ Output: per-candidate verdict with concrete delta + proof + conflict check. Write to `references/effectiveness-gate.md`. **Only TO-APPLY items enter Phase 3.** This is the analog of V1-V4 (which verify MODEL effective at extract) applied to APPLY (which verifies apply effective at integrate) — same rigor, different stage.
137
+
138
+ ## plan-application
139
+ role: planner
140
+ dependsOn: effectiveness-gate
141
+ output: apply-plan.md
142
+
143
+ **Phase 3 (new).** For each TO-APPLY item from `effectiveness-gate.md`, produce a concrete apply plan:
144
+ - **HOW to apply**: AGENTS.md edit, lint rule add, src/ pattern adopt, CONTRIBUTING.md update, scripts/ operational script, or skill in target's skills/ dir
145
+ - **Exact file:line target** in the target project
146
+ - **Tier priority**: Tier 1 (high V3, low V4 cost) → Tier 2 (high V3, medium V4 cost) → Tier 3 (high V3, high V4 cost — pilot first)
147
+ - **Verification gate per item**: `npm run test:critical` + `npm run typecheck` + `npm run build:bundle` must pass after each apply; bundle MD5 must match or be updated; if tests fail → rollback
148
+
149
+ Write to `references/apply-plan.md`. This is the input for the executor in Phase 4.
150
+
151
+ ## apply
152
+ role: executor
153
+ dependsOn: plan-application
154
+ output: applied-changes.md
155
+
156
+ **Phase 4 (new — THE KEY DELIVERABLE).** Read `apply-plan.md`. For each TO-APPLY item, in tier order (Tier 1 first), apply the concrete delta to the target project. **The apply MUST happen here, not be deferred to a "downstream worker"** (this was W1's root cause in the oh-my-pi→pi-crew run — the build phase deferred apply and hung).
157
+
158
+ Per-item protocol:
159
+ 1. Edit the target file(s) as specified in apply-plan.md
160
+ 2. Run verification gate: `npm run test:critical` (or equivalent) + `npm run typecheck` + `npm run build:bundle`
161
+ 3. Capture before/after diff + bundle MD5 (before/after)
162
+ 4. If any gate fails → rollback the change (git checkout the file), log REJECTED in APPLY-LOG, continue with next item
163
+ 5. If passes → log APPLIED in APPLY-LOG with file:line
164
+
165
+ **Anti-loop guard (W2 fix)**: this task has a hard tool-call budget of 50 tool calls and a wall-clock timeout of 15 minutes. If you hit either, STOP and write a partial APPLY-LOG with what was completed + what's remaining. Do NOT re-verify completed items.
166
+
167
+ Write per-item results to `references/applied-changes.md` AND append to `APPLY-LOG.md` in the target project (or the run's APPLY-LOG location).
168
+
169
+ ## verify-target-improved
170
+ role: verifier
171
+ dependsOn: apply
172
+ verify: true
173
+
174
+ **Phase 5 (new — Darwin ratchet).** The output of a distillation is NOT "has skill" — it's "target improved." Read `applied-changes.md` + the target project's git diff. For each APPLIED item, re-verify:
175
+ - Does the change actually improve the target? (concrete, measurable — e.g. "new helper reduces 18 sites to 1 import"; not vibes)
176
+ - Do the tests still pass? (`npm run test:critical`)
177
+ - Did the bundle MD5 change as expected? (intentional change OK; unexpected change → flag)
178
+ - Is the change consistent with target's existing style/conventions? (no foreign code injected)
179
+
180
+ Output: per-item verdict (IMPROVED / NEUTRAL / REGRESSED) + recommendation (KEEP / ROLLBACK). If any REGRESSED → flag for rollback. Write to `references/target-improvement.md`.
181
+
182
+ **Ship-gate**: ≥3 items APPLIED + verified IMPROVED → PASS. Otherwise → FAIL with the gap.
183
+
184
+ ## build
185
+ role: executor
186
+ dependsOn: verify-prune
187
+ output: SKILL.md
188
+
189
+ Read `verified-models.md`. Render the output `<target>-perspective` (person) or `<target>-conventions` (codebase) SKILL.md from the distill skill's template. MUST include: staleness anchors (language+distilled_against for software), the Agentic Protocol (research-before-answer, Step-2 dims DERIVED from the models, F2' third-category rule), the verified models (each with evidence+limitation), expression/code-DNA, ≥3 honest boundaries, sources. For software flavor: wire operational scripts INTO the Agentic Protocol (F13), run `code_dna.py`. Write the SKILL.md into the skill dir.
190
+
191
+ **Anti-loop guard (W2 fix)**: After writing SKILL.md (and FIDELITY.md if fidelity is your job), output `DONE` and stop immediately. Do NOT re-verify, re-read, or re-grep completed files. Hard limits: ≤30 tool calls total, ≤10 min wall-clock. The oh-my-pi→pi-crew run had 11_build loop 14+ times in a verification loop (re-read SKILL.md, re-check claims, re-grep source) and never converged. If you hit either limit, stop and write whatever you've completed.
192
+
193
+ ## fidelity
194
+ role: verifier
195
+ dependsOn: build
196
+ verify: true
197
+
198
+ **Independent fresh-context fidelity check** (the F2' fix — this verifier has NOT seen the synthesis/build reasoning; it reads only the built SKILL.md + ground truth). For person: pose 3 known-stance + 1 NOVEL framework-answerable edge question; score stance-consistency + style + edge-honesty (must flag inference, not fabricate). For codebase: pose a novel "how would this codebase handle X new scenario" question; verify the skill's models give complete consistent guidance + that every factual claim still grep-matches source (V5 re-check). Gate: edge-honesty/accuracy failure = NO-SHIP (return FAIL with the specific failure + which model/claim broke). On PASS, confirm the skill is installable + self-contained. Write the fidelity report to `references/fidelity.md`.