@mmerterden/multi-agent-pipeline 16.25.0 → 16.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +77 -0
  2. package/README.md +1 -1
  3. package/README.tr.md +1 -1
  4. package/install/templates/claude-hooks.json +32 -1
  5. package/package.json +1 -1
  6. package/pipeline/commands/multi-agent/help/SKILL.md +2 -0
  7. package/pipeline/commands/multi-agent/refactor/SKILL.md +23 -1
  8. package/pipeline/commands/multi-agent/search/SKILL.md +28 -0
  9. package/pipeline/commands/multi-agent/setup/SKILL.md +18 -44
  10. package/pipeline/commands/multi-agent/status/SKILL.md +9 -0
  11. package/pipeline/lib/credential-inventory.sh +142 -18
  12. package/pipeline/lib/fetch-crashlytics.sh +123 -28
  13. package/pipeline/multi-agent-refs/features/url-enrichment.md +1 -1
  14. package/pipeline/multi-agent-refs/features/visual-evidence.md +5 -0
  15. package/pipeline/multi-agent-refs/keychain.md +65 -20
  16. package/pipeline/multi-agent-refs/knowledge.md +27 -0
  17. package/pipeline/multi-agent-refs/phases/operations.md +7 -1
  18. package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +3 -1
  19. package/pipeline/multi-agent-refs/phases/phase-4-review.md +1 -8
  20. package/pipeline/multi-agent-refs/phases/phase-7-report.md +11 -21
  21. package/pipeline/multi-agent-refs/picker-contract.md +1 -1
  22. package/pipeline/multi-agent-refs/refactor/observations.md +81 -0
  23. package/pipeline/multi-agent-refs/setup/firebase.md +151 -0
  24. package/pipeline/schemas/learnings-ledger.schema.json +5 -0
  25. package/pipeline/schemas/prefs.schema.json +31 -3
  26. package/pipeline/schemas/skill-observation.schema.json +73 -0
  27. package/pipeline/scripts/capture-flush.sh +158 -0
  28. package/pipeline/scripts/capture-resume.sh +87 -0
  29. package/pipeline/scripts/crush-json.mjs +283 -0
  30. package/pipeline/scripts/firebase-app-discovery.sh +114 -0
  31. package/pipeline/scripts/keychain-save.sh +5 -8
  32. package/pipeline/scripts/keychain.py +76 -14
  33. package/pipeline/scripts/learn-from-transcripts.mjs +625 -0
  34. package/pipeline/scripts/learning-curve.mjs +22 -4
  35. package/pipeline/scripts/learnings-ledger.mjs +86 -12
  36. package/pipeline/scripts/note-session.sh +187 -0
  37. package/pipeline/scripts/observations.mjs +347 -0
  38. package/pipeline/scripts/offload-ref.sh +45 -2
  39. package/pipeline/scripts/pre-commit-check.sh +31 -1
  40. package/pipeline/scripts/scan-agent-config.sh +12 -3
  41. package/pipeline/scripts/skill-siblings.mjs +187 -0
  42. package/pipeline/scripts/triage-memory.mjs +73 -9
  43. package/pipeline/skills/.skill-manifest.json +1 -1
  44. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +23 -1
  45. package/pipeline/skills/shared/core/multi-agent-search/SKILL.md +28 -0
  46. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +37 -6
  47. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +9 -0
@@ -30,6 +30,11 @@
30
30
  * knowledge in a deterministic order, so it can sit in the cacheable
31
31
  * stable prefix of a phase prompt (see multi-agent-refs/prompt-assembly.md).
32
32
  * Every line carries a [L:id] pointer for drill-down. Exit 2 when empty.
33
+ * timeline --anchor <id> [--before N] [--after N] [--repo-slug <slug>]
34
+ * What was learned around an entry. brief/profile rank by
35
+ * relevance ("what applies here"); this answers "what else did
36
+ * we learn at the same time". Append-only, so line adjacency is
37
+ * chronological adjacency.
33
38
  * show --id <id> [--repo-slug <slug>]
34
39
  * Print the full row behind a [L:id] pointer.
35
40
  * from-triage --triage <path> --task <id> [--repo-slug <slug>]
@@ -91,6 +96,7 @@ for (let i = 1; i < argv.length; i++) {
91
96
 
92
97
  const KINDS = new Set(["fact", "convention", "rejected-preference"]);
93
98
  const CONFIDENCES = new Set(["low", "medium", "high"]);
99
+ const SOURCES = new Set(["triage", "transcript-mining", "manual"]);
94
100
 
95
101
  function die(msg, code = 1) {
96
102
  process.stderr.write(`learnings-ledger: ${msg}\n`);
@@ -192,7 +198,7 @@ function writeRow(slug, row) {
192
198
 
193
199
  /** Append one entry unless an entry with the same kind + normalized statement
194
200
  * already exists (idempotent). Returns true when written. */
195
- function addEntry(slug, { kind, statement, scope, task, confidence, diagnosis }, existing) {
201
+ function addEntry(slug, { kind, statement, scope, task, confidence, diagnosis, source }, existing) {
196
202
  // One line, enforced here rather than hoped for. The schema says "the
197
203
  // knowledge itself, in one line", and both blocks render one entry per line -
198
204
  // a statement carrying a newline breaks out of its bullet, and since
@@ -218,6 +224,11 @@ function addEntry(slug, { kind, statement, scope, task, confidence, diagnosis },
218
224
  // one; a bare outcome does not. Stored only when supplied.
219
225
  const diag = oneLine(diagnosis);
220
226
  if (diag) row.diagnosis = diag;
227
+ // Who found it. Optional so old rows stay valid, but written whenever the
228
+ // caller knows: a machine-mined path correction and a model's architectural
229
+ // claim are different evidence, and a metric that pools them can rise while
230
+ // the useful half is flat.
231
+ if (SOURCES.has(source)) row.source = source;
221
232
  writeRow(slug, row);
222
233
  existing.add(seenKey);
223
234
  return true;
@@ -239,6 +250,7 @@ function cmdAdd() {
239
250
  confidence:
240
251
  flags.confidence && flags.confidence !== true ? String(flags.confidence) : "medium",
241
252
  diagnosis: flags.diagnosis && flags.diagnosis !== true ? String(flags.diagnosis) : null,
253
+ source: flags.source && flags.source !== true ? String(flags.source) : "manual",
242
254
  },
243
255
  existing,
244
256
  );
@@ -284,7 +296,14 @@ function cmdFromTriage() {
284
296
  if (
285
297
  addEntry(
286
298
  slug,
287
- { kind: "rejected-preference", statement, scope, task, confidence: "low" },
299
+ {
300
+ kind: "rejected-preference",
301
+ statement,
302
+ scope,
303
+ task,
304
+ confidence: "low",
305
+ source: "triage",
306
+ },
288
307
  existing,
289
308
  )
290
309
  ) {
@@ -292,7 +311,8 @@ function cmdFromTriage() {
292
311
  }
293
312
  }
294
313
  process.stdout.write(JSON.stringify({ ok: true, slug, added: written, skippedBlocking }) + "\n");
295
- process.exit(0);
314
+ process.exitCode = 0;
315
+ return;
296
316
  }
297
317
 
298
318
  function cmdForget() {
@@ -303,7 +323,8 @@ function cmdForget() {
303
323
  const p = ledgerPath(slug);
304
324
  if (!existsSync(p)) {
305
325
  process.stdout.write(JSON.stringify({ ok: true, slug, removed: 0 }) + "\n");
306
- process.exit(0);
326
+ process.exitCode = 0;
327
+ return;
307
328
  }
308
329
  const rows = readLedger(slug);
309
330
  const kept = rows.filter((r) => {
@@ -322,7 +343,8 @@ function cmdForget() {
322
343
  writeFileSync(tmp, kept.length ? body + "\n" : "", "utf8");
323
344
  renameSync(tmp, p);
324
345
  process.stdout.write(JSON.stringify({ ok: true, slug, removed, remaining: kept.length }) + "\n");
325
- process.exit(0);
346
+ process.exitCode = 0;
347
+ return;
326
348
  }
327
349
 
328
350
  const KIND_ORDER = ["fact", "convention", "rejected-preference"];
@@ -469,7 +491,8 @@ function cmdBrief() {
469
491
  process.stdout.write(
470
492
  ["<task-relevant-memory>", header, ...body, "</task-relevant-memory>"].join("\n") + "\n",
471
493
  );
472
- process.exit(0);
494
+ process.exitCode = 0;
495
+ return;
473
496
  }
474
497
 
475
498
  /**
@@ -534,7 +557,8 @@ function cmdProfile() {
534
557
  }
535
558
  lines.push("</repo-profile>");
536
559
  process.stdout.write(lines.join("\n") + "\n");
537
- process.exit(0);
560
+ process.exitCode = 0;
561
+ return;
538
562
  }
539
563
 
540
564
  function cmdShow() {
@@ -545,17 +569,63 @@ function cmdShow() {
545
569
  const row = readLedger(slug).find((r) => rowId(r) === want);
546
570
  if (!row) {
547
571
  process.stdout.write(JSON.stringify({ ok: false, slug, id: want, found: false }) + "\n");
548
- process.exit(2);
572
+ process.exitCode = 2;
573
+ return;
549
574
  }
550
575
  process.stdout.write(JSON.stringify({ ok: true, slug, id: want, row }) + "\n");
551
- process.exit(0);
576
+ process.exitCode = 0;
577
+ return;
578
+ }
579
+
580
+ /**
581
+ * Neighbourhood: what was learned around an entry.
582
+ *
583
+ * `brief` and `profile` rank by relevance, which answers "what applies here".
584
+ * That is a different question from "what else did we learn at the same time",
585
+ * and the second one has no answer without this: the only route was to read the
586
+ * whole ledger, so nobody asked it.
587
+ *
588
+ * The ledger is append-only, so line adjacency is chronological adjacency. No
589
+ * index, no timestamp arithmetic - the file's own order is the answer.
590
+ */
591
+ function cmdTimeline() {
592
+ const id = flags.anchor && flags.anchor !== true ? String(flags.anchor) : null;
593
+ if (!id) die("timeline needs --anchor <id>", 1);
594
+ const slug = flags["repo-slug"] ? String(flags["repo-slug"]) : repoSlug();
595
+ const before = flags.before && flags.before !== true ? Math.max(0, Number(flags.before)) : 3;
596
+ const after = flags.after && flags.after !== true ? Math.max(0, Number(flags.after)) : 3;
597
+ const want = id.startsWith("L:") ? id : `L:${id}`;
598
+ const rows = readLedger(slug);
599
+ const at = rows.findIndex((r) => rowId(r) === want);
600
+ if (at === -1) {
601
+ process.stdout.write(JSON.stringify({ ok: false, slug, anchor: want, found: false }) + "\n");
602
+ process.exitCode = 2;
603
+ return;
604
+ }
605
+ const from = Math.max(0, at - before);
606
+ const to = Math.min(rows.length, at + after + 1);
607
+ const window = rows.slice(from, to).map((r, i) => ({
608
+ id: rowId(r),
609
+ offset: from + i - at,
610
+ kind: r.kind,
611
+ scope: r.scope,
612
+ source: r.source || null,
613
+ source_task: r.source_task || null,
614
+ statement: r.statement,
615
+ }));
616
+ process.stdout.write(
617
+ JSON.stringify({ ok: true, slug, anchor: want, before, after, rows: window }, null, 2) + "\n",
618
+ );
619
+ process.exitCode = 0;
620
+ return;
552
621
  }
553
622
 
554
623
  function cmdPath() {
555
624
  process.stdout.write(
556
625
  ledgerPath(flags["repo-slug"] ? String(flags["repo-slug"]) : repoSlug()) + "\n",
557
626
  );
558
- process.exit(0);
627
+ process.exitCode = 0;
628
+ return;
559
629
  }
560
630
 
561
631
  function cmdStats() {
@@ -566,7 +636,8 @@ function cmdStats() {
566
636
  const out = { ok: true, slug, path: ledgerPath(slug), rows: rows.length, counts };
567
637
  if (existsSync(out.path)) out.bytes = statSync(out.path).size;
568
638
  process.stdout.write(JSON.stringify(out) + "\n");
569
- process.exit(0);
639
+ process.exitCode = 0;
640
+ return;
570
641
  }
571
642
 
572
643
  switch (SUB) {
@@ -585,6 +656,9 @@ switch (SUB) {
585
656
  case "profile":
586
657
  cmdProfile();
587
658
  break;
659
+ case "timeline":
660
+ cmdTimeline();
661
+ break;
588
662
  case "show":
589
663
  cmdShow();
590
664
  break;
@@ -596,7 +670,7 @@ switch (SUB) {
596
670
  break;
597
671
  default:
598
672
  process.stderr.write(
599
- "usage: learnings-ledger.mjs <add|from-triage|forget|brief|profile|show|path|stats> [flags]\n",
673
+ "usage: learnings-ledger.mjs <add|from-triage|forget|brief|profile|show|timeline|path|stats> [flags]\n",
600
674
  );
601
675
  process.exit(64);
602
676
  }
@@ -0,0 +1,187 @@
1
+ #!/usr/bin/env bash
2
+ #
3
+ # note-session.sh - record what a NON-pipeline session ran into.
4
+ #
5
+ # WHY THIS EXISTS
6
+ #
7
+ # rules/outside-the-pipeline.md actively sends the user to work outside a
8
+ # pipeline run: read a ticket, use a stack skill, call an MCP tool. That is the
9
+ # right advice, and it meant every lesson learned in those sessions landed
10
+ # nowhere - none of the five durable stores ever saw them, because all five are
11
+ # written by pipeline phases.
12
+ #
13
+ # So this reads the session's own transcript and keeps the mechanical facts:
14
+ # which repo was touched, which commands failed, which tool calls the user
15
+ # refused. No model, no interpretation.
16
+ #
17
+ # WHAT IS DELIBERATELY NOT KEPT
18
+ #
19
+ # Not the prose. Not command arguments, not file contents, not tool output. A
20
+ # transcript is the least redacted artefact on the machine - it holds whatever
21
+ # the session read, tokens and customer data included - so what survives here is
22
+ # the shape of an event and nothing that carries a payload: a command's first
23
+ # word, a tool's name, an exit code, a count. That is enough for
24
+ # learn-from-transcripts.mjs to correlate later, and it cannot leak a secret
25
+ # because it never copies a value.
26
+ #
27
+ # Usage:
28
+ # ./note-session.sh [--transcript <path>] [--json] [--dry-run]
29
+ #
30
+ # Exit 0 always. It runs from SessionEnd; a hook that fails a session over
31
+ # bookkeeping is worse than the bookkeeping it protects.
32
+
33
+ set -uo pipefail
34
+
35
+ TRANSCRIPT=""
36
+ JSON=0
37
+ DRY=0
38
+ while [ "$#" -gt 0 ]; do
39
+ case "$1" in
40
+ --transcript) TRANSCRIPT="${2:-}"; shift 2 || shift ;;
41
+ --json) JSON=1; shift ;;
42
+ --dry-run) DRY=1; shift ;;
43
+ -h|--help) echo "usage: $0 [--transcript <path>] [--json] [--dry-run]" >&2; exit 0 ;;
44
+ *) shift ;;
45
+ esac
46
+ done
47
+
48
+ # Claude Code exports the active transcript to the hook environment; falling back
49
+ # to "newest under the project's own transcript dir" keeps the script runnable by
50
+ # hand and in a gate.
51
+ if [ -z "$TRANSCRIPT" ]; then
52
+ TRANSCRIPT="${CLAUDE_TRANSCRIPT_PATH:-}"
53
+ fi
54
+ if [ -z "$TRANSCRIPT" ]; then
55
+ SLUG_DIR="$HOME/.claude/projects/$(pwd | sed 's/[^A-Za-z0-9]/-/g')"
56
+ TRANSCRIPT=$(ls -t "$SLUG_DIR"/*.jsonl 2>/dev/null | head -1)
57
+ fi
58
+
59
+ if [ -z "$TRANSCRIPT" ] || [ ! -f "$TRANSCRIPT" ]; then
60
+ [ "$JSON" -eq 1 ] && printf '{"status":"noop","reason":"no-transcript"}\n'
61
+ exit 0
62
+ fi
63
+
64
+ REPO_SLUG=$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)")
65
+ OUT_DIR="$HOME/.claude/memory/multi-agent/$REPO_SLUG"
66
+ OUT="$OUT_DIR/session-notes.jsonl"
67
+
68
+ TRANSCRIPT="$TRANSCRIPT" REPO_SLUG="$REPO_SLUG" OUT="$OUT" DRY="$DRY" JSON="$JSON" \
69
+ python3 - <<'PY'
70
+ import json, os, re, sys
71
+ from collections import Counter
72
+ from datetime import datetime, timezone
73
+
74
+ path = os.environ["TRANSCRIPT"]
75
+
76
+ # Only the head word of a command survives, and only when it looks like a plain
77
+ # program name. `rm -rf /Users/<name>/secret` becomes `rm`; a command that starts
78
+ # with a path or a variable becomes nothing at all.
79
+ SAFE_HEAD = re.compile(r"^[a-z][a-z0-9_.-]{0,31}$")
80
+
81
+ def head_word(cmd):
82
+ """The first real program in a command, or None.
83
+
84
+ `cd` swallows its argument: nearly every command in this codebase opens with
85
+ `cd <repo>`, and a version that skipped only the `cd` then hit the path and
86
+ redacted the whole line - so every failure was recorded with no command at
87
+ all. The path is what must not survive; the program after it is the signal.
88
+ """
89
+ if not isinstance(cmd, str):
90
+ return None
91
+ toks = cmd.strip().split()
92
+ i = 0
93
+ while i < len(toks):
94
+ tok = toks[i]
95
+ if tok in ("sudo", "env", "time", "command", "exec"):
96
+ i += 1
97
+ continue
98
+ if tok in ("cd", "pushd"):
99
+ i += 2 # drop the directory with it
100
+ continue
101
+ if tok in ("&&", ";", "|", "\\"):
102
+ i += 1
103
+ continue
104
+ if tok.startswith(("-", "/", "$", "(", "{", '"', "'", "!")):
105
+ return None
106
+ base = tok.split("/")[-1]
107
+ return base if SAFE_HEAD.match(base) else None
108
+ return None
109
+
110
+ def result_text(block):
111
+ c = block.get("content")
112
+ if isinstance(c, list):
113
+ return " ".join(x.get("text", "") for x in c if isinstance(x, dict))
114
+ return c if isinstance(c, str) else ""
115
+
116
+ pending = {} # tool_use_id -> (tool name, command head)
117
+ failed = Counter() # command head -> failures
118
+ denied = Counter() # tool name -> refusals
119
+ tools = Counter() # tool name -> calls
120
+ errors = 0
121
+
122
+ try:
123
+ fh = open(path, encoding="utf-8")
124
+ except Exception:
125
+ print(json.dumps({"status": "noop", "reason": "unreadable-transcript"}))
126
+ sys.exit(0)
127
+
128
+ with fh:
129
+ for line in fh:
130
+ try:
131
+ d = json.loads(line)
132
+ except Exception:
133
+ continue
134
+ content = (d.get("message") or {}).get("content")
135
+ if not isinstance(content, list):
136
+ continue
137
+ for b in content:
138
+ if not isinstance(b, dict):
139
+ continue
140
+ if b.get("type") == "tool_use":
141
+ name = b.get("name") or "?"
142
+ tools[name] += 1
143
+ cmd = (b.get("input") or {}).get("command") if isinstance(b.get("input"), dict) else None
144
+ pending[b.get("id")] = (name, head_word(cmd))
145
+ elif b.get("type") == "tool_result":
146
+ name, cmd_head = pending.pop(b.get("tool_use_id"), ("?", None))
147
+ text = result_text(b)
148
+ # A refusal is not an error: the tool never ran. It says what the
149
+ # user does not want done, which is the more durable signal.
150
+ if "has been denied" in text:
151
+ denied[name] += 1
152
+ elif b.get("is_error"):
153
+ errors += 1
154
+ if cmd_head:
155
+ failed[cmd_head] += 1
156
+
157
+ row = {
158
+ "v": "1.0.0",
159
+ "ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
160
+ "repo": os.environ["REPO_SLUG"],
161
+ "transcript": os.path.basename(path),
162
+ "source": "outside-pipeline",
163
+ "tool_calls": sum(tools.values()),
164
+ "tools": dict(tools.most_common(8)),
165
+ "errors": errors,
166
+ "failed_commands": dict(failed.most_common(8)),
167
+ "denied_tools": dict(denied),
168
+ }
169
+
170
+ # Nothing happened worth a row. An empty session should not grow the file.
171
+ if row["tool_calls"] == 0 and errors == 0 and not denied:
172
+ print(json.dumps({"status": "noop", "reason": "nothing-observed"}))
173
+ sys.exit(0)
174
+
175
+ if os.environ.get("DRY") == "1":
176
+ print(json.dumps({"status": "dry-run", "row": row}))
177
+ sys.exit(0)
178
+
179
+ out = os.environ["OUT"]
180
+ os.makedirs(os.path.dirname(out), exist_ok=True)
181
+ with open(out, "a", encoding="utf-8") as fh:
182
+ fh.write(json.dumps(row, ensure_ascii=False) + "\n")
183
+
184
+ print(json.dumps({"status": "written", "path": out, "errors": errors,
185
+ "denied": sum(denied.values()), "toolCalls": row["tool_calls"]}))
186
+ PY
187
+ exit 0