@worca/app 1.2.0-rc.3 → 1.3.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/README.md +42 -0
  2. package/agents/memoryDefragmenter.meta.json +24 -0
  3. package/agents/worca-cc-code-reviewer.md +6 -1
  4. package/agents/worca-cc-implementer.md +6 -1
  5. package/agents/worca-cc-memory-defragmenter.md +32 -0
  6. package/agents/worca-cc-planner.md +5 -1
  7. package/package.json +5 -2
  8. package/src/cli/render.mjs +36 -0
  9. package/src/cli/worca-cc.mjs +137 -8
  10. package/src/core/agent-registry.mjs +12 -34
  11. package/src/core/artifacts.mjs +132 -8
  12. package/src/core/ask/catalog.mjs +32 -7
  13. package/src/core/ask/comment-deps.mjs +5 -2
  14. package/src/core/ask/events.mjs +65 -2
  15. package/src/core/ask/limits.mjs +9 -0
  16. package/src/core/ask/mcp-stdio.mjs +10 -0
  17. package/src/core/ask/memory-deps.mjs +107 -0
  18. package/src/core/ask/metrics-deps.mjs +124 -0
  19. package/src/core/ask/metrics-proposal.mjs +175 -0
  20. package/src/core/ask/prompt.mjs +53 -10
  21. package/src/core/ask/proposal.mjs +49 -2
  22. package/src/core/ask/spawn.mjs +21 -4
  23. package/src/core/ask/store.mjs +14 -5
  24. package/src/core/ask/tool-deps.mjs +26 -2
  25. package/src/core/ask/tools.mjs +439 -6
  26. package/src/core/ask/turn.mjs +163 -4
  27. package/src/core/ask/workflow-deps.mjs +226 -0
  28. package/src/core/auto/classify.mjs +352 -0
  29. package/src/core/auto/fingerprint.mjs +141 -0
  30. package/src/core/auto/match.mjs +30 -0
  31. package/src/core/auto/model.mjs +23 -0
  32. package/src/core/auto/proposal.mjs +132 -0
  33. package/src/core/auto/recipes.mjs +75 -0
  34. package/src/core/auto/repo-look.mjs +46 -0
  35. package/src/core/claude-runner.mjs +132 -11
  36. package/src/core/config.mjs +120 -3
  37. package/src/core/db.mjs +44 -1
  38. package/src/core/diff-comments.mjs +55 -9
  39. package/src/core/frontmatter.mjs +75 -0
  40. package/src/core/git-info.mjs +233 -26
  41. package/src/core/graph/builtin-workflows.mjs +50 -0
  42. package/src/core/graph/executor.mjs +11 -3
  43. package/src/core/index-html.mjs +17 -0
  44. package/src/core/memory-store.mjs +441 -0
  45. package/src/core/memory-sync.mjs +300 -0
  46. package/src/core/metrics/ledger.mjs +47 -0
  47. package/src/core/metrics/lock.mjs +117 -0
  48. package/src/core/metrics/read.mjs +303 -0
  49. package/src/core/metrics/record.mjs +389 -0
  50. package/src/core/metrics/sync.mjs +1100 -0
  51. package/src/core/onboarding.mjs +99 -0
  52. package/src/core/orchestrator.mjs +394 -7
  53. package/src/core/phases.mjs +16 -3
  54. package/src/core/pipeline-delete.mjs +1 -1
  55. package/src/core/plugin-store.mjs +2 -10
  56. package/src/core/preflight.mjs +2 -3
  57. package/src/core/projects.mjs +16 -1
  58. package/src/core/run-harness.mjs +458 -32
  59. package/src/core/run-report.mjs +896 -0
  60. package/src/core/settings.mjs +162 -0
  61. package/src/core/sources.mjs +4 -1
  62. package/src/core/store.mjs +5 -0
  63. package/src/core/workflow-export.mjs +2 -0
  64. package/src/core/workflow-share.mjs +1 -0
  65. package/src/core/workflows.mjs +43 -23
  66. package/src/core/workspaces.mjs +37 -8
  67. package/src/shared/graph/agent-meta.mjs +5 -2
  68. package/src/shared/graph/assemble.mjs +455 -0
  69. package/src/shared/graph/flow-layout.mjs +249 -0
  70. package/src/shared/graph/geometry.mjs +48 -28
  71. package/src/shared/graph/isomorphic.mjs +101 -0
  72. package/src/shared/report-reasons.mjs +58 -0
  73. package/src/shared/team-metrics/aggregate.mjs +341 -0
  74. package/src/shared/team-metrics/workspace-match.mjs +13 -0
  75. package/ui/public/about-links.mjs +21 -0
  76. package/ui/public/app.js +3736 -479
  77. package/ui/public/artifact-view.mjs +135 -0
  78. package/ui/public/ask-model.mjs +18 -1
  79. package/ui/public/ask-panel.mjs +1589 -212
  80. package/ui/public/ask-run-card.mjs +209 -0
  81. package/ui/public/assets/worca-logo-mask.png +0 -0
  82. package/ui/public/assets/worca-mark-mask.png +0 -0
  83. package/ui/public/auto-build.mjs +95 -0
  84. package/ui/public/auto-proposal.mjs +174 -0
  85. package/ui/public/comment-thread.mjs +55 -0
  86. package/ui/public/getting-started.mjs +261 -0
  87. package/ui/public/graph/composer.mjs +41 -5
  88. package/ui/public/graph/inspector.mjs +3 -1
  89. package/ui/public/graph/model.mjs +1 -0
  90. package/ui/public/graph/run-hosts.mjs +73 -12
  91. package/ui/public/graph/view.mjs +218 -50
  92. package/ui/public/guide-spot.mjs +215 -0
  93. package/ui/public/index.html +447 -29
  94. package/ui/public/memory-view.mjs +192 -0
  95. package/ui/public/node-tunables.mjs +201 -0
  96. package/ui/public/report-run.mjs +75 -0
  97. package/ui/public/results-view.mjs +25 -0
  98. package/ui/public/source-pane.mjs +16 -2
  99. package/ui/public/stats-view.mjs +2 -2
  100. package/ui/public/style.css +1499 -303
  101. package/ui/public/team-metrics-surfaces.mjs +452 -0
  102. package/ui/public/team-metrics-view.mjs +533 -0
  103. package/ui/public/thinking-orb.mjs +46 -8
  104. package/ui/server.mjs +1304 -194
@@ -0,0 +1,132 @@
1
+ // The `workflow` question payload (spec §5.3/§5.4) and the sanitiser for its
2
+ // answer. Pure: the orchestrator hands in the template (new or matched), the
3
+ // per-node tunables and the registry slice, and gets back exactly what the CLI
4
+ // and the browser render.
5
+ import { buildGraphManifest } from '../../shared/graph/manifest.mjs';
6
+ import { cleanText } from '../../shared/graph/assemble.mjs';
7
+ import { slugify } from '../artifacts.mjs';
8
+ import { GRAPH_DEFAULT_WORKFLOW, AUTO_WORKFLOW_ID, MEMORY_DEFRAG_WORKFLOW_ID } from '../workflows.mjs';
9
+
10
+ const isObject = (v) => Boolean(v) && typeof v === 'object' && !Array.isArray(v);
11
+
12
+ /** Tunables keyed by assembled node ids -> keyed by the matched row's node ids. */
13
+ export function remapTunables(tunables, nodeMap) {
14
+ const out = {};
15
+ for (const [id, sel] of Object.entries(tunables || {})) {
16
+ const to = nodeMap?.get?.(id);
17
+ if (to) out[to] = { ...sel };
18
+ }
19
+ return out;
20
+ }
21
+
22
+ /**
23
+ * @param {{round:number, shape:object, template:object, match:{id:string,name:string}|null,
24
+ * tunables:Record<string,object>, registry:Record<string,object>,
25
+ * models:Array<{id:string,label?:string,efforts?:string[],hidden?:boolean}>, warnings?:Array, costUsd?:number,
26
+ * fingerprint?:string, ignoredProjectOverrides?:boolean}} o
27
+ */
28
+ export function buildProposal({ round, shape, template, match = null, tunables = {}, registry = {}, models = [], warnings = [], costUsd = 0, fingerprint = '', ignoredProjectOverrides = false }) {
29
+ const name = shape?.name || template?.name || 'Auto workflow';
30
+ const agentsByKey = {};
31
+ for (const n of template.nodes || []) if (n.kind === 'agent' && registry[n.key]) agentsByKey[n.key] = registry[n.key];
32
+ // B1 (2026-09-07 review): paint what resolveGraph will RUN (workflows.mjs:664) — an overlay
33
+ // model without an effort suppresses the row's authored effort, so a matched twin's "high"
34
+ // must not be shown next to a classifier-picked model that carries none. The manifest (the
35
+ // graph's band chips, manifest.mjs:145) and the table below read the SAME effective pair:
36
+ // an explicit `effort: ''` overlay is what makes the manifest agree.
37
+ const nodes = {};
38
+ const overlays = {};
39
+ for (const n of template.nodes || []) {
40
+ if (n.kind !== 'agent') continue;
41
+ const meta = registry[n.key] || {};
42
+ const cfg = n.config || {};
43
+ const t = tunables[n.id] || {};
44
+ const effort = t.effort ?? (t.model ? '' : (cfg.effort ?? ''));
45
+ overlays[n.id] = t.model && t.effort === undefined ? { ...t, effort: '' } : { ...t };
46
+ const asks = !!meta.asksQuestions;
47
+ nodes[n.id] = {
48
+ key: n.key,
49
+ label: meta.displayName || n.key,
50
+ model: t.model ?? cfg.model ?? '',
51
+ effort,
52
+ fanOut: !!(t.fanOut ?? cfg.fanOut ?? meta.fanOut ?? false),
53
+ askQuestions: asks ? (meta.questionsLocked ? !!meta.questionsDefault : !!(t.askQuestions ?? cfg.askQuestions ?? meta.questionsDefault ?? false)) : false,
54
+ asksQuestions: asks,
55
+ questionsLocked: asks && !!meta.questionsLocked,
56
+ canFanOut: !!meta.fanOut,
57
+ };
58
+ }
59
+ const manifest = buildGraphManifest({ ...template, id: match ? match.id : AUTO_WORKFLOW_ID, name }, agentsByKey, { overlays: { nodes: overlays } });
60
+ // The DISPATCH order of the AGENT nodes (rank, then launch order) — the manifest
61
+ // already computes it for the run monitor; a reused composer row's graph.nodes order
62
+ // is arbitrary. The steps cells bucket EVERY node — the Task card, End and the gates
63
+ // ride along with `key: null` — so keep only the agent cells.
64
+ const order = (manifest.steps || [])
65
+ .filter((s) => s.kind === 'agents')
66
+ .flatMap((s) => s.nodes.filter((n) => n.key).map((n) => n.id));
67
+ return {
68
+ round,
69
+ name,
70
+ reasoning: shape?.reasoning || '',
71
+ taskKind: shape?.taskKind || 'prompt',
72
+ size: shape?.size || 'medium',
73
+ signals: Array.isArray(shape?.signals) ? [...shape.signals] : [],
74
+ warnings: (warnings || []).map((w) => (typeof w === 'string' ? w : w.message)).filter(Boolean),
75
+ match: match ? { id: match.id, name: match.name } : null,
76
+ manifest,
77
+ order,
78
+ nodes,
79
+ models: models.filter((m) => m && !m.hidden).map((m) => ({ id: m.id, label: m.label || m.id, efforts: [...(m.efforts || [])] })),
80
+ costUsd: Number.isFinite(costUsd) ? costUsd : 0,
81
+ fingerprint: typeof fingerprint === 'string' ? fingerprint : '',
82
+ ignoredProjectOverrides: !!ignoredProjectOverrides,
83
+ };
84
+ }
85
+
86
+ /**
87
+ * @returns {{decision:'accept',name:string,nodes:object}|{decision:'revise',text:string}|{decision:'cancel'}|null}
88
+ */
89
+ export function sanitizeProposalAnswer(payload, { proposal, models = [], registry = {} }) {
90
+ if (!isObject(payload)) return null;
91
+ const d = payload.decision;
92
+ if (d === 'cancel') return { decision: 'cancel' };
93
+ if (d === 'revise') {
94
+ const text = typeof payload.text === 'string' ? payload.text.trim().slice(0, 4096) : '';
95
+ return text ? { decision: 'revise', text } : null;
96
+ }
97
+ if (d !== 'accept') return null;
98
+ const name = cleanText(payload.name, 60) || proposal.name;
99
+ const byId = new Map(models.map((m) => [String(m.id).toLowerCase(), m])); // the FULL catalog: a hidden id still resolves
100
+ const nodes = {};
101
+ for (const [nodeId, sel] of Object.entries(isObject(payload.nodes) ? payload.nodes : {})) {
102
+ const known = proposal.nodes?.[nodeId];
103
+ if (!known || !isObject(sel)) continue;
104
+ const meta = registry[known.key] || {};
105
+ const out = {};
106
+ if (typeof sel.model === 'string') {
107
+ const m = byId.get(sel.model.trim().toLowerCase());
108
+ if (m) out.model = m.id;
109
+ // '' clears the model AND the effort: resolveGraph keeps a row's authored effort
110
+ // when the overlay's model is falsy, which would leave an effort with no model.
111
+ else if (sel.model.trim() === '') { out.model = ''; out.effort = ''; }
112
+ }
113
+ if (typeof sel.effort === 'string') {
114
+ const mid = out.model !== undefined ? out.model : known.model;
115
+ const m = mid ? byId.get(String(mid).toLowerCase()) : null;
116
+ if (m && (m.efforts || []).includes(sel.effort)) out.effort = sel.effort;
117
+ }
118
+ if (typeof sel.fanOut === 'boolean' && meta.fanOut) out.fanOut = sel.fanOut;
119
+ if (typeof sel.askQuestions === 'boolean' && meta.asksQuestions && !meta.questionsLocked) out.askQuestions = sel.askQuestions;
120
+ if (Object.keys(out).length) nodes[nodeId] = out;
121
+ }
122
+ return { decision: 'accept', name, nodes };
123
+ }
124
+
125
+ /** `wf_<slug>` that no row owns; reserved / empty slugs become wf_auto-workflow. */
126
+ export async function mintAutoWorkflowId(name, exists) {
127
+ let stem = `wf_${slugify(name)}`;
128
+ if (stem === GRAPH_DEFAULT_WORKFLOW.id || stem === AUTO_WORKFLOW_ID || stem === MEMORY_DEFRAG_WORKFLOW_ID || stem === 'wf_untitled') stem = 'wf_auto-workflow';
129
+ let id = stem;
130
+ for (let n = 2; await exists(id); n += 1) id = `${stem}-${n}`;
131
+ return id;
132
+ }
@@ -0,0 +1,75 @@
1
+ // The ONE Auto-path module where agent keys may appear (spec D23): prompt
2
+ // guidance for the classifier and the canned shapes the offline mock answers
3
+ // with. The assembler and the orchestrator never read these keys — they read
4
+ // port meta.
5
+
6
+ // Rendered into BOTH selection paths — the Ask system prompt (prompt.mjs renderCatalog)
7
+ // and the classifier system prompt (classify.mjs) — so the sizing principle is stated
8
+ // here once: build UP from the implementer, one rung per concrete signal (2026-09-07 ladder).
9
+ export const RECIPE_GUIDE = [
10
+ '## Recipes (starting points — adapt them to the task)',
11
+ 'Sizing: build the workflow UP from the implementer and add a stage only when a concrete signal in the task itself demands it — every extra stage must earn its cost in time and money; when unsure between two shapes, take the lighter one. Never inflate "size" or "signals" to justify a stage.',
12
+ 'taskKind names what the user GAVE, never how big the work is: prompt = an idea, request or bug report with no plan; plan-partial = a sketch or a plan with gaps; plan-complete-detailed = a complete, detailed plan; plan-complete-small = a complete but small plan. Only plan-complete-* makes the task document the plan of record, so never label a prompt as a plan to reach a lighter rung — the ladder below reaches it directly.',
13
+ 'The ladder (each rung adds ONE stage to the rung before it; take the first rung that fits):',
14
+ '- trivial — implementer only: a well-specified small change whose result the tests can check (a dependency bump, a config or copy change, a one-file fix, a rename, a complete small plan)',
15
+ '- small — implementer ⇄ reviewer: the change is bigger than one bounded edit (several files, a new code path, a public interface, anything a second pair of eyes should check) but the text already says precisely what to build — a complete, detailed plan lands here',
16
+ '- needs a plan — clarify → planner → implementer ⇄ reviewer: the text says WHAT but not HOW, so the implementer would have to explore the codebase and design before writing (a plain prompt whose change is more than trivial)',
17
+ '- big plan — clarify → planner → refiner (selfLoop) → implementer ⇄ reviewer: the plan to be written is large or subtle enough that a second pass on it pays off (many files or subsystems, cross-cutting behaviour, unclear edge cases)',
18
+ '- given plan, large — refiner (selfLoop) → implementer ⇄ reviewer: the user GAVE a complete plan that is large or has gaps worth tightening before implementation; a partial plan (plan-partial) gets a planner in front: planner → refiner (selfLoop) → implementer ⇄ reviewer',
19
+ 'Clarify: it goes only directly in front of a planner (its answers feed the planner), only on a plain prompt (taskKind prompt), and only when a human is in the loop — the trivial and small rungs, a given plan and a run without a human never get it.',
20
+ 'Modifiers (exceptions, never a default — each needs a real signal in the task itself):',
21
+ '- web / UI feature — ONLY for a very big user-facing UI feature (many screens or flows, a new page with complex interaction): append manualTestsChecklist → manualWebUiTesting after the reviewer; its review loops back into the implementer automatically. A CSS tweak, a single component change or any small or medium UI change stays with the reviewer only, and the fingerprint\'s "web-ui likely" hint is context about the repository, never a trigger',
22
+ '- large task — only when the task really spans many files or subsystems: insert decomposer between the last planning stage and the implementer and set the implementer\'s fanOut to true',
23
+ '- risky or large plan — only for an irreversible or high-blast-radius change (data migrations, auth, billing, public APIs) or a plan too big to check by eye: add planReviewer right after the planner (it loops back into the planner)',
24
+ 'Rules: only a clarifier stage asks the user up front; a verifier loops automatically to the nearest earlier stage that can take its verdict (declare "loops" only to override, "loop": false to suppress); give "selfLoop": true to every stage whose card says it can loop on itself; use "parallel" only for stages that do not depend on each other; a stage after a parallel group waits for the whole group on every cycle; a verifier after a group loops back to a stage BEFORE the group unless it reads a member\'s output directly (the assembler enforces this — never loop into a member the verifier does not read); plugin and user agents (the card\'s origin field: plugin:<name> or user) fit wherever their ports match.',
25
+ ].join('\n');
26
+
27
+ const S = (agent, extra = {}) => ({ agent, ...extra });
28
+ const REFINER = () => S('refiner', { selfLoop: true });
29
+
30
+ /** Every rung × modifier the classifier is taught, as shapes the tests run offline. */
31
+ export const RECIPE_SHAPES = Object.freeze([
32
+ // the ladder
33
+ { id: 'trivial', shape: { name: 'Quick fix', taskKind: 'prompt', stages: [S('implementer')] } },
34
+ { id: 'small', shape: { name: 'Implement + review', taskKind: 'prompt', stages: [S('implementer'), S('reviewer')] } },
35
+ { id: 'needs-plan', shape: { name: 'Clarify, plan, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), S('implementer'), S('reviewer')] } },
36
+ { id: 'prompt', shape: { name: 'Clarify, plan, refine, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), REFINER(), S('implementer'), S('reviewer')] } },
37
+ { id: 'plan-partial', shape: { name: 'Plan, refine, implement + review', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('implementer'), S('reviewer')] } },
38
+ { id: 'plan-complete-detailed', shape: { name: 'Implement + review the plan', taskKind: 'plan-complete-detailed', stages: [S('implementer'), S('reviewer')] } },
39
+ { id: 'plan-complete-large', shape: { name: 'Refine, implement + review the plan', taskKind: 'plan-complete-detailed', stages: [REFINER(), S('implementer'), S('reviewer')] } },
40
+ { id: 'plan-complete-small', shape: { name: 'Implement the plan', taskKind: 'plan-complete-small', stages: [S('implementer')] } },
41
+ // modifiers
42
+ { id: 'prompt+web', shape: { name: 'Clarify, plan, refine, implement + web tests', taskKind: 'prompt', stages: [S('clarify'), S('planner'), REFINER(), S('implementer'), S('reviewer'), S('manualTestsChecklist'), S('manualWebUiTesting')] } },
43
+ { id: 'prompt+large', shape: { name: 'Clarify, plan, refine, decompose, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), REFINER(), S('decomposer'), S('implementer', { fanOut: true }), S('reviewer')] } },
44
+ { id: 'prompt+risky', shape: { name: 'Clarify, plan + plan review, refine, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), S('planReviewer'), REFINER(), S('implementer'), S('reviewer')] } },
45
+ { id: 'plan-partial+web+large', shape: { name: 'Plan, refine, decompose, implement + web tests', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('decomposer'), S('implementer', { fanOut: true }), S('reviewer'), S('manualTestsChecklist'), S('manualWebUiTesting')] } },
46
+ { id: 'plan-complete-detailed+web', shape: { name: 'Implement, review + web tests', taskKind: 'plan-complete-detailed', stages: [S('implementer'), S('reviewer'), S('manualTestsChecklist'), S('manualWebUiTesting')] } },
47
+ // A group may sit anywhere; a verifier after it loops to a stage the whole group
48
+ // depends on (or to a member it reads) — the assembler refuses anything else.
49
+ { id: 'parallel-end', shape: { name: 'Plan, refine, implement, review ∥ checklist', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('implementer'), { parallel: [S('reviewer'), S('manualTestsChecklist')] }] } },
50
+ { id: 'parallel-mid', shape: { name: 'Plan, refine, implement, review ∥ checklist, web tests', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('implementer'), { parallel: [S('reviewer'), S('manualTestsChecklist')] }, S('manualWebUiTesting')] } },
51
+ ]);
52
+
53
+ const WEB_RE = /\b(web|ui|page|button|css|react|vue|svelte|browser|frontend|front-end|html|component|modal|dropdown|theme)\b/i;
54
+ const clone = (v) => JSON.parse(JSON.stringify(v));
55
+ const byId = (id) => clone(RECIPE_SHAPES.find((r) => r.id === id).shape);
56
+
57
+ /**
58
+ * The offline classifier: cheap, deterministic heuristics over the task text.
59
+ * Never spawns anything — `npm run smoke`-style runs and the test suite depend on it.
60
+ */
61
+ export function mockShapeFor(taskText, { humanInLoop = true } = {}) {
62
+ const text = String(taskText || '').trim();
63
+ const hasHeading = /^#{1,3}\s+\S/m.test(text);
64
+ let shape;
65
+ if (hasHeading) shape = text.length >= 1200 ? byId('plan-complete-detailed') : byId('plan-complete-small');
66
+ else if (text.length < 80) shape = byId('trivial');
67
+ else if (WEB_RE.test(text)) shape = byId('prompt+web');
68
+ else shape = byId('prompt');
69
+ if (!humanInLoop) shape.stages = shape.stages.filter((s) => s.agent !== 'clarify');
70
+ shape.reasoning = `mock classifier: ${hasHeading ? 'the task is a plan' : text.length < 80 ? 'a trivial prompt' : 'a free-form prompt'}${WEB_RE.test(text) && !hasHeading && text.length >= 80 ? ' for a web feature' : ''}`;
71
+ const web = WEB_RE.test(text) && !hasHeading && text.length >= 80;
72
+ shape.size = hasHeading ? (text.length >= 1200 ? 'large' : 'small') : (text.length < 80 ? 'small' : 'medium');
73
+ shape.signals = hasHeading ? ['plan', text.length >= 1200 ? 'large' : 'trivial'] : (text.length < 80 ? ['trivial'] : (web ? ['web UI'] : []));
74
+ return shape;
75
+ }
@@ -0,0 +1,46 @@
1
+ // A short-lived, read-only DETACHED checkout the Auto classifier may Grep/Glob/Read
2
+ // (spec D6 amendment, 2026-09-07). The run path needs none: the run's own worktree IS
3
+ // the checkout (orchestrator.mjs _autoRound passes this.runCwd). The chat path
4
+ // (propose_workflow) has no worktree, so it borrows one here for the duration of ONE
5
+ // classifier call and removes it in close(). fs + the DB-free git primitives of
6
+ // ../worktree.mjs only — no shell, no hard-coded separators; no agent key is named here (D23).
7
+ import { existsSync, mkdirSync } from 'node:fs';
8
+ import { join } from 'node:path';
9
+ import { randomBytes } from 'node:crypto';
10
+ import { createDetachedWorktree, removeWorktree, isValidSourceRef } from '../worktree.mjs';
11
+
12
+ export const REPO_LOOK_DIRNAME = 'auto-look';
13
+
14
+ /**
15
+ * @param {string} projectDir the project's live checkout — read for HEAD, never used as cwd
16
+ * @param {string} baseDir where the throwaway checkout lives (the chat passes <worcaHome>/tmp/ask)
17
+ * @param {{signal?:AbortSignal|null, log?:(msg:string)=>void}} [o]
18
+ * @returns {Promise<{cwd:string, close:() => Promise<void>}|null>} null ⇒ no git repository, HEAD does
19
+ * not resolve, or git failed — the caller then classifies from the task text only, as before.
20
+ */
21
+ export async function openRepoLook(projectDir, baseDir, { signal = null, log = () => {} } = {}) {
22
+ try {
23
+ if (!projectDir || !baseDir || !existsSync(join(projectDir, '.git'))) return null;
24
+ if (signal?.aborted) throw new Error('aborted before the checkout');
25
+ if (!(await isValidSourceRef(projectDir, 'HEAD'))) return null;
26
+ mkdirSync(baseDir, { recursive: true });
27
+ const worktreeDir = join(baseDir, `${REPO_LOOK_DIRNAME}-${randomBytes(4).toString('hex')}`);
28
+ // The signal is spread in ONLY when present: worktree.mjs's git() forwards it verbatim to
29
+ // child_process.spawn, and Node rejects `options.signal: null` (ERR_INVALID_ARG_TYPE) — git()
30
+ // would swallow that as ok:false and this whole look would silently degrade to text-only.
31
+ await createDetachedWorktree({ projectDir, worktreeDir, ref: 'HEAD', ...(signal ? { signal } : {}) });
32
+ let closed = false;
33
+ const close = async () => {
34
+ if (closed) return;
35
+ closed = true;
36
+ try {
37
+ const res = await removeWorktree({ projectDir, worktreeDir, branch: null, force: true });
38
+ for (const s of (res.steps || []).filter((x) => !x.ok)) log(`repo look: ${s.step} failed for ${worktreeDir}: ${s.stderr || 'unknown error'}`);
39
+ } catch (err) { log(`repo look: could not remove ${worktreeDir}: ${err && err.message ? err.message : err}`); }
40
+ };
41
+ return { cwd: worktreeDir, close };
42
+ } catch (err) {
43
+ log(`repo look unavailable (${err && err.message ? err.message : err}); classifying from the task text only`);
44
+ return null;
45
+ }
46
+ }
@@ -39,7 +39,11 @@ import { effectiveDebugSpawn } from './settings.mjs';
39
39
  import { classifyError, strongestClass } from './recoverable-error.mjs';
40
40
  import { explainUnspawnableClaude, resolveClaudeBin } from './preflight.mjs';
41
41
  import { hostGuardEnabled, hostGuardHookEntry, hostGuardSystemPrompt } from './host-guard.mjs';
42
- import { writeFile, mkdir, appendFile, readFile, access } from 'node:fs/promises';
42
+ // The offline classifier and the shape normalizer the mock ask role answers
43
+ // propose_workflow with — both pure (no DB, no spawn).
44
+ import { mockShapeFor } from './auto/recipes.mjs';
45
+ import { normalizeShape } from '../shared/graph/assemble.mjs';
46
+ import { writeFile, mkdir, appendFile, readFile, access, readdir } from 'node:fs/promises';
43
47
  import { constants as FS, mkdtempSync, writeFileSync, rmSync } from 'node:fs';
44
48
  import { dirname, join } from 'node:path';
45
49
  import { tmpdir } from 'node:os';
@@ -342,7 +346,8 @@ export function mockEnabled(opts) {
342
346
  * @param {number} [o.maxTurns] --max-turns <n> (positive safe integer; else omitted)
343
347
  * @param {number|null} [o.maxBudgetUsd] --max-budget-usd <n> (finite > 0; null/else omitted)
344
348
  * @param {string} [o.appendSubagentSystemPrompt] --append-subagent-system-prompt <text> (Task children only)
345
- * All eight are Ask Worca sandbox options (ask-worca-design.md §6.3) and default-off.
349
+ * @param {string[]} [o.addDirs] --add-dir <dir> per entry (Ask Worca's memory mount; the CLI loads <dir>/.claude/rules only with CLAUDE_CODE_ADDITIONAL_DIRECTORIES_CLAUDE_MD=1 in the env — memory-deps.mjs / spawn.mjs set it)
350
+ * All nine are Ask Worca sandbox options (ask-worca-design.md §6.3) and default-off.
346
351
  * @param {number} [o.argvInlineLimit] override ARGV_INLINE_LIMIT (GH #380; tests force the staged path)
347
352
  * @returns {Promise<{text:string, exitCode:number}>}
348
353
  */
@@ -380,6 +385,7 @@ export async function runClaude(o = {}) {
380
385
  maxTurns,
381
386
  maxBudgetUsd,
382
387
  appendSubagentSystemPrompt,
388
+ addDirs,
383
389
  argvInlineLimit,
384
390
  bin = DEFAULT_BIN,
385
391
  } = o;
@@ -424,6 +430,7 @@ export async function runClaude(o = {}) {
424
430
  maxTurns,
425
431
  maxBudgetUsd,
426
432
  appendSubagentSystemPrompt,
433
+ addDirs,
427
434
  argvInlineLimit,
428
435
  });
429
436
  }
@@ -446,8 +453,9 @@ export async function runClaude(o = {}) {
446
453
  * (docs/run-root-verification.md, branch (a); argv-attested
447
454
  * transcript phase0/out/v1a-rerun.jsonl, with a no-grant
448
455
  * negative control proving the grant is load-bearing).
449
- * `--add-dir` is deliberately absent: it needs an env override to carry memory at
450
- * all (E2) and no shipped feature uses it (§5.3 / §8.18). */
456
+ * `--add-dir` carries Ask Worca's memory mount ONLY (`addDirs`): it needs the
457
+ * CLAUDE_CODE_ADDITIONAL_DIRECTORIES_CLAUDE_MD=1 env override to load memory at all
458
+ * (E2, re-probed 2026-09-13 on 2.1.270), and no pipeline path passes it (§5.3 / §8.18 unchanged). */
451
459
  export function buildClaudeArgs({
452
460
  prompt, systemPrompt, permissionMode, model, effort, allowedTools, resumeSessionId,
453
461
  mcpConfigPath, mcpServerGrants, permissionRules,
@@ -455,7 +463,7 @@ export function buildClaudeArgs({
455
463
  // way in because the legacy body below already owns a local `tools` (the
456
464
  // --allowedTools union).
457
465
  tools: builtinTools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages,
458
- maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, hostGuard,
466
+ maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, hostGuard, addDirs,
459
467
  }, delivery = {}) {
460
468
  // delivery (GH #380, set only by planClaudeInvocation's staged branch):
461
469
  // promptViaStdin -> bare `-p`; the prompt is written to the child's stdin
@@ -516,6 +524,9 @@ export function buildClaudeArgs({
516
524
  if (typeof appendSubagentSystemPrompt === 'string' && appendSubagentSystemPrompt) {
517
525
  args.push('--append-subagent-system-prompt', appendSubagentSystemPrompt);
518
526
  }
527
+ // Native-rules revision (2026-09-13): Ask Worca's memory mount. LAST, so every earlier argv
528
+ // stays a prefix; absent / [] / non-strings ⇒ nothing (the `names` filter above).
529
+ for (const d of names(addDirs)) args.push('--add-dir', d);
519
530
  return args;
520
531
  }
521
532
 
@@ -568,7 +579,7 @@ export function stageClaudeInvocation(opts, { bin = DEFAULT_BIN, limit = ARGV_IN
568
579
  return { ...plan, dir };
569
580
  }
570
581
 
571
- function runReal({ cwd, systemPrompt, prompt, allowedTools, permissionMode, model, effort, onEvent, signal, bin, resumeSessionId, mcpConfigPath, mcpServerGrants, permissionRules, envScrub, envAllowlist, modelEnv, tools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages, maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, argvInlineLimit }) {
582
+ function runReal({ cwd, systemPrompt, prompt, allowedTools, permissionMode, model, effort, onEvent, signal, bin, resumeSessionId, mcpConfigPath, mcpServerGrants, permissionRules, envScrub, envAllowlist, modelEnv, tools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages, maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, addDirs, argvInlineLimit }) {
572
583
  return new Promise((resolveP, rejectP) => {
573
584
  // Per-model routing env (design §4.4), prepared BEFORE argv: reserved keys
574
585
  // are re-dropped here defensively — the write path already rejects them, so
@@ -647,7 +658,7 @@ function runReal({ cwd, systemPrompt, prompt, allowedTools, permissionMode, mode
647
658
  permissionMode, model: wireModel, effort, allowedTools, resumeSessionId,
648
659
  mcpConfigPath, mcpServerGrants, permissionRules,
649
660
  tools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages,
650
- maxTurns, maxBudgetUsd, appendSubagentSystemPrompt,
661
+ maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, addDirs,
651
662
  }, { bin: resolved.bin, limit });
652
663
  } catch (err) {
653
664
  rejectP(new Error(`Failed to stage the claude prompt files: ${err.message}`));
@@ -965,7 +976,7 @@ async function emitLog(onEvent, text) {
965
976
  */
966
977
  export const MOCK_WRITER_ROLES = new Set([
967
978
  'clarify', 'planner-plan', 'refiner', 'decomposer', 'implementer', 'reviewer', 'plan-review',
968
- 'workspace-scan', 'agent-gen', 'workspace-reviewer', 'manual-tests-checklist', 'manual-web-ui-testing',
979
+ 'workspace-scan', 'agent-gen', 'workspace-reviewer', 'manual-tests-checklist', 'manual-web-ui-testing', 'memory-defrag',
969
980
  'generic-producer', 'generic-verifier',
970
981
  ]);
971
982
 
@@ -973,6 +984,10 @@ export const MOCK_WRITER_ROLES = new Set([
973
984
  export const MOCK_ROLE_CLARIFY = 'clarify';
974
985
  export const MOCK_ROLE_DECOMPOSER = 'decomposer';
975
986
 
987
+ /** The memory defragmenter's role, exported like the other two named roles (the switch below
988
+ * still uses the literal string: mock-writer-roles.test.mjs parses the switch arms). */
989
+ export const MOCK_ROLE_MEMORY_DEFRAG = 'memory-defrag';
990
+
976
991
  /**
977
992
  * The mock-fan-out roles (mirror the orchestrator's FANOUT_ELIGIBLE intent): the
978
993
  * roles whose real runs may spawn sub-agents. Keyed by the MOCK_ROLE strings.
@@ -1065,15 +1080,26 @@ async function mockAsk({ markers, prompt, cwd, onEvent, signal, resumeSessionId
1065
1080
  const maxTurns = /\bMOCK_MAX_TURNS\b/.test(userText);
1066
1081
  const maxBudget = /\bMOCK_MAX_BUDGET\b/.test(userText);
1067
1082
  const slow = /\bMOCK_SLOW\b/.test(userText);
1068
- const agents = /\bagents?\b/i.test(userText);
1069
- const propose = /\b(propose|start|run)\b/i.test(userText);
1083
+ // P3 (PD11): a workflow-card EVENT is matched first — it contains the words "workflow" and, when thenRun, "run",
1084
+ // which would otherwise trip the two arms below. Then the workflow trigger, then the run proposal.
1085
+ const wfEvent = /^\s*\[worca event\] workflow card (card_[0-9a-f]{8}) (?:(declined)|saved as (\S+) "([^"]*)"; thenRun=(true|false))/.exec(userText);
1086
+ // A metrics-card EVENT, then the metrics trigger — both before the run arm, whose \brun\b would otherwise fire on
1087
+ // "include my runs"-style prose (it does not, \b stops at the s, but "propose" would).
1088
+ const tmEvent = /^\s*\[worca event\] metrics card (card_[0-9a-f]{8}) (applied|declined|failed)/.exec(userText);
1089
+ // The metrics arm wants a CHANGE, not a question: "metrics" plus a verb of intent ("stop recording my metrics",
1090
+ // "route ... to the metrics home"). A bare "which workspaces use team metrics?" gets the generic echo answer.
1091
+ const metrics = !wfEvent && !tmEvent && /\bmetrics\b/i.test(userText)
1092
+ && /\b(?:stop|start|turn|toggle|switch|record\w*|route|change|enable|disable|set)\b/i.test(userText);
1093
+ const workflow = !wfEvent && !tmEvent && !metrics && /\bworkflow\b/i.test(userText);
1094
+ const agents = !wfEvent && !tmEvent && /\bagents?\b/i.test(userText);
1095
+ const propose = !wfEvent && !tmEvent && !workflow && !metrics && /\b(propose|start|run)\b/i.test(userText);
1070
1096
 
1071
1097
  const SID = resumeSessionId || 'mock-session-ask-1';
1072
1098
  const USAGE = { input_tokens: 10, output_tokens: 20, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 };
1073
1099
  const firstLine = userText.split(/\r?\n/).map((l) => l.trim()).find(Boolean) || '';
1074
1100
  const ANSWER = `[mock] ${firstLine.slice(0, 200)}`;
1075
1101
  const init = { type: 'system', subtype: 'init', session_id: SID, cwd, model: 'mock', permissionMode: 'dontAsk',
1076
- tools: ['Task', 'mcp__worca__list_runs', 'mcp__worca__get_run', 'mcp__worca__propose_run'],
1102
+ tools: ['Task', 'mcp__worca__list_runs', 'mcp__worca__get_run', 'mcp__worca__propose_run', 'mcp__worca__propose_workflow', 'mcp__worca__propose_metrics_change'],
1077
1103
  mcp_servers: [{ name: 'worca', status: 'connected' }], plugins: [], skills: [], slash_commands: [], agents: [], uuid: 'mock-uuid-init' };
1078
1104
  const mstart = (id) => ({ type: 'stream_event', event: { type: 'message_start', message: { id, model: 'mock', role: 'assistant', content: [], usage: USAGE } }, parent_tool_use_id: null, session_id: SID });
1079
1105
  const delta = (t) => ({ type: 'stream_event', event: { type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text: t } }, parent_tool_use_id: null, session_id: SID });
@@ -1110,6 +1136,43 @@ async function mockAsk({ markers, prompt, cwd, onEvent, signal, resumeSessionId
1110
1136
  );
1111
1137
  answerMsg = MSG2;
1112
1138
  }
1139
+ if (workflow) {
1140
+ // The result the real MCP child would return (plan PD1) — the parent re-assembles it with the real registry.
1141
+ const shape = normalizeShape(mockShapeFor(userText, { humanInLoop: true }));
1142
+ const wfInput = { task: userText.slice(0, 2000), projectKey: card.projectKey || null, thenRun: /\brun\b/i.test(userText) };
1143
+ frames.push(delta('[mock] '), delta('building '), delta('a workflow'), atext(MSG1, 'Building a workflow card.'),
1144
+ atool(MSG1, 'toolu_mock_workflow', 'mcp__worca__propose_workflow', wfInput),
1145
+ uresult('toolu_mock_workflow', JSON.stringify({ ok: true, mode: 'task', projectKey: card.projectKey || null, projectName: null, name: shape.name, match: null,
1146
+ warnings: [], summary: '', shape, costUsd: 0, fingerprint: 'top-level: (mock)\nhints: mock', note: '', thenRun: wfInput.thenRun })));
1147
+ answerMsg = MSG2;
1148
+ }
1149
+ if (metrics) {
1150
+ // The MCP child's validation result (metrics-proposal.mjs): the parent re-validates the INPUT and mints the card,
1151
+ // so a mock card always targets the context project's "Include my runs" switch (no git involved when applied).
1152
+ const tmInput = { kind: 'record', projectKey: card.projectKey || null, record: false, note: 'mock: stop recording my runs here' };
1153
+ frames.push(delta('[mock] '), delta('proposing '), delta('a metrics change'), atext(MSG1, 'Proposing a metrics change card.'),
1154
+ atool(MSG1, 'toolu_mock_metrics', 'mcp__worca__propose_metrics_change', tmInput),
1155
+ uresult('toolu_mock_metrics', JSON.stringify({ ok: true, card: { type: 'metrics', ...tmInput } })));
1156
+ answerMsg = MSG2;
1157
+ }
1158
+ if (tmEvent) {
1159
+ const line = tmEvent[2] === 'declined' ? 'Declined — nothing changed.' : tmEvent[2] === 'failed' ? 'The change failed; check the error and try again.' : 'Applied.';
1160
+ frames.push(delta('[mock] '), delta(tmEvent[2]), atext(MSG1, line));
1161
+ answerMsg = MSG2;
1162
+ }
1163
+ if (wfEvent) {
1164
+ // Every event arm answers on MSG2: a second atext on MSG1 would REPLACE the first reply's text.
1165
+ if (wfEvent[2] === 'declined') {
1166
+ frames.push(delta('[mock] '), delta('declined'), atext(MSG1, 'Declined. Want another auto workflow, tell me what to change, or pick a saved workflow?'));
1167
+ } else if (wfEvent[5] === 'true') {
1168
+ frames.push(delta('[mock] '), delta('proposing '), delta('a run'), atext(MSG1, `Proposing a run with "${wfEvent[4]}".`),
1169
+ atool(MSG1, 'toolu_mock_propose', 'mcp__worca__propose_run', { ...card, workflowId: wfEvent[3], brief: `Run with "${wfEvent[4]}"` }),
1170
+ uresult('toolu_mock_propose', JSON.stringify({ ok: true })));
1171
+ } else {
1172
+ frames.push(delta('[mock] '), delta('saved'), atext(MSG1, `Saved "${wfEvent[4]}". Say "run it" when you want a run with it.`));
1173
+ }
1174
+ answerMsg = MSG2;
1175
+ }
1113
1176
  if (propose) {
1114
1177
  frames.push(delta('[mock] '), delta('preparing '), delta('a run'), atext(MSG1, 'Preparing a run card.'),
1115
1178
  atool(MSG1, 'toolu_mock_propose', 'mcp__worca__propose_run', card), uresult('toolu_mock_propose', JSON.stringify({ ok: true })));
@@ -1230,6 +1293,9 @@ async function runMock({ cwd, systemPrompt, prompt, onEvent, signal, resumeSessi
1230
1293
  case 'manual-web-ui-testing':
1231
1294
  text = await mockManualWebUiTesting(m, cycle, onEvent);
1232
1295
  break;
1296
+ case 'memory-defrag':
1297
+ text = await mockMemoryDefrag(m, systemPrompt, onEvent);
1298
+ break;
1233
1299
  case 'generic-producer':
1234
1300
  text = await mockGenericProducer(m, onEvent);
1235
1301
  break;
@@ -1331,6 +1397,61 @@ async function mockGenericProducer(m, onEvent) {
1331
1397
  return `[mock] generic artifact written to ${out}`;
1332
1398
  }
1333
1399
 
1400
+ /** The scope dirs the `## Worca memory` block of a system prompt names (memory-store.mjs
1401
+ * renderMemoryBlock: `<label> — <abs dir>:` lines under the heading; the block is contiguous,
1402
+ * so the first blank line ends it). The mock defragmenter finds its mount exactly the way the
1403
+ * real agent is told to — from its system prompt — so no MOCK marker is needed. Exported for
1404
+ * the parity test against a real renderMemoryBlock output. Both captures are greedy: a project
1405
+ * LABEL may itself contain ` — ` (the renderer's separator is the last one on the line), and a
1406
+ * Windows dir carries a drive colon while the line still ends with `:`. */
1407
+ export function memoryDirsFromPrompt(systemPrompt) {
1408
+ const text = String(systemPrompt || '');
1409
+ const at = text.indexOf('## Worca memory');
1410
+ if (at === -1) return [];
1411
+ const out = [];
1412
+ for (const line of text.slice(at).split(/\r?\n/).slice(1)) {
1413
+ if (!line.trim()) break;
1414
+ const m = line.match(/^(?:Global|Project .*) — (.+):$/);
1415
+ if (m) out.push(m[1]);
1416
+ }
1417
+ return out;
1418
+ }
1419
+
1420
+ /** Memory defragment mock (agent-memory-design.md §7.1): in the FIRST scope dir the system
1421
+ * prompt names, fold the second (sorted) file's body into the first and EMPTY it (amendment
1422
+ * B19 — the memory tool set cannot unlink, so this is the path the real agent takes), then
1423
+ * write the report to MOCK_OUT. A defrag run mounts exactly ONE scope dir. */
1424
+ async function mockMemoryDefrag(m, systemPrompt, onEvent) {
1425
+ const out = m.MOCK_OUT;
1426
+ const dir = memoryDirsFromPrompt(systemPrompt)[0] || null;
1427
+ await emitLog(onEvent, '[mock] memory defragmenter restructuring the mounted scope');
1428
+ const lines = ['# Memory defragment report', ''];
1429
+ let merged = null;
1430
+ if (dir) {
1431
+ const files = (await readdir(dir, { withFileTypes: true })).filter((e) => e.isFile() && e.name.endsWith('.md')).map((e) => e.name).sort();
1432
+ if (files.length < 2) {
1433
+ lines.push(`- ${dir}: ${files.length} file(s), nothing to merge`);
1434
+ } else {
1435
+ const [a, b] = files;
1436
+ const bodyB = (await readFile(join(dir, b), 'utf8')).replace(/^---\n[\s\S]*?\n---\n/, '');
1437
+ const text = `${await readFile(join(dir, a), 'utf8')}\n## Merged from ${b.slice(0, -3)}\n\n${bodyB}`;
1438
+ await writeFile(join(dir, a), text, 'utf8');
1439
+ await writeFile(join(dir, b), '', 'utf8'); // B19: an EMPTIED mount file is a deletion request
1440
+ merged = [a, b];
1441
+ lines.push(`- ${dir}: merged ${b} into ${a}; emptied ${b} (worca removes it at sync-back)`);
1442
+ safeEmit(onEvent, { type: 'tool_use', text: `merged ${join(dir, b)} into ${join(dir, a)}`, raw: { mock: true, file: join(dir, a) } });
1443
+ }
1444
+ } else {
1445
+ lines.push('- no memory scope in the system prompt: nothing to defragment');
1446
+ }
1447
+ if (out) {
1448
+ await ensureDir(out);
1449
+ await writeFile(out, `${lines.join('\n')}\n`, 'utf8');
1450
+ safeEmit(onEvent, { type: 'tool_use', text: `wrote ${out}`, raw: { mock: true, file: out } });
1451
+ }
1452
+ return merged ? `[mock] memory defragment: merged ${merged[1]} into ${merged[0]}` : '[mock] memory defragment: nothing to merge';
1453
+ }
1454
+
1334
1455
  async function mockPlannerPlan(m, onEvent) {
1335
1456
  const out = m.MOCK_OUT;
1336
1457
  const base = m.MOCK_BASE || 'feature';