@worca/app 1.2.0 → 1.3.0-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -0
- package/agents/memoryDefragmenter.meta.json +24 -0
- package/agents/worca-cc-code-reviewer.md +6 -1
- package/agents/worca-cc-implementer.md +6 -1
- package/agents/worca-cc-memory-defragmenter.md +32 -0
- package/agents/worca-cc-planner.md +5 -1
- package/package.json +5 -2
- package/src/cli/render.mjs +36 -0
- package/src/cli/worca-cc.mjs +137 -8
- package/src/core/agent-registry.mjs +12 -34
- package/src/core/artifacts.mjs +132 -8
- package/src/core/ask/catalog.mjs +32 -7
- package/src/core/ask/comment-deps.mjs +5 -2
- package/src/core/ask/events.mjs +65 -2
- package/src/core/ask/limits.mjs +9 -0
- package/src/core/ask/mcp-stdio.mjs +10 -0
- package/src/core/ask/memory-deps.mjs +107 -0
- package/src/core/ask/metrics-deps.mjs +124 -0
- package/src/core/ask/metrics-proposal.mjs +175 -0
- package/src/core/ask/prompt.mjs +53 -10
- package/src/core/ask/proposal.mjs +49 -2
- package/src/core/ask/spawn.mjs +21 -4
- package/src/core/ask/store.mjs +14 -5
- package/src/core/ask/tool-deps.mjs +26 -2
- package/src/core/ask/tools.mjs +439 -6
- package/src/core/ask/turn.mjs +163 -4
- package/src/core/ask/workflow-deps.mjs +226 -0
- package/src/core/auto/classify.mjs +352 -0
- package/src/core/auto/fingerprint.mjs +141 -0
- package/src/core/auto/match.mjs +30 -0
- package/src/core/auto/model.mjs +23 -0
- package/src/core/auto/proposal.mjs +132 -0
- package/src/core/auto/recipes.mjs +75 -0
- package/src/core/auto/repo-look.mjs +46 -0
- package/src/core/claude-runner.mjs +132 -11
- package/src/core/config.mjs +120 -3
- package/src/core/db.mjs +44 -1
- package/src/core/diff-comments.mjs +55 -9
- package/src/core/frontmatter.mjs +75 -0
- package/src/core/git-info.mjs +233 -26
- package/src/core/graph/builtin-workflows.mjs +50 -0
- package/src/core/graph/executor.mjs +11 -3
- package/src/core/index-html.mjs +17 -0
- package/src/core/memory-store.mjs +441 -0
- package/src/core/memory-sync.mjs +300 -0
- package/src/core/metrics/ledger.mjs +47 -0
- package/src/core/metrics/lock.mjs +117 -0
- package/src/core/metrics/read.mjs +303 -0
- package/src/core/metrics/record.mjs +389 -0
- package/src/core/metrics/sync.mjs +1100 -0
- package/src/core/onboarding.mjs +99 -0
- package/src/core/orchestrator.mjs +394 -7
- package/src/core/phases.mjs +16 -3
- package/src/core/pipeline-delete.mjs +1 -1
- package/src/core/plugin-store.mjs +2 -10
- package/src/core/preflight.mjs +2 -3
- package/src/core/projects.mjs +16 -1
- package/src/core/run-harness.mjs +458 -32
- package/src/core/run-report.mjs +896 -0
- package/src/core/settings.mjs +162 -0
- package/src/core/sources.mjs +4 -1
- package/src/core/store.mjs +5 -0
- package/src/core/workflow-export.mjs +2 -0
- package/src/core/workflow-share.mjs +1 -0
- package/src/core/workflows.mjs +43 -23
- package/src/core/workspaces.mjs +37 -8
- package/src/shared/graph/agent-meta.mjs +5 -2
- package/src/shared/graph/assemble.mjs +455 -0
- package/src/shared/graph/flow-layout.mjs +249 -0
- package/src/shared/graph/geometry.mjs +48 -28
- package/src/shared/graph/isomorphic.mjs +101 -0
- package/src/shared/report-reasons.mjs +58 -0
- package/src/shared/team-metrics/aggregate.mjs +341 -0
- package/src/shared/team-metrics/workspace-match.mjs +13 -0
- package/ui/public/about-links.mjs +21 -0
- package/ui/public/app.js +3715 -479
- package/ui/public/artifact-view.mjs +135 -0
- package/ui/public/ask-model.mjs +18 -1
- package/ui/public/ask-panel.mjs +1359 -214
- package/ui/public/ask-run-card.mjs +209 -0
- package/ui/public/assets/worca-logo-mask.png +0 -0
- package/ui/public/assets/worca-mark-mask.png +0 -0
- package/ui/public/auto-build.mjs +95 -0
- package/ui/public/auto-proposal.mjs +174 -0
- package/ui/public/comment-thread.mjs +55 -0
- package/ui/public/getting-started.mjs +261 -0
- package/ui/public/graph/composer.mjs +41 -5
- package/ui/public/graph/inspector.mjs +3 -1
- package/ui/public/graph/model.mjs +1 -0
- package/ui/public/graph/run-hosts.mjs +73 -12
- package/ui/public/graph/view.mjs +218 -50
- package/ui/public/guide-spot.mjs +215 -0
- package/ui/public/index.html +423 -25
- package/ui/public/memory-view.mjs +192 -0
- package/ui/public/node-tunables.mjs +201 -0
- package/ui/public/report-run.mjs +75 -0
- package/ui/public/results-view.mjs +25 -0
- package/ui/public/source-pane.mjs +16 -2
- package/ui/public/stats-view.mjs +2 -2
- package/ui/public/style.css +1450 -303
- package/ui/public/team-metrics-surfaces.mjs +452 -0
- package/ui/public/team-metrics-view.mjs +533 -0
- package/ui/public/thinking-orb.mjs +46 -8
- package/ui/server.mjs +1282 -193
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
// The `workflow` question payload (spec §5.3/§5.4) and the sanitiser for its
|
|
2
|
+
// answer. Pure: the orchestrator hands in the template (new or matched), the
|
|
3
|
+
// per-node tunables and the registry slice, and gets back exactly what the CLI
|
|
4
|
+
// and the browser render.
|
|
5
|
+
import { buildGraphManifest } from '../../shared/graph/manifest.mjs';
|
|
6
|
+
import { cleanText } from '../../shared/graph/assemble.mjs';
|
|
7
|
+
import { slugify } from '../artifacts.mjs';
|
|
8
|
+
import { GRAPH_DEFAULT_WORKFLOW, AUTO_WORKFLOW_ID, MEMORY_DEFRAG_WORKFLOW_ID } from '../workflows.mjs';
|
|
9
|
+
|
|
10
|
+
const isObject = (v) => Boolean(v) && typeof v === 'object' && !Array.isArray(v);
|
|
11
|
+
|
|
12
|
+
/** Tunables keyed by assembled node ids -> keyed by the matched row's node ids. */
|
|
13
|
+
export function remapTunables(tunables, nodeMap) {
|
|
14
|
+
const out = {};
|
|
15
|
+
for (const [id, sel] of Object.entries(tunables || {})) {
|
|
16
|
+
const to = nodeMap?.get?.(id);
|
|
17
|
+
if (to) out[to] = { ...sel };
|
|
18
|
+
}
|
|
19
|
+
return out;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* @param {{round:number, shape:object, template:object, match:{id:string,name:string}|null,
|
|
24
|
+
* tunables:Record<string,object>, registry:Record<string,object>,
|
|
25
|
+
* models:Array<{id:string,label?:string,efforts?:string[],hidden?:boolean}>, warnings?:Array, costUsd?:number,
|
|
26
|
+
* fingerprint?:string, ignoredProjectOverrides?:boolean}} o
|
|
27
|
+
*/
|
|
28
|
+
export function buildProposal({ round, shape, template, match = null, tunables = {}, registry = {}, models = [], warnings = [], costUsd = 0, fingerprint = '', ignoredProjectOverrides = false }) {
|
|
29
|
+
const name = shape?.name || template?.name || 'Auto workflow';
|
|
30
|
+
const agentsByKey = {};
|
|
31
|
+
for (const n of template.nodes || []) if (n.kind === 'agent' && registry[n.key]) agentsByKey[n.key] = registry[n.key];
|
|
32
|
+
// B1 (2026-09-07 review): paint what resolveGraph will RUN (workflows.mjs:664) — an overlay
|
|
33
|
+
// model without an effort suppresses the row's authored effort, so a matched twin's "high"
|
|
34
|
+
// must not be shown next to a classifier-picked model that carries none. The manifest (the
|
|
35
|
+
// graph's band chips, manifest.mjs:145) and the table below read the SAME effective pair:
|
|
36
|
+
// an explicit `effort: ''` overlay is what makes the manifest agree.
|
|
37
|
+
const nodes = {};
|
|
38
|
+
const overlays = {};
|
|
39
|
+
for (const n of template.nodes || []) {
|
|
40
|
+
if (n.kind !== 'agent') continue;
|
|
41
|
+
const meta = registry[n.key] || {};
|
|
42
|
+
const cfg = n.config || {};
|
|
43
|
+
const t = tunables[n.id] || {};
|
|
44
|
+
const effort = t.effort ?? (t.model ? '' : (cfg.effort ?? ''));
|
|
45
|
+
overlays[n.id] = t.model && t.effort === undefined ? { ...t, effort: '' } : { ...t };
|
|
46
|
+
const asks = !!meta.asksQuestions;
|
|
47
|
+
nodes[n.id] = {
|
|
48
|
+
key: n.key,
|
|
49
|
+
label: meta.displayName || n.key,
|
|
50
|
+
model: t.model ?? cfg.model ?? '',
|
|
51
|
+
effort,
|
|
52
|
+
fanOut: !!(t.fanOut ?? cfg.fanOut ?? meta.fanOut ?? false),
|
|
53
|
+
askQuestions: asks ? (meta.questionsLocked ? !!meta.questionsDefault : !!(t.askQuestions ?? cfg.askQuestions ?? meta.questionsDefault ?? false)) : false,
|
|
54
|
+
asksQuestions: asks,
|
|
55
|
+
questionsLocked: asks && !!meta.questionsLocked,
|
|
56
|
+
canFanOut: !!meta.fanOut,
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
const manifest = buildGraphManifest({ ...template, id: match ? match.id : AUTO_WORKFLOW_ID, name }, agentsByKey, { overlays: { nodes: overlays } });
|
|
60
|
+
// The DISPATCH order of the AGENT nodes (rank, then launch order) — the manifest
|
|
61
|
+
// already computes it for the run monitor; a reused composer row's graph.nodes order
|
|
62
|
+
// is arbitrary. The steps cells bucket EVERY node — the Task card, End and the gates
|
|
63
|
+
// ride along with `key: null` — so keep only the agent cells.
|
|
64
|
+
const order = (manifest.steps || [])
|
|
65
|
+
.filter((s) => s.kind === 'agents')
|
|
66
|
+
.flatMap((s) => s.nodes.filter((n) => n.key).map((n) => n.id));
|
|
67
|
+
return {
|
|
68
|
+
round,
|
|
69
|
+
name,
|
|
70
|
+
reasoning: shape?.reasoning || '',
|
|
71
|
+
taskKind: shape?.taskKind || 'prompt',
|
|
72
|
+
size: shape?.size || 'medium',
|
|
73
|
+
signals: Array.isArray(shape?.signals) ? [...shape.signals] : [],
|
|
74
|
+
warnings: (warnings || []).map((w) => (typeof w === 'string' ? w : w.message)).filter(Boolean),
|
|
75
|
+
match: match ? { id: match.id, name: match.name } : null,
|
|
76
|
+
manifest,
|
|
77
|
+
order,
|
|
78
|
+
nodes,
|
|
79
|
+
models: models.filter((m) => m && !m.hidden).map((m) => ({ id: m.id, label: m.label || m.id, efforts: [...(m.efforts || [])] })),
|
|
80
|
+
costUsd: Number.isFinite(costUsd) ? costUsd : 0,
|
|
81
|
+
fingerprint: typeof fingerprint === 'string' ? fingerprint : '',
|
|
82
|
+
ignoredProjectOverrides: !!ignoredProjectOverrides,
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* @returns {{decision:'accept',name:string,nodes:object}|{decision:'revise',text:string}|{decision:'cancel'}|null}
|
|
88
|
+
*/
|
|
89
|
+
export function sanitizeProposalAnswer(payload, { proposal, models = [], registry = {} }) {
|
|
90
|
+
if (!isObject(payload)) return null;
|
|
91
|
+
const d = payload.decision;
|
|
92
|
+
if (d === 'cancel') return { decision: 'cancel' };
|
|
93
|
+
if (d === 'revise') {
|
|
94
|
+
const text = typeof payload.text === 'string' ? payload.text.trim().slice(0, 4096) : '';
|
|
95
|
+
return text ? { decision: 'revise', text } : null;
|
|
96
|
+
}
|
|
97
|
+
if (d !== 'accept') return null;
|
|
98
|
+
const name = cleanText(payload.name, 60) || proposal.name;
|
|
99
|
+
const byId = new Map(models.map((m) => [String(m.id).toLowerCase(), m])); // the FULL catalog: a hidden id still resolves
|
|
100
|
+
const nodes = {};
|
|
101
|
+
for (const [nodeId, sel] of Object.entries(isObject(payload.nodes) ? payload.nodes : {})) {
|
|
102
|
+
const known = proposal.nodes?.[nodeId];
|
|
103
|
+
if (!known || !isObject(sel)) continue;
|
|
104
|
+
const meta = registry[known.key] || {};
|
|
105
|
+
const out = {};
|
|
106
|
+
if (typeof sel.model === 'string') {
|
|
107
|
+
const m = byId.get(sel.model.trim().toLowerCase());
|
|
108
|
+
if (m) out.model = m.id;
|
|
109
|
+
// '' clears the model AND the effort: resolveGraph keeps a row's authored effort
|
|
110
|
+
// when the overlay's model is falsy, which would leave an effort with no model.
|
|
111
|
+
else if (sel.model.trim() === '') { out.model = ''; out.effort = ''; }
|
|
112
|
+
}
|
|
113
|
+
if (typeof sel.effort === 'string') {
|
|
114
|
+
const mid = out.model !== undefined ? out.model : known.model;
|
|
115
|
+
const m = mid ? byId.get(String(mid).toLowerCase()) : null;
|
|
116
|
+
if (m && (m.efforts || []).includes(sel.effort)) out.effort = sel.effort;
|
|
117
|
+
}
|
|
118
|
+
if (typeof sel.fanOut === 'boolean' && meta.fanOut) out.fanOut = sel.fanOut;
|
|
119
|
+
if (typeof sel.askQuestions === 'boolean' && meta.asksQuestions && !meta.questionsLocked) out.askQuestions = sel.askQuestions;
|
|
120
|
+
if (Object.keys(out).length) nodes[nodeId] = out;
|
|
121
|
+
}
|
|
122
|
+
return { decision: 'accept', name, nodes };
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/** `wf_<slug>` that no row owns; reserved / empty slugs become wf_auto-workflow. */
|
|
126
|
+
export async function mintAutoWorkflowId(name, exists) {
|
|
127
|
+
let stem = `wf_${slugify(name)}`;
|
|
128
|
+
if (stem === GRAPH_DEFAULT_WORKFLOW.id || stem === AUTO_WORKFLOW_ID || stem === MEMORY_DEFRAG_WORKFLOW_ID || stem === 'wf_untitled') stem = 'wf_auto-workflow';
|
|
129
|
+
let id = stem;
|
|
130
|
+
for (let n = 2; await exists(id); n += 1) id = `${stem}-${n}`;
|
|
131
|
+
return id;
|
|
132
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
// The ONE Auto-path module where agent keys may appear (spec D23): prompt
|
|
2
|
+
// guidance for the classifier and the canned shapes the offline mock answers
|
|
3
|
+
// with. The assembler and the orchestrator never read these keys — they read
|
|
4
|
+
// port meta.
|
|
5
|
+
|
|
6
|
+
// Rendered into BOTH selection paths — the Ask system prompt (prompt.mjs renderCatalog)
|
|
7
|
+
// and the classifier system prompt (classify.mjs) — so the sizing principle is stated
|
|
8
|
+
// here once: build UP from the implementer, one rung per concrete signal (2026-09-07 ladder).
|
|
9
|
+
export const RECIPE_GUIDE = [
|
|
10
|
+
'## Recipes (starting points — adapt them to the task)',
|
|
11
|
+
'Sizing: build the workflow UP from the implementer and add a stage only when a concrete signal in the task itself demands it — every extra stage must earn its cost in time and money; when unsure between two shapes, take the lighter one. Never inflate "size" or "signals" to justify a stage.',
|
|
12
|
+
'taskKind names what the user GAVE, never how big the work is: prompt = an idea, request or bug report with no plan; plan-partial = a sketch or a plan with gaps; plan-complete-detailed = a complete, detailed plan; plan-complete-small = a complete but small plan. Only plan-complete-* makes the task document the plan of record, so never label a prompt as a plan to reach a lighter rung — the ladder below reaches it directly.',
|
|
13
|
+
'The ladder (each rung adds ONE stage to the rung before it; take the first rung that fits):',
|
|
14
|
+
'- trivial — implementer only: a well-specified small change whose result the tests can check (a dependency bump, a config or copy change, a one-file fix, a rename, a complete small plan)',
|
|
15
|
+
'- small — implementer ⇄ reviewer: the change is bigger than one bounded edit (several files, a new code path, a public interface, anything a second pair of eyes should check) but the text already says precisely what to build — a complete, detailed plan lands here',
|
|
16
|
+
'- needs a plan — clarify → planner → implementer ⇄ reviewer: the text says WHAT but not HOW, so the implementer would have to explore the codebase and design before writing (a plain prompt whose change is more than trivial)',
|
|
17
|
+
'- big plan — clarify → planner → refiner (selfLoop) → implementer ⇄ reviewer: the plan to be written is large or subtle enough that a second pass on it pays off (many files or subsystems, cross-cutting behaviour, unclear edge cases)',
|
|
18
|
+
'- given plan, large — refiner (selfLoop) → implementer ⇄ reviewer: the user GAVE a complete plan that is large or has gaps worth tightening before implementation; a partial plan (plan-partial) gets a planner in front: planner → refiner (selfLoop) → implementer ⇄ reviewer',
|
|
19
|
+
'Clarify: it goes only directly in front of a planner (its answers feed the planner), only on a plain prompt (taskKind prompt), and only when a human is in the loop — the trivial and small rungs, a given plan and a run without a human never get it.',
|
|
20
|
+
'Modifiers (exceptions, never a default — each needs a real signal in the task itself):',
|
|
21
|
+
'- web / UI feature — ONLY for a very big user-facing UI feature (many screens or flows, a new page with complex interaction): append manualTestsChecklist → manualWebUiTesting after the reviewer; its review loops back into the implementer automatically. A CSS tweak, a single component change or any small or medium UI change stays with the reviewer only, and the fingerprint\'s "web-ui likely" hint is context about the repository, never a trigger',
|
|
22
|
+
'- large task — only when the task really spans many files or subsystems: insert decomposer between the last planning stage and the implementer and set the implementer\'s fanOut to true',
|
|
23
|
+
'- risky or large plan — only for an irreversible or high-blast-radius change (data migrations, auth, billing, public APIs) or a plan too big to check by eye: add planReviewer right after the planner (it loops back into the planner)',
|
|
24
|
+
'Rules: only a clarifier stage asks the user up front; a verifier loops automatically to the nearest earlier stage that can take its verdict (declare "loops" only to override, "loop": false to suppress); give "selfLoop": true to every stage whose card says it can loop on itself; use "parallel" only for stages that do not depend on each other; a stage after a parallel group waits for the whole group on every cycle; a verifier after a group loops back to a stage BEFORE the group unless it reads a member\'s output directly (the assembler enforces this — never loop into a member the verifier does not read); plugin and user agents (the card\'s origin field: plugin:<name> or user) fit wherever their ports match.',
|
|
25
|
+
].join('\n');
|
|
26
|
+
|
|
27
|
+
const S = (agent, extra = {}) => ({ agent, ...extra });
|
|
28
|
+
const REFINER = () => S('refiner', { selfLoop: true });
|
|
29
|
+
|
|
30
|
+
/** Every rung × modifier the classifier is taught, as shapes the tests run offline. */
|
|
31
|
+
export const RECIPE_SHAPES = Object.freeze([
|
|
32
|
+
// the ladder
|
|
33
|
+
{ id: 'trivial', shape: { name: 'Quick fix', taskKind: 'prompt', stages: [S('implementer')] } },
|
|
34
|
+
{ id: 'small', shape: { name: 'Implement + review', taskKind: 'prompt', stages: [S('implementer'), S('reviewer')] } },
|
|
35
|
+
{ id: 'needs-plan', shape: { name: 'Clarify, plan, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), S('implementer'), S('reviewer')] } },
|
|
36
|
+
{ id: 'prompt', shape: { name: 'Clarify, plan, refine, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), REFINER(), S('implementer'), S('reviewer')] } },
|
|
37
|
+
{ id: 'plan-partial', shape: { name: 'Plan, refine, implement + review', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('implementer'), S('reviewer')] } },
|
|
38
|
+
{ id: 'plan-complete-detailed', shape: { name: 'Implement + review the plan', taskKind: 'plan-complete-detailed', stages: [S('implementer'), S('reviewer')] } },
|
|
39
|
+
{ id: 'plan-complete-large', shape: { name: 'Refine, implement + review the plan', taskKind: 'plan-complete-detailed', stages: [REFINER(), S('implementer'), S('reviewer')] } },
|
|
40
|
+
{ id: 'plan-complete-small', shape: { name: 'Implement the plan', taskKind: 'plan-complete-small', stages: [S('implementer')] } },
|
|
41
|
+
// modifiers
|
|
42
|
+
{ id: 'prompt+web', shape: { name: 'Clarify, plan, refine, implement + web tests', taskKind: 'prompt', stages: [S('clarify'), S('planner'), REFINER(), S('implementer'), S('reviewer'), S('manualTestsChecklist'), S('manualWebUiTesting')] } },
|
|
43
|
+
{ id: 'prompt+large', shape: { name: 'Clarify, plan, refine, decompose, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), REFINER(), S('decomposer'), S('implementer', { fanOut: true }), S('reviewer')] } },
|
|
44
|
+
{ id: 'prompt+risky', shape: { name: 'Clarify, plan + plan review, refine, implement + review', taskKind: 'prompt', stages: [S('clarify'), S('planner'), S('planReviewer'), REFINER(), S('implementer'), S('reviewer')] } },
|
|
45
|
+
{ id: 'plan-partial+web+large', shape: { name: 'Plan, refine, decompose, implement + web tests', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('decomposer'), S('implementer', { fanOut: true }), S('reviewer'), S('manualTestsChecklist'), S('manualWebUiTesting')] } },
|
|
46
|
+
{ id: 'plan-complete-detailed+web', shape: { name: 'Implement, review + web tests', taskKind: 'plan-complete-detailed', stages: [S('implementer'), S('reviewer'), S('manualTestsChecklist'), S('manualWebUiTesting')] } },
|
|
47
|
+
// A group may sit anywhere; a verifier after it loops to a stage the whole group
|
|
48
|
+
// depends on (or to a member it reads) — the assembler refuses anything else.
|
|
49
|
+
{ id: 'parallel-end', shape: { name: 'Plan, refine, implement, review ∥ checklist', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('implementer'), { parallel: [S('reviewer'), S('manualTestsChecklist')] }] } },
|
|
50
|
+
{ id: 'parallel-mid', shape: { name: 'Plan, refine, implement, review ∥ checklist, web tests', taskKind: 'plan-partial', stages: [S('planner'), REFINER(), S('implementer'), { parallel: [S('reviewer'), S('manualTestsChecklist')] }, S('manualWebUiTesting')] } },
|
|
51
|
+
]);
|
|
52
|
+
|
|
53
|
+
const WEB_RE = /\b(web|ui|page|button|css|react|vue|svelte|browser|frontend|front-end|html|component|modal|dropdown|theme)\b/i;
|
|
54
|
+
const clone = (v) => JSON.parse(JSON.stringify(v));
|
|
55
|
+
const byId = (id) => clone(RECIPE_SHAPES.find((r) => r.id === id).shape);
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* The offline classifier: cheap, deterministic heuristics over the task text.
|
|
59
|
+
* Never spawns anything — `npm run smoke`-style runs and the test suite depend on it.
|
|
60
|
+
*/
|
|
61
|
+
export function mockShapeFor(taskText, { humanInLoop = true } = {}) {
|
|
62
|
+
const text = String(taskText || '').trim();
|
|
63
|
+
const hasHeading = /^#{1,3}\s+\S/m.test(text);
|
|
64
|
+
let shape;
|
|
65
|
+
if (hasHeading) shape = text.length >= 1200 ? byId('plan-complete-detailed') : byId('plan-complete-small');
|
|
66
|
+
else if (text.length < 80) shape = byId('trivial');
|
|
67
|
+
else if (WEB_RE.test(text)) shape = byId('prompt+web');
|
|
68
|
+
else shape = byId('prompt');
|
|
69
|
+
if (!humanInLoop) shape.stages = shape.stages.filter((s) => s.agent !== 'clarify');
|
|
70
|
+
shape.reasoning = `mock classifier: ${hasHeading ? 'the task is a plan' : text.length < 80 ? 'a trivial prompt' : 'a free-form prompt'}${WEB_RE.test(text) && !hasHeading && text.length >= 80 ? ' for a web feature' : ''}`;
|
|
71
|
+
const web = WEB_RE.test(text) && !hasHeading && text.length >= 80;
|
|
72
|
+
shape.size = hasHeading ? (text.length >= 1200 ? 'large' : 'small') : (text.length < 80 ? 'small' : 'medium');
|
|
73
|
+
shape.signals = hasHeading ? ['plan', text.length >= 1200 ? 'large' : 'trivial'] : (text.length < 80 ? ['trivial'] : (web ? ['web UI'] : []));
|
|
74
|
+
return shape;
|
|
75
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
// A short-lived, read-only DETACHED checkout the Auto classifier may Grep/Glob/Read
|
|
2
|
+
// (spec D6 amendment, 2026-09-07). The run path needs none: the run's own worktree IS
|
|
3
|
+
// the checkout (orchestrator.mjs _autoRound passes this.runCwd). The chat path
|
|
4
|
+
// (propose_workflow) has no worktree, so it borrows one here for the duration of ONE
|
|
5
|
+
// classifier call and removes it in close(). fs + the DB-free git primitives of
|
|
6
|
+
// ../worktree.mjs only — no shell, no hard-coded separators; no agent key is named here (D23).
|
|
7
|
+
import { existsSync, mkdirSync } from 'node:fs';
|
|
8
|
+
import { join } from 'node:path';
|
|
9
|
+
import { randomBytes } from 'node:crypto';
|
|
10
|
+
import { createDetachedWorktree, removeWorktree, isValidSourceRef } from '../worktree.mjs';
|
|
11
|
+
|
|
12
|
+
export const REPO_LOOK_DIRNAME = 'auto-look';
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* @param {string} projectDir the project's live checkout — read for HEAD, never used as cwd
|
|
16
|
+
* @param {string} baseDir where the throwaway checkout lives (the chat passes <worcaHome>/tmp/ask)
|
|
17
|
+
* @param {{signal?:AbortSignal|null, log?:(msg:string)=>void}} [o]
|
|
18
|
+
* @returns {Promise<{cwd:string, close:() => Promise<void>}|null>} null ⇒ no git repository, HEAD does
|
|
19
|
+
* not resolve, or git failed — the caller then classifies from the task text only, as before.
|
|
20
|
+
*/
|
|
21
|
+
export async function openRepoLook(projectDir, baseDir, { signal = null, log = () => {} } = {}) {
|
|
22
|
+
try {
|
|
23
|
+
if (!projectDir || !baseDir || !existsSync(join(projectDir, '.git'))) return null;
|
|
24
|
+
if (signal?.aborted) throw new Error('aborted before the checkout');
|
|
25
|
+
if (!(await isValidSourceRef(projectDir, 'HEAD'))) return null;
|
|
26
|
+
mkdirSync(baseDir, { recursive: true });
|
|
27
|
+
const worktreeDir = join(baseDir, `${REPO_LOOK_DIRNAME}-${randomBytes(4).toString('hex')}`);
|
|
28
|
+
// The signal is spread in ONLY when present: worktree.mjs's git() forwards it verbatim to
|
|
29
|
+
// child_process.spawn, and Node rejects `options.signal: null` (ERR_INVALID_ARG_TYPE) — git()
|
|
30
|
+
// would swallow that as ok:false and this whole look would silently degrade to text-only.
|
|
31
|
+
await createDetachedWorktree({ projectDir, worktreeDir, ref: 'HEAD', ...(signal ? { signal } : {}) });
|
|
32
|
+
let closed = false;
|
|
33
|
+
const close = async () => {
|
|
34
|
+
if (closed) return;
|
|
35
|
+
closed = true;
|
|
36
|
+
try {
|
|
37
|
+
const res = await removeWorktree({ projectDir, worktreeDir, branch: null, force: true });
|
|
38
|
+
for (const s of (res.steps || []).filter((x) => !x.ok)) log(`repo look: ${s.step} failed for ${worktreeDir}: ${s.stderr || 'unknown error'}`);
|
|
39
|
+
} catch (err) { log(`repo look: could not remove ${worktreeDir}: ${err && err.message ? err.message : err}`); }
|
|
40
|
+
};
|
|
41
|
+
return { cwd: worktreeDir, close };
|
|
42
|
+
} catch (err) {
|
|
43
|
+
log(`repo look unavailable (${err && err.message ? err.message : err}); classifying from the task text only`);
|
|
44
|
+
return null;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
@@ -39,7 +39,11 @@ import { effectiveDebugSpawn } from './settings.mjs';
|
|
|
39
39
|
import { classifyError, strongestClass } from './recoverable-error.mjs';
|
|
40
40
|
import { explainUnspawnableClaude, resolveClaudeBin } from './preflight.mjs';
|
|
41
41
|
import { hostGuardEnabled, hostGuardHookEntry, hostGuardSystemPrompt } from './host-guard.mjs';
|
|
42
|
-
|
|
42
|
+
// The offline classifier and the shape normalizer the mock ask role answers
|
|
43
|
+
// propose_workflow with — both pure (no DB, no spawn).
|
|
44
|
+
import { mockShapeFor } from './auto/recipes.mjs';
|
|
45
|
+
import { normalizeShape } from '../shared/graph/assemble.mjs';
|
|
46
|
+
import { writeFile, mkdir, appendFile, readFile, access, readdir } from 'node:fs/promises';
|
|
43
47
|
import { constants as FS, mkdtempSync, writeFileSync, rmSync } from 'node:fs';
|
|
44
48
|
import { dirname, join } from 'node:path';
|
|
45
49
|
import { tmpdir } from 'node:os';
|
|
@@ -342,7 +346,8 @@ export function mockEnabled(opts) {
|
|
|
342
346
|
* @param {number} [o.maxTurns] --max-turns <n> (positive safe integer; else omitted)
|
|
343
347
|
* @param {number|null} [o.maxBudgetUsd] --max-budget-usd <n> (finite > 0; null/else omitted)
|
|
344
348
|
* @param {string} [o.appendSubagentSystemPrompt] --append-subagent-system-prompt <text> (Task children only)
|
|
345
|
-
*
|
|
349
|
+
* @param {string[]} [o.addDirs] --add-dir <dir> per entry (Ask Worca's memory mount; the CLI loads <dir>/.claude/rules only with CLAUDE_CODE_ADDITIONAL_DIRECTORIES_CLAUDE_MD=1 in the env — memory-deps.mjs / spawn.mjs set it)
|
|
350
|
+
* All nine are Ask Worca sandbox options (ask-worca-design.md §6.3) and default-off.
|
|
346
351
|
* @param {number} [o.argvInlineLimit] override ARGV_INLINE_LIMIT (GH #380; tests force the staged path)
|
|
347
352
|
* @returns {Promise<{text:string, exitCode:number}>}
|
|
348
353
|
*/
|
|
@@ -380,6 +385,7 @@ export async function runClaude(o = {}) {
|
|
|
380
385
|
maxTurns,
|
|
381
386
|
maxBudgetUsd,
|
|
382
387
|
appendSubagentSystemPrompt,
|
|
388
|
+
addDirs,
|
|
383
389
|
argvInlineLimit,
|
|
384
390
|
bin = DEFAULT_BIN,
|
|
385
391
|
} = o;
|
|
@@ -424,6 +430,7 @@ export async function runClaude(o = {}) {
|
|
|
424
430
|
maxTurns,
|
|
425
431
|
maxBudgetUsd,
|
|
426
432
|
appendSubagentSystemPrompt,
|
|
433
|
+
addDirs,
|
|
427
434
|
argvInlineLimit,
|
|
428
435
|
});
|
|
429
436
|
}
|
|
@@ -446,8 +453,9 @@ export async function runClaude(o = {}) {
|
|
|
446
453
|
* (docs/run-root-verification.md, branch (a); argv-attested
|
|
447
454
|
* transcript phase0/out/v1a-rerun.jsonl, with a no-grant
|
|
448
455
|
* negative control proving the grant is load-bearing).
|
|
449
|
-
* `--add-dir`
|
|
450
|
-
*
|
|
456
|
+
* `--add-dir` carries Ask Worca's memory mount ONLY (`addDirs`): it needs the
|
|
457
|
+
* CLAUDE_CODE_ADDITIONAL_DIRECTORIES_CLAUDE_MD=1 env override to load memory at all
|
|
458
|
+
* (E2, re-probed 2026-09-13 on 2.1.270), and no pipeline path passes it (§5.3 / §8.18 unchanged). */
|
|
451
459
|
export function buildClaudeArgs({
|
|
452
460
|
prompt, systemPrompt, permissionMode, model, effort, allowedTools, resumeSessionId,
|
|
453
461
|
mcpConfigPath, mcpServerGrants, permissionRules,
|
|
@@ -455,7 +463,7 @@ export function buildClaudeArgs({
|
|
|
455
463
|
// way in because the legacy body below already owns a local `tools` (the
|
|
456
464
|
// --allowedTools union).
|
|
457
465
|
tools: builtinTools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages,
|
|
458
|
-
maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, hostGuard,
|
|
466
|
+
maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, hostGuard, addDirs,
|
|
459
467
|
}, delivery = {}) {
|
|
460
468
|
// delivery (GH #380, set only by planClaudeInvocation's staged branch):
|
|
461
469
|
// promptViaStdin -> bare `-p`; the prompt is written to the child's stdin
|
|
@@ -516,6 +524,9 @@ export function buildClaudeArgs({
|
|
|
516
524
|
if (typeof appendSubagentSystemPrompt === 'string' && appendSubagentSystemPrompt) {
|
|
517
525
|
args.push('--append-subagent-system-prompt', appendSubagentSystemPrompt);
|
|
518
526
|
}
|
|
527
|
+
// Native-rules revision (2026-09-13): Ask Worca's memory mount. LAST, so every earlier argv
|
|
528
|
+
// stays a prefix; absent / [] / non-strings ⇒ nothing (the `names` filter above).
|
|
529
|
+
for (const d of names(addDirs)) args.push('--add-dir', d);
|
|
519
530
|
return args;
|
|
520
531
|
}
|
|
521
532
|
|
|
@@ -568,7 +579,7 @@ export function stageClaudeInvocation(opts, { bin = DEFAULT_BIN, limit = ARGV_IN
|
|
|
568
579
|
return { ...plan, dir };
|
|
569
580
|
}
|
|
570
581
|
|
|
571
|
-
function runReal({ cwd, systemPrompt, prompt, allowedTools, permissionMode, model, effort, onEvent, signal, bin, resumeSessionId, mcpConfigPath, mcpServerGrants, permissionRules, envScrub, envAllowlist, modelEnv, tools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages, maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, argvInlineLimit }) {
|
|
582
|
+
function runReal({ cwd, systemPrompt, prompt, allowedTools, permissionMode, model, effort, onEvent, signal, bin, resumeSessionId, mcpConfigPath, mcpServerGrants, permissionRules, envScrub, envAllowlist, modelEnv, tools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages, maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, addDirs, argvInlineLimit }) {
|
|
572
583
|
return new Promise((resolveP, rejectP) => {
|
|
573
584
|
// Per-model routing env (design §4.4), prepared BEFORE argv: reserved keys
|
|
574
585
|
// are re-dropped here defensively — the write path already rejects them, so
|
|
@@ -647,7 +658,7 @@ function runReal({ cwd, systemPrompt, prompt, allowedTools, permissionMode, mode
|
|
|
647
658
|
permissionMode, model: wireModel, effort, allowedTools, resumeSessionId,
|
|
648
659
|
mcpConfigPath, mcpServerGrants, permissionRules,
|
|
649
660
|
tools, strictMcpConfig, settingSources, disableSlashCommands, includePartialMessages,
|
|
650
|
-
maxTurns, maxBudgetUsd, appendSubagentSystemPrompt,
|
|
661
|
+
maxTurns, maxBudgetUsd, appendSubagentSystemPrompt, addDirs,
|
|
651
662
|
}, { bin: resolved.bin, limit });
|
|
652
663
|
} catch (err) {
|
|
653
664
|
rejectP(new Error(`Failed to stage the claude prompt files: ${err.message}`));
|
|
@@ -965,7 +976,7 @@ async function emitLog(onEvent, text) {
|
|
|
965
976
|
*/
|
|
966
977
|
export const MOCK_WRITER_ROLES = new Set([
|
|
967
978
|
'clarify', 'planner-plan', 'refiner', 'decomposer', 'implementer', 'reviewer', 'plan-review',
|
|
968
|
-
'workspace-scan', 'agent-gen', 'workspace-reviewer', 'manual-tests-checklist', 'manual-web-ui-testing',
|
|
979
|
+
'workspace-scan', 'agent-gen', 'workspace-reviewer', 'manual-tests-checklist', 'manual-web-ui-testing', 'memory-defrag',
|
|
969
980
|
'generic-producer', 'generic-verifier',
|
|
970
981
|
]);
|
|
971
982
|
|
|
@@ -973,6 +984,10 @@ export const MOCK_WRITER_ROLES = new Set([
|
|
|
973
984
|
export const MOCK_ROLE_CLARIFY = 'clarify';
|
|
974
985
|
export const MOCK_ROLE_DECOMPOSER = 'decomposer';
|
|
975
986
|
|
|
987
|
+
/** The memory defragmenter's role, exported like the other two named roles (the switch below
|
|
988
|
+
* still uses the literal string: mock-writer-roles.test.mjs parses the switch arms). */
|
|
989
|
+
export const MOCK_ROLE_MEMORY_DEFRAG = 'memory-defrag';
|
|
990
|
+
|
|
976
991
|
/**
|
|
977
992
|
* The mock-fan-out roles (mirror the orchestrator's FANOUT_ELIGIBLE intent): the
|
|
978
993
|
* roles whose real runs may spawn sub-agents. Keyed by the MOCK_ROLE strings.
|
|
@@ -1065,15 +1080,26 @@ async function mockAsk({ markers, prompt, cwd, onEvent, signal, resumeSessionId
|
|
|
1065
1080
|
const maxTurns = /\bMOCK_MAX_TURNS\b/.test(userText);
|
|
1066
1081
|
const maxBudget = /\bMOCK_MAX_BUDGET\b/.test(userText);
|
|
1067
1082
|
const slow = /\bMOCK_SLOW\b/.test(userText);
|
|
1068
|
-
|
|
1069
|
-
|
|
1083
|
+
// P3 (PD11): a workflow-card EVENT is matched first — it contains the words "workflow" and, when thenRun, "run",
|
|
1084
|
+
// which would otherwise trip the two arms below. Then the workflow trigger, then the run proposal.
|
|
1085
|
+
const wfEvent = /^\s*\[worca event\] workflow card (card_[0-9a-f]{8}) (?:(declined)|saved as (\S+) "([^"]*)"; thenRun=(true|false))/.exec(userText);
|
|
1086
|
+
// A metrics-card EVENT, then the metrics trigger — both before the run arm, whose \brun\b would otherwise fire on
|
|
1087
|
+
// "include my runs"-style prose (it does not, \b stops at the s, but "propose" would).
|
|
1088
|
+
const tmEvent = /^\s*\[worca event\] metrics card (card_[0-9a-f]{8}) (applied|declined|failed)/.exec(userText);
|
|
1089
|
+
// The metrics arm wants a CHANGE, not a question: "metrics" plus a verb of intent ("stop recording my metrics",
|
|
1090
|
+
// "route ... to the metrics home"). A bare "which workspaces use team metrics?" gets the generic echo answer.
|
|
1091
|
+
const metrics = !wfEvent && !tmEvent && /\bmetrics\b/i.test(userText)
|
|
1092
|
+
&& /\b(?:stop|start|turn|toggle|switch|record\w*|route|change|enable|disable|set)\b/i.test(userText);
|
|
1093
|
+
const workflow = !wfEvent && !tmEvent && !metrics && /\bworkflow\b/i.test(userText);
|
|
1094
|
+
const agents = !wfEvent && !tmEvent && /\bagents?\b/i.test(userText);
|
|
1095
|
+
const propose = !wfEvent && !tmEvent && !workflow && !metrics && /\b(propose|start|run)\b/i.test(userText);
|
|
1070
1096
|
|
|
1071
1097
|
const SID = resumeSessionId || 'mock-session-ask-1';
|
|
1072
1098
|
const USAGE = { input_tokens: 10, output_tokens: 20, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 };
|
|
1073
1099
|
const firstLine = userText.split(/\r?\n/).map((l) => l.trim()).find(Boolean) || '';
|
|
1074
1100
|
const ANSWER = `[mock] ${firstLine.slice(0, 200)}`;
|
|
1075
1101
|
const init = { type: 'system', subtype: 'init', session_id: SID, cwd, model: 'mock', permissionMode: 'dontAsk',
|
|
1076
|
-
tools: ['Task', 'mcp__worca__list_runs', 'mcp__worca__get_run', 'mcp__worca__propose_run'],
|
|
1102
|
+
tools: ['Task', 'mcp__worca__list_runs', 'mcp__worca__get_run', 'mcp__worca__propose_run', 'mcp__worca__propose_workflow', 'mcp__worca__propose_metrics_change'],
|
|
1077
1103
|
mcp_servers: [{ name: 'worca', status: 'connected' }], plugins: [], skills: [], slash_commands: [], agents: [], uuid: 'mock-uuid-init' };
|
|
1078
1104
|
const mstart = (id) => ({ type: 'stream_event', event: { type: 'message_start', message: { id, model: 'mock', role: 'assistant', content: [], usage: USAGE } }, parent_tool_use_id: null, session_id: SID });
|
|
1079
1105
|
const delta = (t) => ({ type: 'stream_event', event: { type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text: t } }, parent_tool_use_id: null, session_id: SID });
|
|
@@ -1110,6 +1136,43 @@ async function mockAsk({ markers, prompt, cwd, onEvent, signal, resumeSessionId
|
|
|
1110
1136
|
);
|
|
1111
1137
|
answerMsg = MSG2;
|
|
1112
1138
|
}
|
|
1139
|
+
if (workflow) {
|
|
1140
|
+
// The result the real MCP child would return (plan PD1) — the parent re-assembles it with the real registry.
|
|
1141
|
+
const shape = normalizeShape(mockShapeFor(userText, { humanInLoop: true }));
|
|
1142
|
+
const wfInput = { task: userText.slice(0, 2000), projectKey: card.projectKey || null, thenRun: /\brun\b/i.test(userText) };
|
|
1143
|
+
frames.push(delta('[mock] '), delta('building '), delta('a workflow'), atext(MSG1, 'Building a workflow card.'),
|
|
1144
|
+
atool(MSG1, 'toolu_mock_workflow', 'mcp__worca__propose_workflow', wfInput),
|
|
1145
|
+
uresult('toolu_mock_workflow', JSON.stringify({ ok: true, mode: 'task', projectKey: card.projectKey || null, projectName: null, name: shape.name, match: null,
|
|
1146
|
+
warnings: [], summary: '', shape, costUsd: 0, fingerprint: 'top-level: (mock)\nhints: mock', note: '', thenRun: wfInput.thenRun })));
|
|
1147
|
+
answerMsg = MSG2;
|
|
1148
|
+
}
|
|
1149
|
+
if (metrics) {
|
|
1150
|
+
// The MCP child's validation result (metrics-proposal.mjs): the parent re-validates the INPUT and mints the card,
|
|
1151
|
+
// so a mock card always targets the context project's "Include my runs" switch (no git involved when applied).
|
|
1152
|
+
const tmInput = { kind: 'record', projectKey: card.projectKey || null, record: false, note: 'mock: stop recording my runs here' };
|
|
1153
|
+
frames.push(delta('[mock] '), delta('proposing '), delta('a metrics change'), atext(MSG1, 'Proposing a metrics change card.'),
|
|
1154
|
+
atool(MSG1, 'toolu_mock_metrics', 'mcp__worca__propose_metrics_change', tmInput),
|
|
1155
|
+
uresult('toolu_mock_metrics', JSON.stringify({ ok: true, card: { type: 'metrics', ...tmInput } })));
|
|
1156
|
+
answerMsg = MSG2;
|
|
1157
|
+
}
|
|
1158
|
+
if (tmEvent) {
|
|
1159
|
+
const line = tmEvent[2] === 'declined' ? 'Declined — nothing changed.' : tmEvent[2] === 'failed' ? 'The change failed; check the error and try again.' : 'Applied.';
|
|
1160
|
+
frames.push(delta('[mock] '), delta(tmEvent[2]), atext(MSG1, line));
|
|
1161
|
+
answerMsg = MSG2;
|
|
1162
|
+
}
|
|
1163
|
+
if (wfEvent) {
|
|
1164
|
+
// Every event arm answers on MSG2: a second atext on MSG1 would REPLACE the first reply's text.
|
|
1165
|
+
if (wfEvent[2] === 'declined') {
|
|
1166
|
+
frames.push(delta('[mock] '), delta('declined'), atext(MSG1, 'Declined. Want another auto workflow, tell me what to change, or pick a saved workflow?'));
|
|
1167
|
+
} else if (wfEvent[5] === 'true') {
|
|
1168
|
+
frames.push(delta('[mock] '), delta('proposing '), delta('a run'), atext(MSG1, `Proposing a run with "${wfEvent[4]}".`),
|
|
1169
|
+
atool(MSG1, 'toolu_mock_propose', 'mcp__worca__propose_run', { ...card, workflowId: wfEvent[3], brief: `Run with "${wfEvent[4]}"` }),
|
|
1170
|
+
uresult('toolu_mock_propose', JSON.stringify({ ok: true })));
|
|
1171
|
+
} else {
|
|
1172
|
+
frames.push(delta('[mock] '), delta('saved'), atext(MSG1, `Saved "${wfEvent[4]}". Say "run it" when you want a run with it.`));
|
|
1173
|
+
}
|
|
1174
|
+
answerMsg = MSG2;
|
|
1175
|
+
}
|
|
1113
1176
|
if (propose) {
|
|
1114
1177
|
frames.push(delta('[mock] '), delta('preparing '), delta('a run'), atext(MSG1, 'Preparing a run card.'),
|
|
1115
1178
|
atool(MSG1, 'toolu_mock_propose', 'mcp__worca__propose_run', card), uresult('toolu_mock_propose', JSON.stringify({ ok: true })));
|
|
@@ -1230,6 +1293,9 @@ async function runMock({ cwd, systemPrompt, prompt, onEvent, signal, resumeSessi
|
|
|
1230
1293
|
case 'manual-web-ui-testing':
|
|
1231
1294
|
text = await mockManualWebUiTesting(m, cycle, onEvent);
|
|
1232
1295
|
break;
|
|
1296
|
+
case 'memory-defrag':
|
|
1297
|
+
text = await mockMemoryDefrag(m, systemPrompt, onEvent);
|
|
1298
|
+
break;
|
|
1233
1299
|
case 'generic-producer':
|
|
1234
1300
|
text = await mockGenericProducer(m, onEvent);
|
|
1235
1301
|
break;
|
|
@@ -1331,6 +1397,61 @@ async function mockGenericProducer(m, onEvent) {
|
|
|
1331
1397
|
return `[mock] generic artifact written to ${out}`;
|
|
1332
1398
|
}
|
|
1333
1399
|
|
|
1400
|
+
/** The scope dirs the `## Worca memory` block of a system prompt names (memory-store.mjs
|
|
1401
|
+
* renderMemoryBlock: `<label> — <abs dir>:` lines under the heading; the block is contiguous,
|
|
1402
|
+
* so the first blank line ends it). The mock defragmenter finds its mount exactly the way the
|
|
1403
|
+
* real agent is told to — from its system prompt — so no MOCK marker is needed. Exported for
|
|
1404
|
+
* the parity test against a real renderMemoryBlock output. Both captures are greedy: a project
|
|
1405
|
+
* LABEL may itself contain ` — ` (the renderer's separator is the last one on the line), and a
|
|
1406
|
+
* Windows dir carries a drive colon while the line still ends with `:`. */
|
|
1407
|
+
export function memoryDirsFromPrompt(systemPrompt) {
|
|
1408
|
+
const text = String(systemPrompt || '');
|
|
1409
|
+
const at = text.indexOf('## Worca memory');
|
|
1410
|
+
if (at === -1) return [];
|
|
1411
|
+
const out = [];
|
|
1412
|
+
for (const line of text.slice(at).split(/\r?\n/).slice(1)) {
|
|
1413
|
+
if (!line.trim()) break;
|
|
1414
|
+
const m = line.match(/^(?:Global|Project .*) — (.+):$/);
|
|
1415
|
+
if (m) out.push(m[1]);
|
|
1416
|
+
}
|
|
1417
|
+
return out;
|
|
1418
|
+
}
|
|
1419
|
+
|
|
1420
|
+
/** Memory defragment mock (agent-memory-design.md §7.1): in the FIRST scope dir the system
|
|
1421
|
+
* prompt names, fold the second (sorted) file's body into the first and EMPTY it (amendment
|
|
1422
|
+
* B19 — the memory tool set cannot unlink, so this is the path the real agent takes), then
|
|
1423
|
+
* write the report to MOCK_OUT. A defrag run mounts exactly ONE scope dir. */
|
|
1424
|
+
async function mockMemoryDefrag(m, systemPrompt, onEvent) {
|
|
1425
|
+
const out = m.MOCK_OUT;
|
|
1426
|
+
const dir = memoryDirsFromPrompt(systemPrompt)[0] || null;
|
|
1427
|
+
await emitLog(onEvent, '[mock] memory defragmenter restructuring the mounted scope');
|
|
1428
|
+
const lines = ['# Memory defragment report', ''];
|
|
1429
|
+
let merged = null;
|
|
1430
|
+
if (dir) {
|
|
1431
|
+
const files = (await readdir(dir, { withFileTypes: true })).filter((e) => e.isFile() && e.name.endsWith('.md')).map((e) => e.name).sort();
|
|
1432
|
+
if (files.length < 2) {
|
|
1433
|
+
lines.push(`- ${dir}: ${files.length} file(s), nothing to merge`);
|
|
1434
|
+
} else {
|
|
1435
|
+
const [a, b] = files;
|
|
1436
|
+
const bodyB = (await readFile(join(dir, b), 'utf8')).replace(/^---\n[\s\S]*?\n---\n/, '');
|
|
1437
|
+
const text = `${await readFile(join(dir, a), 'utf8')}\n## Merged from ${b.slice(0, -3)}\n\n${bodyB}`;
|
|
1438
|
+
await writeFile(join(dir, a), text, 'utf8');
|
|
1439
|
+
await writeFile(join(dir, b), '', 'utf8'); // B19: an EMPTIED mount file is a deletion request
|
|
1440
|
+
merged = [a, b];
|
|
1441
|
+
lines.push(`- ${dir}: merged ${b} into ${a}; emptied ${b} (worca removes it at sync-back)`);
|
|
1442
|
+
safeEmit(onEvent, { type: 'tool_use', text: `merged ${join(dir, b)} into ${join(dir, a)}`, raw: { mock: true, file: join(dir, a) } });
|
|
1443
|
+
}
|
|
1444
|
+
} else {
|
|
1445
|
+
lines.push('- no memory scope in the system prompt: nothing to defragment');
|
|
1446
|
+
}
|
|
1447
|
+
if (out) {
|
|
1448
|
+
await ensureDir(out);
|
|
1449
|
+
await writeFile(out, `${lines.join('\n')}\n`, 'utf8');
|
|
1450
|
+
safeEmit(onEvent, { type: 'tool_use', text: `wrote ${out}`, raw: { mock: true, file: out } });
|
|
1451
|
+
}
|
|
1452
|
+
return merged ? `[mock] memory defragment: merged ${merged[1]} into ${merged[0]}` : '[mock] memory defragment: nothing to merge';
|
|
1453
|
+
}
|
|
1454
|
+
|
|
1334
1455
|
async function mockPlannerPlan(m, onEvent) {
|
|
1335
1456
|
const out = m.MOCK_OUT;
|
|
1336
1457
|
const base = m.MOCK_BASE || 'feature';
|