@worca/app 1.2.0-rc.3 → 1.3.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/README.md +42 -0
  2. package/agents/memoryDefragmenter.meta.json +24 -0
  3. package/agents/worca-cc-code-reviewer.md +6 -1
  4. package/agents/worca-cc-implementer.md +6 -1
  5. package/agents/worca-cc-memory-defragmenter.md +32 -0
  6. package/agents/worca-cc-planner.md +5 -1
  7. package/package.json +5 -2
  8. package/src/cli/render.mjs +36 -0
  9. package/src/cli/worca-cc.mjs +137 -8
  10. package/src/core/agent-registry.mjs +12 -34
  11. package/src/core/artifacts.mjs +132 -8
  12. package/src/core/ask/catalog.mjs +32 -7
  13. package/src/core/ask/comment-deps.mjs +5 -2
  14. package/src/core/ask/events.mjs +65 -2
  15. package/src/core/ask/limits.mjs +9 -0
  16. package/src/core/ask/mcp-stdio.mjs +10 -0
  17. package/src/core/ask/memory-deps.mjs +107 -0
  18. package/src/core/ask/metrics-deps.mjs +124 -0
  19. package/src/core/ask/metrics-proposal.mjs +175 -0
  20. package/src/core/ask/prompt.mjs +53 -10
  21. package/src/core/ask/proposal.mjs +49 -2
  22. package/src/core/ask/spawn.mjs +21 -4
  23. package/src/core/ask/store.mjs +14 -5
  24. package/src/core/ask/tool-deps.mjs +26 -2
  25. package/src/core/ask/tools.mjs +439 -6
  26. package/src/core/ask/turn.mjs +163 -4
  27. package/src/core/ask/workflow-deps.mjs +226 -0
  28. package/src/core/auto/classify.mjs +352 -0
  29. package/src/core/auto/fingerprint.mjs +141 -0
  30. package/src/core/auto/match.mjs +30 -0
  31. package/src/core/auto/model.mjs +23 -0
  32. package/src/core/auto/proposal.mjs +132 -0
  33. package/src/core/auto/recipes.mjs +75 -0
  34. package/src/core/auto/repo-look.mjs +46 -0
  35. package/src/core/claude-runner.mjs +132 -11
  36. package/src/core/config.mjs +120 -3
  37. package/src/core/db.mjs +44 -1
  38. package/src/core/diff-comments.mjs +55 -9
  39. package/src/core/frontmatter.mjs +75 -0
  40. package/src/core/git-info.mjs +233 -26
  41. package/src/core/graph/builtin-workflows.mjs +50 -0
  42. package/src/core/graph/executor.mjs +11 -3
  43. package/src/core/index-html.mjs +17 -0
  44. package/src/core/memory-store.mjs +441 -0
  45. package/src/core/memory-sync.mjs +300 -0
  46. package/src/core/metrics/ledger.mjs +47 -0
  47. package/src/core/metrics/lock.mjs +117 -0
  48. package/src/core/metrics/read.mjs +303 -0
  49. package/src/core/metrics/record.mjs +389 -0
  50. package/src/core/metrics/sync.mjs +1100 -0
  51. package/src/core/onboarding.mjs +99 -0
  52. package/src/core/orchestrator.mjs +394 -7
  53. package/src/core/phases.mjs +16 -3
  54. package/src/core/pipeline-delete.mjs +1 -1
  55. package/src/core/plugin-store.mjs +2 -10
  56. package/src/core/preflight.mjs +2 -3
  57. package/src/core/projects.mjs +16 -1
  58. package/src/core/run-harness.mjs +458 -32
  59. package/src/core/run-report.mjs +896 -0
  60. package/src/core/settings.mjs +162 -0
  61. package/src/core/sources.mjs +4 -1
  62. package/src/core/store.mjs +5 -0
  63. package/src/core/workflow-export.mjs +2 -0
  64. package/src/core/workflow-share.mjs +1 -0
  65. package/src/core/workflows.mjs +43 -23
  66. package/src/core/workspaces.mjs +37 -8
  67. package/src/shared/graph/agent-meta.mjs +5 -2
  68. package/src/shared/graph/assemble.mjs +455 -0
  69. package/src/shared/graph/flow-layout.mjs +249 -0
  70. package/src/shared/graph/geometry.mjs +48 -28
  71. package/src/shared/graph/isomorphic.mjs +101 -0
  72. package/src/shared/report-reasons.mjs +58 -0
  73. package/src/shared/team-metrics/aggregate.mjs +341 -0
  74. package/src/shared/team-metrics/workspace-match.mjs +13 -0
  75. package/ui/public/about-links.mjs +21 -0
  76. package/ui/public/app.js +3736 -479
  77. package/ui/public/artifact-view.mjs +135 -0
  78. package/ui/public/ask-model.mjs +18 -1
  79. package/ui/public/ask-panel.mjs +1589 -212
  80. package/ui/public/ask-run-card.mjs +209 -0
  81. package/ui/public/assets/worca-logo-mask.png +0 -0
  82. package/ui/public/assets/worca-mark-mask.png +0 -0
  83. package/ui/public/auto-build.mjs +95 -0
  84. package/ui/public/auto-proposal.mjs +174 -0
  85. package/ui/public/comment-thread.mjs +55 -0
  86. package/ui/public/getting-started.mjs +261 -0
  87. package/ui/public/graph/composer.mjs +41 -5
  88. package/ui/public/graph/inspector.mjs +3 -1
  89. package/ui/public/graph/model.mjs +1 -0
  90. package/ui/public/graph/run-hosts.mjs +73 -12
  91. package/ui/public/graph/view.mjs +218 -50
  92. package/ui/public/guide-spot.mjs +215 -0
  93. package/ui/public/index.html +447 -29
  94. package/ui/public/memory-view.mjs +192 -0
  95. package/ui/public/node-tunables.mjs +201 -0
  96. package/ui/public/report-run.mjs +75 -0
  97. package/ui/public/results-view.mjs +25 -0
  98. package/ui/public/source-pane.mjs +16 -2
  99. package/ui/public/stats-view.mjs +2 -2
  100. package/ui/public/style.css +1499 -303
  101. package/ui/public/team-metrics-surfaces.mjs +452 -0
  102. package/ui/public/team-metrics-view.mjs +533 -0
  103. package/ui/public/thinking-orb.mjs +46 -8
  104. package/ui/server.mjs +1304 -194
package/README.md CHANGED
@@ -19,6 +19,19 @@ Claude Code — all running the same engine. See the
19
19
 
20
20
  ## How a run works
21
21
 
22
+ Pick **Auto** (`--workflow auto` on the CLI; the web picker's Auto entry ships with
23
+ the UI update) and worca picks the workflow for you: it classifies the task (a
24
+ free-form prompt, a partial plan, or a complete plan; web feature or not; trivial or
25
+ large) from the task text, the attached files and a small offline fingerprint of the
26
+ repository, knowing every agent by its metadata and the front matter of its agent
27
+ file (never the agent's full instructions), assembles a matching workflow from those
28
+ agents, reuses a saved workflow when one has exactly that shape — otherwise the
29
+ proposal is saved as a new workflow when you accept it — and, with *Human in the
30
+ loop* on, shows you the proposal first so you can accept it, ask for changes in plain
31
+ text, or cancel. Turn the switch off and the run decides on its own, asks no
32
+ questions, and never stops for you. Pick any saved workflow instead to skip all of
33
+ this.
34
+
22
35
  1. **Clarify** — instead of assuming, the planner turns hidden decisions into
23
36
  multiple-choice questions (2–4 options plus free text). Your answers are
24
37
  appended to the plan so reviewers see them.
@@ -145,6 +158,18 @@ and durations, the clarify Q&A, agent transcripts, and logs:
145
158
 
146
159
  ![Statistics — spend, time worked, outcomes, and per-day charts](docs/screenshots/stats.png)
147
160
 
161
+ ### Team metrics
162
+
163
+ - **A shared, git-backed record** — every finished run is pushed as one file to an orphan
164
+ `worca-metrics` branch on the project's own `origin`; there is no separate metrics server.
165
+ - **Opt-in per project, with delegation** — enable it on the repository itself, or delegate to
166
+ another project (or a workspace's metrics home) that already records.
167
+ - **A Team metrics page** — project and workspace scope, spend/runs/duration/autonomy/review
168
+ KPIs, breakdowns and a CSV export.
169
+ - **`worca metrics push`** — flush pending run records from the CLI, e.g. on a headless machine.
170
+
171
+ See [`docs/team-metrics.md`](docs/team-metrics.md).
172
+
148
173
  ### Models
149
174
 
150
175
  - **Bring your own models** — register any model id (a proxy, a fine-tune, an
@@ -215,6 +240,12 @@ worca ui --port 4318 --open # another port; open the browser when up
215
240
  `status` remember the port of the last started UI, so they usually need no
216
241
  flag. See `worca ui help`.
217
242
 
243
+ **Appearance.** Settings › General › Appearance picks **System** (follow the
244
+ operating system), **Light** or **Dark** — one setting for every browser that
245
+ opens this Worca. The web UI relies on CSS `light-dark()` (and `::backdrop`
246
+ inheriting the dialog's scheme), so it needs Chrome/Edge 123, Firefox 120 or
247
+ Safari 17.5 (or newer).
248
+
218
249
  ### CLI
219
250
 
220
251
  ```bash
@@ -224,12 +255,21 @@ worca --project /path/to/your/project --prompt "Add a /search endpoint"
224
255
  # use a markdown brief as the prompt
225
256
  worca --project /path/to/your/project --file ./brief.md --title "Search feature"
226
257
 
258
+ # let worca pick the workflow for the task (Auto), review the proposal first
259
+ worca --project /path/to/your/project --prompt "Add a /search endpoint" --workflow auto
260
+
261
+ # Auto run with no proposal and no questions (loop-budget, recovery, cost and error pauses still apply; add --yes when nothing can answer them, e.g. in CI)
262
+ worca --project /path/to/your/project --prompt "Add a /search endpoint" --workflow auto --no-human
263
+
227
264
  # pause with Ctrl+C, continue later (survives restarts)
228
265
  worca resume <pipelineId>
229
266
 
230
267
  # offline demo — full pipeline, no tokens
231
268
  worca --project /path/to/your/project --prompt "demo task" --mock --yes
232
269
 
270
+ # flush pending team-metrics run records (headless machines with no UI server)
271
+ worca metrics push
272
+
233
273
  # share a saved pipeline: as JSON, or as a plugin folder bundling your agents + skills
234
274
  worca workflow export wf_my-flow --format json --out my-flow.json
235
275
  worca workflow import my-flow.json
@@ -265,6 +305,8 @@ The skill starts the same deterministic orchestrator.
265
305
 
266
306
  - [Architecture](docs/ARCHITECTURE.md) — the whole stack in one picture
267
307
  - [Guardrails](docs/guardrails.md) — policy model, enforcement, limitations
308
+ - [Team metrics](docs/team-metrics.md) — git-backed, team-wide run records
309
+ - [Getting started](docs/getting-started.md) — the in-app checklist, welcome and spotlight guides
268
310
  - [Storage](docs/storage.md) — where state lives, project keys, migration
269
311
  - [Releasing](docs/RELEASING.md) — how `@worca/app` versions are published
270
312
  - [Contributing](CONTRIBUTING.md) — developing Worca from source
@@ -0,0 +1,24 @@
1
+ {
2
+ "key": "memoryDefragmenter",
3
+ "metaVersion": 2,
4
+ "mockRole": "memory-defrag",
5
+ "sideEffect": "memory",
6
+ "inputs": [
7
+ { "id": "task", "type": "md", "required": true,
8
+ "directive": "Defragment the memory scope named below. Work only inside the mounted memory directory given in your system prompt (the `## Worca memory` block) \u2014 it sits under the project directory at .claude/rules/worca/; never touch anything else in the project directory." }
9
+ ],
10
+ "outputs": [
11
+ { "id": "report", "type": "md", "filename": "defrag-report.md" },
12
+ { "id": "done", "type": "void" }
13
+ ],
14
+ "domain": "shared",
15
+ "displayName": "Memory defragment",
16
+ "description": "Restructures one memory scope: merges duplicates, splits overgrown topics, drops stale rules, tightens hooks.",
17
+ "color": "violet",
18
+ "icon": "<path d=\"M4 6h16M4 12h16M4 18h10\" stroke-linecap=\"round\"/><path d=\"M17 15l2 2 4-4\" stroke-linecap=\"round\" stroke-linejoin=\"round\"/>",
19
+ "agentFile": "worca-cc-memory-defragmenter.md",
20
+ "runnerType": "producer",
21
+ "fanOut": false,
22
+ "asksQuestions": false,
23
+ "order": 8
24
+ }
@@ -61,9 +61,14 @@ After writing both files, emit a short assistant note with the absolute paths of
61
61
 
62
62
  ## Output contract reminders
63
63
  - The review JSON must be valid and match the shape above (`severity` from {critical, major, minor, suggestion}); it is parsed by `safeParseJson` / `readReview`.
64
- - Base findings on the real `git diff`, not assumptions. Write only to the two absolute paths given.
64
+ - Base findings on the real `git diff`, not assumptions. The review artifacts go only to the two absolute paths given — the one other place you may write is the memory directory your system prompt's `## Worca memory` block names (see the section below).
65
65
  - Keep prose in the assistant message minimal; the markdown + JSON are your real output.
66
66
 
67
+ ## Worca memory
68
+ After the verdict is written: if you flagged a defect class that will recur in this area — a class, never this diff's individual bugs — record the rule that prevents it in the memory directory your system prompt's `## Worca memory` block names, following the WRITE TRIGGER there.
69
+ It qualifies only if it cost a cycle (or would cost the next reviewer one) or contradicted what the implementer assumed, AND will still be true next month.
70
+ At most 1–2 files per run, and prefer editing an existing file over adding one. Memory is never a substitute for the review itself — every finding of this cycle belongs in the review markdown and the review JSON.
71
+
67
72
  ## Workspace runs
68
73
  You are NOT used for workspace runs: a workspace pipeline substitutes the **Workspace Reviewer** (`workspaceReviewer`), which fans out one reviewer per changed member and synthesizes one merged verdict. If you ever see a `## Workspace Context` block in your task, review only your single cwd's diff as usual.
69
74
 
@@ -66,11 +66,16 @@ If (and only if) you had to deviate, append a brief, factual note so it survives
66
66
  ## Quality bar
67
67
  - No TODOs, stubs, placeholders, or commented-out dead code in what you ship.
68
68
  - Match the project's existing style and structure exactly.
69
- - Only the files the plan (implement) or the review (fix) require should change.
69
+ - Only the files the plan (implement) or the review (fix) require should change (the memory directory your system prompt names is not part of the change set).
70
70
  - All tests green before you finish.
71
71
 
72
72
  After finishing, emit a concise assistant note summarizing: mode, which plan steps or review issues you handled, the tests you added/ran and their result, and any deviations (or "No deviations"). This summary is returned to the orchestrator.
73
73
 
74
+ ## Worca memory
75
+ Once the suite is green: if this run hit a trap, an unstated invariant or a verification recipe that cost you a cycle — or that contradicted what you assumed when you started — and it will still be true next month, record it in the memory directory your system prompt's `## Worca memory` block names, following the WRITE TRIGGER there.
76
+ In **fix** mode the review that bounced you is the highest-signal source: the rule that would have prevented the finding, never the finding itself.
77
+ At most 1–2 files per run, and prefer editing an existing file over adding one. Memory is never a substitute for this run's own outputs — DEVIATIONS.md, the tests you wrote and your final note still carry everything about THIS run.
78
+
74
79
  ## Workspace runs
75
80
  When the task prompt carries a `## Workspace Context` block, your task names ONE plan task plus the project(s) it touches (its `Projects:` tag) and a `## Workspace projects` block gives each member's worktree directory. Edit ONLY the named project(s), inside their named worktree path(s) (cwd into the worktree) — touch no other member repo — and apply the same strict TDD as a single-project run.
76
81
 
@@ -0,0 +1,32 @@
1
+ ---
2
+ name: worca-cc-memory-defragmenter
3
+ description: Memory defragmenter for worca. Restructures ONE scope of worca's durable memory (global or one project) inside the run's memory mount — merges duplicate topics, splits overgrown files, removes stale or contradicted rules, tightens hooks — and writes a short report. Never touches the project directory. Invoked by the Memory defragment workflow, never directly by a human.
4
+ tools: Read, Write, Edit, Glob, Grep
5
+ model: inherit
6
+ ---
7
+
8
+ You are the **Memory defragment** agent. worca keeps durable rules and preferences as markdown files — one topic per file, YAML frontmatter (`name`, `description`, optional `paths`; worca stamps `source` and `updated`). Your system prompt carries a `## Worca memory` block naming the scope directory you may work in (its absolute path) — a Memory defragment run mounts exactly one; that directory is the only place you read from or write to. Claude Code has already loaded every file of it into your context as rules, but editing needs a fresh Read of the file. The scope directory sits INSIDE your cwd (under `.claude/rules/worca/`); everything else in the project directory is off limits: do not read it, do not edit it, do not create anything there.
9
+
10
+ ## Ports
11
+
12
+ The engine binds every port to an absolute path in the task prompt — never hardcode filenames.
13
+
14
+ - **in `task`** (md) — which scope to defragment and why (the user's request).
15
+ - **out `report`** (md) — `defrag-report.md`: what you merged, split, removed and why. Write it to the exact path the task prompt gives.
16
+
17
+ ## What to do
18
+
19
+ 1. Read EVERY `*.md` file of the scope directory (Glob it; the bodies are in your context already, but Edit needs the file read in this session).
20
+ 2. Decide the target structure. Rules:
21
+ - **One topic per file.** Merge files that cover the same topic into one; keep the older, better-named file and fold the other's body in, then remove the file you folded in (step 3).
22
+ - **Split** a file that has grown past one topic (or past ~8 KB) into focused files with short kebab-case names (`[A-Za-z0-9._-]`, no leading dot, no `.md` in the name field).
23
+ - **Remove** a rule that is stale, superseded or contradicted; when two rules conflict, the one with the newer `updated` stamp wins.
24
+ - **Tighten hooks.** Every `description` is ONE line ≤ 160 characters that says WHEN the file is worth reading, not what it contains.
25
+ - **Keep names** whenever the topic survives (other files and people refer to them). Never rename to a name that differs only by letter case from an existing one.
26
+ - **Keep the frontmatter** (`name`, `description`, `paths` when present, any extra keys). Do not write `source` or `updated` — worca stamps them.
27
+ - Stay within the caps: at most 50 files in the scope, each under 32 KB (aim for under 8 KB).
28
+ - Preserve facts. Defragmenting is restructuring, not rewriting: keep every rule that is still true, in fewer and clearer files.
29
+ 3. Apply the changes with Write / Edit. To REMOVE a file (merged away, stale), make it EMPTY: Edit it, replacing its entire content — frontmatter included — with nothing. worca's sync-back treats an empty file in the mount as a deletion (an absent file too). Never leave a file that only carries frontmatter.
30
+ 4. Write `defrag-report.md` to the `report` output path: a heading, then three lists — **Merged** (`b.md → a.md: why`), **Split** (`x.md → x-1.md, x-2.md: why`), **Removed** (`stale.md: why`) — and one closing line with the file count before and after. Keep it under 60 lines.
31
+
32
+ Never write progress notes, run summaries or anything about this run into the memory scope. Never read or write outside the scope directory except the report path.
@@ -69,9 +69,13 @@ After writing the file, emit a short assistant note confirming the absolute plan
69
69
  This is a variant of PLAN mode. When the task prompt names a plan-review path — a `## Revise to address the review` block carrying a `Review to address: <path>` line — a reviewer found blocking issues with the previous plan. Read the prior plan AND that review, then write a fresh plan version (to the same given output path) that addresses EVERY critical and major finding. Treat it as a cold re-plan from scratch, not an in-place patch of the old plan, and preserve the `## Clarifications (Q&A)` section. All PLAN requirements still apply.
70
70
 
71
71
  ## Output contract reminders
72
- - Write files with absolute paths taken from the prompt. Never write outside the pipeline dir / the given plan path.
72
+ - Write files with absolute paths taken from the prompt. Never write outside the pipeline dir / the given plan path — the one exception is the memory directory your system prompt's `## Worca memory` block names (see the section below).
73
73
  - Keep assistant chatter minimal; your real output is the file you write.
74
74
 
75
+ ## Worca memory
76
+ After the plan is written: if exploring the codebase surfaced a constraint the plan had to design around — one that cost you a cycle, or that contradicted what you assumed when you started, and will still be true next month — record it in the memory directory your system prompt's `## Worca memory` block names, following the WRITE TRIGGER there.
77
+ At most 1–2 files per run, and prefer editing an existing file over adding one. Memory is never a substitute for the plan — every decision this run made belongs in the plan itself, including its "## Clarifications (Q&A)" section.
78
+
75
79
  ## Workspace runs
76
80
  When the task prompt carries a `## Workspace Context` block, you are planning across a SET of member projects. Treat that block as a point-in-time, frozen interconnection description (it does not change mid-run). Fan out one read-only investigator per member project to survey it, then write ONE unified plan whose every task is tagged `Projects: <projectKey>[, ...]` naming the project(s) it touches; honor the description's change-coordination notes and suggested change order.
77
81
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@worca/app",
3
- "version": "1.2.0-rc.3",
3
+ "version": "1.3.0-rc.2",
4
4
  "description": "Worca — deterministic multi-agent pipeline that drives Claude Code (headless) through Plan -> Refine -> Implement -> Review, with a CLI, an installable /worca skill, and a web UI.",
5
5
  "license": "MIT",
6
6
  "author": "Sinisha Djukic",
@@ -44,12 +44,15 @@
44
44
  "cli": "node --disable-warning=ExperimentalWarning src/cli/worca-cc.mjs",
45
45
  "install:agents": "node scripts/install.mjs",
46
46
  "smoke": "WORCA_MOCK=1 WORCA_HOME=.worca-cc-smoke node --disable-warning=ExperimentalWarning src/cli/worca-cc.mjs --project sandbox --prompt \"demo task\" --mock --yes",
47
+ "smoke:auto": "WORCA_MOCK=1 WORCA_HOME=.worca-cc-smoke node --disable-warning=ExperimentalWarning src/cli/worca-cc.mjs --project sandbox --prompt \"demo task\" --mock --yes --workflow auto",
47
48
  "smoke:workspace": "WORCA_MOCK=1 WORCA_HOME=.worca-cc-smoke node --disable-warning=ExperimentalWarning scripts/smoke-workspace.mjs",
48
49
  "smoke:plugin": "WORCA_MOCK=1 WORCA_HOME=.worca-cc-smoke node --disable-warning=ExperimentalWarning scripts/smoke-plugin.mjs",
49
50
  "ask:fixtures": "node --disable-warning=ExperimentalWarning scripts/ask-capture-fixtures.mjs",
50
51
  "verify:composer": "node --disable-warning=ExperimentalWarning scripts/verify-composer-cdp.mjs",
51
52
  "verify:run-monitor": "node --disable-warning=ExperimentalWarning scripts/verify-run-monitor-cdp.mjs",
52
- "test": "rm -rf .worca-cc-test && PATH=\"$PWD/test/helpers/no-real-claude:$PATH\" WORCA_HOME=.worca-cc-test node --disable-warning=ExperimentalWarning --test test/*.mjs; s=$?; node test/helpers/no-real-claude/assert-none.mjs || s=1; exit $s"
53
+ "verify:theme": "node --disable-warning=ExperimentalWarning scripts/verify-theme-cdp.mjs",
54
+ "verify:memory": "node --disable-warning=ExperimentalWarning scripts/verify-memory-cdp.mjs",
55
+ "test": "rm -rf .worca-cc-test && PATH=\"$PWD/test/helpers/no-real-claude:$PATH\" WORCA_HOME=.worca-cc-test node --disable-warning=ExperimentalWarning --test --test-timeout=300000 test/*.mjs; s=$?; node test/helpers/no-real-claude/assert-none.mjs || s=1; exit $s"
53
56
  },
54
57
  "dependencies": {
55
58
  "@highlightjs/cdn-assets": "11.12.0",
@@ -146,3 +146,39 @@ export function formatRunSummary(state) {
146
146
  }));
147
147
  return lines;
148
148
  }
149
+
150
+ /**
151
+ * The interactive Auto proposal (spec §9): what the question panel shows in the
152
+ * browser, as text. Pure — the payload is `question.workflow` (auto/proposal.mjs).
153
+ * Stages follow the proposal's DISPATCH `order` (a reused composer row's node order
154
+ * is arbitrary); every model/plugin-authored string in the payload was cleaned by
155
+ * the assembler (single line, no control characters), so it is safe to print.
156
+ * @param {object} w the proposal
157
+ * @returns {string[]} lines
158
+ */
159
+ export function formatWorkflowProposal(w) {
160
+ const p = w && typeof w === 'object' ? w : {};
161
+ const where = p.match
162
+ ? `(same shape as your saved workflow "${p.match.name}" — Accept reuses it)`
163
+ : `(no saved workflow has this shape — Accept saves it as "${p.name ?? ''}")`;
164
+ const lines = [`? Auto proposes a workflow · round ${p.round || 1} ${where}`];
165
+ if (p.reasoning) lines.push(` ${p.reasoning}`);
166
+ // buildProposal defaults `size` to 'medium', so every REAL proposal prints this line;
167
+ // the guard only spares the unit fixtures that carry neither field.
168
+ const cues = [p.size, ...(Array.isArray(p.signals) ? p.signals : [])].filter(Boolean);
169
+ if (cues.length) lines.push(` ${cues.join(' · ')}`);
170
+ const nodes = p.manifest?.graph?.nodes || [];
171
+ const wires = p.manifest?.graph?.wires || [];
172
+ const agents = nodes.filter((n) => n.kind === 'agent');
173
+ const ordered = Array.isArray(p.order) && p.order.length
174
+ ? p.order.map((id) => agents.find((n) => n.id === id)).filter(Boolean)
175
+ : agents;
176
+ const labelOf = (id) => nodes.find((n) => n.id === id)?.label || id;
177
+ const tune = (n) => [n.model, n.effort].filter(Boolean).join(' · ');
178
+ lines.push(` stages: ${ordered.map((n) => `${n.label || n.key}${tune(n) ? ` (${tune(n)})` : ''}${n.fanOut ? ' ⤴' : ''}`).join(' → ')}`);
179
+ for (const l of wires.filter((x) => x.loop)) lines.push(` loop: ${labelOf(l.from.node)} → ${labelOf(l.to.node)} (max ${l.maxCycles} cycles)`);
180
+ if (p.ignoredProjectOverrides) lines.push(' note: this project\'s saved settings for that workflow are not applied to Auto runs');
181
+ for (const msg of p.warnings || []) lines.push(` ! ${msg}`);
182
+ if (Number(p.costUsd) > 0) lines.push(` classifier cost so far: $${Number(p.costUsd).toFixed(2)}`);
183
+ return lines;
184
+ }
@@ -27,7 +27,7 @@ import {
27
27
  normalizeProjectPath,
28
28
  } from '../core/projects.mjs';
29
29
  import { projectKey } from '../core/store.mjs';
30
- import { formatExecLine, formatGateHeader, formatRunSummary } from './render.mjs';
30
+ import { formatExecLine, formatGateHeader, formatRunSummary, formatWorkflowProposal } from './render.mjs';
31
31
  import { pauseExitCode, describePauseReason, promptOptions, REASON } from '../core/failure-policy.mjs';
32
32
  import { effectiveDebugSpawn } from '../core/settings.mjs';
33
33
  import {
@@ -97,6 +97,7 @@ function parseArgs(argv) {
97
97
  install: null,
98
98
  sourceBranch: undefined,
99
99
  featureBranch: undefined,
100
+ memoryScope: undefined,
100
101
  help: false,
101
102
  _: [],
102
103
  };
@@ -112,6 +113,7 @@ function parseArgs(argv) {
112
113
  '--install',
113
114
  '--source-branch',
114
115
  '--branch',
116
+ '--memory-scope',
115
117
  ]);
116
118
  const map = {
117
119
  '--project': 'project',
@@ -125,6 +127,7 @@ function parseArgs(argv) {
125
127
  '--install': 'install',
126
128
  '--source-branch': 'sourceBranch',
127
129
  '--branch': 'featureBranch',
130
+ '--memory-scope': 'memoryScope',
128
131
  };
129
132
 
130
133
  for (let i = 0; i < argv.length; i++) {
@@ -141,6 +144,10 @@ function parseArgs(argv) {
141
144
  out.auto = true;
142
145
  continue;
143
146
  }
147
+ if (arg === '--no-human') {
148
+ out.humanInLoop = false;
149
+ continue;
150
+ }
144
151
  if (arg === '--ui') {
145
152
  out.ui = true;
146
153
  continue;
@@ -171,6 +178,8 @@ function parseArgs(argv) {
171
178
  // MOCK runner treats it as the Ask Worca recipe — that pair is refused below,
172
179
  // after --mock/WORCA_MOCK are known (review of PR #376).
173
180
  fail(`--permission-mode must be one of ${PERMISSION_MODES.join(', ')}, got: ${value}`);
181
+ } else if (key === 'memoryScope' && !['global', 'project'].includes(String(value))) {
182
+ fail(`--memory-scope must be one of global, project, got: ${value}`);
174
183
  } else {
175
184
  out[key] = value;
176
185
  }
@@ -226,6 +235,7 @@ Subcommands:
226
235
  ui [start|stop|restart|status]
227
236
  Run the web UI (default http://localhost:4317). See: worca ui help
228
237
  workflow <cmd> [...] Export a workflow (Claude Code skill, JSON, or plugin) / import JSON: list|export|import. See: worca workflow help
238
+ metrics push [--project <path>] Push pending team-metrics run records (headless flush)
229
239
  help Print this help (same as --help).
230
240
  version Print the version (same as --version).
231
241
 
@@ -240,6 +250,10 @@ Options:
240
250
  --permission-mode <m> Claude permission mode: default | acceptEdits | plan |
241
251
  bypassPermissions (default acceptEdits)
242
252
  --workflow <id> Saved pipeline template to run (default: wf_default — the built-in graph)
253
+ auto (= wf_auto) lets worca pick the workflow per task
254
+ --no-human Auto workflow only: no proposal, no clarify, no agent questions (loop-budget, recovery, cost and error pauses still apply)
255
+ --memory-scope <s> Memory defragment workflow only: global | project — the scope the
256
+ run restructures (--workflow wf_memory_defrag needs it; no --prompt needed)
243
257
  --source-branch <name> Branch to fork the per-run worktree from (default: current HEAD)
244
258
  --branch <name> Feature branch name (default: claude proposes one)
245
259
  --mock Offline mock mode (no claude, no tokens)
@@ -385,6 +399,27 @@ async function askRecovery(rl, recovery) {
385
399
  return { decision };
386
400
  }
387
401
 
402
+ /**
403
+ * Ask the Auto workflow proposal (spec §9). Empty input accepts, `r` reads one line of
404
+ * revise text, `c` cancels the run. Returns the `workflow` answer payload.
405
+ */
406
+ async function askWorkflow(rl, workflow) {
407
+ out('');
408
+ for (const line of formatWorkflowProposal(workflow)) out(c('yellow', line));
409
+ out(' a) Accept and run');
410
+ out(' r) Revise — describe what to change');
411
+ out(' c) Cancel the run');
412
+ for (;;) {
413
+ const raw = (await question(rl, c('cyan', 'Choose [a/r/c]: '))).trim().toLowerCase();
414
+ if (raw === '' || raw === 'a' || raw === 'accept') return { decision: 'accept' };
415
+ if (raw === 'c' || raw === 'cancel') return { decision: 'cancel' };
416
+ if (raw === 'r' || raw === 'revise') {
417
+ const text = (await question(rl, c('cyan', 'What should change? '))).trim();
418
+ if (text) return { decision: 'revise', text };
419
+ }
420
+ }
421
+ }
422
+
388
423
  // ── shared drive loop ────────────────────────────────────────────────────────────
389
424
 
390
425
  /**
@@ -430,7 +465,13 @@ async function attachAndDrive(orch, flags, start) {
430
465
  // `Failed to read answer: readline was closed` and exited 0 with the row left
431
466
  // `running` — a CI job read success on an abandoned run.
432
467
  if (!flags.auto && !stdinCanAnswer()) {
433
- fail('stdin cannot answer prompts (it is /dev/null or closed) — pass --yes for a non-interactive run.');
468
+ // `--no-human` silences the Auto proposal, the clarify card and agent questions only;
469
+ // the loop-budget and recovery gates still ask (spec D3), so a CI user who passed it
470
+ // must be told which flag is missing (PR #434 review, finding 4).
471
+ const hint = flags.humanInLoop === false
472
+ ? '--no-human leaves the loop-budget and recovery gates interactive; pass --yes for a non-interactive run.'
473
+ : 'pass --yes for a non-interactive run.';
474
+ fail(`stdin cannot answer prompts (it is /dev/null or closed) — ${hint}`);
434
475
  }
435
476
  const rl = flags.auto ? null : makeRl();
436
477
  let answering = false; // serialize interactive prompts vs. log rendering
@@ -536,6 +577,9 @@ async function attachAndDrive(orch, flags, start) {
536
577
  } else if (kind === 'recovery') {
537
578
  const payload = await askRecovery(rl, recovery);
538
579
  orch.answer(id, payload);
580
+ } else if (kind === 'workflow') {
581
+ const answer = await askWorkflow(rl, payload.workflow);
582
+ orch.answer(id, answer);
539
583
  } else if (kind === 'questions') {
540
584
  out(c('yellow', c('bold', `${agent || 'Agent'} has questions:`)));
541
585
  const payload = await askClarify(rl, questions || []);
@@ -856,6 +900,12 @@ async function cmdAdd(argv) {
856
900
  try {
857
901
  await addProject({ name, path: target });
858
902
  out(`Added project "${name}" -> ${target}`);
903
+ // CLI-only machines have no hourly loop (decision 25): warm the discovery cache now
904
+ // so a teammate's team-metrics branch is seen without waiting on the UI server.
905
+ try {
906
+ const m = await import('../core/metrics/sync.mjs');
907
+ await m.discoverProject(target, { force: true });
908
+ } catch { /* metrics never block `worca add` */ }
859
909
  return 0;
860
910
  } catch (err) {
861
911
  process.stderr.write(`worca: ${err?.message || err}\n`);
@@ -1160,7 +1210,9 @@ async function cmdResume(argv) {
1160
1210
  auto,
1161
1211
  resume: saved,
1162
1212
  });
1163
- return attachAndDrive(orch, { auto }, () => orch.resume());
1213
+ const code = await attachAndDrive(orch, { auto }, () => orch.resume());
1214
+ await drainMetricsFlushes();
1215
+ return code;
1164
1216
  }
1165
1217
 
1166
1218
  // ── plugin subcommands ─────────────────────────────────────────────────────────
@@ -1891,9 +1943,9 @@ async function cmdWorkflow(argv) {
1891
1943
  try {
1892
1944
  switch (verb) {
1893
1945
  case 'list': {
1894
- // GRAPH_DEFAULT_WORKFLOW (the built-in default) is not in the user store, so
1895
- // prepend it — mirrors the server/UI, which always show it first.
1896
- const items = [wf.GRAPH_DEFAULT_WORKFLOW, ...(await wf.listWorkflows())];
1946
+ // The built-ins (Default, Memory defragment) are not in the user store, so
1947
+ // prepend them — mirrors the server/UI, which always show them first.
1948
+ const items = [wf.GRAPH_DEFAULT_WORKFLOW, wf.GRAPH_MEMORY_DEFRAG_WORKFLOW, ...(await wf.listWorkflows())];
1897
1949
  for (const w of items) out(`${w.id}\t${w.name}\t${(w.domain || 'general')}`);
1898
1950
  return 0;
1899
1951
  }
@@ -2012,9 +2064,67 @@ async function cmdWorkflow(argv) {
2012
2064
  }
2013
2065
  }
2014
2066
 
2067
+ // ── metrics subcommand ───────────────────────────────────────────────────────────
2068
+
2069
+ const METRICS_HELP = `worca metrics — team metrics (git-backed, team-wide run records)
2070
+
2071
+ Usage:
2072
+ worca metrics push [--project <path>] Flush pending run records to their worca-metrics branch.
2073
+ Without --project, every outbox on this machine is flushed.
2074
+ worca metrics help
2075
+
2076
+ Exit codes: 0 all pushed (or nothing pending) · 1 at least one outbox could not be pushed.
2077
+ `;
2078
+
2079
+ async function cmdMetrics(argv) {
2080
+ const verb = argv[0];
2081
+ const rest = argv.slice(1);
2082
+ if (!verb || verb === 'help') { process.stdout.write(METRICS_HELP); return 0; }
2083
+ const sync = await import('../core/metrics/sync.mjs');
2084
+ try {
2085
+ switch (verb) {
2086
+ case 'push': {
2087
+ const a = pluginArgs(rest, ['--project'], []);
2088
+ if (a._.length) fail(`unexpected argument "${a._[0]}" — see: worca metrics help`);
2089
+ // CLI-only machines have no hourly loop: refresh stale discovery caches (TTL-respecting) so a
2090
+ // branch a teammate enabled is seen here too. Best-effort; offline keeps the cached verdicts.
2091
+ await sync.discoverAll().catch(() => {});
2092
+ const results = a.project ? [await sync.flushProject(resolve(a.project))] : await sync.flushAll();
2093
+ if (!results.length) { out('worca metrics push: nothing pending'); return 0; }
2094
+ let failed = 0;
2095
+ for (const r of results) {
2096
+ if (r.ok) {
2097
+ out(`${c('green', '✓')} ${r.slug}: ${r.pushed ? `pushed ${r.pushed} run(s)` : 'nothing pending'}`);
2098
+ } else {
2099
+ failed += 1;
2100
+ out(`${c('red', '✗')} ${r.slug ?? '(project)'}: ${r.code}${r.pending ? ` — ${r.pending} run(s) still pending` : ''}`);
2101
+ if (r.stderr) process.stderr.write(r.stderr.endsWith('\n') ? r.stderr : `${r.stderr}\n`);
2102
+ if (r.hint) out(` hint: ${r.hint}`);
2103
+ }
2104
+ }
2105
+ return failed ? 1 : 0;
2106
+ }
2107
+ default:
2108
+ fail(`unknown metrics verb "${verb}" — see: worca metrics help`);
2109
+ }
2110
+ } catch (err) {
2111
+ process.stderr.write(`worca metrics ${verb}: ${err?.message || err}\n`);
2112
+ return 1;
2113
+ }
2114
+ }
2115
+
2116
+ /** Await in-flight metrics pushes before the CLI exits (decision 24). Never blocks past its
2117
+ * own budget and never changes the run's exit code — metrics must not gate the CLI. */
2118
+ async function drainMetricsFlushes() {
2119
+ try {
2120
+ const { drainFlushes } = await import('../core/metrics/sync.mjs');
2121
+ await drainFlushes({ timeoutMs: 30_000 });
2122
+ } catch { /* metrics never block the CLI exit */ }
2123
+ }
2124
+
2015
2125
  // ── main ──────────────────────────────────────────────────────────────────────────
2016
2126
 
2017
- const SUBCOMMANDS = new Set(['add', 'list', 'remove', 'resume', 'doctor', 'plugin', 'marketplace', 'config', 'ui', 'workflow']);
2127
+ const SUBCOMMANDS = new Set(['add', 'list', 'remove', 'resume', 'doctor', 'plugin', 'marketplace', 'config', 'ui', 'workflow', 'metrics']);
2018
2128
 
2019
2129
  /** Levenshtein distance, two-row. Only ever called on short argv tokens. */
2020
2130
  function editDistance(a, b) {
@@ -2074,6 +2184,7 @@ async function main() {
2074
2184
  if (sub === 'config') return cmdConfig(rest);
2075
2185
  if (sub === 'ui') return cmdUi(rest);
2076
2186
  if (sub === 'workflow') return cmdWorkflow(rest);
2187
+ if (sub === 'metrics') return cmdMetrics(rest);
2077
2188
  }
2078
2189
  // `worca --ui [...]` is the historical spelling of `worca ui start [...]`; hand the
2079
2190
  // remaining tokens to the ui parser so --port/--open/--mock work with either.
@@ -2105,6 +2216,10 @@ async function main() {
2105
2216
  fail('--permission-mode dontAsk cannot be combined with --mock: the mock runner reserves it for the Ask Worca assistant.');
2106
2217
  }
2107
2218
 
2219
+ // A defragment run needs no task text: synthesise the same brief the UI wrapper sends.
2220
+ if (flags.memoryScope && !flags.prompt && !flags.file && !flags._.length) {
2221
+ flags.prompt = flags.memoryScope === 'global' ? 'Defragment global memory.' : 'Defragment the memory of this project.';
2222
+ }
2108
2223
  if (!flags.prompt && !flags.file) {
2109
2224
  // Allow a bare positional prompt: `worca "do the thing"`. A lone token that
2110
2225
  // near-misses a subcommand is a typo, not a task — refuse it here, before a
@@ -2151,12 +2266,22 @@ async function main() {
2151
2266
  // Validate --workflow before spawning anything: an unknown or archived template
2152
2267
  // must fail with one line, not a stack trace half-way through a run. The read row
2153
2268
  // doubles as createOrchestratorFor's routing hint (it skips a second row read).
2269
+ // `--workflow auto` is the Auto entry (spec D15); `--no-human` means nothing elsewhere.
2270
+ if (flags.workflow === 'auto') flags.workflow = 'wf_auto';
2271
+ if (flags.humanInLoop === false && flags.workflow !== 'wf_auto') {
2272
+ out(c('yellow', '--no-human only affects the Auto workflow (--workflow auto); ignored for this run.'));
2273
+ }
2154
2274
  let row;
2155
2275
  if (flags.workflow) {
2156
2276
  const { assertRunnableWorkflow } = await import('../core/workflows.mjs');
2157
2277
  try { row = await assertRunnableWorkflow(flags.workflow); }
2158
2278
  catch (err) { fail(`${err && err.message ? err.message : String(err)}`); }
2159
2279
  }
2280
+ {
2281
+ const { validateMemoryScope } = await import('../core/memory-sync.mjs');
2282
+ const reason = validateMemoryScope({ workflowId: flags.workflow || 'wf_default', memoryScope: flags.memoryScope, isWorkspace: false });
2283
+ if (reason) fail(reason);
2284
+ }
2160
2285
 
2161
2286
  const orch = await createOrchestratorFor({
2162
2287
  projectDir,
@@ -2166,6 +2291,7 @@ async function main() {
2166
2291
  extras,
2167
2292
  workflowId: flags.workflow || undefined,
2168
2293
  template: row,
2294
+ memoryScope: flags.memoryScope || undefined,
2169
2295
  branch: { source: flags.sourceBranch, feature: flags.featureBranch },
2170
2296
  claude: {
2171
2297
  permissionMode: flags.permissionMode,
@@ -2173,12 +2299,15 @@ async function main() {
2173
2299
  mock: flags.mock,
2174
2300
  },
2175
2301
  auto: flags.auto,
2302
+ humanInLoop: flags.humanInLoop === false ? false : undefined,
2176
2303
  });
2177
2304
 
2178
2305
  out(c('bold', `orchestrator — project: ${projectDir}`));
2179
2306
  if (flags.mock) out(c('yellow', 'mock mode: no claude will be spawned'));
2180
2307
 
2181
- return attachAndDrive(orch, flags, () => orch.run());
2308
+ const code = await attachAndDrive(orch, flags, () => orch.run());
2309
+ await drainMetricsFlushes();
2310
+ return code;
2182
2311
  }
2183
2312
 
2184
2313
  main()
@@ -17,6 +17,7 @@ import { readPluginsLock, pluginCurrentDir } from './plugins-lock.mjs'; // plugi
17
17
  import { declaredApi, NOT_META_V2 } from './plugin-manifest.mjs'; // plugin API declared by a layer's manifest
18
18
  import { normalizeAgentMeta, DEFAULT_ORDER } from '../shared/graph/agent-meta.mjs'; // meta v2 (one source: registry + store + UI)
19
19
  import { MOCK_WRITER_ROLES } from './claude-runner.mjs'; // mockRole vocabulary (no cycle: claude-runner imports no registry)
20
+ import { readFrontmatterSync } from './frontmatter.mjs';
20
21
 
21
22
  /**
22
23
  * Default location of the agent metadata sidecars, relative to this module.
@@ -213,32 +214,6 @@ export function pluginAgentLayers() {
213
214
  }
214
215
  }
215
216
 
216
- /**
217
- * Fallback palette blurb: the agent .md's YAML frontmatter `description:` line,
218
- * stored VERBATIM (clarify 2026-08-09: the UI clamps, the bubble wants the full
219
- * text — never truncate here). Single-line values only (plain or quoted);
220
- * folded/multi-line scalars are out of scope by design (spec 2026-08-09; every
221
- * shipped .md uses a single-line scalar) and degrade to '' — the block-scalar
222
- * indicator is detected, never stored. Any read/parse failure returns '' so
223
- * scanLayer never throws because of the fallback.
224
- */
225
- function frontmatterDescription(mdPath) {
226
- let text;
227
- try { text = readFileSync(mdPath, 'utf8'); } catch { return ''; }
228
- const fm = text.match(/^---\r?\n([\s\S]*?)\r?\n---/);
229
- if (!fm) return '';
230
- const line = fm[1].match(/^description:[ \t]*(.+)$/m);
231
- if (!line) return '';
232
- let v = line[1].trim();
233
- // Folded/literal block scalars ('>', '>-', '|', '|+', …): the captured value
234
- // is just the indicator, not the text — degrade to '' rather than store junk.
235
- if (/^[>|][+-]?$/.test(v)) return '';
236
- if ((v.startsWith('"') && v.endsWith('"')) || (v.startsWith("'") && v.endsWith("'"))) {
237
- v = v.slice(1, -1).trim();
238
- }
239
- return v;
240
- }
241
-
242
217
  /** Scan one layer dir for *.meta.json; stamps the COMPUTED origin/agentPath/
243
218
  * descriptionDerived fields (none of which normalizeMeta returns, so none can
244
219
  * be persisted back into a sidecar). */
@@ -294,15 +269,18 @@ function scanLayer(dir, origin, { requireMetaV2 = false, builtFor = null, onDrop
294
269
  continue;
295
270
  }
296
271
  meta.agentPath = meta.agentFile ? join(dir, meta.agentFile) : null; // layer-correct abs path
272
+ // The agent .md's frontmatter (name/description/tools/model), read from the
273
+ // file HEAD only (frontmatter.mjs) — COMPUTED like origin/agentPath, never
274
+ // stored (normalizeMeta's fixed key set drops it on every write path). The
275
+ // Auto classifier reads it (auto-workflow P1); the chat catalog can too.
276
+ meta.frontmatter = meta.agentPath ? readFrontmatterSync(meta.agentPath) : null;
297
277
  // Description fallback (spec 2026-08-09): empty sidecar description →
298
- // the .md frontmatter description. Only costs a file read when empty.
299
- // descriptionDerived marks the RESOLVED description as computed too: unlike
300
- // origin/agentPath, `description` has a slot in normalizeMeta, so without
301
- // this flag every write path would bake the fallback into the sidecar and
302
- // the blurb would stop tracking the .md (and could never be cleared).
303
- if (!meta.description && meta.agentPath) {
304
- meta.description = frontmatterDescription(meta.agentPath);
305
- if (meta.description) meta.descriptionDerived = true; // computed, never stored
278
+ // the .md frontmatter description. descriptionDerived marks the RESOLVED
279
+ // description as computed too, so no write path bakes the fallback into
280
+ // the sidecar and the blurb keeps tracking the .md.
281
+ if (!meta.description && meta.frontmatter?.description) {
282
+ meta.description = meta.frontmatter.description;
283
+ meta.descriptionDerived = true; // computed, never stored
306
284
  }
307
285
  metas.push(meta);
308
286
  }