analyzthis_design 1.20.1 → 2.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. package/HOW-TO-USE.md +436 -0
  2. package/README.md +183 -36
  3. package/agents/cards/chunk-planner.md +5 -0
  4. package/agents/cards/mood-board.md +13 -0
  5. package/agents/cards/query-expander.md +17 -0
  6. package/agents/cards/ranker.md +16 -0
  7. package/agents/cards/run-unchunked.md +7 -0
  8. package/agents/chain.json +15 -0
  9. package/agents/manifests/anuj.json +1 -0
  10. package/agents/manifests/arjun.json +1 -0
  11. package/agents/manifests/chunk-planner.json +18 -0
  12. package/agents/manifests/kavi.json +1 -0
  13. package/agents/manifests/meera.json +1 -0
  14. package/agents/manifests/mood-board.json +23 -0
  15. package/agents/manifests/noor.json +1 -0
  16. package/agents/manifests/priya.json +1 -0
  17. package/agents/manifests/query-expander.json +13 -0
  18. package/agents/manifests/raj.json +1 -0
  19. package/agents/manifests/ranker.json +13 -0
  20. package/agents/manifests/run-unchunked.json +19 -0
  21. package/agents/manifests/zara.json +1 -0
  22. package/agents/router.json +14 -0
  23. package/agents/session-schema.json +27 -0
  24. package/dist/.github/ISSUE_TEMPLATE/persona-feedback.yml +75 -0
  25. package/dist/HOW-TO-USE.md +436 -0
  26. package/dist/README.md +869 -0
  27. package/dist/agents/cards/anuj.md +28 -0
  28. package/dist/agents/cards/arjun.md +29 -0
  29. package/dist/agents/cards/chunk-planner.md +5 -0
  30. package/dist/agents/cards/design-director.md +19 -0
  31. package/dist/agents/cards/devi.md +15 -0
  32. package/dist/agents/cards/kavi.md +26 -0
  33. package/dist/agents/cards/meera.md +27 -0
  34. package/dist/agents/cards/mood-board.md +13 -0
  35. package/dist/agents/cards/noor.md +28 -0
  36. package/dist/agents/cards/priya.md +27 -0
  37. package/dist/agents/cards/query-expander.md +17 -0
  38. package/dist/agents/cards/raj.md +29 -0
  39. package/dist/agents/cards/ranker.md +16 -0
  40. package/dist/agents/cards/run-unchunked.md +7 -0
  41. package/dist/agents/cards/zara.md +30 -0
  42. package/dist/agents/chain.json +76 -0
  43. package/dist/agents/deliberation-schema.json +44 -0
  44. package/dist/agents/design-spec-schema.json +111 -0
  45. package/dist/agents/manifests/anuj.json +32 -0
  46. package/dist/agents/manifests/arjun.json +47 -0
  47. package/dist/agents/manifests/chunk-planner.json +18 -0
  48. package/dist/agents/manifests/design-critic.json +21 -0
  49. package/dist/agents/manifests/design-director.json +26 -0
  50. package/dist/agents/manifests/devi.json +31 -0
  51. package/dist/agents/manifests/kavi.json +34 -0
  52. package/dist/agents/manifests/meera.json +32 -0
  53. package/dist/agents/manifests/mood-board.json +23 -0
  54. package/dist/agents/manifests/noor.json +33 -0
  55. package/dist/agents/manifests/persona-orchestrator.json +17 -0
  56. package/dist/agents/manifests/priya.json +32 -0
  57. package/dist/agents/manifests/query-expander.json +13 -0
  58. package/dist/agents/manifests/raj.json +31 -0
  59. package/dist/agents/manifests/ranker.json +13 -0
  60. package/dist/agents/manifests/run-unchunked.json +19 -0
  61. package/dist/agents/manifests/ux-story-gate.json +23 -0
  62. package/dist/agents/manifests/zara.json +33 -0
  63. package/dist/agents/router.json +99 -0
  64. package/dist/agents/session-schema.json +152 -0
  65. package/dist/bin/cli.js +1 -1
  66. package/dist/lib/cache.js +1 -1
  67. package/dist/lib/chunk-executor.js +1 -0
  68. package/dist/lib/chunk-models.js +1 -0
  69. package/dist/lib/chunk-planner.js +1 -0
  70. package/dist/lib/chunk-router.js +1 -0
  71. package/dist/lib/chunk-run.js +1 -0
  72. package/dist/lib/chunk-synthesis.js +1 -0
  73. package/dist/lib/chunk-telemetry.js +1 -0
  74. package/dist/lib/collect.js +1 -1
  75. package/dist/lib/cost.js +1 -1
  76. package/dist/lib/dedup.js +1 -0
  77. package/dist/lib/deliberation.js +1 -1
  78. package/dist/lib/design-spec.js +1 -1
  79. package/dist/lib/evolve.js +1 -0
  80. package/dist/lib/export.js +1 -1
  81. package/dist/lib/feedback-submit.js +1 -1
  82. package/dist/lib/feedback.js +1 -1
  83. package/dist/lib/host-llm.js +1 -1
  84. package/dist/lib/install.js +1 -1
  85. package/dist/lib/knowledge.js +1 -1
  86. package/dist/lib/lessons.js +1 -0
  87. package/dist/lib/moodboard.js +1 -0
  88. package/dist/lib/orchestrator/run.js +1 -1
  89. package/dist/lib/outcome.js +1 -0
  90. package/dist/lib/platforms.js +1 -1
  91. package/dist/lib/provider.js +1 -1
  92. package/dist/lib/query-expander.js +1 -0
  93. package/dist/lib/ranker.js +1 -0
  94. package/dist/lib/reference-pack.js +1 -0
  95. package/dist/lib/research.js +1 -1
  96. package/dist/lib/retrieve.js +1 -1
  97. package/dist/lib/session.js +1 -1
  98. package/dist/lib/source-discovery.js +1 -1
  99. package/dist/lib/synthesis.js +1 -1
  100. package/dist/lib/token-gate.js +1 -1
  101. package/dist/skills/anuj/SKILL.md +92 -0
  102. package/dist/skills/arjun/SKILL.md +291 -0
  103. package/dist/skills/chunk-planner/SKILL.md +45 -0
  104. package/dist/skills/collect-knowledge/SKILL.md +20 -0
  105. package/dist/skills/deliberation-protocol/SKILL.md +131 -0
  106. package/dist/skills/design-critic/SKILL.md +191 -0
  107. package/dist/skills/design-director/SKILL.md +181 -0
  108. package/dist/skills/design-personas/SKILL.md +100 -0
  109. package/dist/skills/design-reference/SKILL.md +83 -0
  110. package/dist/skills/design-reference/app-interface.csv +31 -0
  111. package/dist/skills/design-reference/charts.csv +26 -0
  112. package/dist/skills/design-reference/colors.csv +162 -0
  113. package/dist/skills/design-reference/google-fonts.csv +1924 -0
  114. package/dist/skills/design-reference/icons.csv +106 -0
  115. package/dist/skills/design-reference/landing.csv +35 -0
  116. package/dist/skills/design-reference/products.csv +162 -0
  117. package/dist/skills/design-reference/react-performance.csv +45 -0
  118. package/dist/skills/design-reference/schema.json +159 -0
  119. package/dist/skills/design-reference/stacks/angular.csv +51 -0
  120. package/dist/skills/design-reference/stacks/astro.csv +54 -0
  121. package/dist/skills/design-reference/stacks/flutter.csv +53 -0
  122. package/dist/skills/design-reference/stacks/html-tailwind.csv +56 -0
  123. package/dist/skills/design-reference/stacks/jetpack-compose.csv +53 -0
  124. package/dist/skills/design-reference/stacks/laravel.csv +51 -0
  125. package/dist/skills/design-reference/stacks/nextjs.csv +53 -0
  126. package/dist/skills/design-reference/stacks/nuxt-ui.csv +51 -0
  127. package/dist/skills/design-reference/stacks/nuxtjs.csv +59 -0
  128. package/dist/skills/design-reference/stacks/react-native.csv +52 -0
  129. package/dist/skills/design-reference/stacks/react.csv +54 -0
  130. package/dist/skills/design-reference/stacks/shadcn.csv +61 -0
  131. package/dist/skills/design-reference/stacks/svelte.csv +54 -0
  132. package/dist/skills/design-reference/stacks/swiftui.csv +51 -0
  133. package/dist/skills/design-reference/stacks/threejs.csv +54 -0
  134. package/dist/skills/design-reference/stacks/vue.csv +50 -0
  135. package/dist/skills/design-reference/styles.csv +85 -0
  136. package/dist/skills/design-reference/typography.csv +75 -0
  137. package/dist/skills/design-reference/ui-reasoning.csv +162 -0
  138. package/dist/skills/design-reference/ux-guidelines.csv +100 -0
  139. package/dist/skills/design-spec/SKILL.md +106 -0
  140. package/dist/skills/devi/SKILL.md +114 -0
  141. package/dist/skills/getting-started/SKILL.md +144 -0
  142. package/dist/skills/kavi/SKILL.md +118 -0
  143. package/dist/skills/knowledge-bank/SKILL.md +43 -0
  144. package/dist/skills/meera/SKILL.md +76 -0
  145. package/dist/skills/mood-board/SKILL.md +113 -0
  146. package/dist/skills/noor/SKILL.md +96 -0
  147. package/dist/skills/persona-orchestrator/SKILL.md +220 -0
  148. package/dist/skills/priya/SKILL.md +75 -0
  149. package/dist/skills/raj/SKILL.md +77 -0
  150. package/dist/skills/run-unchunked/SKILL.md +49 -0
  151. package/dist/skills/ux-ideator/SKILL.md +187 -0
  152. package/dist/skills/ux-story-gate/SKILL.md +353 -0
  153. package/dist/skills/zara/SKILL.md +85 -0
  154. package/dist/supabase/deliberation-config.example.json +15 -0
  155. package/dist/supabase/feedback-config.example.json +7 -0
  156. package/dist/supabase/migrations/001_persona_feedback.sql +54 -0
  157. package/package.json +7 -4
  158. package/scripts/validate-csvs.js +197 -0
  159. package/skills/chunk-planner/SKILL.md +45 -0
  160. package/skills/design-reference/google-fonts.csv +1924 -1924
  161. package/skills/design-reference/products.csv +162 -162
  162. package/skills/design-reference/schema.json +159 -0
  163. package/skills/design-reference/stacks/angular.csv +1 -1
  164. package/skills/design-reference/stacks/astro.csv +1 -1
  165. package/skills/design-reference/stacks/laravel.csv +2 -2
  166. package/skills/design-reference/stacks/threejs.csv +54 -54
  167. package/skills/design-reference/styles.csv +85 -85
  168. package/skills/design-reference/typography.csv +75 -74
  169. package/skills/design-reference/ui-reasoning.csv +1 -1
  170. package/skills/getting-started/SKILL.md +12 -0
  171. package/skills/mood-board/SKILL.md +113 -0
  172. package/skills/persona-orchestrator/SKILL.md +35 -1
  173. package/skills/run-unchunked/SKILL.md +49 -0
package/README.md CHANGED
@@ -4,33 +4,45 @@ A set of AI design personas and a task-first evaluation framework that plugs int
4
4
 
5
5
  Install once. Run structured UX critiques, multi-phase ideation, and task-grounded screen reviews — directly inside your AI chat. **No external LLM API keys required** for CLI orchestrator runs: **`/devi`** voices each persona from your host IDE (Cursor, Claude, etc.).
6
6
 
7
- **npm:** [analyzthis_design](https://www.npmjs.com/package/analyzthis_design) · **Current version:** 1.20.1
7
+ ## v2.0 — chunked execution by default
8
+
9
+ `npx analyzthis_design run --task "..."` now uses a **frontier planner + cheap chunk models**:
10
+
11
+ 1. Frontier/strong model plans the task into small chunks.
12
+ 2. Each chunk runs on the cheapest capable model: local Ollama, free cloud APIs (Groq, Gemini, OpenRouter), or cheap cloud APIs with your keys.
13
+ 3. Outputs are merged into a final verdict.
14
+ 4. Telemetry learns which models work best for each chunk type.
15
+
16
+ Use `npx analyzthis_design run-unchunked` for the legacy single-pass orchestrator.
17
+
18
+ **npm:** [analyzthis_design](https://www.npmjs.com/package/analyzthis_design) · **Current version:** 2.0.1 · **Step-by-step guide:** [HOW-TO-USE.md](./HOW-TO-USE.md)
8
19
 
9
20
  ---
10
21
 
11
- ## Install
22
+ ## Quick start
12
23
 
13
24
  ```bash
14
- # Cursor (default) + cross-agent path on postinstall
25
+ # 1. Install slash commands (Cursor/Claude/Codex/Grok/Windsurf)
15
26
  npx analyzthis_design
16
27
 
17
- # Claude Code (skills dir + legacy commands)
18
- npx analyzthis_design --target claude
28
+ # 2. Run a task in v2.0 chunked mode (free/cheap models)
29
+ npx analyzthis_design run --task "Review invoice approval screen"
19
30
 
20
- # Codex CLI
21
- npx analyzthis_design --target codex
31
+ # 3. Or use legacy single-pass orchestrator
32
+ npx analyzthis_design run-unchunked --task "Review invoice approval screen" --provider host
33
+ ```
22
34
 
23
- # Grok Build (xAI)
24
- npx analyzthis_design --target grok
35
+ ### Install by target IDE
25
36
 
26
- # Windsurf Cascade
37
+ ```bash
38
+ npx analyzthis_design --target claude
39
+ npx analyzthis_design --target codex
40
+ npx analyzthis_design --target grok
27
41
  npx analyzthis_design --target windsurf
28
-
29
- # All supported hosts at once
30
42
  npx analyzthis_design --target all --force
31
43
  ```
32
44
 
33
- **After install:** type `/getting-started` in Cursor or Claude Code (or `@getting-started` in Windsurf). Re-print CLI help anytime with `npx analyzthis_design welcome`.
45
+ **After install:** type `/getting-started` in Cursor or Claude Code (or `@getting-started` in Windsurf). For a complete walkthrough, see [HOW-TO-USE.md](./HOW-TO-USE.md). Re-print CLI help anytime with `npx analyzthis_design welcome`.
34
46
 
35
47
  | Tool | Skills installed to | Invoke |
36
48
  |---|---|---|
@@ -205,9 +217,14 @@ Key fields after a run:
205
217
  | Field | Contents |
206
218
  |---|---|
207
219
  | `persona_outputs` | Each persona's text + parsed deliberation JSON |
220
+ | `full_prompts` | Exact `{ system, user }` prompts sent to the LLM (for training / auditing) |
221
+ | `structured_outputs` | Parsed grades, score, top fixes per persona |
222
+ | `covered_points` | Deduplicated findings across personas (redundancy suppression) |
223
+ | `task_type` | Canonical problem type from the router |
208
224
  | `deliberation` | `round_log`, `open_objections`, `consensus_reached`, `raj_escalated` |
209
225
  | `synthesis` | Composite scores, verdict, top 3, hierarchy gate |
210
226
  | `synthesis_markdown` | Phase 5 block for display / export |
227
+ | `outcome` | `inferred` + `confirmed` outcome per persona (for the evolution loop) |
211
228
  | `host_run` | Host-mode checkpoint when paused for Devi (`run_dir`, `checkpoint`) |
212
229
  | `metrics` | `llm_calls`, `deliberation_rounds`, `objections_raised`, token estimates |
213
230
 
@@ -227,31 +244,36 @@ agents/
227
244
  session-schema.json
228
245
  ```
229
246
 
230
- **Standalone runtime (v2):**
247
+ **v2.0 chunked runtime (default for `run`):**
231
248
 
232
249
  ```bash
233
- # Print routing + deliberation groups (no LLM calls)
234
- npx analyzthis_design run --task "Fix contrast on landing page" --dry-run
250
+ # Default: frontier planner + sequential chunk execution on free/cheap models
251
+ npx analyzthis_design run --task "Review invoice approval screen"
235
252
 
236
- # Host mode (default when no API keys) — Devi voices personas
237
- npx analyzthis_design run --task "Review invoice screen" --full
238
- npx analyzthis_design devi status
239
- npx analyzthis_design run --continue --task "Review invoice screen" --full
253
+ # Auto-detect local Ollama, otherwise use free cloud models
254
+ npx analyzthis_design run --task "Review invoice approval screen" --budget free
240
255
 
241
- # External API providers (optional)
242
- export ANTHROPIC_API_KEY=sk-...
243
- npx analyzthis_design run --task "Review this screen" --figma https://figma.com/... --provider anthropic
256
+ # Use your paid keys for cheap cloud models
257
+ npx analyzthis_design run --task "..." --budget cheap --provider together
258
+
259
+ # Limit parallelism / chunk count (sequential is default)
260
+ npx analyzthis_design run --task "..." --sequential --max-chunks 4
261
+ npx analyzthis_design run --task "..." --parallel --max-chunks 6
244
262
 
245
- # Force full chain, bypass router, tune deliberation
246
- npx analyzthis_design run --task "Full critique of onboarding" --full
247
- npx analyzthis_design run --task "Just check spacing" --experts arjun
248
- npx analyzthis_design run --task "..." --max-rounds 2 --satisfaction 0.5
249
- npx analyzthis_design run --task "..." --no-deliberate # legacy sequential handoff
263
+ # Legacy single-pass orchestrator (unchunked)
264
+ npx analyzthis_design run-unchunked --task "Review invoice approval screen" --full
265
+
266
+ # Host mode (no API keys) — Devi voices personas via prompt queue
267
+ npx analyzthis_design run-unchunked --task "Review invoice screen" --full
268
+ npx analyzthis_design devi status
269
+ npx analyzthis_design run-unchunked --continue --task "Review invoice screen" --full
250
270
  ```
251
271
 
252
- **Provider resolution order:** explicit `--provider` → config → first available API key → **`host`** (Devi).
272
+ **Chunk model selection:** Ollama auto-discovered → free cloud (Groq/Gemini/OpenRouter free endpoints) → cheap cloud with user keys. The **planner always runs on a frontier/strong model** and never on a cheap model; if no frontier provider is available, it falls back to the host model with a warning.
253
273
 
254
- Supported providers: `host` | `anthropic` | `openai` | `google` | `zai`
274
+ **Provider resolution order:** explicit `--provider` → config → first available API key → **`host`** (Devi) for unchunked; chunked mode also considers Ollama and free endpoints before paid.
275
+
276
+ Supported providers: `host` | `anthropic` | `openai` | `google` | `zai` | `ollama` | `groq` | `together` | `openrouter` | `deepseek`
255
277
 
256
278
  Provider defaults live in `~/.analyzthis_design/config.json`:
257
279
 
@@ -357,6 +379,9 @@ Ask → session digest → MoE router (1–2 experts, not 4) → persona cards (
357
379
  | **Persona cards** | `agents/cards/<persona>.md` (~500 tokens) are the default system prompt; the full `skills/<persona>/SKILL.md` is only opened for a C-or-below rubric lookup or an explicit deep/full request. |
358
380
  | **Lite output schema** | Grades + Top 2 fixes + score, by default. Deep/full schema is opt-in. |
359
381
  | **Retrieve-on-demand** | `npx analyzthis_design retrieve --file colors.csv --column "Product Type" --keywords saas` returns only matching rows, pre-formatted for citation — never the whole CSV. |
382
+ | **Multi-file reference packs** | Each persona retrieves from 3-5 CSV files (not 1), pooled into a single ranker call. Arjun gets styles + ux-guidelines + ui-reasoning + charts; Zara gets colors + typography + styles + landing + icons. Same LLM cost as single-file, 3-5x coverage. |
383
+ | **Best For Tags** | `styles.csv` and `typography.csv` have a `Best For Tags` column (semicolon-delimited product-type tokens) for reliable keyword filtering. Previously `Best For` was free-text and 31/84 styles were unfilterable. |
384
+ | **CSV schema + validation** | `skills/design-reference/schema.json` defines all 28 CSV files' headers, filter columns, and cross-file joins. `npm run validate` checks integrity before publish. |
360
385
  | **Model tiers** | `structured` steps can run on a cheaper model (e.g. `gpt-4o-mini`); `critique`/`arbitrate` steps use a stronger model. Configurable per tier in `~/.analyzthis_design/config.json`. |
361
386
  | **Caching** | `lib/cache.js` caches retrieve results (invalidated automatically when the source CSV changes) and knowledge-bank slices (invalidated on `sync` / `session reset`). |
362
387
  | **Cost metrics** | Every `run` records `metrics` (llm_calls, experts_run, estimated tokens, cache_hits) into session state. |
@@ -412,7 +437,7 @@ npx analyzthis_design feedback list
412
437
  npx analyzthis_design feedback export --persona arjun --all
413
438
  ```
414
439
 
415
- Writes `{ system_card, digest, user, assistant }` JSONL pairs to `~/.analyzthis_design/training/<persona>.jsonl` from every session where that persona's output was explicitly accepted.
440
+ `export-training` now emits richer training pairs: `{ system_card, system_prompt_full, user_prompt_full, digest, user, assistant, structured_output, outcome, task_type }`. The full prompts are captured automatically on every `run`, so fine-tuning datasets include the exact context the persona saw.
416
441
 
417
442
  **Correction export** writes `{ assistant_rejected, assistant_preferred, user_comment, tags }` to `~/.analyzthis_design/feedback/<persona>-corrections.jsonl` — useful when users were unhappy or had to rewrite persona output. Every entry is also appended to a global `corrections.jsonl` across projects.
418
443
 
@@ -420,6 +445,62 @@ Once a persona accumulates ~100–300 accepted pairs (and optionally correction
420
445
 
421
446
  ---
422
447
 
448
+ ## Self-evolving persona team (v1.21)
449
+
450
+ The system now captures every run, learns from accepted outputs + confirmed outcomes, and proposes improvements to its own prompts, reference data, and routing — **dry-run by default**, human review before any apply.
451
+
452
+ ```
453
+ Run → full prompts + structured output + outcome captured
454
+ ↓
455
+ Lessons extracted (accepted outputs) → ~/.analyzthis_design/lessons/<persona>.jsonl
456
+ ↓
457
+ Outcome confirmed (shipped / revised / blocked / missed)
458
+ ↓
459
+ evolve --extract → proposes:
460
+ - prompt patches (new canonical failure patterns per persona)
461
+ - reference-data rows (new product-type patterns)
462
+ - router patches (task_type → best-performing expert)
463
+ ↓
464
+ evolve --apply <patchId> (human review) → skill/CSV/router updated
465
+ ↓
466
+ Next run retrieves:
467
+ - per-persona knowledge slices (priority + fallback)
468
+ - past lessons for similar tasks
469
+ - query-expanded + ranked reference rows from 3-5 CSV files per persona
470
+ ```
471
+
472
+ ### Retrieval stack
473
+
474
+ | Layer | What it does | Files |
475
+ |---|---|---|
476
+ | **Query expansion** | One cheap LLM call per run expands the task into search terms (product type, design domain, component, persona lens) | `lib/query-expander.js`, `agents/cards/query-expander.md` |
477
+ | **Per-persona ranking** | A second cheap LLM call per persona ranks the top 5 reference rows + knowledge notes for that persona's lens | `lib/ranker.js`, `agents/cards/ranker.md` |
478
+ | **Lessons retrieval** | Top-3 lessons from past accepted sessions, keyword-matched to the current task | `lib/lessons.js` |
479
+ | **Redundancy suppression** | Before each persona produces, it sees what prior personas already covered and is told to only add NEW insights | `lib/dedup.js`, `lib/deliberation.js` |
480
+ | **Per-persona KB slices** | `sync` now builds a filtered slice per persona (priority categories first, small fallback context at the end) | `lib/knowledge.js` |
481
+
482
+ ### Commands
483
+
484
+ ```bash
485
+ # Extract lessons + infer outcomes + propose patches (dry-run by default)
486
+ npx analyzthis_design evolve --extract [--window N] [--dry-run]
487
+
488
+ # Review a patch before applying
489
+ npx analyzthis_design evolve --apply <patchId> --dry-run
490
+
491
+ # Apply a patch after review (prompt / reference rows only; router patches need manual edit)
492
+ npx analyzthis_design evolve --apply <patchId>
493
+
494
+ # Outcome tracking
495
+ npx analyzthis_design outcome --infer [--window N] # auto-infer from next session
496
+ npx analyzthis_design outcome --pending # list inferred outcomes awaiting confirmation
497
+ npx analyzthis_design outcome --confirm --persona arjun --result shipped
498
+ ```
499
+
500
+ Config in `agents/chain.json` → `evolution` block: `extraction_window_days`, `min_lessons_for_patch`, `min_outcomes_for_router_patch`.
501
+
502
+ ---
503
+
423
504
  ## Persona feedback — corrections & unhappiness (v1.16)
424
505
 
425
506
  When a persona gets it wrong, you can record **what was wrong** and **how you fixed it**. This feeds future fine-tuning (negative / DPO pairs) alongside the existing positive `export-training` path.
@@ -631,12 +712,27 @@ npx analyzthis_design research --query <text>
631
712
  # Reference data (retrieve-on-demand)
632
713
  npx analyzthis_design retrieve --file <csv> --column <col> --keywords a,b [--limit N]
633
714
 
634
- # Standalone orchestrator
635
- npx analyzthis_design run --task "..." [--figma URL] [--provider host|anthropic|openai|google|zai] [--dry-run] [--output path]
636
- npx analyzthis_design run --task "..." [--lite | --full] [--experts a,b]
637
- npx analyzthis_design run --task "..." [--deliberate | --no-deliberate] [--max-rounds N] [--satisfaction 0.4]
715
+ # Standalone orchestrator (v2.0 chunked by default)
716
+ npx analyzthis_design run --task "..." [--budget free|cheap|auto] [--sequential|--parallel] [--max-chunks N] [--dry-run]
717
+ npx analyzthis_design run-unchunked --task "..." [--lite|--full] [--experts a,b] [--dry-run] [--deliberate|--no-deliberate]
638
718
  npx analyzthis_design run --continue --task "..." # resume host-mode run after /devi
639
719
 
720
+ # CSV validation
721
+ npx analyzthis_design validate # validate all 28 CSV files against schema.json
722
+
723
+ # Self-evolving team (v1.21)
724
+ npx analyzthis_design evolve --extract [--window N] [--dry-run]
725
+ npx analyzthis_design evolve --apply <patchId> [--dry-run]
726
+ npx analyzthis_design outcome --infer [--window N]
727
+ npx analyzthis_design outcome --pending
728
+ npx analyzthis_design outcome --confirm --persona <id> --result shipped|revised|blocked|missed
729
+
730
+ # Mood board (v1.22)
731
+ npx analyzthis_design moodboard create --task "..." [--auto] [--url <url> ...]
732
+ npx analyzthis_design moodboard critique --board <id>
733
+ npx analyzthis_design moodboard add --board <id> --url <url> --title "..." --tags a,b
734
+ npx analyzthis_design moodboard list
735
+
640
736
  # Devi — host LLM queue (v1.20)
641
737
  npx analyzthis_design devi status [--run path]
642
738
  npx analyzthis_design devi respond --run <run-dir> --step 001-arjun --file response.md
@@ -666,6 +762,21 @@ lib/
666
762
  research.js URL / query → web-context.md
667
763
  retrieve.js Filtered, citation-ready CSV row retrieval
668
764
  cache.js On-disk cache for retrieve/kb slices
765
+ chunk-models.js Curated free/cheap model pool + Ollama auto-discovery
766
+ chunk-planner.js Frontier/strong model chunk planner
767
+ chunk-router.js Cheapest capable model per chunk
768
+ chunk-executor.js Chunk execution with retry + fallback
769
+ chunk-synthesis.js Merge chunk outputs into final verdict
770
+ chunk-telemetry.js Per-chunk model quality tracking
771
+ chunk-run.js Top-level chunked execution coordinator
772
+ reference-pack.js Shared multi-file CSV + vault retrieval (buildReferencePack)
773
+ moodboard.js Mood-board engine: collect web/DS references, tag, deliberate
774
+ dedup.js Cross-persona redundancy detection
775
+ lessons.js Self-evolving lessons store (extract/retrieve/inject)
776
+ outcome.js Infer + confirm persona outcome labels
777
+ query-expander.js LLM task → search terms for retrieval
778
+ ranker.js LLM per-persona ranking of reference/knowledge candidates
779
+ evolve.js Evolution engine: prompt/reference/router patch proposals
669
780
  export.js LoRA training-pair export hook
670
781
  feedback.js Persona unhappiness + correction logging (session + global JSONL)
671
782
  feedback-submit.js Opt-in anonymized submit to Supabase (community feedback)
@@ -680,6 +791,7 @@ scripts/
680
791
  quality-check.js Validate persona outputs vs skill + deliberation protocol
681
792
  demo-fictional-deliberation.js Dry-run walkthrough for FlowPay scenario
682
793
  scripts/obfuscate.js Build step → dist/
794
+ scripts/validate-csvs.js CSV integrity validation against schema.json
683
795
  skills/
684
796
  devi/ Host LLM runtime — voices personas from pending prompts
685
797
  kavi/ Kavi — Knowledge Archivist (/kavi)
@@ -699,11 +811,46 @@ skills/
699
811
 
700
812
  - Node.js 16+
701
813
  - Any Agent Skills–compatible host: [Cursor](https://cursor.com), [Claude Code](https://code.claude.com), Codex CLI, [Grok Build](https://x.ai), or Windsurf Cascade
702
- - **CLI `run`:** works without API keys via **`/devi`** host mode (default). Optional keys for automated API runs: `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`
814
+ - **CLI `run`:** works without API keys via **`/devi`** host mode (default). Optional keys for automated API runs: `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`, `GROQ_API_KEY`, `TOGETHER_API_KEY`, `OPENROUTER_API_KEY`, `DEEPSEEK_API_KEY`. Local Ollama auto-detected at `localhost:11434`.
703
815
  - **Kavi `collect` enrichment:** optional — same keys as above; without keys, draft vault + sync still run
704
816
 
705
817
  ---
706
818
 
819
+ ## What's new in v2.0
820
+
821
+ | Feature | Description |
822
+ |---------|-------------|
823
+ | **Chunked execution by default** | Frontier planner → cheap chunk models (Ollama, free/cheap cloud) → synthesis |
824
+ | **Model router** | Auto-discovers Ollama + uses curated free/cheap cloud model pool |
825
+ | **Chunk telemetry** | Tracks per-model success rate so router improves over time |
826
+ | **Legacy mode preserved** | `npx analyzthis_design run-unchunked` for the original single-pass orchestrator |
827
+ | **Planner never cheap** | Planner always uses frontier/host; chunk models are cost-optimized |
828
+ | **Multi-file reference retrieval** | Each persona retrieves from 3-5 CSV files (not 1) via a shared ranker — same LLM cost, 3-5x coverage |
829
+ | **All 16 stacks detected** | `detectStack()` covers all 16 stack CSVs (flutter, swiftui, laravel, threejs, etc.) — was 8 |
830
+ | **CSV schema + validation** | `npm run validate` checks all 28 CSV files against `schema.json` before publish |
831
+ | **Best For Tags** | styles.csv + typography.csv have normalized tag columns for reliable keyword filtering |
832
+
833
+ ## What's new in v1.22
834
+
835
+ | Feature | Description |
836
+ |---------|-------------|
837
+ | **Mood boards** | `/mood-board` collects web references + design-system patterns, tags them, and runs team deliberation |
838
+ | **User-contributed references** | User can add URLs/notes to a board and rerun the critique loop |
839
+ | **Visual direction setting** | Arjun, Meera, Priya, Zara, Noor score references via the UX Honeycomb rigor matrix |
840
+ | **Workspace artifacts** | `moodboard/{boardId}.json` is written into the project for the user to inspect |
841
+
842
+ ## What's new in v1.21
843
+
844
+ | Feature | Description |
845
+ |---------|-------------|
846
+ | **Self-evolving team** | `evolve --extract` harvests lessons + outcomes; proposes prompt/CSV/router patches |
847
+ | **Per-persona knowledge slices** | `sync` builds filtered context per persona (priority categories + fallback) |
848
+ | **Query expansion + ranking** | Cheap LLM calls broaden retrieval and rank references per persona lens |
849
+ | **Lessons store** | Past accepted fixes are retrieved for similar future tasks |
850
+ | **Outcome tracking** | `outcome --confirm` / `--infer` labels whether a persona's output actually shipped |
851
+ | **Full-prompt training export** | `export-training` now emits the exact `{ system, user }` prompts + structured output |
852
+ | **Redundancy suppression** | Personas see what prior personas already covered and add only new insights |
853
+
707
854
  ## What's new in v1.20
708
855
 
709
856
  | Feature | Description |
@@ -0,0 +1,5 @@
1
+ You are the `/chunk-planner` explainer. You describe how the v2.0 chunked execution planner works.
2
+
3
+ You do not execute chunks or replace routing decisions. You help the user inspect and understand chunk plans.
4
+
5
+ When asked about a specific plan, read `session-state.json` and look under `chunk_run.plan`.
@@ -0,0 +1,13 @@
1
+ You are the Mood Board conductor. You do not have a design opinion of your own.
2
+
3
+ Your job is to:
4
+ 1. Collect visual/textual references for a design direction.
5
+ 2. Tag references with style, mood, surface, and relevant keywords.
6
+ 3. Pull matching design-system tokens, styles, and patterns from the reference CSVs.
7
+ 4. Run Arjun, Meera, Priya, Zara, and Noor through adversarial deliberation over the board.
8
+ 5. Write `board.json` to the session and a workspace copy the user can open.
9
+ 6. Accept user-contributed references and rerun the loop.
10
+
11
+ Always assess-only: never write code or final wireframes. Produce the structured board artifact and a synthesis of the team's direction.
12
+
13
+ Cite design-system rows using the format: `[filename, row N: "exact quoted value"]`.
@@ -0,0 +1,17 @@
1
+ # Query Expander (card)
2
+
3
+ You expand a design task into search terms for reference-data retrieval.
4
+
5
+ **Allowed:** Read the task + persona lens; output a JSON array of 5-10 search terms.
6
+ **Forbidden:** Output anything other than the JSON array. No prose, no explanation.
7
+
8
+ **Output format:**
9
+ ```json
10
+ ["term1", "term2", "term3", "term4", "term5"]
11
+ ```
12
+
13
+ Terms should be:
14
+ - Product-type keywords (saas, fintech, e-commerce, dashboard, etc.)
15
+ - Design-domain keywords (contrast, hierarchy, spacing, accessibility, etc.)
16
+ - Component/pattern keywords (modal, table, onboarding, empty state, etc.)
17
+ - Persona-specific lens keywords (e.g. for Arjun: color, typography, gestalt; for Priya: state, API, performance)
@@ -0,0 +1,16 @@
1
+ # Ranker (card)
2
+
3
+ You rank reference candidates for a specific design persona's lens.
4
+
5
+ **Allowed:** Read the task, persona lens, and candidate list; output a JSON array of candidate numbers in priority order (most relevant first).
6
+ **Forbidden:** Output anything other than the JSON array. No prose.
7
+
8
+ **Output format:**
9
+ ```json
10
+ [3, 7, 1, 5, 2]
11
+ ```
12
+
13
+ Rank by:
14
+ - Relevance to the persona's lens (UX, business, feasibility, delight, IA, etc.)
15
+ - Relevance to the specific task
16
+ - Specificity (more specific = higher rank than generic)
@@ -0,0 +1,7 @@
1
+ You are the `/run-unchunked` skill. You invoke the legacy non-chunked orchestrator.
2
+
3
+ Use this only when the user explicitly asks for unchunked mode or a quick single-expert run.
4
+
5
+ Do not replace the default chunked orchestrator. Do not run the chunk planner.
6
+
7
+ Assess-only: do not implement code unless mode is build_approved.
package/agents/chain.json CHANGED
@@ -57,5 +57,20 @@
57
57
  "re_evaluation": {
58
58
  "mode": "delta_only",
59
59
  "description": "On any follow-up after REVISE, re-run only the persona(s) assigned to the prior Top 3 actionable changes — never the full chain. See design-critic Re-evaluation Protocol."
60
+ },
61
+ "evolution": {
62
+ "lessons_dir": "~/.analyzthis_design/lessons",
63
+ "extraction_window_days": 7,
64
+ "min_lessons_for_patch": 5,
65
+ "min_outcomes_for_router_patch": 10,
66
+ "dry_run_default": true
67
+ },
68
+ "chunk_models": {
69
+ "description": "v2.0 chunked execution model pool. Planner is always frontier/strong; chunk models are auto-discovered (Ollama) plus curated cloud free/cheap models.",
70
+ "planner": { "provider": "anthropic", "model": "claude-sonnet-4-20250514" },
71
+ "max_chunks": 6,
72
+ "default_budget": "auto",
73
+ "default_mode": "sequential",
74
+ "allow_cheap_paid": true
60
75
  }
61
76
  }
@@ -24,6 +24,7 @@
24
24
  "handoff_from": [],
25
25
  "handoff_to": ["arjun"],
26
26
  "requires_session_state": true,
27
+ "knowledge_categories": ["design", "tech"],
27
28
  "effort_overrides": {
28
29
  "trivial": { "when": "shortcuts-only audit", "max_output_tokens": 700 },
29
30
  "hard": { "when": "bulk-action state design", "max_output_tokens": 1500 }
@@ -25,6 +25,7 @@
25
25
  "handoff_from": ["noor"],
26
26
  "handoff_to": ["meera"],
27
27
  "requires_session_state": true,
28
+ "knowledge_categories": ["brand", "design", "research"],
28
29
  "hard_gates": ["ds_gate", "information_hierarchy_gate"],
29
30
  "scoped_modes": {
30
31
  "arjun_color_system_only": {
@@ -0,0 +1,18 @@
1
+ {
2
+ "id": "chunk-planner",
3
+ "name": "Chunk Planner",
4
+ "tier": "router",
5
+ "max_output_tokens": 2500,
6
+ "system_card": "agents/cards/chunk-planner.md",
7
+ "system_skill": "skills/chunk-planner/SKILL.md",
8
+ "allowed_jobs": [
9
+ "Explain the chunked execution planner",
10
+ "Inspect chunk plans from session state"
11
+ ],
12
+ "forbidden_jobs": [
13
+ "Replace the router or chain",
14
+ "Execute chunks directly"
15
+ ],
16
+ "hard_gates": [],
17
+ "parallel_safe_with": []
18
+ }
@@ -26,6 +26,7 @@
26
26
  "handoff_from": [],
27
27
  "handoff_to": [],
28
28
  "requires_session_state": true,
29
+ "knowledge_categories": ["prd", "brand", "product", "design", "research", "tech", "web"],
29
30
  "effort_overrides": {
30
31
  "standard": { "when": "enrichment batch (default)", "max_output_tokens": 4000 },
31
32
  "trivial": { "when": "draft-only / --no-enrich (no LLM)", "max_output_tokens": 500 }
@@ -24,6 +24,7 @@
24
24
  "handoff_from": ["arjun"],
25
25
  "handoff_to": ["priya"],
26
26
  "requires_session_state": true,
27
+ "knowledge_categories": ["product", "research"],
27
28
  "effort_overrides": {
28
29
  "trivial": { "when": "single GTM lever / retention check", "max_output_tokens": 500 },
29
30
  "hard": { "when": "north-star tradeoff or cross-metric conflict", "max_output_tokens": 1200 }
@@ -0,0 +1,23 @@
1
+ {
2
+ "id": "mood-board",
3
+ "name": "Mood Board",
4
+ "tier": "structured",
5
+ "max_output_tokens": 1800,
6
+ "system_card": "agents/cards/mood-board.md",
7
+ "system_skill": "skills/mood-board/SKILL.md",
8
+ "allowed_jobs": [
9
+ "collect web references for visual direction",
10
+ "tag references with style, mood, surface",
11
+ "retrieve design-system patterns for the task",
12
+ "orchestrate team deliberation over references",
13
+ "write board.json artifacts",
14
+ "accept user-contributed references and rerun"
15
+ ],
16
+ "forbidden_jobs": [
17
+ "produce final wireframes or code",
18
+ "replace the design-critic chain for screen review",
19
+ "auto-apply design system changes"
20
+ ],
21
+ "hard_gates": [],
22
+ "parallel_safe_with": []
23
+ }
@@ -25,6 +25,7 @@
25
25
  "handoff_from": [],
26
26
  "handoff_to": ["arjun", "anuj"],
27
27
  "requires_session_state": true,
28
+ "knowledge_categories": ["design", "product"],
28
29
  "effort_overrides": {
29
30
  "trivial": { "when": "hierarchy-only declaration", "max_output_tokens": 700 },
30
31
  "hard": { "when": "multi-screen IA", "max_output_tokens": 1500 }
@@ -24,6 +24,7 @@
24
24
  "handoff_from": ["meera"],
25
25
  "handoff_to": ["zara"],
26
26
  "requires_session_state": true,
27
+ "knowledge_categories": ["tech"],
27
28
  "effort_overrides": {
28
29
  "trivial": { "when": "single feature T-shirt sizing (structured extract)", "max_output_tokens": 500 },
29
30
  "hard": { "when": "architecture / state-machine risk", "max_output_tokens": 900 }
@@ -0,0 +1,13 @@
1
+ {
2
+ "id": "query-expander",
3
+ "role": "retrieval_assistant",
4
+ "system_card": "agents/cards/query-expander.md",
5
+ "tier": "structured",
6
+ "max_output_tokens": 200,
7
+ "parallel_safe_with": [],
8
+ "allowed_jobs": ["query expansion for reference retrieval"],
9
+ "forbidden_jobs": ["critique", "wireframe", "design generation"],
10
+ "effort_overrides": {
11
+ "trivial": { "when": "always", "max_output_tokens": 200 }
12
+ }
13
+ }
@@ -23,6 +23,7 @@
23
23
  "handoff_from": ["zara", "arjun", "meera", "priya", "noor", "anuj"],
24
24
  "handoff_to": [],
25
25
  "requires_session_state": true,
26
+ "knowledge_categories": ["prd", "product", "design"],
26
27
  "effort_overrides": {
27
28
  "standard": { "when": "principle ranking (one verdict, never first)", "max_output_tokens": 1200 },
28
29
  "hard": { "when": "cross-persona stalemate", "max_output_tokens": 1200 }
@@ -0,0 +1,13 @@
1
+ {
2
+ "id": "ranker",
3
+ "role": "retrieval_assistant",
4
+ "system_card": "agents/cards/ranker.md",
5
+ "tier": "structured",
6
+ "max_output_tokens": 200,
7
+ "parallel_safe_with": [],
8
+ "allowed_jobs": ["rank reference candidates for persona relevance"],
9
+ "forbidden_jobs": ["critique", "wireframe", "design generation"],
10
+ "effort_overrides": {
11
+ "trivial": { "when": "always", "max_output_tokens": 200 }
12
+ }
13
+ }
@@ -0,0 +1,19 @@
1
+ {
2
+ "id": "run-unchunked",
3
+ "name": "Run Unchunked",
4
+ "tier": "structured",
5
+ "max_output_tokens": 900,
6
+ "system_card": "agents/cards/run-unchunked.md",
7
+ "system_skill": "skills/run-unchunked/SKILL.md",
8
+ "allowed_jobs": [
9
+ "Run the legacy non-chunked orchestrator",
10
+ "Execute a single-pass persona chain",
11
+ "Compare against chunked output"
12
+ ],
13
+ "forbidden_jobs": [
14
+ "Replace the default chunked orchestrator",
15
+ "Run the chunk planner"
16
+ ],
17
+ "hard_gates": [],
18
+ "parallel_safe_with": []
19
+ }
@@ -25,6 +25,7 @@
25
25
  "handoff_from": ["priya"],
26
26
  "handoff_to": [],
27
27
  "requires_session_state": true,
28
+ "knowledge_categories": ["brand", "design"],
28
29
  "effort_overrides": {
29
30
  "standard": { "when": "delight pass (one peak moment)", "max_output_tokens": 1200 },
30
31
  "hard": { "when": "novel interaction design", "max_output_tokens": 1200 }
@@ -67,6 +67,13 @@
67
67
  "route_to": ["raj"],
68
68
  "never_route_to": []
69
69
  },
70
+ {
71
+ "problem_type": "chunked_run",
72
+ "signals": ["chunked", "chunk planner", "run-chunked", "cheap models", "free models", "ollama"],
73
+ "route_to": ["chunk-planner"],
74
+ "never_route_to": ["design-critic_chain"],
75
+ "notes": "Explain or inspect the v2.0 chunked execution planner."
76
+ },
70
77
  {
71
78
  "problem_type": "full_screen_review",
72
79
  "signals": ["full screen review", "critique this screen", "review this page"],
@@ -80,6 +87,13 @@
80
87
  "route_to": ["kavi"],
81
88
  "never_route_to": ["design-critic_chain", "zara", "arjun", "meera", "priya"],
82
89
  "notes": "Producer only — run /kavi or `npx analyzthis_design collect`. Never part of a critique chain."
90
+ },
91
+ {
92
+ "problem_type": "mood_board",
93
+ "signals": ["mood board", "mood-board", "/mood-board", "visual direction", "references", "inspiration", "style direction", "design direction"],
94
+ "route_to": ["mood-board"],
95
+ "never_route_to": ["design-critic_chain", "zara", "noor_alone"],
96
+ "notes": "Visual direction setting — collect references, tag them, pull DS patterns, run team deliberation. Not a screen critique."
83
97
  }
84
98
  ]
85
99
  }