@imfusion/web-ui 0.6.4-dev.50.g71d59d59 → 0.6.4-dev.55.g47fb2f84

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,2 @@
1
+ #!/usr/bin/env -S npx tsx
2
+ export {};
@@ -0,0 +1,12 @@
1
+ import { Benchmark, RunResult } from './types';
2
+ type AggregateOptions = {
3
+ skillFamily: string;
4
+ runsPerScenario: number;
5
+ executorModel: string;
6
+ graderModel: string;
7
+ graderVotes: number;
8
+ tier: string;
9
+ };
10
+ export declare function aggregateBenchmark(runs: RunResult[], options: AggregateOptions): Benchmark;
11
+ export declare function mergeBenchmarkUpdate(previous: Benchmark | undefined, update: Benchmark, updatedEvalIds: string[]): Benchmark;
12
+ export {};
@@ -0,0 +1,2 @@
1
+ #!/usr/bin/env -S npx tsx
2
+ export {};
@@ -0,0 +1,29 @@
1
+ export declare const EVALS_DIR: string;
2
+ export declare const NO_SKILLS: boolean;
3
+ export declare const EVAL_ROOT: string;
4
+ export declare const TEMPLATE: string;
5
+ export declare const CODEX_HOME_DIR: string;
6
+ export declare const RESULTS_DIR: string;
7
+ export declare const REPORT_PATH: string;
8
+ export declare const SCENARIOS_DIR: string;
9
+ export declare const FIXTURES_DIR: string;
10
+ export declare const SKILLS_DIR: string;
11
+ export declare const SKILL_FAMILY: string;
12
+ export declare const WORKSPACE_LABEL: string;
13
+ export declare const DEFAULT_FIXTURE = "fresh-install";
14
+ export declare const ROUND_TIMEOUT_MS = 600000;
15
+ export declare const MAX_TURNS = 60;
16
+ export declare const HOST_PLANS_DIR: string;
17
+ type TierName = string;
18
+ type Tier = {
19
+ runs: number;
20
+ graderVotes: number;
21
+ };
22
+ type EvalConfig = {
23
+ executorModel?: string;
24
+ graderModel?: string;
25
+ defaultTier?: TierName;
26
+ tiers?: Record<TierName, Tier>;
27
+ };
28
+ export declare const CONFIG: EvalConfig;
29
+ export {};
@@ -0,0 +1,7 @@
1
+ import { HostPlan, ToolUse, TranscriptEntry } from './types.ts';
2
+ export declare function isHostPlanPath(path: string): boolean;
3
+ export declare function extractToolUses(transcript: TranscriptEntry[]): ToolUse[];
4
+ export declare function extractSkillLoads(toolUses: ToolUse[]): string[];
5
+ export declare function findHostPlans(toolUses: ToolUse[]): HostPlan[];
6
+ export declare function renderConversation(transcript: TranscriptEntry[], maxChars?: number): string;
7
+ export declare function renderWorkspaceFiles(workdir: string, hostPlans?: HostPlan[], maxChars?: number): string;
@@ -0,0 +1,32 @@
1
+ import { TranscriptEntry } from './types.ts';
2
+ export type RoundOutput = {
3
+ session_id: string | null;
4
+ result?: string;
5
+ duration_ms?: number;
6
+ total_cost_usd?: number;
7
+ usage?: {
8
+ input_tokens?: number;
9
+ output_tokens?: number;
10
+ };
11
+ modelUsage?: Record<string, unknown>;
12
+ subtype?: string;
13
+ num_turns?: number;
14
+ timedOut?: boolean;
15
+ timeoutMs?: number;
16
+ };
17
+ export declare function runClaude(prompt: string, { cwd, resumeSessionId, maxTurns, model, permissionMode, timeoutMs }: {
18
+ cwd: string;
19
+ resumeSessionId?: string | null;
20
+ maxTurns?: number;
21
+ model?: string | null;
22
+ permissionMode?: string;
23
+ timeoutMs?: number;
24
+ }): RoundOutput;
25
+ export declare function runCodex(prompt: string, { cwd, resumeSessionId }: {
26
+ cwd: string;
27
+ resumeSessionId?: string | null;
28
+ }): RoundOutput;
29
+ export declare function parseJsonl(raw: string): Record<string, unknown>[];
30
+ export declare function readTranscript(workdir: string, sessionId: string): string;
31
+ export declare function findCodexRollout(threadId: string): string;
32
+ export declare function codexEntries(lines: Record<string, unknown>[]): TranscriptEntry[];
@@ -0,0 +1,10 @@
1
+ import { AssertionResult, Scenario, ToolUse } from './types.ts';
2
+ export declare function gradeRouting(label: "Skill" | "Tool", config: {
3
+ expect?: string[];
4
+ forbid?: string[];
5
+ } | undefined, actual: string[]): AssertionResult[];
6
+ export declare function gradeTranscript(scenario: Scenario, rawTranscript: string): AssertionResult[];
7
+ export declare function gradeWorkspace(scenario: Scenario, workdir: string): AssertionResult[];
8
+ export declare function gradeWrites(scenario: Scenario, toolUses: ToolUse[], workdir: string): AssertionResult[];
9
+ export declare function gradeReport(scenario: Scenario, workdir: string, initialReport: string | null): AssertionResult[];
10
+ export declare function gradeWithLlm(scenario: Scenario, conversation: string, workspaceFiles: string, voteCount: number): AssertionResult[];
@@ -0,0 +1,2 @@
1
+ #!/usr/bin/env -S npx tsx
2
+ export declare function writeReportMarkdown(): string | null;
@@ -0,0 +1,4 @@
1
+ import { Benchmark } from './types.ts';
2
+ export declare function printReport(benchmark: Benchmark, { allEvidence }?: {
3
+ allEvidence?: boolean | undefined;
4
+ }): void;
@@ -0,0 +1,2 @@
1
+ #!/usr/bin/env -S npx tsx
2
+ export {};
@@ -0,0 +1,3 @@
1
+ import { Host, RunResult, Scenario } from './types.ts';
2
+ export declare function evalIdFor(scenario: Scenario, host: Host): string;
3
+ export declare function runScenario(scenario: Scenario, host: Host, runNumber: number, runsRoot: string, model: string | null, graderVotes: number): RunResult;
@@ -0,0 +1,140 @@
1
+ export declare const HOSTS: readonly ["claude", "codex"];
2
+ export type Host = (typeof HOSTS)[number];
3
+ export type PathRule = {
4
+ pattern: string;
5
+ text: string;
6
+ flags?: string;
7
+ /** Named paths to search instead of the `src/**` default. */
8
+ files?: string[];
9
+ };
10
+ export type Scenario = {
11
+ id: string;
12
+ description: string;
13
+ rounds: string[];
14
+ /** Skills this scenario belongs to without asserting they loaded. */
15
+ covers?: string[];
16
+ skills?: {
17
+ expect?: string[];
18
+ forbid?: string[];
19
+ };
20
+ tools?: {
21
+ expect?: string[];
22
+ forbid?: string[];
23
+ };
24
+ writes?: {
25
+ only?: string[];
26
+ expect?: string[];
27
+ };
28
+ workspace?: {
29
+ forbidPatterns?: PathRule[];
30
+ expectPatterns?: PathRule[];
31
+ };
32
+ transcript?: {
33
+ expectStrings?: string[];
34
+ forbidStrings?: string[];
35
+ };
36
+ report?: {
37
+ path: string;
38
+ headings?: string[];
39
+ preserveFrom?: string;
40
+ };
41
+ assertions?: string[];
42
+ fixture?: string;
43
+ setup?: string;
44
+ hosts?: Host[];
45
+ maxTurns?: number;
46
+ roundTimeoutMinutes?: number;
47
+ permissionMode?: string;
48
+ };
49
+ export type AssertionResult = {
50
+ text: string;
51
+ passed: boolean;
52
+ method: "deterministic" | "llm";
53
+ evidence: string;
54
+ };
55
+ export type ToolUse = {
56
+ type: "tool_use";
57
+ name: string;
58
+ input?: Record<string, unknown>;
59
+ };
60
+ export type TranscriptEntry = {
61
+ type: string;
62
+ message?: {
63
+ content?: unknown;
64
+ };
65
+ };
66
+ export type HostPlan = {
67
+ path: string;
68
+ content: string;
69
+ };
70
+ export type RoundMeta = {
71
+ prompt: string;
72
+ session_id: string | null;
73
+ duration_ms?: number;
74
+ total_cost_usd?: number;
75
+ usage?: {
76
+ input_tokens?: number;
77
+ output_tokens?: number;
78
+ };
79
+ models?: string[];
80
+ subtype?: string;
81
+ num_turns?: number;
82
+ timed_out?: boolean;
83
+ };
84
+ export type RunResult = {
85
+ eval_id: string;
86
+ executor_host: Host;
87
+ fixture: string;
88
+ description: string;
89
+ run_number: number;
90
+ models: string[];
91
+ result: {
92
+ pass_rate: number;
93
+ passed: number;
94
+ failed: number;
95
+ total: number;
96
+ time_seconds: number;
97
+ turns: number;
98
+ tokens: number;
99
+ cost_usd: number;
100
+ };
101
+ expectations: {
102
+ text: string;
103
+ passed: boolean;
104
+ evidence: string;
105
+ }[];
106
+ truncated: boolean;
107
+ timed_out: boolean;
108
+ notes: string;
109
+ };
110
+ export type PerEvalSummary = {
111
+ pass_rate_mean: number;
112
+ pass_rate_stddev: number;
113
+ time_seconds_mean: number;
114
+ turns_mean: number;
115
+ tokens_mean: number;
116
+ cost_usd_total: number;
117
+ };
118
+ export type Benchmark = {
119
+ metadata: BenchmarkMetadata;
120
+ runs: RunResult[];
121
+ run_summary: {
122
+ per_eval: Record<string, PerEvalSummary>;
123
+ overall_pass_rate: number;
124
+ total_cost_usd: number;
125
+ };
126
+ notes: string[];
127
+ };
128
+ export type BenchmarkMetadata = {
129
+ skill_family: string;
130
+ timestamp: string;
131
+ evals_run: string[];
132
+ runs_per_scenario: number;
133
+ executor_model: string;
134
+ executor_hosts: string[];
135
+ grader_model: string;
136
+ grader_votes: number;
137
+ tier: string;
138
+ models_observed: string[];
139
+ fixtures: string[];
140
+ };
@@ -0,0 +1,2 @@
1
+ export declare function copyTemplate(dest: string): void;
2
+ export declare function applyFixture(dest: string, fixture: string): void;
@@ -0,0 +1 @@
1
+ export declare function writeGenerated(path: string, contents: string): string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@imfusion/web-ui",
3
- "version": "0.6.4-dev.50.g71d59d59",
3
+ "version": "0.6.4-dev.55.g47fb2f84",
4
4
  "description": "The official Web UI component library for ImFusion web apps",
5
5
  "author": "ImFusion GmbH",
6
6
  "homepage": "https://imfusion.com",
@@ -90,10 +90,11 @@
90
90
  "test:unit": "vitest run --project unit",
91
91
  "test:stories": "vitest run --project storybook",
92
92
  "test:watch": "vitest",
93
- "skills:eval": "node src/llms/evals/run.mjs",
94
- "skills:eval:dev": "WEB_UI_SKILL_EVAL_ROOT=/private/tmp/web-ui-dev-skill-evals WEB_UI_SKILL_EVAL_RESULTS_DIR=.agents/evals/results WEB_UI_SKILL_EVAL_REPORT_PATH=.agents/evals/REPORT.md WEB_UI_SKILL_EVAL_SKILL_FAMILY=web-ui-dev WEB_UI_SKILL_EVAL_WORKSPACE_LABEL='development repository' node src/llms/evals/run.mjs --scenarios-dir .agents/evals/scenarios --setup .agents/evals/setup-env.sh",
95
- "skills:eval:compare": "node src/llms/evals/compare.mjs",
96
- "skills:eval:report": "node src/llms/evals/report.mjs",
93
+ "skills:eval": "tsx src/llms/evals/runner/run.ts",
94
+ "skills:eval:dev": "WEB_UI_SKILL_EVAL_ROOT=/private/tmp/web-ui-dev-skill-evals WEB_UI_SKILL_EVAL_RESULTS_DIR=.agents/evals/results WEB_UI_SKILL_EVAL_REPORT_PATH=.agents/evals/REPORT.md WEB_UI_SKILL_EVAL_SKILL_FAMILY=web-ui-dev WEB_UI_SKILL_EVAL_WORKSPACE_LABEL='development repository' tsx src/llms/evals/runner/run.ts --scenarios-dir .agents/evals/scenarios --setup .agents/evals/setup-env.sh",
95
+ "skills:eval:compare": "tsx src/llms/evals/runner/compare.ts",
96
+ "skills:eval:affected": "tsx src/llms/evals/runner/affected.ts",
97
+ "skills:eval:report": "tsx src/llms/evals/runner/report.ts",
97
98
  "git:config": "git config core.hooksPath .githooks && git config pull.rebase true && git config merge.ff only"
98
99
  },
99
100
  "peerDependencies": {
@@ -13,27 +13,30 @@ copy of every convention.
13
13
 
14
14
  ## 1. Decide whether guidance is needed
15
15
 
16
- Skip a companion when the task is explicit and small, an existing local pattern already solves it, and the library usage is
17
- already correct. Do the work. Look up an API silently only when you are unsure.
16
+ Skip a companion when the task is explicit and small, an existing local pattern already solves it in the file being changed,
17
+ and the library usage is already correct. Do the work. Look up an API silently only when you are unsure.
18
18
 
19
- Use a companion when the task involves a choice, a missing setup piece, a new file, or a library convention the project may
20
- not already have.
19
+ Use a companion when the task involves a choice, a missing setup piece, a new UI file, or a library convention the project
20
+ may not already have. Open a companion by invoking its skill; reading one of its files directly does not load its workflow.
21
21
 
22
22
  ## 2. Route the task
23
23
 
24
- | Task | Open |
25
- | ----------------------------------------------------------------- | ------------------------------------------------------ |
26
- | Understand library setup, brand assets, theming, or usage | Packaged user guides below |
27
- | Look up a component, part, prop, default, or icon | `imf-web-ui-components` |
28
- | Choose components or shape a screen or flow | `imf-web-ui-ux` |
29
- | Write or update documentation | `/documentation-writer`, then `imf-web-ui-conventions` |
30
- | Write a wrapper, custom UI, CSS, data layer, validation, or tests | `imf-web-ui-conventions` |
31
- | Install the library or bootstrap project tooling | `imf-web-ui-setup` |
32
- | Inspect an existing project without changing it | `imf-web-ui-audit` |
33
- | Update the package, skills, or hooks | `imf-web-ui-update` |
34
-
35
- A screen often needs both `imf-web-ui-ux` and `imf-web-ui-components`, in that order. Setup and audit are for project-wide
36
- questions, not every one-file edit.
24
+ | Task | Open |
25
+ | -------------------------------------------------------------------- | ------------------------------------------------------ |
26
+ | Understand library setup, brand assets, theming, or usage | Packaged user guides below |
27
+ | Look up a component, part, prop, default, or icon | `imf-web-ui-components` |
28
+ | Choose components or shape a screen, form, or flow | `imf-web-ui-ux` |
29
+ | Write or update documentation | `/documentation-writer`, then `imf-web-ui-conventions` |
30
+ | Write a wrapper, custom UI, CSS, data layer, validation, or tests | `imf-web-ui-conventions` |
31
+ | Choose a routing, data, form, or validation library for the project | `imf-web-ui-conventions` |
32
+ | Install the library or bootstrap project tooling | `imf-web-ui-setup` |
33
+ | Diagnose a broken install: unstyled output, missing tokens, no theme | `imf-web-ui-setup` |
34
+ | Inspect an existing project without changing it | `imf-web-ui-audit` |
35
+ | Update the package, skills, or hooks | `imf-web-ui-update` |
36
+
37
+ A new form opens `imf-web-ui-ux`, even when its fields and action are already specified. A screen often needs both
38
+ `imf-web-ui-ux` and `imf-web-ui-components`, in that order. Setup and audit are for project-wide questions, not every
39
+ one-file edit.
37
40
 
38
41
  ### Read a packaged user guide
39
42
 
@@ -58,8 +61,9 @@ props and the conventions `tokens` topic for exact token names and defaults.
58
61
  The host project's existing conventions win. The companion skills fill gaps; they do not justify refactoring a working
59
62
  styling system, state library, or folder structure.
60
63
 
61
- When a task needs TanStack Router, Query, Form, Table, or Store and the project has no incumbent, propose the matching
62
- library and read its current documentation with `npx @tanstack/cli` before using it.
64
+ When a task needs TanStack Router, Query, Form, Table, or Store and the project has no incumbent, open
65
+ `imf-web-ui-conventions` for the house position on that layer, then read the library's current documentation with
66
+ `npx @tanstack/cli` before using it. The recommendation lives in the conventions topics, not here.
63
67
 
64
68
  ## 4. Ask only when the choice depends on missing context
65
69
 
@@ -10,22 +10,35 @@ allowed-tools: Read Glob Grep
10
10
 
11
11
  # Audit a consumer project
12
12
 
13
- This is a read-only audit. Compare the project with the selected `imf-web-ui-conventions` topics, report evidence, and end
14
- with work the human can approve. The baseline is an ImFusion default, not a universal law; a deliberate project choice is a
15
- deviation, not a defect.
13
+ This is a read-only audit. Its job is to align a project with its conventions: compare the project against the selected
14
+ `imf-web-ui-conventions` topics, report evidence, and end with work the human can approve.
15
+
16
+ A gap against a convention is high severity. The project agreed to those conventions, so the finding carries no argument
17
+ about whether it matters, and a deliberate reason to differ is the only thing that lowers it. Say what that reason is.
18
+
19
+ Anything else you notice belongs under `Suggestions`: what running the code revealed, what an investigation turned up, what a
20
+ reader would tidy on the way past. Report it, keep it low, and do not let it crowd the sections a consumer has to act on.
16
21
 
17
22
  ## Workflow
18
23
 
19
24
  1. Resolve the argument. Bare means `full`; an unknown topic is an error, not a reason to widen the scope.
20
25
  2. Enter plan mode unless this audit is being called as verification by another skill or the host has no plan mode.
21
- 3. Assign one investigator to each in-scope topic. An investigator checks every section in that topic and writes nothing. If
22
- the host cannot dispatch agents, inspect the topics inline in the same order.
26
+ 3. Read the report template and existing reviewer notes, then dispatch one investigator per in-scope topic. Tell each
27
+ investigator to use only `Read`, `Glob`, and `Grep`, with no shell commands or writes. An investigator checks every
28
+ section in that topic and writes nothing. Once dispatch starts, the host performs no repository inspection: the returned
29
+ reports are the evidence, and the host merges them without reading or searching the files again. Gathering the evidence
30
+ twice spends context on the second pass and reports what the host still remembers. Inspect a topic inline only after its
31
+ dispatch has been tried and refused, and say which attempt failed.
23
32
  4. Merge the evidence into the shared report format in
24
- [`templates/REPORT.md`](../imf-web-ui-conventions/templates/REPORT.md).
25
- 5. Put each finding's `Next action` into an ordered plan, grouped by topic and cheapest first. Present the report and wait
26
- for approval; an audit does not edit the project.
33
+ [`templates/REPORT.md`](../imf-web-ui-conventions/templates/REPORT.md). Before presenting or writing it, check that every
34
+ Broken, Missing, Deviation, and Unverified entry has a severity, evidence, impact, and next action; do not compact
35
+ Unverified entries.
36
+ 5. Put each finding's `Next action` into an ordered plan, grouped by topic and cheapest first. The plan is separate from the
37
+ report contract: present it in the host plan or response, never as another heading in `AUDIT_REPORT.md`. Present the
38
+ report and wait for approval; an audit does not edit the project.
27
39
 
28
- If the caller needs a durable report, write `AUDIT_REPORT.md` and preserve everything under `## Reviewer notes` verbatim.
40
+ If the caller needs a durable report, write `AUDIT_REPORT.md` with exactly the template's H2 headings and preserve everything
41
+ under `## Reviewer notes` verbatim.
29
42
 
30
43
  ## Finding format
31
44
 
@@ -37,8 +50,9 @@ Classify every result as one of these:
37
50
  - **Present**: the rule is met, with evidence.
38
51
  - **Unverified**: static inspection cannot establish it.
39
52
 
40
- Use the report contract's severity, `path:line` evidence, concrete impact, and smallest next action. Do not promote an
41
- optional tool or a working alternative to a missing finding.
53
+ Use the report contract's severity, `path:line` evidence, concrete impact, and smallest next action. Unverified entries use
54
+ that same structure: unavailable runtime evidence, the impact of the unknown, and the check that resolves it. Do not promote
55
+ an optional tool or a working alternative to a missing finding.
42
56
 
43
57
  ## Topics
44
58
 
@@ -87,7 +87,7 @@ A component whose identity entry says it comes from `@imfusion/web-ui/integratio
87
87
  listed optional peer explicitly, then import from that path:
88
88
 
89
89
  ```tsx
90
- import { Code } from "@imfusion/web-ui/integrations/code-highlight";
90
+ import { ImageDisplayOptions } from "@imfusion/web-ui/integrations/image-display-options";
91
91
  ```
92
92
 
93
93
  ## Data grids
@@ -12,8 +12,23 @@ Broken, Missing, Deviations, and Unverified entries use:
12
12
  - Impact: concrete consequence
13
13
  - Next action: smallest selectable follow-up
14
14
 
15
- Present entries name the topic and evidence path. Deviations are selectable follow-up work, not defects. Optional tools are not
16
- missing findings.
15
+ Present entries name the topic and evidence path. Optional tools are not missing findings.
16
+
17
+ Where a finding belongs follows from where it came from, not from how strongly you hold it.
18
+
19
+ A convention names it:
20
+ - Broken: the convention is followed but does not work. Something is unmounted, unwired, or fails at runtime.
21
+ - Missing: a convention expects it and the project has nothing there.
22
+ - Deviations: it works, and differs from what a convention says.
23
+
24
+ These three are high severity by default. The project agreed to the conventions; a gap against them is not a preference.
25
+ Drop the severity only when the project shows a deliberate reason to differ, and say what the reason is.
26
+
27
+ Nothing names it:
28
+ - Suggestions: real improvements that no convention asks for. What running the code revealed, what an investigation turned
29
+ up, what a reader would tidy on their way past. Never high severity, and never a reason to hold up the work.
30
+
31
+ - Unverified: it could not be checked from the files available.
17
32
  -->
18
33
 
19
34
  ## Verdict
@@ -32,6 +47,10 @@ None.
32
47
 
33
48
  None.
34
49
 
50
+ ## Suggestions
51
+
52
+ None.
53
+
35
54
  ## Present
36
55
 
37
56
  None.
@@ -85,7 +85,9 @@ export function getDetails(options?: QueryOptions<User>) {
85
85
  }
86
86
  ```
87
87
 
88
- Callers choose `ensureQueryData`, `useSuspenseQuery`, or another Query API.
88
+ Callers choose `ensureQueryData`, `useSuspenseQuery`, or another Query API. Mutation factories likewise return
89
+ `mutationOptions`, own their mutation key and parsing, and are passed to `useMutation`; do not inline those options at the
90
+ call site.
89
91
 
90
92
  ## Router context access
91
93
 
@@ -8,5 +8,7 @@ Read:
8
8
  node_modules/@imfusion/web-ui/src/llms/tokens.gen.json
9
9
  ```
10
10
 
11
- It contains the shipped token names and authored defaults. Look up the exact entry instead of guessing a plausible
12
- `--imf-ui-*` name or copying a token list into the project.
11
+ It contains the shipped token names and authored defaults. Before writing CSS, check every concrete `--imf-ui-*` name the
12
+ file will use against that index or the relevant component stylesheet. A family with one entry does not imply numbered
13
+ variants. When no matching token exists, use an existing documented seam or say that the token is absent instead of guessing
14
+ a plausible name.
@@ -5,7 +5,7 @@ description:
5
5
  needs library wiring, tooling, git, npm, authentication, structure, data, testing, docs, or agent tooling. Inspect first,
6
6
  propose concrete files, and write only what the user approves. Use imf-web-ui-audit for read-only checks."
7
7
  argument-hint: "[full|library-setup|tooling|git|npm-project|authentication|project-structure|docs-structure|data|testing|agent-tooling]"
8
- allowed-tools: Read Glob Grep Write
8
+ allowed-tools: Read Glob Grep Write Skill
9
9
  ---
10
10
 
11
11
  # Set up a consumer project
@@ -15,16 +15,25 @@ working choices.
15
15
 
16
16
  ## Workflow
17
17
 
18
- 1. Resolve the argument. Bare means `full`. A named topic limits the assessment to that topic. If the argument is unknown,
19
- list the supported topics and point to `imf-web-ui-audit`.
18
+ 1. Resolve the scope. A named topic limits the assessment to that topic. A bare invocation means `full`, unless the request
19
+ is about one topic, in which case that topic is the scope: answer what was asked and offer the wider sweep rather than
20
+ running it. If the argument is unknown, list the supported topics and point to `imf-web-ui-audit`.
20
21
  2. Enter the host's plan mode before inspecting or proposing changes.
21
22
  3. Read the selected convention topics and inspect the repository with `Read`, `Glob`, and `Grep` only.
22
23
  4. Put a complete proposal in the host plan, using [`templates/REPORT.md`](../imf-web-ui-conventions/templates/REPORT.md).
23
24
  For new or updated documentation, invoke `/documentation-writer` before writing and use the `docs-structure` topic for
24
25
  repository-specific boundaries. Name every file to create or change, include its intended content, and cite the repository
25
26
  evidence behind the choice.
26
- 5. Wait for explicit approval. After approval, write only the listed files.
27
- 6. Run `imf-web-ui-audit full` as a follow-up and review its findings before calling setup complete.
27
+ 5. Wait for explicit approval. After approval, write only the listed files. Before calling setup complete, ensure every
28
+ package imported by those files and every tool named by their scripts is declared in `package.json`.
29
+ 6. Offer `imf-web-ui-audit full` as an optional next step for checking the wider project, and run it only if the user says
30
+ yes. Setup is complete once the approved files are written; the audit reviews ground setup did not touch.
31
+
32
+ ## Audit handoff
33
+
34
+ When the approved request includes an audit, the next tool call after the final setup write invokes `imf-web-ui-audit`
35
+ through the host's skill tool. Do not read audit files, inspect the project again, or dispatch audit investigators first;
36
+ reproducing the audit workflow does not load the skill.
28
37
 
29
38
  ## Greenfield `full` setup
30
39
 
@@ -20,8 +20,10 @@ Never overwrite user changes or commit without fresh approval.
20
20
  if another lockfile is the active one.
21
21
  3. Record `git status --short`, the installed Web UI version, the installed skill targets, hooks, registrations, and
22
22
  `AGENTS.md` fence.
23
- 4. Check for another npm process and stop on a conflict. A dirty file this workflow needs is also a conflict. Do not reset,
24
- clean, restore, or hide it.
23
+ 4. Check for another npm process and stop on a conflict. Uncommitted work is also a conflict: report it and stop before the
24
+ first mutating command, rather than deciding it falls outside this workflow's paths. What the update touches is not
25
+ knowable until it runs, and the person can commit, stash, or wave it through in one reply. Never reset, clean, restore, or
26
+ hide their work.
25
27
  5. With `--dry-run`, stop after reporting what would happen. Do not install, format, stage, commit, or ask for commit
26
28
  approval.
27
29
 
@@ -15,7 +15,9 @@ when it is used across many call sites so API movement stays local.
15
15
 
16
16
  ## Ask only what matters
17
17
 
18
- If the request is vague and a human can answer, ask these questions one at a time and stop once the choice is clear:
18
+ When the request is vague and a human is there to answer, the first reply is a question and nothing else. No default, no
19
+ options table, no recommendation held in reserve: any of those end the conversation, because the person now has an answer and
20
+ no reason to reply. Ask one at a time, and stop once the choice is clear:
19
21
 
20
22
  1. Who uses the screen, and how often?
21
23
  2. What is its one primary action?
@@ -23,8 +25,8 @@ If the request is vague and a human can answer, ask these questions one at a tim
23
25
  4. What should users see when it is empty, loading, or failing?
24
26
  5. Is it a full page or part of another flow?
25
27
 
26
- If the prompt already answers a question, do not ask it again. If nobody can answer, make the conservative choice and state
27
- the assumption.
28
+ If the prompt already answers a question, do not ask it again. Only when nobody is there to answer, or a question has gone
29
+ unanswered, make the conservative choice and state the assumption.
28
30
 
29
31
  ## Choose a surface
30
32