@imfusion/web-ui 0.6.4-dev.51.gf38b0107 → 0.6.4-dev.57.g664ba66a
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/llms/evals/runner/affected.d.ts +2 -0
- package/dist/llms/evals/runner/benchmark.d.ts +12 -0
- package/dist/llms/evals/runner/compare.d.ts +2 -0
- package/dist/llms/evals/runner/config.d.ts +29 -0
- package/dist/llms/evals/runner/evidence.d.ts +7 -0
- package/dist/llms/evals/runner/execute.d.ts +32 -0
- package/dist/llms/evals/runner/grade.d.ts +10 -0
- package/dist/llms/evals/runner/report-md.d.ts +2 -0
- package/dist/llms/evals/runner/report.d.ts +4 -0
- package/dist/llms/evals/runner/run.d.ts +2 -0
- package/dist/llms/evals/runner/scenario.d.ts +3 -0
- package/dist/llms/evals/runner/types.d.ts +140 -0
- package/dist/llms/evals/runner/workspace.d.ts +2 -0
- package/dist/llms/evals/runner/write-generated.d.ts +1 -0
- package/package.json +6 -5
- package/src/llms/skills/imf-web-ui/SKILL.md +23 -19
- package/src/llms/skills/imf-web-ui-audit/SKILL.md +25 -11
- package/src/llms/skills/imf-web-ui-components/SKILL.md +1 -1
- package/src/llms/skills/imf-web-ui-conventions/templates/REPORT.md +21 -2
- package/src/llms/skills/imf-web-ui-conventions/topics/data.md +3 -1
- package/src/llms/skills/imf-web-ui-conventions/topics/tokens.md +4 -2
- package/src/llms/skills/imf-web-ui-setup/SKILL.md +14 -5
- package/src/llms/skills/imf-web-ui-update/SKILL.md +4 -2
- package/src/llms/skills/imf-web-ui-ux/SKILL.md +5 -3
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { Benchmark, RunResult } from './types';
|
|
2
|
+
type AggregateOptions = {
|
|
3
|
+
skillFamily: string;
|
|
4
|
+
runsPerScenario: number;
|
|
5
|
+
executorModel: string;
|
|
6
|
+
graderModel: string;
|
|
7
|
+
graderVotes: number;
|
|
8
|
+
tier: string;
|
|
9
|
+
};
|
|
10
|
+
export declare function aggregateBenchmark(runs: RunResult[], options: AggregateOptions): Benchmark;
|
|
11
|
+
export declare function mergeBenchmarkUpdate(previous: Benchmark | undefined, update: Benchmark, updatedEvalIds: string[]): Benchmark;
|
|
12
|
+
export {};
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
export declare const EVALS_DIR: string;
|
|
2
|
+
export declare const NO_SKILLS: boolean;
|
|
3
|
+
export declare const EVAL_ROOT: string;
|
|
4
|
+
export declare const TEMPLATE: string;
|
|
5
|
+
export declare const CODEX_HOME_DIR: string;
|
|
6
|
+
export declare const RESULTS_DIR: string;
|
|
7
|
+
export declare const REPORT_PATH: string;
|
|
8
|
+
export declare const SCENARIOS_DIR: string;
|
|
9
|
+
export declare const FIXTURES_DIR: string;
|
|
10
|
+
export declare const SKILLS_DIR: string;
|
|
11
|
+
export declare const SKILL_FAMILY: string;
|
|
12
|
+
export declare const WORKSPACE_LABEL: string;
|
|
13
|
+
export declare const DEFAULT_FIXTURE = "fresh-install";
|
|
14
|
+
export declare const ROUND_TIMEOUT_MS = 600000;
|
|
15
|
+
export declare const MAX_TURNS = 60;
|
|
16
|
+
export declare const HOST_PLANS_DIR: string;
|
|
17
|
+
type TierName = string;
|
|
18
|
+
type Tier = {
|
|
19
|
+
runs: number;
|
|
20
|
+
graderVotes: number;
|
|
21
|
+
};
|
|
22
|
+
type EvalConfig = {
|
|
23
|
+
executorModel?: string;
|
|
24
|
+
graderModel?: string;
|
|
25
|
+
defaultTier?: TierName;
|
|
26
|
+
tiers?: Record<TierName, Tier>;
|
|
27
|
+
};
|
|
28
|
+
export declare const CONFIG: EvalConfig;
|
|
29
|
+
export {};
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { HostPlan, ToolUse, TranscriptEntry } from './types.ts';
|
|
2
|
+
export declare function isHostPlanPath(path: string): boolean;
|
|
3
|
+
export declare function extractToolUses(transcript: TranscriptEntry[]): ToolUse[];
|
|
4
|
+
export declare function extractSkillLoads(toolUses: ToolUse[]): string[];
|
|
5
|
+
export declare function findHostPlans(toolUses: ToolUse[]): HostPlan[];
|
|
6
|
+
export declare function renderConversation(transcript: TranscriptEntry[], maxChars?: number): string;
|
|
7
|
+
export declare function renderWorkspaceFiles(workdir: string, hostPlans?: HostPlan[], maxChars?: number): string;
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { TranscriptEntry } from './types.ts';
|
|
2
|
+
export type RoundOutput = {
|
|
3
|
+
session_id: string | null;
|
|
4
|
+
result?: string;
|
|
5
|
+
duration_ms?: number;
|
|
6
|
+
total_cost_usd?: number;
|
|
7
|
+
usage?: {
|
|
8
|
+
input_tokens?: number;
|
|
9
|
+
output_tokens?: number;
|
|
10
|
+
};
|
|
11
|
+
modelUsage?: Record<string, unknown>;
|
|
12
|
+
subtype?: string;
|
|
13
|
+
num_turns?: number;
|
|
14
|
+
timedOut?: boolean;
|
|
15
|
+
timeoutMs?: number;
|
|
16
|
+
};
|
|
17
|
+
export declare function runClaude(prompt: string, { cwd, resumeSessionId, maxTurns, model, permissionMode, timeoutMs }: {
|
|
18
|
+
cwd: string;
|
|
19
|
+
resumeSessionId?: string | null;
|
|
20
|
+
maxTurns?: number;
|
|
21
|
+
model?: string | null;
|
|
22
|
+
permissionMode?: string;
|
|
23
|
+
timeoutMs?: number;
|
|
24
|
+
}): RoundOutput;
|
|
25
|
+
export declare function runCodex(prompt: string, { cwd, resumeSessionId }: {
|
|
26
|
+
cwd: string;
|
|
27
|
+
resumeSessionId?: string | null;
|
|
28
|
+
}): RoundOutput;
|
|
29
|
+
export declare function parseJsonl(raw: string): Record<string, unknown>[];
|
|
30
|
+
export declare function readTranscript(workdir: string, sessionId: string): string;
|
|
31
|
+
export declare function findCodexRollout(threadId: string): string;
|
|
32
|
+
export declare function codexEntries(lines: Record<string, unknown>[]): TranscriptEntry[];
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { AssertionResult, Scenario, ToolUse } from './types.ts';
|
|
2
|
+
export declare function gradeRouting(label: "Skill" | "Tool", config: {
|
|
3
|
+
expect?: string[];
|
|
4
|
+
forbid?: string[];
|
|
5
|
+
} | undefined, actual: string[]): AssertionResult[];
|
|
6
|
+
export declare function gradeTranscript(scenario: Scenario, rawTranscript: string): AssertionResult[];
|
|
7
|
+
export declare function gradeWorkspace(scenario: Scenario, workdir: string): AssertionResult[];
|
|
8
|
+
export declare function gradeWrites(scenario: Scenario, toolUses: ToolUse[], workdir: string): AssertionResult[];
|
|
9
|
+
export declare function gradeReport(scenario: Scenario, workdir: string, initialReport: string | null): AssertionResult[];
|
|
10
|
+
export declare function gradeWithLlm(scenario: Scenario, conversation: string, workspaceFiles: string, voteCount: number): AssertionResult[];
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import { Host, RunResult, Scenario } from './types.ts';
|
|
2
|
+
export declare function evalIdFor(scenario: Scenario, host: Host): string;
|
|
3
|
+
export declare function runScenario(scenario: Scenario, host: Host, runNumber: number, runsRoot: string, model: string | null, graderVotes: number): RunResult;
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
export declare const HOSTS: readonly ["claude", "codex"];
|
|
2
|
+
export type Host = (typeof HOSTS)[number];
|
|
3
|
+
export type PathRule = {
|
|
4
|
+
pattern: string;
|
|
5
|
+
text: string;
|
|
6
|
+
flags?: string;
|
|
7
|
+
/** Named paths to search instead of the `src/**` default. */
|
|
8
|
+
files?: string[];
|
|
9
|
+
};
|
|
10
|
+
export type Scenario = {
|
|
11
|
+
id: string;
|
|
12
|
+
description: string;
|
|
13
|
+
rounds: string[];
|
|
14
|
+
/** Skills this scenario belongs to without asserting they loaded. */
|
|
15
|
+
covers?: string[];
|
|
16
|
+
skills?: {
|
|
17
|
+
expect?: string[];
|
|
18
|
+
forbid?: string[];
|
|
19
|
+
};
|
|
20
|
+
tools?: {
|
|
21
|
+
expect?: string[];
|
|
22
|
+
forbid?: string[];
|
|
23
|
+
};
|
|
24
|
+
writes?: {
|
|
25
|
+
only?: string[];
|
|
26
|
+
expect?: string[];
|
|
27
|
+
};
|
|
28
|
+
workspace?: {
|
|
29
|
+
forbidPatterns?: PathRule[];
|
|
30
|
+
expectPatterns?: PathRule[];
|
|
31
|
+
};
|
|
32
|
+
transcript?: {
|
|
33
|
+
expectStrings?: string[];
|
|
34
|
+
forbidStrings?: string[];
|
|
35
|
+
};
|
|
36
|
+
report?: {
|
|
37
|
+
path: string;
|
|
38
|
+
headings?: string[];
|
|
39
|
+
preserveFrom?: string;
|
|
40
|
+
};
|
|
41
|
+
assertions?: string[];
|
|
42
|
+
fixture?: string;
|
|
43
|
+
setup?: string;
|
|
44
|
+
hosts?: Host[];
|
|
45
|
+
maxTurns?: number;
|
|
46
|
+
roundTimeoutMinutes?: number;
|
|
47
|
+
permissionMode?: string;
|
|
48
|
+
};
|
|
49
|
+
export type AssertionResult = {
|
|
50
|
+
text: string;
|
|
51
|
+
passed: boolean;
|
|
52
|
+
method: "deterministic" | "llm";
|
|
53
|
+
evidence: string;
|
|
54
|
+
};
|
|
55
|
+
export type ToolUse = {
|
|
56
|
+
type: "tool_use";
|
|
57
|
+
name: string;
|
|
58
|
+
input?: Record<string, unknown>;
|
|
59
|
+
};
|
|
60
|
+
export type TranscriptEntry = {
|
|
61
|
+
type: string;
|
|
62
|
+
message?: {
|
|
63
|
+
content?: unknown;
|
|
64
|
+
};
|
|
65
|
+
};
|
|
66
|
+
export type HostPlan = {
|
|
67
|
+
path: string;
|
|
68
|
+
content: string;
|
|
69
|
+
};
|
|
70
|
+
export type RoundMeta = {
|
|
71
|
+
prompt: string;
|
|
72
|
+
session_id: string | null;
|
|
73
|
+
duration_ms?: number;
|
|
74
|
+
total_cost_usd?: number;
|
|
75
|
+
usage?: {
|
|
76
|
+
input_tokens?: number;
|
|
77
|
+
output_tokens?: number;
|
|
78
|
+
};
|
|
79
|
+
models?: string[];
|
|
80
|
+
subtype?: string;
|
|
81
|
+
num_turns?: number;
|
|
82
|
+
timed_out?: boolean;
|
|
83
|
+
};
|
|
84
|
+
export type RunResult = {
|
|
85
|
+
eval_id: string;
|
|
86
|
+
executor_host: Host;
|
|
87
|
+
fixture: string;
|
|
88
|
+
description: string;
|
|
89
|
+
run_number: number;
|
|
90
|
+
models: string[];
|
|
91
|
+
result: {
|
|
92
|
+
pass_rate: number;
|
|
93
|
+
passed: number;
|
|
94
|
+
failed: number;
|
|
95
|
+
total: number;
|
|
96
|
+
time_seconds: number;
|
|
97
|
+
turns: number;
|
|
98
|
+
tokens: number;
|
|
99
|
+
cost_usd: number;
|
|
100
|
+
};
|
|
101
|
+
expectations: {
|
|
102
|
+
text: string;
|
|
103
|
+
passed: boolean;
|
|
104
|
+
evidence: string;
|
|
105
|
+
}[];
|
|
106
|
+
truncated: boolean;
|
|
107
|
+
timed_out: boolean;
|
|
108
|
+
notes: string;
|
|
109
|
+
};
|
|
110
|
+
export type PerEvalSummary = {
|
|
111
|
+
pass_rate_mean: number;
|
|
112
|
+
pass_rate_stddev: number;
|
|
113
|
+
time_seconds_mean: number;
|
|
114
|
+
turns_mean: number;
|
|
115
|
+
tokens_mean: number;
|
|
116
|
+
cost_usd_total: number;
|
|
117
|
+
};
|
|
118
|
+
export type Benchmark = {
|
|
119
|
+
metadata: BenchmarkMetadata;
|
|
120
|
+
runs: RunResult[];
|
|
121
|
+
run_summary: {
|
|
122
|
+
per_eval: Record<string, PerEvalSummary>;
|
|
123
|
+
overall_pass_rate: number;
|
|
124
|
+
total_cost_usd: number;
|
|
125
|
+
};
|
|
126
|
+
notes: string[];
|
|
127
|
+
};
|
|
128
|
+
export type BenchmarkMetadata = {
|
|
129
|
+
skill_family: string;
|
|
130
|
+
timestamp: string;
|
|
131
|
+
evals_run: string[];
|
|
132
|
+
runs_per_scenario: number;
|
|
133
|
+
executor_model: string;
|
|
134
|
+
executor_hosts: string[];
|
|
135
|
+
grader_model: string;
|
|
136
|
+
grader_votes: number;
|
|
137
|
+
tier: string;
|
|
138
|
+
models_observed: string[];
|
|
139
|
+
fixtures: string[];
|
|
140
|
+
};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare function writeGenerated(path: string, contents: string): string;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@imfusion/web-ui",
|
|
3
|
-
"version": "0.6.4-dev.
|
|
3
|
+
"version": "0.6.4-dev.57.g664ba66a",
|
|
4
4
|
"description": "The official Web UI component library for ImFusion web apps",
|
|
5
5
|
"author": "ImFusion GmbH",
|
|
6
6
|
"homepage": "https://imfusion.com",
|
|
@@ -90,10 +90,11 @@
|
|
|
90
90
|
"test:unit": "vitest run --project unit",
|
|
91
91
|
"test:stories": "vitest run --project storybook",
|
|
92
92
|
"test:watch": "vitest",
|
|
93
|
-
"skills:eval": "
|
|
94
|
-
"skills:eval:dev": "WEB_UI_SKILL_EVAL_ROOT=/private/tmp/web-ui-dev-skill-evals WEB_UI_SKILL_EVAL_RESULTS_DIR=.agents/evals/results WEB_UI_SKILL_EVAL_REPORT_PATH=.agents/evals/REPORT.md WEB_UI_SKILL_EVAL_SKILL_FAMILY=web-ui-dev WEB_UI_SKILL_EVAL_WORKSPACE_LABEL='development repository'
|
|
95
|
-
"skills:eval:compare": "
|
|
96
|
-
"skills:eval:
|
|
93
|
+
"skills:eval": "tsx src/llms/evals/runner/run.ts",
|
|
94
|
+
"skills:eval:dev": "WEB_UI_SKILL_EVAL_ROOT=/private/tmp/web-ui-dev-skill-evals WEB_UI_SKILL_EVAL_RESULTS_DIR=.agents/evals/results WEB_UI_SKILL_EVAL_REPORT_PATH=.agents/evals/REPORT.md WEB_UI_SKILL_EVAL_SKILL_FAMILY=web-ui-dev WEB_UI_SKILL_EVAL_WORKSPACE_LABEL='development repository' tsx src/llms/evals/runner/run.ts --scenarios-dir .agents/evals/scenarios --setup .agents/evals/setup-env.sh",
|
|
95
|
+
"skills:eval:compare": "tsx src/llms/evals/runner/compare.ts",
|
|
96
|
+
"skills:eval:affected": "tsx src/llms/evals/runner/affected.ts",
|
|
97
|
+
"skills:eval:report": "tsx src/llms/evals/runner/report.ts",
|
|
97
98
|
"git:config": "git config core.hooksPath .githooks && git config pull.rebase true && git config merge.ff only"
|
|
98
99
|
},
|
|
99
100
|
"peerDependencies": {
|
|
@@ -13,27 +13,30 @@ copy of every convention.
|
|
|
13
13
|
|
|
14
14
|
## 1. Decide whether guidance is needed
|
|
15
15
|
|
|
16
|
-
Skip a companion when the task is explicit and small, an existing local pattern already solves it
|
|
17
|
-
already correct. Do the work. Look up an API silently only when you are unsure.
|
|
16
|
+
Skip a companion when the task is explicit and small, an existing local pattern already solves it in the file being changed,
|
|
17
|
+
and the library usage is already correct. Do the work. Look up an API silently only when you are unsure.
|
|
18
18
|
|
|
19
|
-
Use a companion when the task involves a choice, a missing setup piece, a new file, or a library convention the project
|
|
20
|
-
not already have.
|
|
19
|
+
Use a companion when the task involves a choice, a missing setup piece, a new UI file, or a library convention the project
|
|
20
|
+
may not already have. Open a companion by invoking its skill; reading one of its files directly does not load its workflow.
|
|
21
21
|
|
|
22
22
|
## 2. Route the task
|
|
23
23
|
|
|
24
|
-
| Task
|
|
25
|
-
|
|
|
26
|
-
| Understand library setup, brand assets, theming, or usage
|
|
27
|
-
| Look up a component, part, prop, default, or icon
|
|
28
|
-
| Choose components or shape a screen or flow
|
|
29
|
-
| Write or update documentation
|
|
30
|
-
| Write a wrapper, custom UI, CSS, data layer, validation, or tests
|
|
31
|
-
|
|
|
32
|
-
|
|
|
33
|
-
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
24
|
+
| Task | Open |
|
|
25
|
+
| -------------------------------------------------------------------- | ------------------------------------------------------ |
|
|
26
|
+
| Understand library setup, brand assets, theming, or usage | Packaged user guides below |
|
|
27
|
+
| Look up a component, part, prop, default, or icon | `imf-web-ui-components` |
|
|
28
|
+
| Choose components or shape a screen, form, or flow | `imf-web-ui-ux` |
|
|
29
|
+
| Write or update documentation | `/documentation-writer`, then `imf-web-ui-conventions` |
|
|
30
|
+
| Write a wrapper, custom UI, CSS, data layer, validation, or tests | `imf-web-ui-conventions` |
|
|
31
|
+
| Choose a routing, data, form, or validation library for the project | `imf-web-ui-conventions` |
|
|
32
|
+
| Install the library or bootstrap project tooling | `imf-web-ui-setup` |
|
|
33
|
+
| Diagnose a broken install: unstyled output, missing tokens, no theme | `imf-web-ui-setup` |
|
|
34
|
+
| Inspect an existing project without changing it | `imf-web-ui-audit` |
|
|
35
|
+
| Update the package, skills, or hooks | `imf-web-ui-update` |
|
|
36
|
+
|
|
37
|
+
A new form opens `imf-web-ui-ux`, even when its fields and action are already specified. A screen often needs both
|
|
38
|
+
`imf-web-ui-ux` and `imf-web-ui-components`, in that order. Setup and audit are for project-wide questions, not every
|
|
39
|
+
one-file edit.
|
|
37
40
|
|
|
38
41
|
### Read a packaged user guide
|
|
39
42
|
|
|
@@ -58,8 +61,9 @@ props and the conventions `tokens` topic for exact token names and defaults.
|
|
|
58
61
|
The host project's existing conventions win. The companion skills fill gaps; they do not justify refactoring a working
|
|
59
62
|
styling system, state library, or folder structure.
|
|
60
63
|
|
|
61
|
-
When a task needs TanStack Router, Query, Form, Table, or Store and the project has no incumbent,
|
|
62
|
-
|
|
64
|
+
When a task needs TanStack Router, Query, Form, Table, or Store and the project has no incumbent, open
|
|
65
|
+
`imf-web-ui-conventions` for the house position on that layer, then read the library's current documentation with
|
|
66
|
+
`npx @tanstack/cli` before using it. The recommendation lives in the conventions topics, not here.
|
|
63
67
|
|
|
64
68
|
## 4. Ask only when the choice depends on missing context
|
|
65
69
|
|
|
@@ -10,22 +10,35 @@ allowed-tools: Read Glob Grep
|
|
|
10
10
|
|
|
11
11
|
# Audit a consumer project
|
|
12
12
|
|
|
13
|
-
This is a read-only audit.
|
|
14
|
-
with work the human can approve.
|
|
15
|
-
|
|
13
|
+
This is a read-only audit. Its job is to align a project with its conventions: compare the project against the selected
|
|
14
|
+
`imf-web-ui-conventions` topics, report evidence, and end with work the human can approve.
|
|
15
|
+
|
|
16
|
+
A gap against a convention is high severity. The project agreed to those conventions, so the finding carries no argument
|
|
17
|
+
about whether it matters, and a deliberate reason to differ is the only thing that lowers it. Say what that reason is.
|
|
18
|
+
|
|
19
|
+
Anything else you notice belongs under `Suggestions`: what running the code revealed, what an investigation turned up, what a
|
|
20
|
+
reader would tidy on the way past. Report it, keep it low, and do not let it crowd the sections a consumer has to act on.
|
|
16
21
|
|
|
17
22
|
## Workflow
|
|
18
23
|
|
|
19
24
|
1. Resolve the argument. Bare means `full`; an unknown topic is an error, not a reason to widen the scope.
|
|
20
25
|
2. Enter plan mode unless this audit is being called as verification by another skill or the host has no plan mode.
|
|
21
|
-
3.
|
|
22
|
-
|
|
26
|
+
3. Read the report template and existing reviewer notes, then dispatch one investigator per in-scope topic. Tell each
|
|
27
|
+
investigator to use only `Read`, `Glob`, and `Grep`, with no shell commands or writes. An investigator checks every
|
|
28
|
+
section in that topic and writes nothing. Once dispatch starts, the host performs no repository inspection: the returned
|
|
29
|
+
reports are the evidence, and the host merges them without reading or searching the files again. Gathering the evidence
|
|
30
|
+
twice spends context on the second pass and reports what the host still remembers. Inspect a topic inline only after its
|
|
31
|
+
dispatch has been tried and refused, and say which attempt failed.
|
|
23
32
|
4. Merge the evidence into the shared report format in
|
|
24
|
-
[`templates/REPORT.md`](../imf-web-ui-conventions/templates/REPORT.md).
|
|
25
|
-
|
|
26
|
-
|
|
33
|
+
[`templates/REPORT.md`](../imf-web-ui-conventions/templates/REPORT.md). Before presenting or writing it, check that every
|
|
34
|
+
Broken, Missing, Deviation, and Unverified entry has a severity, evidence, impact, and next action; do not compact
|
|
35
|
+
Unverified entries.
|
|
36
|
+
5. Put each finding's `Next action` into an ordered plan, grouped by topic and cheapest first. The plan is separate from the
|
|
37
|
+
report contract: present it in the host plan or response, never as another heading in `AUDIT_REPORT.md`. Present the
|
|
38
|
+
report and wait for approval; an audit does not edit the project.
|
|
27
39
|
|
|
28
|
-
If the caller needs a durable report, write `AUDIT_REPORT.md`
|
|
40
|
+
If the caller needs a durable report, write `AUDIT_REPORT.md` with exactly the template's H2 headings and preserve everything
|
|
41
|
+
under `## Reviewer notes` verbatim.
|
|
29
42
|
|
|
30
43
|
## Finding format
|
|
31
44
|
|
|
@@ -37,8 +50,9 @@ Classify every result as one of these:
|
|
|
37
50
|
- **Present**: the rule is met, with evidence.
|
|
38
51
|
- **Unverified**: static inspection cannot establish it.
|
|
39
52
|
|
|
40
|
-
Use the report contract's severity, `path:line` evidence, concrete impact, and smallest next action.
|
|
41
|
-
|
|
53
|
+
Use the report contract's severity, `path:line` evidence, concrete impact, and smallest next action. Unverified entries use
|
|
54
|
+
that same structure: unavailable runtime evidence, the impact of the unknown, and the check that resolves it. Do not promote
|
|
55
|
+
an optional tool or a working alternative to a missing finding.
|
|
42
56
|
|
|
43
57
|
## Topics
|
|
44
58
|
|
|
@@ -87,7 +87,7 @@ A component whose identity entry says it comes from `@imfusion/web-ui/integratio
|
|
|
87
87
|
listed optional peer explicitly, then import from that path:
|
|
88
88
|
|
|
89
89
|
```tsx
|
|
90
|
-
import {
|
|
90
|
+
import { ImageDisplayOptions } from "@imfusion/web-ui/integrations/image-display-options";
|
|
91
91
|
```
|
|
92
92
|
|
|
93
93
|
## Data grids
|
|
@@ -12,8 +12,23 @@ Broken, Missing, Deviations, and Unverified entries use:
|
|
|
12
12
|
- Impact: concrete consequence
|
|
13
13
|
- Next action: smallest selectable follow-up
|
|
14
14
|
|
|
15
|
-
Present entries name the topic and evidence path.
|
|
16
|
-
|
|
15
|
+
Present entries name the topic and evidence path. Optional tools are not missing findings.
|
|
16
|
+
|
|
17
|
+
Where a finding belongs follows from where it came from, not from how strongly you hold it.
|
|
18
|
+
|
|
19
|
+
A convention names it:
|
|
20
|
+
- Broken: the convention is followed but does not work. Something is unmounted, unwired, or fails at runtime.
|
|
21
|
+
- Missing: a convention expects it and the project has nothing there.
|
|
22
|
+
- Deviations: it works, and differs from what a convention says.
|
|
23
|
+
|
|
24
|
+
These three are high severity by default. The project agreed to the conventions; a gap against them is not a preference.
|
|
25
|
+
Drop the severity only when the project shows a deliberate reason to differ, and say what the reason is.
|
|
26
|
+
|
|
27
|
+
Nothing names it:
|
|
28
|
+
- Suggestions: real improvements that no convention asks for. What running the code revealed, what an investigation turned
|
|
29
|
+
up, what a reader would tidy on their way past. Never high severity, and never a reason to hold up the work.
|
|
30
|
+
|
|
31
|
+
- Unverified: it could not be checked from the files available.
|
|
17
32
|
-->
|
|
18
33
|
|
|
19
34
|
## Verdict
|
|
@@ -32,6 +47,10 @@ None.
|
|
|
32
47
|
|
|
33
48
|
None.
|
|
34
49
|
|
|
50
|
+
## Suggestions
|
|
51
|
+
|
|
52
|
+
None.
|
|
53
|
+
|
|
35
54
|
## Present
|
|
36
55
|
|
|
37
56
|
None.
|
|
@@ -85,7 +85,9 @@ export function getDetails(options?: QueryOptions<User>) {
|
|
|
85
85
|
}
|
|
86
86
|
```
|
|
87
87
|
|
|
88
|
-
Callers choose `ensureQueryData`, `useSuspenseQuery`, or another Query API.
|
|
88
|
+
Callers choose `ensureQueryData`, `useSuspenseQuery`, or another Query API. Mutation factories likewise return
|
|
89
|
+
`mutationOptions`, own their mutation key and parsing, and are passed to `useMutation`; do not inline those options at the
|
|
90
|
+
call site.
|
|
89
91
|
|
|
90
92
|
## Router context access
|
|
91
93
|
|
|
@@ -8,5 +8,7 @@ Read:
|
|
|
8
8
|
node_modules/@imfusion/web-ui/src/llms/tokens.gen.json
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
-
It contains the shipped token names and authored defaults.
|
|
12
|
-
|
|
11
|
+
It contains the shipped token names and authored defaults. Before writing CSS, check every concrete `--imf-ui-*` name the
|
|
12
|
+
file will use against that index or the relevant component stylesheet. A family with one entry does not imply numbered
|
|
13
|
+
variants. When no matching token exists, use an existing documented seam or say that the token is absent instead of guessing
|
|
14
|
+
a plausible name.
|
|
@@ -5,7 +5,7 @@ description:
|
|
|
5
5
|
needs library wiring, tooling, git, npm, authentication, structure, data, testing, docs, or agent tooling. Inspect first,
|
|
6
6
|
propose concrete files, and write only what the user approves. Use imf-web-ui-audit for read-only checks."
|
|
7
7
|
argument-hint: "[full|library-setup|tooling|git|npm-project|authentication|project-structure|docs-structure|data|testing|agent-tooling]"
|
|
8
|
-
allowed-tools: Read Glob Grep Write
|
|
8
|
+
allowed-tools: Read Glob Grep Write Skill
|
|
9
9
|
---
|
|
10
10
|
|
|
11
11
|
# Set up a consumer project
|
|
@@ -15,16 +15,25 @@ working choices.
|
|
|
15
15
|
|
|
16
16
|
## Workflow
|
|
17
17
|
|
|
18
|
-
1. Resolve the
|
|
19
|
-
|
|
18
|
+
1. Resolve the scope. A named topic limits the assessment to that topic. A bare invocation means `full`, unless the request
|
|
19
|
+
is about one topic, in which case that topic is the scope: answer what was asked and offer the wider sweep rather than
|
|
20
|
+
running it. If the argument is unknown, list the supported topics and point to `imf-web-ui-audit`.
|
|
20
21
|
2. Enter the host's plan mode before inspecting or proposing changes.
|
|
21
22
|
3. Read the selected convention topics and inspect the repository with `Read`, `Glob`, and `Grep` only.
|
|
22
23
|
4. Put a complete proposal in the host plan, using [`templates/REPORT.md`](../imf-web-ui-conventions/templates/REPORT.md).
|
|
23
24
|
For new or updated documentation, invoke `/documentation-writer` before writing and use the `docs-structure` topic for
|
|
24
25
|
repository-specific boundaries. Name every file to create or change, include its intended content, and cite the repository
|
|
25
26
|
evidence behind the choice.
|
|
26
|
-
5. Wait for explicit approval. After approval, write only the listed files.
|
|
27
|
-
|
|
27
|
+
5. Wait for explicit approval. After approval, write only the listed files. Before calling setup complete, ensure every
|
|
28
|
+
package imported by those files and every tool named by their scripts is declared in `package.json`.
|
|
29
|
+
6. Offer `imf-web-ui-audit full` as an optional next step for checking the wider project, and run it only if the user says
|
|
30
|
+
yes. Setup is complete once the approved files are written; the audit reviews ground setup did not touch.
|
|
31
|
+
|
|
32
|
+
## Audit handoff
|
|
33
|
+
|
|
34
|
+
When the approved request includes an audit, the next tool call after the final setup write invokes `imf-web-ui-audit`
|
|
35
|
+
through the host's skill tool. Do not read audit files, inspect the project again, or dispatch audit investigators first;
|
|
36
|
+
reproducing the audit workflow does not load the skill.
|
|
28
37
|
|
|
29
38
|
## Greenfield `full` setup
|
|
30
39
|
|
|
@@ -20,8 +20,10 @@ Never overwrite user changes or commit without fresh approval.
|
|
|
20
20
|
if another lockfile is the active one.
|
|
21
21
|
3. Record `git status --short`, the installed Web UI version, the installed skill targets, hooks, registrations, and
|
|
22
22
|
`AGENTS.md` fence.
|
|
23
|
-
4. Check for another npm process and stop on a conflict.
|
|
24
|
-
|
|
23
|
+
4. Check for another npm process and stop on a conflict. Uncommitted work is also a conflict: report it and stop before the
|
|
24
|
+
first mutating command, rather than deciding it falls outside this workflow's paths. What the update touches is not
|
|
25
|
+
knowable until it runs, and the person can commit, stash, or wave it through in one reply. Never reset, clean, restore, or
|
|
26
|
+
hide their work.
|
|
25
27
|
5. With `--dry-run`, stop after reporting what would happen. Do not install, format, stage, commit, or ask for commit
|
|
26
28
|
approval.
|
|
27
29
|
|
|
@@ -15,7 +15,9 @@ when it is used across many call sites so API movement stays local.
|
|
|
15
15
|
|
|
16
16
|
## Ask only what matters
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
When the request is vague and a human is there to answer, the first reply is a question and nothing else. No default, no
|
|
19
|
+
options table, no recommendation held in reserve: any of those end the conversation, because the person now has an answer and
|
|
20
|
+
no reason to reply. Ask one at a time, and stop once the choice is clear:
|
|
19
21
|
|
|
20
22
|
1. Who uses the screen, and how often?
|
|
21
23
|
2. What is its one primary action?
|
|
@@ -23,8 +25,8 @@ If the request is vague and a human can answer, ask these questions one at a tim
|
|
|
23
25
|
4. What should users see when it is empty, loading, or failing?
|
|
24
26
|
5. Is it a full page or part of another flow?
|
|
25
27
|
|
|
26
|
-
If the prompt already answers a question, do not ask it again.
|
|
27
|
-
the assumption.
|
|
28
|
+
If the prompt already answers a question, do not ask it again. Only when nobody is there to answer, or a question has gone
|
|
29
|
+
unanswered, make the conservative choice and state the assumption.
|
|
28
30
|
|
|
29
31
|
## Choose a surface
|
|
30
32
|
|