@ultimat3/cli 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +100 -0
- package/package.json +60 -0
- package/src/app-agents-md.ts +27 -0
- package/src/app-boundaries.ts +206 -0
- package/src/app-evals.ts +74 -0
- package/src/app-load.ts +136 -0
- package/src/app-manifest.ts +137 -0
- package/src/app-openapi.ts +12 -0
- package/src/app-root.ts +57 -0
- package/src/bin.ts +17 -0
- package/src/boundary-cuts.ts +219 -0
- package/src/budgets.ts +92 -0
- package/src/cmd-build.ts +109 -0
- package/src/cmd-db.ts +187 -0
- package/src/cmd-deploy.ts +124 -0
- package/src/cmd-dev.ts +286 -0
- package/src/cmd-doctor.ts +178 -0
- package/src/cmd-errors.ts +99 -0
- package/src/cmd-fix.ts +126 -0
- package/src/cmd-generate.ts +434 -0
- package/src/cmd-help.ts +94 -0
- package/src/cmd-i18n.ts +212 -0
- package/src/cmd-jobs.ts +237 -0
- package/src/cmd-manifest.ts +97 -0
- package/src/cmd-mcp.ts +176 -0
- package/src/cmd-new.ts +133 -0
- package/src/cmd-planned.ts +119 -0
- package/src/cmd-policy.ts +136 -0
- package/src/cmd-registries.ts +195 -0
- package/src/cmd-routes.ts +73 -0
- package/src/cmd-tasks.ts +151 -0
- package/src/cmd-test.ts +109 -0
- package/src/cmd-verify.ts +265 -0
- package/src/command.ts +33 -0
- package/src/dev-assets.ts +177 -0
- package/src/dev-dashboard.ts +242 -0
- package/src/dev-hooks.ts +51 -0
- package/src/dev-policy.ts +82 -0
- package/src/dev-queue.ts +109 -0
- package/src/dev-render.ts +129 -0
- package/src/dev-replicator.ts +92 -0
- package/src/dev-roles.ts +246 -0
- package/src/dev-runtime.ts +203 -0
- package/src/dev-services.ts +75 -0
- package/src/dev-traces.ts +141 -0
- package/src/dispatch.ts +98 -0
- package/src/drift.ts +86 -0
- package/src/error-catalog.ts +156 -0
- package/src/error-contract.ts +212 -0
- package/src/errors.ts +367 -0
- package/src/exec.ts +70 -0
- package/src/hold.ts +48 -0
- package/src/i18n-audit.ts +183 -0
- package/src/index.ts +179 -0
- package/src/jobs-drain.ts +151 -0
- package/src/jobs-json.ts +134 -0
- package/src/jobs-report.ts +132 -0
- package/src/jobs-table.ts +34 -0
- package/src/json-merge.ts +40 -0
- package/src/mcp-db-target.ts +50 -0
- package/src/mcp-errors.ts +99 -0
- package/src/mcp-host.ts +282 -0
- package/src/mcp-test-output.ts +57 -0
- package/src/messages.ts +119 -0
- package/src/output.ts +174 -0
- package/src/parse.ts +243 -0
- package/src/policy-facts.ts +196 -0
- package/src/policy-fixture.ts +71 -0
- package/src/registry.ts +73 -0
- package/src/scaffold-fixture.ts +69 -0
- package/src/scaffold-typecheck.ts +240 -0
- package/src/source-files.ts +38 -0
- package/src/table.ts +19 -0
- package/src/tasks-facts.ts +113 -0
- package/src/templates/action.ts +193 -0
- package/src/templates/admin.ts +46 -0
- package/src/templates/catalog-json.ts +17 -0
- package/src/templates/entity.ts +157 -0
- package/src/templates/index.ts +23 -0
- package/src/templates/job.ts +148 -0
- package/src/templates/locales.ts +93 -0
- package/src/templates/naming.ts +97 -0
- package/src/templates/policy.ts +120 -0
- package/src/templates/query.ts +116 -0
- package/src/templates/resource.ts +199 -0
- package/src/templates/route.ts +138 -0
- package/src/templates/scaffold-app.ts +320 -0
- package/src/templates/scaffold-docs.ts +156 -0
- package/src/templates/scaffold-i18n.ts +149 -0
- package/src/templates/scaffold-icon.ts +54 -0
- package/src/templates/scaffold-package-shape.ts +49 -0
- package/src/templates/scaffold-repo.ts +427 -0
- package/src/test-select.ts +130 -0
- package/src/test-shards.ts +188 -0
- package/src/thrown-by.ts +24 -0
- package/src/ts-scan.ts +217 -0
- package/src/verify-step.ts +83 -0
- package/src/verify-tests.ts +166 -0
- package/src/version-loader.ts +16 -0
- package/src/workspace-checks.ts +288 -0
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
// Spending the selected test files: the LPT plan that balances them across worker processes, the
|
|
2
|
+
// argv each child gets, and the command that reproduces one shard exactly. Split out of
|
|
3
|
+
// cmd-test.ts because a printed reproduction is only true if it carries every input to the split —
|
|
4
|
+
// that rule is this file's, and argv parsing is that one's.
|
|
5
|
+
|
|
6
|
+
import { docsFor } from './errors';
|
|
7
|
+
import type { Runner } from './exec';
|
|
8
|
+
import { execOutput } from './exec';
|
|
9
|
+
import { msg } from './messages';
|
|
10
|
+
import type { CommandResult, Finding, JsonValue, StepResult } from './output';
|
|
11
|
+
import type { TestFile } from './test-select';
|
|
12
|
+
import { bySizeThenPath } from './test-select';
|
|
13
|
+
import type { TestType } from './verify-tests';
|
|
14
|
+
|
|
15
|
+
export interface Shard {
|
|
16
|
+
readonly index: number;
|
|
17
|
+
readonly files: readonly string[];
|
|
18
|
+
readonly bytes: number;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Largest-first greedy bin packing (LPT). Deterministic — the total order is (size desc, path asc),
|
|
23
|
+
* so the filesystem's scan order never reaches the assignment and a CI failure on worker 3 is the
|
|
24
|
+
* same worker 3 locally. Balanced — every file lands in the currently emptiest bin, which bounds a
|
|
25
|
+
* bin at average + largest file; round-robin or hashing can pile every slow file onto one worker.
|
|
26
|
+
* The inner scan is O(files × workers), and workers is a core count, so a heap would only add
|
|
27
|
+
* allocation.
|
|
28
|
+
*/
|
|
29
|
+
export function planShards(files: readonly TestFile[], workers: number): readonly Shard[] {
|
|
30
|
+
const count = Math.max(1, Math.min(Math.trunc(workers), files.length));
|
|
31
|
+
const loads = new Array<number>(count).fill(0);
|
|
32
|
+
const buckets: string[][] = Array.from({ length: count }, () => []);
|
|
33
|
+
for (const file of [...files].sort(bySizeThenPath)) {
|
|
34
|
+
let target = 0;
|
|
35
|
+
for (let i = 1; i < count; i += 1) if ((loads[i] ?? 0) < (loads[target] ?? 0)) target = i;
|
|
36
|
+
buckets[target]?.push(file.path);
|
|
37
|
+
loads[target] = (loads[target] ?? 0) + file.bytes;
|
|
38
|
+
}
|
|
39
|
+
return buckets.map((paths, index) => ({
|
|
40
|
+
index,
|
|
41
|
+
files: [...paths].sort(),
|
|
42
|
+
bytes: loads[index] ?? 0,
|
|
43
|
+
}));
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Explicit file list, so the child never re-globs and can never pick up another shard's files. */
|
|
47
|
+
export const shardArgs = (shard: Shard): readonly string[] => ['bun', 'test', ...shard.files];
|
|
48
|
+
|
|
49
|
+
const SHELL_SAFE = /^[\w@%+=:,./-]+$/;
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* POSIX single-quoting for the values a caller supplies. A `--filter` holding a space, a `$` or a
|
|
53
|
+
* `;` pastes back as two arguments or as a second command, so an unquoted line would run something
|
|
54
|
+
* other than the run it claims to reproduce. `'\''` is the only escape a single-quoted string has.
|
|
55
|
+
*/
|
|
56
|
+
export const quoteArg = (value: string): string =>
|
|
57
|
+
SHELL_SAFE.test(value) ? value : `'${value.split("'").join("'\\''")}'`;
|
|
58
|
+
|
|
59
|
+
export interface ReproduceOptions {
|
|
60
|
+
/** The *effective* worker count: `planShards` clamps to the file count, and the split follows. */
|
|
61
|
+
readonly workers: number;
|
|
62
|
+
readonly filter?: string;
|
|
63
|
+
readonly type?: TestType;
|
|
64
|
+
/** Files `--sample` kept, so the rerun samples the same corpus instead of the whole type. */
|
|
65
|
+
readonly sample?: number;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Every input to the split, printed back. The type and `--filter` decide which files exist to
|
|
70
|
+
* shard, `--sample` decides how many of them survive, `--workers` decides the bins — drop any one
|
|
71
|
+
* and the command still runs, over a different file set, which reproduces nothing.
|
|
72
|
+
*/
|
|
73
|
+
export function reproduceFor(shard: Shard, options: ReproduceOptions): string {
|
|
74
|
+
return [
|
|
75
|
+
'x test',
|
|
76
|
+
...(options.type === undefined ? [] : [quoteArg(options.type)]),
|
|
77
|
+
...(options.filter === undefined ? [] : ['--filter', quoteArg(options.filter)]),
|
|
78
|
+
...(options.sample === undefined ? [] : ['--sample', String(options.sample)]),
|
|
79
|
+
'--workers',
|
|
80
|
+
String(options.workers),
|
|
81
|
+
'--worker',
|
|
82
|
+
String(shard.index),
|
|
83
|
+
].join(' ');
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export interface RunShardsOptions {
|
|
87
|
+
readonly root: string;
|
|
88
|
+
readonly runner: Runner;
|
|
89
|
+
readonly files: readonly TestFile[];
|
|
90
|
+
readonly workers: number;
|
|
91
|
+
/** Run exactly one shard of the same split, not a one-worker run of everything. */
|
|
92
|
+
readonly only?: number;
|
|
93
|
+
readonly filter?: string;
|
|
94
|
+
readonly type?: TestType;
|
|
95
|
+
/**
|
|
96
|
+
* Set when `--sample` narrowed `files`: `kept` is what survived, `total` what discovery found.
|
|
97
|
+
* `kept` is carried rather than counted from the shards that ran, because `--worker N` runs one
|
|
98
|
+
* shard of the sample and would otherwise report that shard's size as the corpus.
|
|
99
|
+
*/
|
|
100
|
+
readonly sample?: { readonly kept: number; readonly total: number };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** The reproduction's inputs, resolved once: `workers` is the split's real width, not the ask. */
|
|
104
|
+
const planOf = (options: RunShardsOptions, workers: number): ReproduceOptions => ({
|
|
105
|
+
workers,
|
|
106
|
+
...(options.filter === undefined ? {} : { filter: options.filter }),
|
|
107
|
+
...(options.type === undefined ? {} : { type: options.type }),
|
|
108
|
+
...(options.sample === undefined ? {} : { sample: options.sample.kept }),
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
const failureOf = (shard: Shard, code: number, plan: ReproduceOptions): Finding => ({
|
|
112
|
+
code: 'X_TEST_SHARD_FAILED',
|
|
113
|
+
cause: `shard ${shard.index} of ${plan.workers} exited ${code} (${shard.files.length} file(s))`,
|
|
114
|
+
fix: reproduceFor(shard, plan),
|
|
115
|
+
docs: docsFor('X_TEST_SHARD_FAILED'),
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
export async function runShards(options: RunShardsOptions): Promise<CommandResult> {
|
|
119
|
+
const shards = planShards(options.files, options.workers);
|
|
120
|
+
const only = options.only;
|
|
121
|
+
const chosen = only === undefined ? shards : shards.filter((shard) => shard.index === only);
|
|
122
|
+
const started = performance.now();
|
|
123
|
+
// All shards at once: wall-clock is the whole point, and each one owns a separate database.
|
|
124
|
+
const runs = await Promise.all(
|
|
125
|
+
chosen.map(async (shard) => ({
|
|
126
|
+
shard,
|
|
127
|
+
result: await options.runner(shardArgs(shard), {
|
|
128
|
+
cwd: options.root,
|
|
129
|
+
env: { ULTIMATE_TEST_WORKER: String(shard.index) },
|
|
130
|
+
}),
|
|
131
|
+
})),
|
|
132
|
+
);
|
|
133
|
+
const durationMs = Math.round(performance.now() - started);
|
|
134
|
+
const plan = planOf(options, shards.length);
|
|
135
|
+
const steps: readonly StepResult[] = runs.map(({ shard, result }) => ({
|
|
136
|
+
name: `shard ${shard.index} · ${shard.files.length} files`,
|
|
137
|
+
ok: result.ok,
|
|
138
|
+
durationMs: result.durationMs,
|
|
139
|
+
findings: result.ok ? [] : [failureOf(shard, result.code, plan)],
|
|
140
|
+
output: execOutput(result),
|
|
141
|
+
}));
|
|
142
|
+
const failed = runs.filter((run) => !run.result.ok).map((run) => run.shard.index);
|
|
143
|
+
const fileCount = chosen.reduce((total, shard) => total + shard.files.length, 0);
|
|
144
|
+
const type = options.type;
|
|
145
|
+
const typeParam = type === undefined ? {} : { type };
|
|
146
|
+
const sample = options.sample;
|
|
147
|
+
const data: JsonValue = {
|
|
148
|
+
...typeParam,
|
|
149
|
+
workers: shards.length,
|
|
150
|
+
files: fileCount,
|
|
151
|
+
durationMs,
|
|
152
|
+
...(options.filter === undefined ? {} : { filter: options.filter }),
|
|
153
|
+
...(sample === undefined ? {} : { sample: { kept: sample.kept, total: sample.total } }),
|
|
154
|
+
shards: runs.map(({ shard, result }) => ({
|
|
155
|
+
index: shard.index,
|
|
156
|
+
files: shard.files.length,
|
|
157
|
+
bytes: shard.bytes,
|
|
158
|
+
ok: result.ok,
|
|
159
|
+
exitCode: result.code,
|
|
160
|
+
durationMs: result.durationMs,
|
|
161
|
+
reproduce: reproduceFor(shard, plan),
|
|
162
|
+
})),
|
|
163
|
+
failed,
|
|
164
|
+
};
|
|
165
|
+
return {
|
|
166
|
+
ok: failed.length === 0,
|
|
167
|
+
command: 'test',
|
|
168
|
+
summary:
|
|
169
|
+
failed.length === 0
|
|
170
|
+
? msg(type === undefined ? 'cli.test.pass' : 'cli.test.type.pass', {
|
|
171
|
+
...typeParam,
|
|
172
|
+
files: fileCount,
|
|
173
|
+
workers: chosen.length,
|
|
174
|
+
ms: durationMs,
|
|
175
|
+
})
|
|
176
|
+
: msg(type === undefined ? 'cli.test.fail' : 'cli.test.type.fail', {
|
|
177
|
+
...typeParam,
|
|
178
|
+
failed: failed.length,
|
|
179
|
+
workers: chosen.length,
|
|
180
|
+
}),
|
|
181
|
+
steps,
|
|
182
|
+
...(sample === undefined
|
|
183
|
+
? {}
|
|
184
|
+
: { lines: [msg('cli.test.sampled', { ...sample, type: type ?? 'all' })] }),
|
|
185
|
+
data,
|
|
186
|
+
exitCode: failed.length === 0 ? 0 : 1,
|
|
187
|
+
};
|
|
188
|
+
}
|
package/src/thrown-by.ts
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
// One `thrownBy` for every test in this package. What a caller acts on is the thrown *shape* —
|
|
2
|
+
// the code, the cause and the fix — never the class, so a test that asserts on the class would
|
|
3
|
+
// pass while the fix line rots. Shared rather than copied: a second copy asserts less, silently.
|
|
4
|
+
|
|
5
|
+
import { expect } from 'bun:test';
|
|
6
|
+
|
|
7
|
+
/** The three fields every `UltimateError` carries, as a test reads them off the thrown value. */
|
|
8
|
+
export interface ThrownShape {
|
|
9
|
+
readonly code?: string;
|
|
10
|
+
readonly cause?: string;
|
|
11
|
+
readonly fix?: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export function thrownBy(call: () => unknown): ThrownShape {
|
|
15
|
+
try {
|
|
16
|
+
call();
|
|
17
|
+
} catch (error) {
|
|
18
|
+
return error as ThrownShape;
|
|
19
|
+
}
|
|
20
|
+
// Not `throw new Error(...)`: a bare throw carries no code and no fix, and this is the one
|
|
21
|
+
// failure a test author most needs named. `expect.unreachable` fails through the runner, which
|
|
22
|
+
// prints the assertion the caller wrote instead of a stack from inside this helper.
|
|
23
|
+
return expect.unreachable('expected a throw');
|
|
24
|
+
}
|
package/src/ts-scan.ts
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
// Reading two things out of TypeScript source without a parser: the strings a `fix:` can evaluate
|
|
2
|
+
// to, and the `X_*` codes a package declares. Deliberately not `tsc` — a regex over a masked file
|
|
3
|
+
// is the whole job. Masking is the load-bearing part: the contract's own 3-line rendering appears
|
|
4
|
+
// verbatim in doc blocks and template literals, and a scanner that reads it as code invents work.
|
|
5
|
+
|
|
6
|
+
export interface SourceSite {
|
|
7
|
+
/** Repo-relative file the site was read from. */
|
|
8
|
+
readonly at: string;
|
|
9
|
+
readonly line: number;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
export interface FixSite extends SourceSite {
|
|
13
|
+
/** The literal exactly as written, `${…}` included. */
|
|
14
|
+
readonly fix: string;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
export interface CodeSite extends SourceSite {
|
|
18
|
+
readonly code: string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
const QUOTES = new Set(["'", '"', '`']);
|
|
22
|
+
const OPENERS = new Set(['(', '[', '{']);
|
|
23
|
+
const CLOSERS = new Set([')', ']', '}']);
|
|
24
|
+
const WORD = /[\w$]/;
|
|
25
|
+
|
|
26
|
+
/** After one of these words a `/` opens a regex; after any other identifier it divides. */
|
|
27
|
+
const REGEX_AFTER_WORDS = new Set(
|
|
28
|
+
'await case delete do else in instanceof new of return throw typeof void yield'.split(' '),
|
|
29
|
+
);
|
|
30
|
+
|
|
31
|
+
/** Index just past the closing quote of the literal opening at `from`, or the end of the text. */
|
|
32
|
+
function endOfLiteral(text: string, from: number): number {
|
|
33
|
+
const quote = text[from] as string;
|
|
34
|
+
for (let i = from + 1; i < text.length; i += 1) {
|
|
35
|
+
if (text[i] === '\\') i += 1;
|
|
36
|
+
else if (text[i] === quote) return i + 1;
|
|
37
|
+
}
|
|
38
|
+
return text.length;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Whether the `/` at `at` opens a regex rather than divides — the call no scanner without a parser
|
|
43
|
+
* avoids. A regex cannot follow what ends an expression: an identifier that is not one of the words
|
|
44
|
+
* above, a number, `)`, `]`, a string's closing quote. Every other position is an operator's and
|
|
45
|
+
* opens one; `</` and `/>` are JSX delimiters. Read from the masked prefix, so a comment is space.
|
|
46
|
+
*/
|
|
47
|
+
function opensRegex(out: readonly string[], at: number): boolean {
|
|
48
|
+
if (out[at + 1] === '>') return false;
|
|
49
|
+
let i = at - 1;
|
|
50
|
+
while (i >= 0 && /\s/.test(out[i] as string)) i -= 1;
|
|
51
|
+
if (i < 0) return true;
|
|
52
|
+
const ch = out[i] as string;
|
|
53
|
+
if (ch === '<' || ch === ')' || ch === ']' || QUOTES.has(ch)) return false;
|
|
54
|
+
if (!WORD.test(ch)) return true;
|
|
55
|
+
let start = i;
|
|
56
|
+
while (start >= 0 && WORD.test(out[start] as string)) start -= 1;
|
|
57
|
+
return REGEX_AFTER_WORDS.has(out.slice(start + 1, i + 1).join(''));
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Index just past the closing `/` of the regex opening at `from`, or `from + 1` when it does not
|
|
62
|
+
* close on its own line — a literal may not span one, so an unterminated candidate was a division
|
|
63
|
+
* or a JSX delimiter after all. A `/` inside a `[…]` class does not close the literal.
|
|
64
|
+
*/
|
|
65
|
+
function endOfRegex(text: string, from: number): number {
|
|
66
|
+
let inClass = false;
|
|
67
|
+
let escaped = false;
|
|
68
|
+
for (let i = from + 1; i < text.length; i += 1) {
|
|
69
|
+
const ch = text[i];
|
|
70
|
+
if (ch === '\n') break;
|
|
71
|
+
if (escaped) escaped = false;
|
|
72
|
+
else if (ch === '\\') escaped = true;
|
|
73
|
+
else if (inClass) inClass = ch !== ']';
|
|
74
|
+
else if (ch === '[') inClass = true;
|
|
75
|
+
else if (ch === '/') return i + 1;
|
|
76
|
+
}
|
|
77
|
+
return from + 1;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Comments — and optionally string contents — replaced by spaces, newlines kept so line numbers
|
|
82
|
+
* survive and quote delimiters kept so the caller can still find where a literal starts and ends.
|
|
83
|
+
* A regex body is masked the same way: `/(['"`])/` holds three quotes that delimit nothing, and
|
|
84
|
+
* reading one as an opening quote desyncs every literal after it.
|
|
85
|
+
*/
|
|
86
|
+
function blankRegions(text: string, strings: boolean): string {
|
|
87
|
+
const out = [...text];
|
|
88
|
+
const blank = (from: number, to: number): void => {
|
|
89
|
+
for (let n = from; n < to; n += 1) if (out[n] !== '\n') out[n] = ' ';
|
|
90
|
+
};
|
|
91
|
+
let i = 0;
|
|
92
|
+
while (i < text.length) {
|
|
93
|
+
const ch = text[i] as string;
|
|
94
|
+
if (ch === '/' && (text[i + 1] === '/' || text[i + 1] === '*')) {
|
|
95
|
+
const line = text[i + 1] === '/';
|
|
96
|
+
const end = line ? text.indexOf('\n', i) : text.indexOf('*/', i + 2);
|
|
97
|
+
const stop = end === -1 ? text.length : line ? end : end + 2;
|
|
98
|
+
blank(i, stop);
|
|
99
|
+
i = stop;
|
|
100
|
+
continue;
|
|
101
|
+
}
|
|
102
|
+
// Not code: a regex body or a literal. `end === i + 1` blanks nothing and steps one char.
|
|
103
|
+
const end =
|
|
104
|
+
ch === '/' && opensRegex(out, i)
|
|
105
|
+
? endOfRegex(text, i)
|
|
106
|
+
: QUOTES.has(ch)
|
|
107
|
+
? endOfLiteral(text, i)
|
|
108
|
+
: i + 1;
|
|
109
|
+
if (strings) blank(i + 1, end - 1);
|
|
110
|
+
i = end;
|
|
111
|
+
}
|
|
112
|
+
return out.join('');
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Comments gone, string literals intact — what a scan for declared codes reads. */
|
|
116
|
+
export const stripComments = (text: string): string => blankRegions(text, false);
|
|
117
|
+
|
|
118
|
+
/** Comments and string contents gone, delimiters kept — what a scan for code structure reads. */
|
|
119
|
+
export const maskLiterals = (text: string): string => blankRegions(text, true);
|
|
120
|
+
|
|
121
|
+
const lineOf = (text: string, index: number): number => {
|
|
122
|
+
let line = 1;
|
|
123
|
+
for (let i = 0; i < index; i += 1) if (text[i] === '\n') line += 1;
|
|
124
|
+
return line;
|
|
125
|
+
};
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Every string literal in the value expression starting at `from`, at the expression's own bracket
|
|
129
|
+
* depth. Spanning the expression is what makes `cond ? 'a' : 'b'` and `table[k] ?? 'c'` checkable
|
|
130
|
+
* instead of silently skipped; the depth rule is what keeps `command.join(' ')`'s separator and
|
|
131
|
+
* `table['key']`'s key out — an argument is not a fix.
|
|
132
|
+
*/
|
|
133
|
+
function valueLiterals(masked: string, source: string, from: number): readonly FixSite[] {
|
|
134
|
+
const found: { value: string; index: number }[] = [];
|
|
135
|
+
let depth = 0;
|
|
136
|
+
for (let i = from; i < masked.length; i += 1) {
|
|
137
|
+
const ch = masked[i] as string;
|
|
138
|
+
if (QUOTES.has(ch)) {
|
|
139
|
+
const end = endOfLiteral(masked, i);
|
|
140
|
+
if (depth === 0) found.push({ value: source.slice(i + 1, end - 1), index: i });
|
|
141
|
+
i = end - 1;
|
|
142
|
+
} else if (OPENERS.has(ch)) depth += 1;
|
|
143
|
+
else if (CLOSERS.has(ch)) {
|
|
144
|
+
if (depth === 0) break;
|
|
145
|
+
depth -= 1;
|
|
146
|
+
} else if (depth === 0 && (ch === ',' || ch === ';')) break;
|
|
147
|
+
}
|
|
148
|
+
return found.map((literal) => ({
|
|
149
|
+
at: '',
|
|
150
|
+
line: lineOf(masked, literal.index),
|
|
151
|
+
fix: literal.value,
|
|
152
|
+
}));
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** The lookbehind rejects member access: `cond ? e.fix : ''` is a ternary, not a declaration. */
|
|
156
|
+
const FIX_KEY = /(?<![.\w$])fix\s*:\s*/g;
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Every string a `fix:` can evaluate to. Searched over the masked source, so a `fix:` written
|
|
160
|
+
* inside a doc comment or interpolated into a message is not mistaken for a declaration. A `fix`
|
|
161
|
+
* computed at runtime — a bare identifier, a parameter, a table lookup with no literal fallback —
|
|
162
|
+
* has nothing to read and is beyond a static scan; the gate says so rather than guessing.
|
|
163
|
+
*/
|
|
164
|
+
export function scanFixes(source: string, at: string): readonly FixSite[] {
|
|
165
|
+
const masked = maskLiterals(source);
|
|
166
|
+
const sites: FixSite[] = [];
|
|
167
|
+
for (const key of masked.matchAll(FIX_KEY)) {
|
|
168
|
+
const start = key.index + key[0].length;
|
|
169
|
+
for (const literal of valueLiterals(masked, source, start)) sites.push({ ...literal, at });
|
|
170
|
+
}
|
|
171
|
+
return sites;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** `packages/<pkg>/src/errors.ts` and core's `error-codes.ts`: one file per package, by rule. */
|
|
175
|
+
export const isCodeRegistry = (path: string): boolean =>
|
|
176
|
+
/(?:^|\/)(?:errors|error-codes)\.ts$/.test(path);
|
|
177
|
+
|
|
178
|
+
const CODE_AT_KEY = /\bcode\s*[:=]\s*(['"`])(X_[A-Z0-9_]+)\1/g;
|
|
179
|
+
const CODE_LITERAL = /(['"`])(X_[A-Z0-9_]+)\1/g;
|
|
180
|
+
const CODE_KEY = /^[\t ]*(X_[A-Z0-9_]+)\s*:/gm;
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* Codes this file declares: every `code:` / `code =` throw site, plus — in a package's own code
|
|
184
|
+
* registry — every entry of its code list or title table, whichever shape it uses. A registry is
|
|
185
|
+
* the only place a bare `X_*` literal is a declaration; anywhere else it is a reference (an env
|
|
186
|
+
* var named `X_BUILD_ID`, an HTTP status map keyed by code) and collecting it would invent a code.
|
|
187
|
+
*/
|
|
188
|
+
export function scanCodes(source: string, at: string): readonly CodeSite[] {
|
|
189
|
+
const text = stripComments(source);
|
|
190
|
+
const sites = new Map<string, CodeSite>();
|
|
191
|
+
const add = (code: string, index: number): void => {
|
|
192
|
+
if (!sites.has(code)) sites.set(code, { at, line: lineOf(text, index), code });
|
|
193
|
+
};
|
|
194
|
+
for (const match of text.matchAll(CODE_AT_KEY)) add(match[2] as string, match.index);
|
|
195
|
+
if (isCodeRegistry(at)) {
|
|
196
|
+
for (const match of text.matchAll(CODE_LITERAL)) add(match[2] as string, match.index);
|
|
197
|
+
for (const match of text.matchAll(CODE_KEY)) add(match[1] as string, match.index);
|
|
198
|
+
}
|
|
199
|
+
return [...sites.values()];
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
const BORROWED_LIST = /BORROWED_ERROR_CODES[^=]*=[^[]*\[([^\]]*)\]/g;
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* The codes a registry names and says are not its own. Every borrower declares them in one place
|
|
206
|
+
* and one shape — `export const CLI_BORROWED_ERROR_CODES = ['X_NOT_IMPLEMENTED'] as const` — so
|
|
207
|
+
* "who owns this code?" is answerable from source rather than guessed at. Without it the answer
|
|
208
|
+
* for a code eleven packages throw and one titles is whichever file sorts first, which is how
|
|
209
|
+
* `X_NOT_IMPLEMENTED` came to be attributed to `storage` instead of `core`.
|
|
210
|
+
*/
|
|
211
|
+
export function scanBorrowedCodes(source: string): ReadonlySet<string> {
|
|
212
|
+
const borrowed = new Set<string>();
|
|
213
|
+
for (const list of stripComments(source).matchAll(BORROWED_LIST)) {
|
|
214
|
+
for (const code of (list[1] ?? '').matchAll(CODE_LITERAL)) borrowed.add(code[2] as string);
|
|
215
|
+
}
|
|
216
|
+
return borrowed;
|
|
217
|
+
}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
// The shape of one gate step: its name, what it checks, whether it applies here, and how a host
|
|
2
|
+
// repo feeds it findings it could not produce on its own. Split from the step list so a step
|
|
3
|
+
// implementation can live beside the code it checks without importing the list.
|
|
4
|
+
|
|
5
|
+
import type { ExecResult, Runner } from './exec';
|
|
6
|
+
import { execOutput } from './exec';
|
|
7
|
+
import type { Finding } from './output';
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Every step of the gate, in cost order — cheapest and most informative first, and never a check
|
|
11
|
+
* whose result would be meaningless because an earlier one failed. This list is the definition of
|
|
12
|
+
* shippable: the framework repo and a generated app run exactly it, whole, or not at all.
|
|
13
|
+
*/
|
|
14
|
+
export const VERIFY_STEP_NAMES = [
|
|
15
|
+
'typecheck',
|
|
16
|
+
'lint',
|
|
17
|
+
'boundaries',
|
|
18
|
+
'filesize',
|
|
19
|
+
'package-shape',
|
|
20
|
+
'errors',
|
|
21
|
+
'unit',
|
|
22
|
+
'contract',
|
|
23
|
+
'live',
|
|
24
|
+
'job',
|
|
25
|
+
'e2e',
|
|
26
|
+
'eval',
|
|
27
|
+
'drift',
|
|
28
|
+
'contract-diff',
|
|
29
|
+
'budgets',
|
|
30
|
+
'manifest',
|
|
31
|
+
'roadmap',
|
|
32
|
+
] as const;
|
|
33
|
+
|
|
34
|
+
export type VerifyStepName = (typeof VERIFY_STEP_NAMES)[number];
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* A rule the host repo enforces inside an existing step — the framework monorepo's package tier
|
|
38
|
+
* table under `boundaries`, its generated manifest under `manifest`. A host adds findings to a
|
|
39
|
+
* step; it can never add, remove, reorder or skip one.
|
|
40
|
+
*/
|
|
41
|
+
export type HostCheck = (root: string) => Promise<readonly Finding[]>;
|
|
42
|
+
|
|
43
|
+
export interface VerifyContext {
|
|
44
|
+
readonly root: string;
|
|
45
|
+
readonly runner: Runner;
|
|
46
|
+
readonly hostChecks?: Partial<Record<VerifyStepName, HostCheck>>;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export interface StepOutcome {
|
|
50
|
+
readonly ok: boolean;
|
|
51
|
+
readonly findings: readonly Finding[];
|
|
52
|
+
readonly output?: string;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export interface VerifyStep {
|
|
56
|
+
readonly name: VerifyStepName;
|
|
57
|
+
readonly summary: string;
|
|
58
|
+
/** Returns false to record the step as skipped rather than passed. */
|
|
59
|
+
applies?(ctx: VerifyContext): Promise<boolean>;
|
|
60
|
+
run(ctx: VerifyContext): Promise<StepOutcome>;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export const passed: StepOutcome = { ok: true, findings: [] };
|
|
64
|
+
|
|
65
|
+
export function fromExec(result: ExecResult, finding: Omit<Finding, 'docs'>): StepOutcome {
|
|
66
|
+
if (result.ok) return { ok: true, findings: [], output: execOutput(result) };
|
|
67
|
+
return {
|
|
68
|
+
ok: false,
|
|
69
|
+
findings: [{ ...finding, docs: `https://ultimate.dev/errors/${finding.code}` }],
|
|
70
|
+
output: execOutput(result),
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export const fromFindings = (findings: readonly Finding[]): StepOutcome => ({
|
|
75
|
+
ok: findings.length === 0,
|
|
76
|
+
findings,
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
/** What the host repo contributes to this step, or nothing. */
|
|
80
|
+
export const hostFindings = async (
|
|
81
|
+
ctx: VerifyContext,
|
|
82
|
+
step: VerifyStepName,
|
|
83
|
+
): Promise<readonly Finding[]> => (await ctx.hostChecks?.[step]?.(ctx.root)) ?? [];
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
// One `bun test` invocation per test type, so every type reports on its own line of the gate. A
|
|
2
|
+
// test's type is its filename suffix — `*.contract.test.ts`, `*.live.test.ts`, `*.job.test.ts`,
|
|
3
|
+
// `*.e2e.test.ts` (or any file under an `e2e/` directory), `*.eval.test.ts`. Everything else is a
|
|
4
|
+
// unit test, which is why the unit step is the only one that selects by exclusion.
|
|
5
|
+
//
|
|
6
|
+
// `eval` carries one rule beyond its suite — every prompt must have an eval — so it is the only
|
|
7
|
+
// step here that can fail with no test file of its own.
|
|
8
|
+
|
|
9
|
+
// Bun ships no `Bun.*` equivalent for either: `existsSync` answers whether this root is an app,
|
|
10
|
+
// and `join` builds the host-separator path to its config file.
|
|
11
|
+
import { existsSync } from 'node:fs';
|
|
12
|
+
import { join } from 'node:path';
|
|
13
|
+
import { checkEvalBaselines, checkEvalCoverage, checkEvalRecording } from './app-evals';
|
|
14
|
+
import { APP_CONFIG_FILE } from './app-root';
|
|
15
|
+
import type { StepOutcome, VerifyContext, VerifyStep } from './verify-step';
|
|
16
|
+
import { fromExec, fromFindings } from './verify-step';
|
|
17
|
+
|
|
18
|
+
export const TEST_TYPES = ['unit', 'contract', 'live', 'job', 'e2e', 'eval'] as const;
|
|
19
|
+
|
|
20
|
+
export type TestType = (typeof TEST_TYPES)[number];
|
|
21
|
+
|
|
22
|
+
interface TestSuite {
|
|
23
|
+
readonly summary: string;
|
|
24
|
+
/** Substring `bun test` matches against each file path. */
|
|
25
|
+
readonly filter: string;
|
|
26
|
+
/** Globs that decide whether this type exists here at all. */
|
|
27
|
+
readonly globs: readonly string[];
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
const TYPED_SUFFIXES = '{contract,live,job,e2e,eval}';
|
|
31
|
+
|
|
32
|
+
const SUITES: Readonly<Record<Exclude<TestType, 'unit'>, TestSuite>> = {
|
|
33
|
+
contract: {
|
|
34
|
+
summary: 'action/query schemas, policy denials, emitted OpenAPI and MCP shapes',
|
|
35
|
+
filter: '.contract.test.',
|
|
36
|
+
globs: ['**/*.contract.test.{ts,tsx}'],
|
|
37
|
+
},
|
|
38
|
+
live: {
|
|
39
|
+
summary: 'live-query snapshots, incremental patches, reconnect deltas',
|
|
40
|
+
filter: '.live.test.',
|
|
41
|
+
globs: ['**/*.live.test.{ts,tsx}'],
|
|
42
|
+
},
|
|
43
|
+
job: {
|
|
44
|
+
summary: 'step replay, idempotency dedupe, retry/backoff, outbox atomicity',
|
|
45
|
+
filter: '.job.test.',
|
|
46
|
+
globs: ['**/*.job.test.{ts,tsx}'],
|
|
47
|
+
},
|
|
48
|
+
e2e: {
|
|
49
|
+
summary: 'the built output, incl. offline and SW update',
|
|
50
|
+
filter: 'e2e',
|
|
51
|
+
globs: ['**/*.e2e.test.{ts,tsx}', '**/e2e/**/*.test.{ts,tsx}'],
|
|
52
|
+
},
|
|
53
|
+
eval: {
|
|
54
|
+
summary: 'LLM output scored against thresholds',
|
|
55
|
+
filter: '.eval.test.',
|
|
56
|
+
globs: ['**/*.eval.test.{ts,tsx}'],
|
|
57
|
+
},
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Build output, and nested projects that carry their own `x verify`. `examples/**` is the second
|
|
62
|
+
* kind: the reference app is gated by its own run of this same step list, so collecting it here
|
|
63
|
+
* would report one app failure on two different gates. The patterns are relative to the run's
|
|
64
|
+
* root, so this excludes nothing when the app itself is the root.
|
|
65
|
+
*/
|
|
66
|
+
const NEVER_A_TEST = ['**/dist/**', '**/build/**', '**/examples/**'];
|
|
67
|
+
|
|
68
|
+
const ignoreFlags = (patterns: readonly string[]): readonly string[] =>
|
|
69
|
+
patterns.map((pattern) => `--path-ignore-patterns=${pattern}`);
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* The substring that decides a file's type — the same one the step's `bun test` runs with, so
|
|
73
|
+
* `x test <type>` and the gate's `<type>` step can never disagree about what a contract test is.
|
|
74
|
+
*/
|
|
75
|
+
export const typeFilterOf = (type: Exclude<TestType, 'unit'>): string => SUITES[type].filter;
|
|
76
|
+
|
|
77
|
+
/** Unit is everything the typed suites do not claim, so no test can fall between two steps. */
|
|
78
|
+
export const testStepCommand = (type: TestType): readonly string[] =>
|
|
79
|
+
type === 'unit'
|
|
80
|
+
? [
|
|
81
|
+
'bun',
|
|
82
|
+
'test',
|
|
83
|
+
...ignoreFlags([...NEVER_A_TEST, '**/e2e/**', `**/*.${TYPED_SUFFIXES}.test.*`]),
|
|
84
|
+
]
|
|
85
|
+
: ['bun', 'test', ...ignoreFlags(NEVER_A_TEST), SUITES[type].filter];
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Whether a step applies has to be decided by the same rule that decides what it runs. When the
|
|
89
|
+
* two drifted, a suite that lived only under an ignored path made its step apply and then fail
|
|
90
|
+
* with "no test files matched" — a red gate reporting a suite that, by its own rule, is not here.
|
|
91
|
+
*/
|
|
92
|
+
const NEVER_A_TEST_GLOBS = NEVER_A_TEST.map((pattern) => new Bun.Glob(pattern));
|
|
93
|
+
|
|
94
|
+
const ignoredPath = (path: string): boolean =>
|
|
95
|
+
path.includes('node_modules') || NEVER_A_TEST_GLOBS.some((glob) => glob.match(path));
|
|
96
|
+
|
|
97
|
+
const exists = async (root: string, globs: readonly string[]): Promise<boolean> => {
|
|
98
|
+
for (const pattern of globs) {
|
|
99
|
+
for await (const path of new Bun.Glob(pattern).scan({ cwd: root, absolute: false })) {
|
|
100
|
+
if (!ignoredPath(path)) return true;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
return false;
|
|
104
|
+
};
|
|
105
|
+
|
|
106
|
+
const runType = async (ctx: VerifyContext, type: TestType): Promise<StepOutcome> => {
|
|
107
|
+
const command = testStepCommand(type);
|
|
108
|
+
const result = await ctx.runner(command, { cwd: ctx.root });
|
|
109
|
+
return fromExec(result, {
|
|
110
|
+
code: 'X_TEST_FAILED',
|
|
111
|
+
cause: `one or more ${type} tests failed`,
|
|
112
|
+
fix: command.join(' '),
|
|
113
|
+
});
|
|
114
|
+
};
|
|
115
|
+
|
|
116
|
+
const isApp = (root: string): boolean => existsSync(join(root, APP_CONFIG_FILE));
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* The eval step carries rules beyond "the suite is green": every prompt must have an eval, every
|
|
120
|
+
* eval must have a baseline to gate against, and the run must not be recording. So it is the one
|
|
121
|
+
* test step that applies with no suite of its own — an app whose only prompt has no eval file
|
|
122
|
+
* would otherwise skip this step and report a green gate over untested code.
|
|
123
|
+
*/
|
|
124
|
+
const evalStep: VerifyStep = {
|
|
125
|
+
name: 'eval',
|
|
126
|
+
summary: SUITES.eval.summary,
|
|
127
|
+
applies: async (ctx) => isApp(ctx.root) || (await exists(ctx.root, SUITES.eval.globs)),
|
|
128
|
+
async run(ctx) {
|
|
129
|
+
// First, and instead of the suite: under recording every eval writes the numbers it just
|
|
130
|
+
// measured and passes, so running it here would rewrite the committed baselines during the
|
|
131
|
+
// gate — damage a red step does not undo.
|
|
132
|
+
const recording = checkEvalRecording();
|
|
133
|
+
if (recording.length > 0) return fromFindings(recording);
|
|
134
|
+
const declarations = isApp(ctx.root)
|
|
135
|
+
? [...(await checkEvalCoverage(ctx.root)), ...(await checkEvalBaselines(ctx.root))]
|
|
136
|
+
: [];
|
|
137
|
+
if (!(await exists(ctx.root, SUITES.eval.globs))) return fromFindings(declarations);
|
|
138
|
+
const suite = await runType(ctx, 'eval');
|
|
139
|
+
return {
|
|
140
|
+
ok: suite.ok && declarations.length === 0,
|
|
141
|
+
findings: [...declarations, ...suite.findings],
|
|
142
|
+
...(suite.output === undefined ? {} : { output: suite.output }),
|
|
143
|
+
};
|
|
144
|
+
},
|
|
145
|
+
};
|
|
146
|
+
|
|
147
|
+
const stepFor = (type: TestType): VerifyStep => {
|
|
148
|
+
if (type === 'unit') {
|
|
149
|
+
return {
|
|
150
|
+
name: 'unit',
|
|
151
|
+
summary: 'pure logic — no database, no network',
|
|
152
|
+
run: (ctx) => runType(ctx, 'unit'),
|
|
153
|
+
};
|
|
154
|
+
}
|
|
155
|
+
if (type === 'eval') return evalStep;
|
|
156
|
+
const suite = SUITES[type];
|
|
157
|
+
return {
|
|
158
|
+
name: type,
|
|
159
|
+
summary: suite.summary,
|
|
160
|
+
applies: (ctx) => exists(ctx.root, suite.globs),
|
|
161
|
+
run: (ctx) => runType(ctx, type),
|
|
162
|
+
};
|
|
163
|
+
};
|
|
164
|
+
|
|
165
|
+
/** In cost order: unit needs nothing, e2e needs a build. */
|
|
166
|
+
export const TEST_STEPS: readonly VerifyStep[] = TEST_TYPES.map(stepFor);
|