@hyperfixi/testing-framework 2.5.1 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -455,4 +455,4 @@ declare function createExpectAPI(): ExpectAPI;
455
455
  */
456
456
  declare function createAssertAPI(): AssertAPI;
457
457
 
458
- export { type AccessibilityCheckResult as A, type BrowserTestOptions as B, type CoverageData as C, createAssertAPI as D, type E2EAction as E, createExpectAPI as F, type MockFunction as M, type PageObject as P, type SpyFunction as S, type TestCase as T, type VisualTestConfig as V, type AccessibilityIncomplete as a, type AccessibilityNode as b, type AccessibilityPass as c, type AccessibilityResult as d, type AccessibilityViolation as e, type AssertAPI as f, AssertionError as g, type BrowserType as h, type E2EStep as i, type ExpectAPI as j, type ExpectMatcher as k, type PerformanceMetrics as l, type TestConfig as m, type TestContext as n, type TestDataProvider as o, type TestDiscoveryOptions as p, type TestEnvironment as q, type TestError as r, type TestFixture as s, type TestFunction as t, type TestLog as u, type TestReporter as v, type TestResult as w, type TestRunner as x, type TestStatus as y, type TestSuite as z };
458
+ export { type AccessibilityCheckResult as A, type BrowserType as B, type CoverageData as C, createAssertAPI as D, type E2EAction as E, createExpectAPI as F, type MockFunction as M, type PerformanceMetrics as P, type SpyFunction as S, type TestRunner as T, type VisualTestConfig as V, type TestSuite as a, type TestConfig as b, type TestResult as c, type TestCase as d, type TestContext as e, type PageObject as f, type TestFunction as g, type TestEnvironment as h, type AccessibilityIncomplete as i, type AccessibilityNode as j, type AccessibilityPass as k, type AccessibilityResult as l, type AccessibilityViolation as m, type AssertAPI as n, AssertionError as o, type BrowserTestOptions as p, type E2EStep as q, type ExpectAPI as r, type ExpectMatcher as s, type TestDataProvider as t, type TestDiscoveryOptions as u, type TestError as v, type TestFixture as w, type TestLog as x, type TestReporter as y, type TestStatus as z };
@@ -455,4 +455,4 @@ declare function createExpectAPI(): ExpectAPI;
455
455
  */
456
456
  declare function createAssertAPI(): AssertAPI;
457
457
 
458
- export { type AccessibilityCheckResult as A, type BrowserTestOptions as B, type CoverageData as C, createAssertAPI as D, type E2EAction as E, createExpectAPI as F, type MockFunction as M, type PageObject as P, type SpyFunction as S, type TestCase as T, type VisualTestConfig as V, type AccessibilityIncomplete as a, type AccessibilityNode as b, type AccessibilityPass as c, type AccessibilityResult as d, type AccessibilityViolation as e, type AssertAPI as f, AssertionError as g, type BrowserType as h, type E2EStep as i, type ExpectAPI as j, type ExpectMatcher as k, type PerformanceMetrics as l, type TestConfig as m, type TestContext as n, type TestDataProvider as o, type TestDiscoveryOptions as p, type TestEnvironment as q, type TestError as r, type TestFixture as s, type TestFunction as t, type TestLog as u, type TestReporter as v, type TestResult as w, type TestRunner as x, type TestStatus as y, type TestSuite as z };
458
+ export { type AccessibilityCheckResult as A, type BrowserType as B, type CoverageData as C, createAssertAPI as D, type E2EAction as E, createExpectAPI as F, type MockFunction as M, type PerformanceMetrics as P, type SpyFunction as S, type TestRunner as T, type VisualTestConfig as V, type TestSuite as a, type TestConfig as b, type TestResult as c, type TestCase as d, type TestContext as e, type PageObject as f, type TestFunction as g, type TestEnvironment as h, type AccessibilityIncomplete as i, type AccessibilityNode as j, type AccessibilityPass as k, type AccessibilityResult as l, type AccessibilityViolation as m, type AssertAPI as n, AssertionError as o, type BrowserTestOptions as p, type E2EStep as q, type ExpectAPI as r, type ExpectMatcher as s, type TestDataProvider as t, type TestDiscoveryOptions as u, type TestError as v, type TestFixture as w, type TestLog as x, type TestReporter as y, type TestStatus as z };
@@ -1 +1 @@
1
- export { g as AssertionError, D as createAssertAPI, F as createExpectAPI } from './assertions-CthVynH6.mjs';
1
+ export { o as AssertionError, D as createAssertAPI, F as createExpectAPI } from './assertions-CsGP61iW.mjs';
@@ -1 +1 @@
1
- export { g as AssertionError, D as createAssertAPI, F as createExpectAPI } from './assertions-CthVynH6.js';
1
+ export { o as AssertionError, D as createAssertAPI, F as createExpectAPI } from './assertions-CsGP61iW.js';
package/dist/index.d.mts CHANGED
@@ -1,5 +1,5 @@
1
- import { x as TestRunner, z as TestSuite, m as TestConfig, w as TestResult, T as TestCase, n as TestContext, l as PerformanceMetrics, M as MockFunction, P as PageObject, S as SpyFunction, t as TestFunction, q as TestEnvironment, h as BrowserType } from './assertions-CthVynH6.mjs';
2
- export { A as AccessibilityCheckResult, a as AccessibilityIncomplete, b as AccessibilityNode, c as AccessibilityPass, d as AccessibilityResult, e as AccessibilityViolation, f as AssertAPI, g as AssertionError, B as BrowserTestOptions, C as CoverageData, E as E2EAction, i as E2EStep, j as ExpectAPI, k as ExpectMatcher, o as TestDataProvider, p as TestDiscoveryOptions, r as TestError, s as TestFixture, u as TestLog, v as TestReporter, y as TestStatus, V as VisualTestConfig, D as createAssertAPI, F as createExpectAPI } from './assertions-CthVynH6.mjs';
1
+ import { T as TestRunner, a as TestSuite, b as TestConfig, c as TestResult, d as TestCase, e as TestContext, P as PerformanceMetrics, M as MockFunction, f as PageObject, S as SpyFunction, g as TestFunction, h as TestEnvironment, B as BrowserType } from './assertions-CsGP61iW.mjs';
2
+ export { A as AccessibilityCheckResult, i as AccessibilityIncomplete, j as AccessibilityNode, k as AccessibilityPass, l as AccessibilityResult, m as AccessibilityViolation, n as AssertAPI, o as AssertionError, p as BrowserTestOptions, C as CoverageData, E as E2EAction, q as E2EStep, r as ExpectAPI, s as ExpectMatcher, t as TestDataProvider, u as TestDiscoveryOptions, v as TestError, w as TestFixture, x as TestLog, y as TestReporter, z as TestStatus, V as VisualTestConfig, D as createAssertAPI, F as createExpectAPI } from './assertions-CsGP61iW.mjs';
3
3
 
4
4
  /**
5
5
  * Test Runner
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
- import { x as TestRunner, z as TestSuite, m as TestConfig, w as TestResult, T as TestCase, n as TestContext, l as PerformanceMetrics, M as MockFunction, P as PageObject, S as SpyFunction, t as TestFunction, q as TestEnvironment, h as BrowserType } from './assertions-CthVynH6.js';
2
- export { A as AccessibilityCheckResult, a as AccessibilityIncomplete, b as AccessibilityNode, c as AccessibilityPass, d as AccessibilityResult, e as AccessibilityViolation, f as AssertAPI, g as AssertionError, B as BrowserTestOptions, C as CoverageData, E as E2EAction, i as E2EStep, j as ExpectAPI, k as ExpectMatcher, o as TestDataProvider, p as TestDiscoveryOptions, r as TestError, s as TestFixture, u as TestLog, v as TestReporter, y as TestStatus, V as VisualTestConfig, D as createAssertAPI, F as createExpectAPI } from './assertions-CthVynH6.js';
1
+ import { T as TestRunner, a as TestSuite, b as TestConfig, c as TestResult, d as TestCase, e as TestContext, P as PerformanceMetrics, M as MockFunction, f as PageObject, S as SpyFunction, g as TestFunction, h as TestEnvironment, B as BrowserType } from './assertions-CsGP61iW.js';
2
+ export { A as AccessibilityCheckResult, i as AccessibilityIncomplete, j as AccessibilityNode, k as AccessibilityPass, l as AccessibilityResult, m as AccessibilityViolation, n as AssertAPI, o as AssertionError, p as BrowserTestOptions, C as CoverageData, E as E2EAction, q as E2EStep, r as ExpectAPI, s as ExpectMatcher, t as TestDataProvider, u as TestDiscoveryOptions, v as TestError, w as TestFixture, x as TestLog, y as TestReporter, z as TestStatus, V as VisualTestConfig, D as createAssertAPI, F as createExpectAPI } from './assertions-CsGP61iW.js';
3
3
 
4
4
  /**
5
5
  * Test Runner
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hyperfixi/testing-framework",
3
- "version": "2.5.1",
3
+ "version": "2.7.0",
4
4
  "description": "Cross-platform behavior testing suite for LokaScript applications",
5
5
  "main": "dist/index.js",
6
6
  "module": "dist/index.mjs",
@@ -44,6 +44,7 @@
44
44
  "scripts": {
45
45
  "build": "tsup",
46
46
  "dev": "tsup --watch",
47
+ "pretest": "../../scripts/ensure-fresh.sh ../intent ../framework ../semantic ../patterns-reference ../core",
47
48
  "test": "vitest run",
48
49
  "test:watch": "vitest",
49
50
  "test:coverage": "vitest run --coverage",
@@ -55,7 +56,7 @@
55
56
  "typecheck": "tsc --noEmit",
56
57
  "lint": "eslint src --ext ts,tsx",
57
58
  "lint:fix": "eslint src --ext ts,tsx --fix",
58
- "test:check": "vitest run --reporter=dot 2>&1 | tail -5"
59
+ "test:check": "VITEST_QUIET=1 bash ../../scripts/vitest-run.sh --reporter=dot"
59
60
  },
60
61
  "keywords": [
61
62
  "hyperscript",
@@ -71,9 +72,9 @@
71
72
  "author": "LokaScript Contributors",
72
73
  "license": "MIT",
73
74
  "dependencies": {
74
- "@hyperfixi/core": "^2.5.1",
75
- "@hyperfixi/patterns-reference": "^2.5.1",
76
- "@lokascript/semantic": "^2.5.1",
75
+ "@hyperfixi/core": "^2.7.0",
76
+ "@hyperfixi/patterns-reference": "^2.7.0",
77
+ "@lokascript/semantic": "^2.7.0",
77
78
  "diff": "^8.0.3",
78
79
  "esbuild": "^0.28.0",
79
80
  "happy-dom": "^20.8.9",
@@ -119,5 +120,8 @@
119
120
  "bugs": {
120
121
  "url": "https://github.com/codetalcott/hyperfixi/issues"
121
122
  },
122
- "homepage": "https://github.com/codetalcott/hyperfixi/tree/main/packages/testing-framework#readme"
123
+ "homepage": "https://github.com/codetalcott/hyperfixi/tree/main/packages/testing-framework#readme",
124
+ "engines": {
125
+ "node": ">=24"
126
+ }
123
127
  }
@@ -121,6 +121,16 @@ export async function findBundleForLanguages(
121
121
  }
122
122
 
123
123
  if (!bestBundle) {
124
+ // Fallback: a per-language bundle `browser-{lang}.{lang}.global.js` may exist
125
+ // even when no predefined group covers the language (these are emitted by the
126
+ // semantic build for every individual language).
127
+ if (languages.length === 1) {
128
+ const name = `browser-${languages[0]}`;
129
+ const p = getBundlePath(name);
130
+ if (existsSync(p)) {
131
+ return { name, path: p, languages, size: statSync(p).size, exists: true };
132
+ }
133
+ }
124
134
  return null;
125
135
  }
126
136
 
@@ -163,8 +173,13 @@ export async function getBundleInfo(bundleName: string): Promise<BundleInfo | nu
163
173
  export async function buildBundle(options: BundleBuildOptions): Promise<BundleInfo> {
164
174
  const { languages, groupName } = options;
165
175
 
166
- // Generate bundle using the generate-bundle.mjs script
167
- const semanticRoot = join(process.cwd(), 'packages/semantic');
176
+ // Generate bundle using the generate-bundle.mjs script.
177
+ // Anchor to this file's location (not process.cwd()): when the gate runs via
178
+ // `npm run … --workspace=@hyperfixi/testing-framework`, cwd is the workspace
179
+ // package dir, so `cwd()/packages/semantic` resolves to a non-existent path and
180
+ // exec() fails with `spawn /bin/sh ENOENT`. getBundlePath() already resolves the
181
+ // semantic root this way — mirror it here.
182
+ const semanticRoot = join(__dirname, '../../..', 'semantic');
168
183
  const scriptPath = join(semanticRoot, 'scripts/generate-bundle.mjs');
169
184
 
170
185
  let bundleName: string;
@@ -210,32 +225,35 @@ export async function selectBundle(
210
225
  languages: LanguageCode[],
211
226
  build: boolean = false
212
227
  ): Promise<BundleInfo> {
213
- // First, try to find existing bundle
214
- let bundle = await findBundleForLanguages(languages);
228
+ // First, try to find an existing bundle
229
+ const bundle = await findBundleForLanguages(languages);
215
230
 
216
- // If bundle exists and we don't need to build, use it
217
- if (bundle && bundle.exists && !build) {
231
+ if (bundle && bundle.exists) {
218
232
  return bundle;
219
233
  }
220
234
 
221
- // If build is requested or no bundle exists, build it
222
- if (build || !bundle || !bundle.exists) {
223
- // Determine if we can use a predefined group
235
+ // Build only on explicit request. The regression gate parses in-process (via
236
+ // parseSemantic), so the bundle is consumed only for size/existence reporting —
237
+ // a missing display bundle must NOT silently trigger a heavy on-demand semantic
238
+ // rebuild as a side effect of running tests.
239
+ if (build) {
224
240
  const groupName = findGroupForLanguages(languages);
225
-
226
- bundle = await buildBundle({
241
+ const built = await buildBundle({
227
242
  languages,
228
243
  groupName: groupName !== null ? groupName : undefined,
229
244
  outputPath: undefined,
230
245
  updateConfig: false, // Don't modify configs during testing
231
246
  });
247
+ if (!built.exists) {
248
+ throw new Error(`Bundle not found: ${built.path}`);
249
+ }
250
+ return built;
232
251
  }
233
252
 
234
- if (!bundle.exists) {
235
- throw new Error(`Bundle not found: ${bundle.path}`);
236
- }
237
-
238
- return bundle;
253
+ // Not building: return what we know (possibly non-existent). The caller reports
254
+ // it as unavailable and continues with the in-process parse.
255
+ const name = `browser-${languages.join('-')}`;
256
+ return bundle ?? { name, path: getBundlePath(name), languages, size: 0, exists: false };
239
257
  }
240
258
 
241
259
  /**
@@ -5,10 +5,62 @@
5
5
  * Command-line interface for running multilingual tests.
6
6
  */
7
7
 
8
+ import * as fs from 'node:fs';
9
+ import * as path from 'node:path';
10
+ import { fileURLToPath } from 'node:url';
11
+ import { checkDbStamp, getDefaultDbPath } from '@hyperfixi/patterns-reference';
8
12
  import { TestOrchestrator } from './orchestrator';
9
- import { RegressionReporter } from './reporters/regression-reporter';
10
13
  import type { TestConfig, LanguageCode } from './types';
11
14
 
15
+ /**
16
+ * Packages whose dist/ this harness executes (directly or via imports), in
17
+ * topological build order. The sibling of the patterns.db provenance stamp:
18
+ * a sweep against a stale dist scores code that differs from the checkout.
19
+ */
20
+ const DIST_GUARD_PACKAGES = [
21
+ 'intent',
22
+ 'framework',
23
+ 'semantic',
24
+ 'i18n',
25
+ 'patterns-reference',
26
+ 'core',
27
+ ];
28
+
29
+ /** True if any .ts under dir is newer than builtAt (early-exit walk). */
30
+ function hasNewerTs(dir: string, builtAt: number): boolean {
31
+ for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
32
+ const p = path.join(dir, entry.name);
33
+ if (entry.isDirectory()) {
34
+ if (hasNewerTs(p, builtAt)) return true;
35
+ } else if (entry.name.endsWith('.ts') && fs.statSync(p).mtimeMs > builtAt) {
36
+ return true;
37
+ }
38
+ }
39
+ return false;
40
+ }
41
+
42
+ /**
43
+ * Names of guard packages whose src/ is newer than their built dist/ (or whose
44
+ * dist/ is missing entirely). Same staleness semantics as
45
+ * scripts/ensure-fresh.sh, but read-only — the CLI refuses instead of
46
+ * rebuilding, so a gate/baseline run never silently mutates build state.
47
+ */
48
+ function findStaleDists(): string[] {
49
+ const here = path.dirname(fileURLToPath(import.meta.url));
50
+ const packagesRoot = path.resolve(here, '../../..');
51
+ const stale: string[] = [];
52
+ for (const name of DIST_GUARD_PACKAGES) {
53
+ const pkg = path.join(packagesRoot, name);
54
+ const srcDir = path.join(pkg, 'src');
55
+ const marker = path.join(pkg, 'dist', 'index.js');
56
+ if (!fs.existsSync(srcDir)) continue;
57
+ if (!fs.existsSync(marker) || hasNewerTs(srcDir, fs.statSync(marker).mtimeMs)) {
58
+ stale.push(name);
59
+ }
60
+ }
61
+ return stale;
62
+ }
63
+
12
64
  /**
13
65
  * Parse command-line arguments
14
66
  */
@@ -19,6 +71,11 @@ function parseArgs(): TestConfig {
19
71
  quickModeLimit: 10,
20
72
  };
21
73
 
74
+ // Track whether quick mode was explicitly requested, so --regression can safely
75
+ // upgrade the default-quick run to full (where the fidelity/degenerate ratchet
76
+ // actually runs) without overriding an explicit --quick.
77
+ let explicitQuick = false;
78
+
22
79
  for (let i = 0; i < args.length; i++) {
23
80
  const arg = args[i];
24
81
 
@@ -48,12 +105,16 @@ function parseArgs(): TestConfig {
48
105
  break;
49
106
 
50
107
  case '--mode':
51
- case '-m':
52
- config.mode = args[++i] as 'quick' | 'full';
108
+ case '-m': {
109
+ const mode = args[++i] as 'quick' | 'full';
110
+ config.mode = mode;
111
+ if (mode === 'quick') explicitQuick = true;
53
112
  break;
113
+ }
54
114
 
55
115
  case '--quick':
56
116
  config.mode = 'quick';
117
+ explicitQuick = true;
57
118
  break;
58
119
 
59
120
  case '--full':
@@ -70,6 +131,12 @@ function parseArgs(): TestConfig {
70
131
  config.regression = true;
71
132
  break;
72
133
 
134
+ case '--baseline': {
135
+ const path = args[++i];
136
+ if (path) config.baselinePath = path;
137
+ break;
138
+ }
139
+
73
140
  case '--confidence':
74
141
  case '-c': {
75
142
  const conf = args[++i];
@@ -100,9 +167,10 @@ function parseArgs(): TestConfig {
100
167
  break;
101
168
 
102
169
  case '--save-baseline':
103
- // Special flag to save results as baseline
104
- runAndSaveBaseline();
105
- return config; // Early return to avoid running tests below
170
+ // Run the suite, then persist the results as the committed baseline
171
+ // (handled in main() — see saveBaseline below).
172
+ config.saveBaseline = true;
173
+ break;
106
174
 
107
175
  default:
108
176
  if (arg && arg.startsWith('-')) {
@@ -113,6 +181,20 @@ function parseArgs(): TestConfig {
113
181
  }
114
182
  }
115
183
 
184
+ // The regression gate ratchets on degenerate/fidelity passes, which are only
185
+ // computed in full mode. A default-quick --regression run silently checks just
186
+ // parse rate — a much weaker gate. Upgrade to full unless the caller explicitly
187
+ // asked for quick (in which case warn that the fidelity ratchet is skipped).
188
+ if (config.regression && config.mode === 'quick') {
189
+ if (explicitQuick) {
190
+ console.warn(
191
+ '⚠ --regression in --quick mode only checks parse rate; the degenerate/fidelity ratchet requires --full.'
192
+ );
193
+ } else {
194
+ config.mode = 'full';
195
+ }
196
+ }
197
+
116
198
  return config;
117
199
  }
118
200
 
@@ -135,7 +217,11 @@ OPTIONS:
135
217
  --quick Quick mode (10 patterns per language)
136
218
  --full Full mode (all patterns)
137
219
  -v, --verbose Enable verbose output
138
- -r, --regression Compare to baseline and report regressions
220
+ -r, --regression Gate on regressions vs baseline (exit 1 if any).
221
+ Implies --full (the degenerate/fidelity ratchet
222
+ needs full mode); pass --quick to force parse-rate-
223
+ only and skip it.
224
+ --baseline <path> Baseline file (default: ./baselines/multilingual-priority.json)
139
225
  -c, --confidence <n> Minimum confidence threshold (0-1)
140
226
  --verified-only Only test verified translations
141
227
  --categories <cats> Filter by categories (comma-separated)
@@ -162,49 +248,283 @@ EXAMPLES:
162
248
  }
163
249
 
164
250
  /**
165
- * Run tests and save as baseline
251
+ * Main CLI entry point
166
252
  */
167
- async function runAndSaveBaseline(): Promise<void> {
253
+ async function main(): Promise<void> {
168
254
  try {
169
- // Parse config without --save-baseline
170
- const args = process.argv.slice(2).filter(arg => arg !== '--save-baseline');
171
- process.argv = [...process.argv.slice(0, 2), ...args];
172
-
173
255
  const config = parseArgs();
174
- config.regression = true; // Enable regression reporter
256
+
257
+ // --save-baseline needs the regression reporter wired so it can persist.
258
+ if (config.saveBaseline) config.regression = true;
259
+
260
+ // DB freshness guard: refuse to run a regression/baseline compare against a
261
+ // patterns.db generated from *different* source than is currently checked out
262
+ // (the cross-branch "phantom regression" footgun). The committed baseline only
263
+ // pairs with a DB synced from the matching source. CI re-syncs before the gate,
264
+ // so it always passes; locally this catches a stale DB after a branch switch or
265
+ // an un-re-synced source edit.
266
+ if (config.regression) {
267
+ const dbPath = getDefaultDbPath();
268
+ const stamp = checkDbStamp(dbPath);
269
+ if (stamp.status === 'stale') {
270
+ console.error(
271
+ '\n✗ patterns.db is STALE — it was generated from different source than is currently\n' +
272
+ ' checked out, so a comparison against the committed baseline would report phantom\n' +
273
+ ' regressions. Re-sync it before running the gate:\n\n' +
274
+ ' npm run db:init:force --prefix packages/patterns-reference\n' +
275
+ ' npm run sync:translations --prefix packages/patterns-reference\n'
276
+ );
277
+ process.exit(1);
278
+ } else if (stamp.status === 'unstamped') {
279
+ console.warn(
280
+ '⚠ patterns.db has no provenance stamp (generated before the freshness guard); ' +
281
+ 'cannot verify it is fresh. Re-sync if results look surprising.'
282
+ );
283
+ }
284
+
285
+ // Dist freshness guard (the db stamp's sibling): this CLI is normally
286
+ // invoked via `npx tsx`, which skips the pretest ensure-fresh hooks, and
287
+ // it executes the parser stack from dist/. A gate or --save-baseline run
288
+ // against a stale dist scores code that differs from the checkout — the
289
+ // "baseline saved from a transitional build" footgun (roadmap §7g/§10).
290
+ // Refuse rather than auto-rebuild so the run never mutates build state.
291
+ const staleDists = findStaleDists();
292
+ if (staleDists.length > 0) {
293
+ console.error(
294
+ `\n✗ Stale dist/ in: ${staleDists.join(', ')} — src/ is newer than the built output,\n` +
295
+ ' so this run would execute code that differs from the checkout (and a saved\n' +
296
+ ' baseline would not reproduce). Rebuild first:\n\n' +
297
+ staleDists.map(p => ` npm run build --prefix packages/${p}`).join('\n') +
298
+ '\n\n (or rebuild the whole stack: npm run test:multilingual:build-deps)\n'
299
+ );
300
+ process.exit(1);
301
+ }
302
+ }
175
303
 
176
304
  const orchestrator = new TestOrchestrator(config);
177
305
  const results = await orchestrator.run();
178
306
 
179
- // Save as baseline
180
- const regressionReporter = orchestrator.getRegressionReporter();
181
- if (regressionReporter) {
182
- regressionReporter.saveAsBaseline(results);
183
- } else {
184
- // Create reporter and save
185
- const reporter = new RegressionReporter();
307
+ // --save-baseline: write the current results as the committed baseline, done.
308
+ if (config.saveBaseline) {
309
+ const reporter = orchestrator.getRegressionReporter();
310
+ if (!reporter) {
311
+ console.error('✗ Could not create regression reporter to save baseline.');
312
+ process.exit(1);
313
+ }
186
314
  reporter.saveAsBaseline(results);
315
+ process.exit(0);
187
316
  }
188
317
 
189
- process.exit(0);
190
- } catch (error) {
191
- console.error('Error:', error instanceof Error ? error.message : String(error));
192
- process.exit(1);
193
- }
194
- }
318
+ // Exit code:
319
+ // - `--regression`: gate on regressions vs the committed baseline, NOT the
320
+ // absolute pass rate. The full multilingual sweep has a KNOWN, documented
321
+ // gap (newer feature patterns lack complete non-English coverage), so a
322
+ // 100%-or-fail gate would never go green. A language fails the gate when
323
+ // its parse rate drops more than REGRESSION_TOLERANCE_PTS below baseline.
324
+ // We deliberately gate on the RATE, not on per-pattern flips: a handful of
325
+ // borderline-confidence patterns can flip pass/fail between builds without
326
+ // a real regression (the per-pattern `newFailures` is kept for reporting
327
+ // only). The baseline must be generated against a freshly `populate`d
328
+ // patterns.db (CI re-populates), or every language reads as shifted.
329
+ // Missing baseline → fail.
330
+ // - otherwise: the normal absolute pass/fail (used by the quick-mode gate,
331
+ // which is genuinely expected at 100%).
332
+ const REGRESSION_TOLERANCE_PTS = 2;
333
+ let exitCode: number;
334
+ if (config.regression) {
335
+ const reporter = orchestrator.getRegressionReporter();
336
+ if (!reporter?.hasBaseline()) {
337
+ console.error(
338
+ `✗ --regression set but no baseline found at ${config.baselinePath ?? './baselines/multilingual-priority.json'}. ` +
339
+ `Generate one with --save-baseline.`
340
+ );
341
+ exitCode = 1;
342
+ } else {
343
+ const allResults = reporter.getRegressionResults();
344
+ const regressed = allResults.filter(r => r.parseRateDelta < -REGRESSION_TOLERANCE_PTS);
345
+
346
+ // Fidelity ratchet: faithful baseline passes that became degenerate
347
+ // (parse non-null but lost most of the English command structure). A
348
+ // small tolerance absorbs residual baseline/DB noise — consistent with
349
+ // the parse-rate tolerance above — while catching real backsliding (a
350
+ // transformer change that degrades a whole cluster). Regenerate the
351
+ // baseline (with --save-baseline) after an intentional fidelity change.
352
+ const FIDELITY_REGRESSION_TOLERANCE = 3;
353
+ const fidelityRegressions = allResults.flatMap(r =>
354
+ r.newDegeneratePasses.map(id => `${r.language}/${id}`)
355
+ );
356
+
357
+ // R0 — correctness ratchet: catch *silent command-drops* the degenerate
358
+ // ratchet misses (a faithful 1.0 pass that becomes lossy, 0.5 ≤ fid < 1.0).
359
+ // Two complementary signals: (1) per-pattern faithful→lossy flips (precise),
360
+ // and (2) a per-language avgFidelity drop (coarse backstop — also catches
361
+ // lossy→more-lossy that the per-pattern flip can't see). Both are guarded by
362
+ // the baseline carrying the new `lossyPasses`/`avgFidelity` data, so an
363
+ // un-regenerated baseline never retro-flags. avgFidelity is deterministic
364
+ // (parse-derived, independent of the patterns.db confidence column), so the
365
+ // tolerance can be tight.
366
+ const LOSSY_REGRESSION_TOLERANCE = 3;
367
+ // 0.02 ≈ six single-pattern drops in a ~154-pattern language — absorbs any
368
+ // rare populate jitter while still catching real per-language cluster
369
+ // regressions (typically ≥0.03). The per-pattern lossy ratchet above is the
370
+ // precise primary signal; this is the coarse lossy→more-lossy backstop.
371
+ const AVG_FIDELITY_DROP_TOLERANCE = 0.02;
372
+ const lossyRegressions = allResults.flatMap(r =>
373
+ r.newLossyPasses.map(id => `${r.language}/${id}`)
374
+ );
375
+ const fidelityDrops = allResults.filter(
376
+ r => r.avgFidelityDelta < -AVG_FIDELITY_DROP_TOLERANCE
377
+ );
378
+
379
+ // R0-precision ratchet: same semantics as the avgFidelity ratchet, on
380
+ // the precision signal (fraction of each parse's actions justified by the
381
+ // en reference). A drop means a parser/render change started injecting
382
+ // phantom/spurious commands — invisible to recall. Deltas are 0 unless
383
+ // the baseline carries avgPrecision, so an un-regenerated baseline never
384
+ // retro-flags.
385
+ const AVG_PRECISION_DROP_TOLERANCE = 0.02;
386
+ const precisionDrops = allResults.filter(
387
+ r => r.avgPrecisionDelta < -AVG_PRECISION_DROP_TOLERANCE
388
+ );
389
+
390
+ // R1 — role-fidelity ratchet (§8): same semantics as the avgFidelity
391
+ // ratchet, on the role-recall signal (action.role:valueType vs the en
392
+ // reference). Deltas are 0 unless the baseline carries avgRoleFidelity,
393
+ // so an un-regenerated baseline never retro-flags. Slightly looser
394
+ // tolerance: role signatures are bigger sets, so populate jitter moves
395
+ // them a bit more than action sets.
396
+ const AVG_ROLE_FIDELITY_DROP_TOLERANCE = 0.02;
397
+ const roleFidelityDrops = allResults.filter(
398
+ r => r.avgRoleFidelityDelta < -AVG_ROLE_FIDELITY_DROP_TOLERANCE
399
+ );
400
+
401
+ // R2 — execution ratchet (§8): curated-subset patterns whose jsdom DOM
402
+ // effects matched the en reference in the baseline but diverge now.
403
+ // Tolerance 0: execution is binary and the harness is deterministic
404
+ // (two full probe sweeps were byte-identical), so ANY pass→fail flip is
405
+ // a real behavioral regression. Guarded by the baseline carrying
406
+ // executionFailures, so an un-regenerated baseline never retro-flags.
407
+ const executionRegressions = allResults.flatMap(r =>
408
+ r.newExecutionFailures.map(id => `${r.language}/${id}`)
409
+ );
410
+
411
+ let failed = false;
412
+
413
+ if (regressed.length > 0) {
414
+ console.error(
415
+ `\n✗ Regression vs baseline in ${regressed.length} language(s) ` +
416
+ `(parse rate dropped > ${REGRESSION_TOLERANCE_PTS}pts):`
417
+ );
418
+ for (const r of regressed) {
419
+ const fails = r.newFailures.length
420
+ ? ` — newly failing: ${r.newFailures.join(', ')}`
421
+ : '';
422
+ console.error(` ${r.language}: ΔparseRate ${r.parseRateDelta.toFixed(1)}pts${fails}`);
423
+ }
424
+ failed = true;
425
+ }
195
426
 
196
- /**
197
- * Main CLI entry point
198
- */
199
- async function main(): Promise<void> {
200
- try {
201
- const config = parseArgs();
427
+ if (fidelityRegressions.length > FIDELITY_REGRESSION_TOLERANCE) {
428
+ console.error(
429
+ `\n✗ Fidelity regression vs baseline: ${fidelityRegressions.length} faithful pass(es) ` +
430
+ `became degenerate (tolerance ${FIDELITY_REGRESSION_TOLERANCE}):`
431
+ );
432
+ for (const id of fidelityRegressions) console.error(` ${id}`);
433
+ console.error(
434
+ ` (parse non-null but lost >50% of the English command structure — ` +
435
+ `if intentional, regenerate the baseline with --save-baseline)`
436
+ );
437
+ failed = true;
438
+ } else if (fidelityRegressions.length > 0) {
439
+ console.warn(
440
+ `\n⚠ ${fidelityRegressions.length} fidelity regression(s) within tolerance ` +
441
+ `(${FIDELITY_REGRESSION_TOLERANCE}): ${fidelityRegressions.join(', ')}`
442
+ );
443
+ }
202
444
 
203
- const orchestrator = new TestOrchestrator(config);
204
- const results = await orchestrator.run();
445
+ if (lossyRegressions.length > LOSSY_REGRESSION_TOLERANCE) {
446
+ console.error(
447
+ `\n✗ Correctness regression vs baseline: ${lossyRegressions.length} faithful pass(es) ` +
448
+ `became lossy (silently dropped a command; tolerance ${LOSSY_REGRESSION_TOLERANCE}):`
449
+ );
450
+ for (const id of lossyRegressions) console.error(` ${id}`);
451
+ console.error(
452
+ ` (still parses but lost ≥1 command vs the English reference — ` +
453
+ `if intentional, regenerate the baseline with --save-baseline)`
454
+ );
455
+ failed = true;
456
+ } else if (lossyRegressions.length > 0) {
457
+ console.warn(
458
+ `\n⚠ ${lossyRegressions.length} correctness regression(s) within tolerance ` +
459
+ `(${LOSSY_REGRESSION_TOLERANCE}): ${lossyRegressions.join(', ')}`
460
+ );
461
+ }
462
+
463
+ if (fidelityDrops.length > 0) {
464
+ console.error(
465
+ `\n✗ avgFidelity dropped > ${AVG_FIDELITY_DROP_TOLERANCE} in ` +
466
+ `${fidelityDrops.length} language(s):`
467
+ );
468
+ for (const r of fidelityDrops) {
469
+ console.error(` ${r.language}: ΔavgFidelity ${r.avgFidelityDelta.toFixed(4)}`);
470
+ }
471
+ console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
472
+ failed = true;
473
+ }
474
+
475
+ if (precisionDrops.length > 0) {
476
+ console.error(
477
+ `\n✗ avgPrecision dropped > ${AVG_PRECISION_DROP_TOLERANCE} in ` +
478
+ `${precisionDrops.length} language(s) — a parse/render started injecting ` +
479
+ `phantom commands the source never had:`
480
+ );
481
+ for (const r of precisionDrops) {
482
+ console.error(` ${r.language}: ΔavgPrecision ${r.avgPrecisionDelta.toFixed(4)}`);
483
+ }
484
+ console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
485
+ failed = true;
486
+ }
205
487
 
206
- // Exit with appropriate code
207
- const exitCode = results.summary.overallStatus === 'pass' ? 0 : 1;
488
+ if (roleFidelityDrops.length > 0) {
489
+ console.error(
490
+ `\n✗ avgRoleFidelity dropped > ${AVG_ROLE_FIDELITY_DROP_TOLERANCE} in ` +
491
+ `${roleFidelityDrops.length} language(s):`
492
+ );
493
+ for (const r of roleFidelityDrops) {
494
+ console.error(
495
+ ` ${r.language}: ΔavgRoleFidelity ${r.avgRoleFidelityDelta.toFixed(4)}`
496
+ );
497
+ }
498
+ console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
499
+ failed = true;
500
+ }
501
+
502
+ if (executionRegressions.length > 0) {
503
+ console.error(
504
+ `\n✗ Execution regression vs baseline (R2): ${executionRegressions.length} ` +
505
+ `curated pattern(s) no longer reproduce the en reference's DOM effects:`
506
+ );
507
+ for (const id of executionRegressions) console.error(` ${id}`);
508
+ console.error(
509
+ ` (executed faithfully in the baseline — if intentional, regenerate ` +
510
+ `the baseline with --save-baseline)`
511
+ );
512
+ failed = true;
513
+ }
514
+
515
+ if (failed) {
516
+ exitCode = 1;
517
+ } else {
518
+ console.log(
519
+ `\n✓ No regression vs baseline ` +
520
+ `(parse-rate ${REGRESSION_TOLERANCE_PTS}pts, fidelity + correctness + execution ratchets).`
521
+ );
522
+ exitCode = 0;
523
+ }
524
+ }
525
+ } else {
526
+ exitCode = results.summary.overallStatus === 'pass' ? 0 : 1;
527
+ }
208
528
  process.exit(exitCode);
209
529
  } catch (error) {
210
530
  console.error('Error:', error instanceof Error ? error.message : String(error));