@ontrails/regrade 1.0.0-beta.32 → 1.0.0-beta.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,16 @@ import {
5
5
  escapeRegExp,
6
6
  matchesAnyPathGlob,
7
7
  } from '@ontrails/core';
8
- import { readFileSync, writeFileSync } from 'node:fs';
8
+ import { createHash } from 'node:crypto';
9
+ import {
10
+ dirname,
11
+ extname,
12
+ isAbsolute,
13
+ join,
14
+ normalize,
15
+ relative,
16
+ } from 'node:path';
17
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
9
18
  import { z } from 'zod';
10
19
 
11
20
  import { collectDownstreamSources } from './collect.js';
@@ -19,12 +28,36 @@ import { buildRegradeScanSummary } from './scan-summary.js';
19
28
 
20
29
  export type VocabularyVerdict = 'deferred' | 'modified' | 'skipped';
21
30
 
31
+ export const vocabularyDispositionValues = [
32
+ 'code-context-out-of-engine',
33
+ 'docs-only',
34
+ 'explicit-preserve',
35
+ 'forward-pointer',
36
+ 'ignored-by-scope',
37
+ 'in-family-modified',
38
+ 'in-family-unresolved',
39
+ 'out-of-family',
40
+ 'preserve-current-live-api',
41
+ ] as const;
42
+
43
+ export type VocabularyDisposition =
44
+ (typeof vocabularyDispositionValues)[number];
45
+
46
+ const vocabularyDispositions = new Set<string>(vocabularyDispositionValues);
47
+
22
48
  export interface VocabularyPreserveRule {
49
+ readonly disposition?: VocabularyDisposition;
50
+ readonly forms?: readonly string[];
23
51
  readonly pattern: string;
24
52
  readonly reason?: string;
25
53
  readonly paths?: readonly string[];
26
54
  }
27
55
 
56
+ export interface VocabularyPreserveInventoryEntry extends VocabularyPreserveRule {
57
+ readonly evidence: readonly string[];
58
+ readonly source: 'derived-live-api';
59
+ }
60
+
28
61
  export interface VocabularyRegradeScope {
29
62
  readonly exclude?: readonly string[];
30
63
  readonly extensions?: readonly string[];
@@ -38,6 +71,8 @@ export interface VocabularyRegradeScope {
38
71
  }
39
72
 
40
73
  export interface VocabularyRegradePlan {
74
+ readonly caseSensitive?: boolean;
75
+ readonly deferForms?: readonly string[];
41
76
  readonly from: string;
42
77
  readonly id?: string;
43
78
  readonly intent?: string;
@@ -51,6 +86,7 @@ export interface VocabularyRegradePlan {
51
86
  export interface VocabularyOccurrence {
52
87
  readonly column: number;
53
88
  readonly context: string;
89
+ readonly disposition: VocabularyDisposition;
54
90
  readonly end: number;
55
91
  readonly form: string;
56
92
  readonly line: number;
@@ -69,6 +105,9 @@ export interface VocabularyRunLedger {
69
105
 
70
106
  export interface VocabularyRunGate {
71
107
  readonly remaining: number;
108
+ readonly remainingByDisposition: Partial<
109
+ Readonly<Record<VocabularyDisposition, number>>
110
+ >;
72
111
  readonly reasons: readonly string[];
73
112
  readonly status: 'green' | 'open';
74
113
  }
@@ -76,6 +115,9 @@ export interface VocabularyRunGate {
76
115
  export interface VocabularyRunReport {
77
116
  readonly applied: number;
78
117
  readonly deferred: number;
118
+ readonly dispositions: Partial<
119
+ Readonly<Record<VocabularyDisposition, number>>
120
+ >;
79
121
  readonly filesChanged: number;
80
122
  readonly gate: VocabularyRunGate;
81
123
  readonly modified: number;
@@ -86,9 +128,38 @@ export interface VocabularyRunReport {
86
128
  export interface VocabularyRegradeRun {
87
129
  readonly ledger: VocabularyRunLedger;
88
130
  readonly plan: VocabularyRegradePlan;
131
+ readonly preserveInventory?: readonly VocabularyPreserveInventoryEntry[];
89
132
  readonly report: VocabularyRunReport;
90
133
  }
91
134
 
135
+ export const VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION = 1;
136
+
137
+ export interface VocabularyTransitionRecordEnvironment {
138
+ readonly commitSha?: string;
139
+ readonly engineVersion?: string;
140
+ readonly graphHash?: string;
141
+ readonly root: string;
142
+ }
143
+
144
+ export interface VocabularyTransitionRecord {
145
+ readonly environment: VocabularyTransitionRecordEnvironment;
146
+ readonly kind: 'vocabulary-transition-record';
147
+ readonly recordPath: string;
148
+ readonly report: Omit<RegradeReport, 'record'>;
149
+ readonly schemaVersion: typeof VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION;
150
+ readonly transition: {
151
+ readonly from: string;
152
+ readonly id: string;
153
+ readonly to: string;
154
+ };
155
+ }
156
+
157
+ export interface VocabularyTransitionRecordSummary {
158
+ readonly path: string;
159
+ readonly schemaVersion: typeof VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION;
160
+ readonly status: 'candidate' | 'applied' | 'checked';
161
+ }
162
+
92
163
  interface SourceFile {
93
164
  readonly absolutePath: string;
94
165
  readonly path: string;
@@ -99,6 +170,13 @@ interface SourceOccurrence extends VocabularyOccurrence {
99
170
  readonly absolutePath: string;
100
171
  }
101
172
 
173
+ interface SourceOccurrenceDraft extends Omit<
174
+ SourceOccurrence,
175
+ 'disposition' | 'reason' | 'verdict'
176
+ > {
177
+ readonly contextColumn: number;
178
+ }
179
+
102
180
  interface VocabularyEvaluation {
103
181
  readonly entries: readonly RegradeReportEntry[];
104
182
  readonly occurrences: readonly SourceOccurrence[];
@@ -125,30 +203,73 @@ const VOCABULARY_SOURCE_EXTENSIONS = Object.freeze([
125
203
  const uniqueSorted = (values: readonly string[]): readonly string[] =>
126
204
  [...new Set(values)].toSorted((a, b) => a.localeCompare(b));
127
205
 
128
- const isVocabularyTokenCharacter = (value: string): boolean =>
129
- /[A-Za-z0-9_$-]/.test(value);
206
+ const vocabularyDispositionCounts = (
207
+ occurrences: readonly VocabularyOccurrence[]
208
+ ): Partial<Readonly<Record<VocabularyDisposition, number>>> => {
209
+ const counts = new Map<VocabularyDisposition, number>();
210
+ for (const occurrence of occurrences) {
211
+ counts.set(
212
+ occurrence.disposition,
213
+ (counts.get(occurrence.disposition) ?? 0) + 1
214
+ );
215
+ }
216
+ return Object.fromEntries(
217
+ [...counts.entries()].toSorted(([left], [right]) =>
218
+ left.localeCompare(right)
219
+ )
220
+ );
221
+ };
222
+
223
+ const isVocabularyTokenCharacter = (
224
+ value: string,
225
+ routeLike: boolean
226
+ ): boolean => /[A-Za-z0-9_$-]/.test(value) || (routeLike && value === '/');
227
+
228
+ const isVocabularyTokenCharacterAt = (
229
+ source: string,
230
+ index: number,
231
+ routeLike: boolean
232
+ ): boolean => {
233
+ if (index < 0 || index >= source.length) {
234
+ return false;
235
+ }
236
+ const value = source.at(index) ?? '';
237
+ if (isVocabularyTokenCharacter(value, routeLike)) {
238
+ return true;
239
+ }
240
+ if (!routeLike || value !== '.') {
241
+ return false;
242
+ }
243
+ return (
244
+ isVocabularyTokenCharacter(source.at(index - 1) ?? '', false) &&
245
+ isVocabularyTokenCharacter(source.at(index + 1) ?? '', false)
246
+ );
247
+ };
130
248
 
131
249
  const hasWordBoundary = (
132
250
  source: string,
133
251
  start: number,
134
- end: number
252
+ end: number,
253
+ form: string
135
254
  ): boolean => {
136
- const before = start === 0 ? '' : (source.at(start - 1) ?? '');
137
- const after = end >= source.length ? '' : (source.at(end) ?? '');
255
+ const routeLike = form.includes('/');
138
256
  return (
139
- !isVocabularyTokenCharacter(before) && !isVocabularyTokenCharacter(after)
257
+ !isVocabularyTokenCharacterAt(source, start - 1, routeLike) &&
258
+ !isVocabularyTokenCharacterAt(source, end, routeLike)
140
259
  );
141
260
  };
142
261
 
143
262
  const expandVocabularyNeighborSpan = (
144
263
  source: string,
145
264
  start: number,
146
- end: number
265
+ end: number,
266
+ form: string
147
267
  ): { readonly end: number; readonly start: number } => {
268
+ const routeLike = form.includes('/');
148
269
  let expandedStart = start;
149
270
  while (
150
271
  expandedStart > 0 &&
151
- isVocabularyTokenCharacter(source.at(expandedStart - 1) ?? '')
272
+ isVocabularyTokenCharacterAt(source, expandedStart - 1, routeLike)
152
273
  ) {
153
274
  expandedStart -= 1;
154
275
  }
@@ -156,7 +277,7 @@ const expandVocabularyNeighborSpan = (
156
277
  let expandedEnd = end;
157
278
  while (
158
279
  expandedEnd < source.length &&
159
- isVocabularyTokenCharacter(source.at(expandedEnd) ?? '')
280
+ isVocabularyTokenCharacterAt(source, expandedEnd, routeLike)
160
281
  ) {
161
282
  expandedEnd += 1;
162
283
  }
@@ -181,15 +302,155 @@ const lineColumnForOffset = (
181
302
  return { column, line };
182
303
  };
183
304
 
184
- const contextForOffset = (
305
+ const contextDetailsForOffset = (
185
306
  source: string,
186
307
  start: number,
187
308
  end: number
188
- ): string => {
309
+ ): { readonly context: string; readonly contextColumn: number } => {
189
310
  const lineStart = source.lastIndexOf('\n', start - 1) + 1;
190
311
  const nextLine = source.indexOf('\n', end);
191
312
  const lineEnd = nextLine === -1 ? source.length : nextLine;
192
- return source.slice(lineStart, lineEnd).trim();
313
+ const rawLine = source.slice(lineStart, lineEnd);
314
+ const leadingTrimmed = rawLine.length - rawLine.trimStart().length;
315
+ return {
316
+ context: rawLine.trim(),
317
+ contextColumn: start - lineStart - leadingTrimmed + 1,
318
+ };
319
+ };
320
+
321
+ const isMarkdownPath = (path: string): boolean =>
322
+ path.endsWith('.md') || path.endsWith('.mdx');
323
+
324
+ const packageRouteCodeExtensions = new Set([
325
+ '.cjs',
326
+ '.cts',
327
+ '.js',
328
+ '.jsx',
329
+ '.mjs',
330
+ '.mts',
331
+ '.ts',
332
+ '.tsx',
333
+ ]);
334
+
335
+ const isPackageRouteCodeOccurrence = (path: string, form: string): boolean =>
336
+ packageRouteCodeExtensions.has(extname(path)) &&
337
+ /^@[^/]+\/[^/]+(?:\/.*)?$/.test(form);
338
+
339
+ const isPackageManifestPath = (path: string): boolean =>
340
+ path === 'package.json' || path.endsWith('/package.json');
341
+
342
+ const sourceLineBoundsForOffset = (
343
+ source: string,
344
+ start: number,
345
+ end: number
346
+ ): { readonly lineEnd: number; readonly lineStart: number } => {
347
+ const lineStart = source.lastIndexOf('\n', start - 1) + 1;
348
+ const nextLine = source.indexOf('\n', end);
349
+ return { lineEnd: nextLine === -1 ? source.length : nextLine, lineStart };
350
+ };
351
+
352
+ const markdownBacktickRuns = (value: string): readonly RegExpMatchArray[] => [
353
+ ...value.matchAll(/(?<!\\)`+/g),
354
+ ];
355
+
356
+ const isMarkdownInlineCodeContext = (
357
+ source: string,
358
+ start: number,
359
+ end: number
360
+ ): boolean => {
361
+ const { lineEnd, lineStart } = sourceLineBoundsForOffset(source, start, end);
362
+ const line = source.slice(lineStart, lineEnd);
363
+ const relativeStart = start - lineStart;
364
+ const relativeEnd = end - lineStart;
365
+ let openRun: { readonly length: number; readonly start: number } | undefined;
366
+
367
+ for (const run of markdownBacktickRuns(line)) {
368
+ const runStart = run.index ?? 0;
369
+ const [value] = run;
370
+ const runLength = value.length;
371
+ if (openRun === undefined) {
372
+ openRun = { length: runLength, start: runStart };
373
+ continue;
374
+ }
375
+ if (runLength !== openRun.length) {
376
+ continue;
377
+ }
378
+ if (
379
+ openRun.start + openRun.length <= relativeStart &&
380
+ relativeEnd <= runStart
381
+ ) {
382
+ return true;
383
+ }
384
+ openRun = undefined;
385
+ }
386
+
387
+ return false;
388
+ };
389
+
390
+ const markdownFenceLinePattern = /^\s*(?:>\s*){0,8}(```|~~~)/;
391
+
392
+ const isMarkdownFenceContext = (source: string, start: number): boolean => {
393
+ const before = source.slice(0, start);
394
+ let fenced = false;
395
+ for (const line of before.split('\n')) {
396
+ if (markdownFenceLinePattern.test(line)) {
397
+ fenced = !fenced;
398
+ }
399
+ }
400
+ return fenced;
401
+ };
402
+
403
+ const isMarkdownCodeContext = (
404
+ file: SourceFile,
405
+ start: number,
406
+ end: number
407
+ ): boolean =>
408
+ isMarkdownPath(file.path) &&
409
+ (isMarkdownInlineCodeContext(file.source, start, end) ||
410
+ isMarkdownFenceContext(file.source, start));
411
+
412
+ const vocabularyOccurrenceReason = (
413
+ preserveRule: VocabularyPreserveRule | undefined,
414
+ markdownCodeContext: boolean,
415
+ defaultReason: string
416
+ ): string => {
417
+ if (preserveRule !== undefined) {
418
+ return preserveRule.reason ?? 'preserved-by-plan';
419
+ }
420
+ if (markdownCodeContext) {
421
+ return 'markdown-code-context';
422
+ }
423
+ return defaultReason;
424
+ };
425
+
426
+ const capturedVocabularyVerdict = (
427
+ preserveRule: VocabularyPreserveRule | undefined,
428
+ markdownCodeContext: boolean
429
+ ): VocabularyVerdict => {
430
+ if (preserveRule !== undefined) {
431
+ return 'skipped';
432
+ }
433
+ if (markdownCodeContext) {
434
+ return 'deferred';
435
+ }
436
+ return 'modified';
437
+ };
438
+
439
+ const vocabularyOccurrenceDisposition = (
440
+ verdict: VocabularyVerdict,
441
+ preserveRule: VocabularyPreserveRule | undefined,
442
+ markdownCodeContext: boolean
443
+ ): VocabularyDisposition => {
444
+ if (preserveRule !== undefined) {
445
+ return preserveRule.disposition ?? 'explicit-preserve';
446
+ }
447
+ if (markdownCodeContext) {
448
+ return 'code-context-out-of-engine';
449
+ }
450
+ if (verdict === 'modified') {
451
+ return 'in-family-modified';
452
+ }
453
+ return 'in-family-unresolved';
193
454
  };
194
455
 
195
456
  const preserveCase = (sourceForm: string, replacement: string): string => {
@@ -203,16 +464,78 @@ const preserveCase = (sourceForm: string, replacement: string): string => {
203
464
  return replacement;
204
465
  };
205
466
 
206
- const pluralize = (value: string): string =>
207
- value.endsWith('s') || value.endsWith('x') || value.endsWith('ch')
208
- ? `${value}es`
209
- : `${value}s`;
467
+ const isSimpleVocabularyWord = (value: string): boolean =>
468
+ /^[A-Za-z]+$/.test(value);
210
469
 
211
- const defaultVocabularyForms = (from: string, to: string) =>
212
- new Map<string, string>([
213
- [from, to],
214
- [pluralize(from), pluralize(to)],
215
- ]);
470
+ const endsWithConsonantY = (value: string): boolean => {
471
+ const penultimate = value.at(-2);
472
+ return (
473
+ value.endsWith('y') &&
474
+ penultimate !== undefined &&
475
+ !/[aeiou]/.test(penultimate)
476
+ );
477
+ };
478
+
479
+ const pluralize = (value: string): string => {
480
+ const lower = value.toLowerCase();
481
+ let lowerForm: string;
482
+ if (endsWithConsonantY(lower)) {
483
+ lowerForm = `${lower.slice(0, -1)}ies`;
484
+ } else if (
485
+ lower.endsWith('s') ||
486
+ lower.endsWith('x') ||
487
+ lower.endsWith('ch')
488
+ ) {
489
+ lowerForm = `${lower}es`;
490
+ } else {
491
+ lowerForm = `${lower}s`;
492
+ }
493
+ return preserveCase(value, lowerForm);
494
+ };
495
+
496
+ const pastTenseForm = (value: string): string => {
497
+ const lower = value.toLowerCase();
498
+ let lowerForm: string;
499
+ if (endsWithConsonantY(lower)) {
500
+ lowerForm = `${lower.slice(0, -1)}ied`;
501
+ } else if (lower.endsWith('e')) {
502
+ lowerForm = `${lower}d`;
503
+ } else {
504
+ lowerForm = `${lower}ed`;
505
+ }
506
+ return preserveCase(value, lowerForm);
507
+ };
508
+
509
+ const presentParticipleForm = (value: string): string => {
510
+ const lower = value.toLowerCase();
511
+ let lowerForm: string;
512
+ if (lower.endsWith('ie')) {
513
+ lowerForm = `${lower.slice(0, -2)}ying`;
514
+ } else if (lower.endsWith('e') && !lower.endsWith('ee')) {
515
+ lowerForm = `${lower.slice(0, -1)}ing`;
516
+ } else {
517
+ lowerForm = `${lower}ing`;
518
+ }
519
+ return preserveCase(value, lowerForm);
520
+ };
521
+
522
+ const defaultDeferredVocabularyForms = (from: string): readonly string[] => {
523
+ if (!isSimpleVocabularyWord(from)) {
524
+ return [];
525
+ }
526
+ return uniqueSorted([
527
+ pastTenseForm(from),
528
+ presentParticipleForm(from),
529
+ ]).filter((form) => form !== from && form !== pluralize(from));
530
+ };
531
+
532
+ const defaultVocabularyForms = (from: string, to: string) => {
533
+ const forms = new Map<string, string>([[from, to]]);
534
+ if (isSimpleVocabularyWord(from) && isSimpleVocabularyWord(to)) {
535
+ forms.set(pluralize(from), pluralize(to));
536
+ }
537
+ return forms;
538
+ };
216
539
 
217
540
  const normalizedOverrideEntries = (
218
541
  overrides: Readonly<Record<string, string>> | undefined
@@ -221,6 +544,11 @@ const normalizedOverrideEntries = (
221
544
  left.localeCompare(right)
222
545
  );
223
546
 
547
+ const formIdentityForPlan = (
548
+ plan: VocabularyRegradePlan,
549
+ form: string
550
+ ): string => (plan.caseSensitive === true ? form : form.toLowerCase());
551
+
224
552
  const targetFormsForPlan = (
225
553
  plan: VocabularyRegradePlan
226
554
  ): Map<string, string> => {
@@ -231,6 +559,20 @@ const targetFormsForPlan = (
231
559
  return forms;
232
560
  };
233
561
 
562
+ const deferFormsForPlan = (plan: VocabularyRegradePlan): readonly string[] => {
563
+ const overrideForms = new Set(
564
+ normalizedOverrideEntries(plan.overrides).map(([form]) =>
565
+ formIdentityForPlan(plan, form)
566
+ )
567
+ );
568
+ return uniqueSorted([
569
+ ...defaultDeferredVocabularyForms(plan.from).filter(
570
+ (form) => !overrideForms.has(formIdentityForPlan(plan, form))
571
+ ),
572
+ ...(plan.deferForms ?? []),
573
+ ]);
574
+ };
575
+
234
576
  const validateVocabularyPlan = (
235
577
  plan: VocabularyRegradePlan
236
578
  ): Result<void, ValidationError> => {
@@ -260,6 +602,15 @@ const validateVocabularyPlan = (
260
602
  );
261
603
  }
262
604
  }
605
+ for (const form of deferFormsForPlan(plan)) {
606
+ if (form.trim().length === 0) {
607
+ return Result.err(
608
+ new ValidationError(
609
+ 'Vocabulary Regrade plan deferForms entries cannot be empty.'
610
+ )
611
+ );
612
+ }
613
+ }
263
614
  for (const rule of plan.preserve ?? []) {
264
615
  if (rule.pattern.trim().length === 0) {
265
616
  return Result.err(
@@ -268,10 +619,83 @@ const validateVocabularyPlan = (
268
619
  )
269
620
  );
270
621
  }
622
+ if (
623
+ rule.disposition !== undefined &&
624
+ !vocabularyDispositions.has(rule.disposition)
625
+ ) {
626
+ return Result.err(
627
+ new ValidationError(
628
+ `Vocabulary Regrade plan preserve disposition "${rule.disposition}" is not supported.`
629
+ )
630
+ );
631
+ }
632
+ if (rule.forms?.some((form) => form.trim().length === 0) === true) {
633
+ return Result.err(
634
+ new ValidationError(
635
+ 'Vocabulary Regrade plan preserve forms cannot be empty.'
636
+ )
637
+ );
638
+ }
271
639
  }
272
640
  return Result.ok();
273
641
  };
274
642
 
643
+ const validatePreserveInventory = (
644
+ inventory: readonly VocabularyPreserveInventoryEntry[] | undefined
645
+ ): Result<void, ValidationError> => {
646
+ for (const entry of inventory ?? []) {
647
+ if (entry.pattern.trim().length === 0) {
648
+ return Result.err(
649
+ new ValidationError(
650
+ 'Vocabulary Regrade preserve inventory patterns cannot be empty.'
651
+ )
652
+ );
653
+ }
654
+ if (entry.forms?.some((form) => form.trim().length === 0) === true) {
655
+ return Result.err(
656
+ new ValidationError(
657
+ 'Vocabulary Regrade preserve inventory forms cannot be empty.'
658
+ )
659
+ );
660
+ }
661
+ if (
662
+ entry.disposition !== undefined &&
663
+ !vocabularyDispositions.has(entry.disposition)
664
+ ) {
665
+ return Result.err(
666
+ new ValidationError(
667
+ `Vocabulary Regrade preserve inventory disposition "${entry.disposition}" is not supported.`
668
+ )
669
+ );
670
+ }
671
+ if (entry.evidence.length === 0) {
672
+ return Result.err(
673
+ new ValidationError(
674
+ 'Vocabulary Regrade preserve inventory entries need evidence.'
675
+ )
676
+ );
677
+ }
678
+ }
679
+ return Result.ok();
680
+ };
681
+
682
+ const effectivePlanForRun = (
683
+ plan: VocabularyRegradePlan,
684
+ preserveInventory: readonly VocabularyPreserveInventoryEntry[] | undefined
685
+ ): VocabularyRegradePlan => {
686
+ if (preserveInventory === undefined || preserveInventory.length === 0) {
687
+ return plan;
688
+ }
689
+
690
+ return {
691
+ ...plan,
692
+ preserve: [...(plan.preserve ?? []), ...preserveInventory],
693
+ };
694
+ };
695
+
696
+ const vocabularyScanFlags = (plan: VocabularyRegradePlan): string =>
697
+ plan.caseSensitive === true ? 'g' : 'gi';
698
+
275
699
  const includedByScope = (
276
700
  path: string,
277
701
  scope: VocabularyRegradeScope | undefined
@@ -289,27 +713,202 @@ const compilePreservePattern = (pattern: string): RegExp => {
289
713
  }
290
714
  };
291
715
 
716
+ const globalPreservePattern = (pattern: RegExp): RegExp => {
717
+ const flags = pattern.flags.includes('g')
718
+ ? pattern.flags
719
+ : `${pattern.flags}g`;
720
+ return new RegExp(pattern.source, flags);
721
+ };
722
+
723
+ const patternOverlapsOccurrence = (
724
+ pattern: RegExp,
725
+ occurrence: SourceOccurrenceDraft
726
+ ): boolean => {
727
+ const occurrenceStart = occurrence.contextColumn - 1;
728
+ const occurrenceEnd = occurrenceStart + occurrence.form.length;
729
+
730
+ for (const match of occurrence.context.matchAll(
731
+ globalPreservePattern(pattern)
732
+ )) {
733
+ const matchStart = match.index ?? 0;
734
+ const matchEnd = matchStart + match[0].length;
735
+ if (
736
+ matchStart !== matchEnd &&
737
+ occurrenceStart < matchEnd &&
738
+ matchStart < occurrenceEnd
739
+ ) {
740
+ return true;
741
+ }
742
+ }
743
+
744
+ return false;
745
+ };
746
+
292
747
  const preserveRuleForOccurrence = (
293
- occurrence: Omit<SourceOccurrence, 'reason' | 'verdict'>,
748
+ occurrence: SourceOccurrenceDraft,
294
749
  plan: VocabularyRegradePlan
295
750
  ): VocabularyPreserveRule | undefined =>
296
751
  plan.preserve?.find((rule) => {
752
+ if (rule.forms !== undefined && !rule.forms.includes(occurrence.form)) {
753
+ return false;
754
+ }
297
755
  if (
298
756
  rule.paths !== undefined &&
299
757
  !matchesAnyPathGlob(occurrence.path, rule.paths)
300
758
  ) {
301
759
  return false;
302
760
  }
303
- return (
304
- compilePreservePattern(rule.pattern).test(occurrence.context) ||
305
- compilePreservePattern(rule.pattern).test(occurrence.form)
306
- );
761
+ const pattern = compilePreservePattern(rule.pattern);
762
+ if (
763
+ pattern.test(occurrence.form) ||
764
+ patternOverlapsOccurrence(pattern, occurrence)
765
+ ) {
766
+ return true;
767
+ }
768
+ return rule.forms === undefined && pattern.test(occurrence.context);
307
769
  });
308
770
 
309
- const occurrencesForFile = (
771
+ const occurrenceOverlaps = (
772
+ occurrences: readonly {
773
+ readonly end: number;
774
+ readonly start: number;
775
+ }[],
776
+ start: number,
777
+ end: number
778
+ ): boolean =>
779
+ occurrences.some(
780
+ (occurrence) => start < occurrence.end && occurrence.start < end
781
+ );
782
+
783
+ const occurrenceDraftForSpan = (
784
+ file: SourceFile,
785
+ start: number,
786
+ end: number,
787
+ form = file.source.slice(start, end)
788
+ ): SourceOccurrenceDraft => {
789
+ const { column, line } = lineColumnForOffset(file.source, start);
790
+ const context = contextDetailsForOffset(file.source, start, end);
791
+ return {
792
+ absolutePath: file.absolutePath,
793
+ column,
794
+ context: context.context,
795
+ contextColumn: context.contextColumn,
796
+ end,
797
+ form,
798
+ line,
799
+ path: file.path,
800
+ start,
801
+ };
802
+ };
803
+
804
+ const deferredOccurrenceFromDraft = (
805
+ file: SourceFile,
806
+ plan: VocabularyRegradePlan,
807
+ baseOccurrence: SourceOccurrenceDraft,
808
+ reason = 'unclassified-neighbor'
809
+ ): SourceOccurrence => {
810
+ const preserveRule = preserveRuleForOccurrence(baseOccurrence, plan);
811
+ const markdownCodeContext = isMarkdownCodeContext(
812
+ file,
813
+ baseOccurrence.start,
814
+ baseOccurrence.end
815
+ );
816
+ const verdict = preserveRule === undefined ? 'deferred' : 'skipped';
817
+ return {
818
+ absolutePath: baseOccurrence.absolutePath,
819
+ column: baseOccurrence.column,
820
+ context: baseOccurrence.context,
821
+ disposition: vocabularyOccurrenceDisposition(
822
+ verdict,
823
+ preserveRule,
824
+ markdownCodeContext
825
+ ),
826
+ end: baseOccurrence.end,
827
+ form: baseOccurrence.form,
828
+ line: baseOccurrence.line,
829
+ path: baseOccurrence.path,
830
+ reason: vocabularyOccurrenceReason(
831
+ preserveRule,
832
+ markdownCodeContext,
833
+ reason
834
+ ),
835
+ start: baseOccurrence.start,
836
+ verdict,
837
+ };
838
+ };
839
+
840
+ const exactDeferredFormOccurrencesForFile = (
841
+ file: SourceFile,
842
+ plan: VocabularyRegradePlan,
843
+ deferForms: readonly string[],
844
+ targetFormSpans: readonly {
845
+ readonly end: number;
846
+ readonly start: number;
847
+ }[]
848
+ ): readonly SourceOccurrence[] => {
849
+ const occurrences: SourceOccurrence[] = [];
850
+ const authoredDeferForms = new Set(
851
+ (plan.deferForms ?? []).map((form) => formIdentityForPlan(plan, form))
852
+ );
853
+ for (const form of deferForms) {
854
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
855
+ for (const match of file.source.matchAll(pattern)) {
856
+ const start = match.index ?? 0;
857
+ const end = start + match[0].length;
858
+ const isAuthoredDefer = authoredDeferForms.has(
859
+ formIdentityForPlan(plan, form)
860
+ );
861
+ if (
862
+ !hasWordBoundary(file.source, start, end, form) ||
863
+ occurrenceOverlaps(occurrences, start, end) ||
864
+ (!isAuthoredDefer && occurrenceOverlaps(targetFormSpans, start, end))
865
+ ) {
866
+ continue;
867
+ }
868
+ occurrences.push(
869
+ deferredOccurrenceFromDraft(
870
+ file,
871
+ plan,
872
+ occurrenceDraftForSpan(file, start, end, match[0]),
873
+ 'deferred-form'
874
+ )
875
+ );
876
+ }
877
+ }
878
+ return occurrences;
879
+ };
880
+
881
+ const targetFormSpansForFile = (
310
882
  file: SourceFile,
311
883
  plan: VocabularyRegradePlan,
312
884
  targetForms: Map<string, string>
885
+ ): readonly {
886
+ readonly end: number;
887
+ readonly start: number;
888
+ }[] => {
889
+ const spans: { end: number; start: number }[] = [];
890
+ for (const form of targetForms.keys()) {
891
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
892
+ for (const match of file.source.matchAll(pattern)) {
893
+ const start = match.index ?? 0;
894
+ const end = start + match[0].length;
895
+ if (
896
+ !hasWordBoundary(file.source, start, end, form) ||
897
+ occurrenceOverlaps(spans, start, end)
898
+ ) {
899
+ continue;
900
+ }
901
+ spans.push({ end, start });
902
+ }
903
+ }
904
+ return spans;
905
+ };
906
+
907
+ const occurrencesForFile = (
908
+ file: SourceFile,
909
+ plan: VocabularyRegradePlan,
910
+ targetForms: Map<string, string>,
911
+ deferredOccurrences: readonly SourceOccurrence[]
313
912
  ): readonly SourceOccurrence[] => {
314
913
  const occurrences: SourceOccurrence[] = [];
315
914
  const candidates: SourceOccurrence[] = [];
@@ -318,18 +917,23 @@ const occurrencesForFile = (
318
917
  );
319
918
 
320
919
  for (const [form, replacement] of forms) {
321
- const pattern = new RegExp(escapeRegExp(form), 'gi');
920
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
322
921
  for (const match of file.source.matchAll(pattern)) {
323
922
  const start = match.index ?? 0;
324
923
  const end = start + match[0].length;
325
- if (!hasWordBoundary(file.source, start, end)) {
924
+ if (
925
+ !hasWordBoundary(file.source, start, end, form) ||
926
+ occurrenceOverlaps(deferredOccurrences, start, end)
927
+ ) {
326
928
  continue;
327
929
  }
328
930
  const { column, line } = lineColumnForOffset(file.source, start);
931
+ const context = contextDetailsForOffset(file.source, start, end);
329
932
  const baseOccurrence = {
330
933
  absolutePath: file.absolutePath,
331
934
  column,
332
- context: contextForOffset(file.source, start, end),
935
+ context: context.context,
936
+ contextColumn: context.contextColumn,
333
937
  end,
334
938
  form: match[0],
335
939
  line,
@@ -337,15 +941,54 @@ const occurrencesForFile = (
337
941
  start,
338
942
  };
339
943
  const preserveRule = preserveRuleForOccurrence(baseOccurrence, plan);
944
+ const markdownCodeContext = isMarkdownCodeContext(file, start, end);
945
+ const packageRouteCodeContext = isPackageRouteCodeOccurrence(
946
+ file.path,
947
+ form
948
+ );
949
+ let verdict = capturedVocabularyVerdict(
950
+ preserveRule,
951
+ markdownCodeContext
952
+ );
953
+ const packageManifestContext = isPackageManifestPath(file.path);
954
+ if (
955
+ preserveRule === undefined &&
956
+ (packageRouteCodeContext || packageManifestContext)
957
+ ) {
958
+ verdict = 'deferred';
959
+ }
960
+ let capturedReason = 'captured-form';
961
+ if (packageRouteCodeContext) {
962
+ capturedReason = 'package-route-ast-required';
963
+ } else if (packageManifestContext) {
964
+ capturedReason = 'package-manifest-structured-edit-required';
965
+ }
340
966
  candidates.push({
341
- ...baseOccurrence,
342
- reason:
343
- preserveRule?.reason ??
344
- (preserveRule === undefined ? 'captured-form' : 'preserved-by-plan'),
345
- ...(preserveRule === undefined
967
+ absolutePath: baseOccurrence.absolutePath,
968
+ column: baseOccurrence.column,
969
+ context: baseOccurrence.context,
970
+ disposition: vocabularyOccurrenceDisposition(
971
+ verdict,
972
+ preserveRule,
973
+ markdownCodeContext
974
+ ),
975
+ end: baseOccurrence.end,
976
+ form: baseOccurrence.form,
977
+ line: baseOccurrence.line,
978
+ path: baseOccurrence.path,
979
+ reason: vocabularyOccurrenceReason(
980
+ preserveRule,
981
+ markdownCodeContext,
982
+ capturedReason
983
+ ),
984
+ ...(preserveRule === undefined &&
985
+ !markdownCodeContext &&
986
+ !packageRouteCodeContext &&
987
+ !packageManifestContext
346
988
  ? { replacement: preserveCase(match[0], replacement) }
347
989
  : {}),
348
- verdict: preserveRule === undefined ? 'modified' : 'skipped',
990
+ start: baseOccurrence.start,
991
+ verdict,
349
992
  });
350
993
  }
351
994
  }
@@ -376,71 +1019,68 @@ const deferredOccurrencesForFile = (
376
1019
  plan: VocabularyRegradePlan,
377
1020
  targetForms: Map<string, string>
378
1021
  ): readonly SourceOccurrence[] => {
1022
+ const deferForms = deferFormsForPlan(plan);
1023
+ const targetFormSpans = targetFormSpansForFile(file, plan, targetForms);
379
1024
  const knownForms = new Set(
380
- [...targetForms.keys()].flatMap((form) => [form, form.toLowerCase()])
1025
+ plan.caseSensitive === true
1026
+ ? [...targetForms.keys(), ...deferForms]
1027
+ : [...targetForms.keys(), ...deferForms].flatMap((form) => [
1028
+ form,
1029
+ form.toLowerCase(),
1030
+ ])
381
1031
  );
382
1032
  const lowerFrom = plan.from.toLowerCase();
383
1033
  const tokenPattern = /[A-Za-z_$][A-Za-z0-9_$-]*/g;
384
- const occurrences: SourceOccurrence[] = [];
385
- const pushOccurrence = (
386
- baseOccurrence: Omit<SourceOccurrence, 'reason' | 'verdict'>
387
- ): void => {
388
- const preserveRule = preserveRuleForOccurrence(baseOccurrence, plan);
389
- occurrences.push({
390
- ...baseOccurrence,
391
- reason:
392
- preserveRule?.reason ??
393
- (preserveRule === undefined
394
- ? 'unclassified-neighbor'
395
- : 'preserved-by-plan'),
396
- verdict: preserveRule === undefined ? 'deferred' : 'skipped',
397
- });
398
- };
1034
+ const occurrences = [
1035
+ ...exactDeferredFormOccurrencesForFile(
1036
+ file,
1037
+ plan,
1038
+ deferForms,
1039
+ targetFormSpans
1040
+ ),
1041
+ ];
399
1042
 
400
1043
  for (const form of targetForms.keys()) {
401
- const pattern = new RegExp(escapeRegExp(form), 'gi');
1044
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
402
1045
  for (const match of file.source.matchAll(pattern)) {
403
1046
  const matchStart = match.index ?? 0;
404
1047
  const matchEnd = matchStart + match[0].length;
405
- if (hasWordBoundary(file.source, matchStart, matchEnd)) {
1048
+ if (hasWordBoundary(file.source, matchStart, matchEnd, form)) {
406
1049
  continue;
407
1050
  }
408
1051
  const { end, start } = expandVocabularyNeighborSpan(
409
1052
  file.source,
410
1053
  matchStart,
411
- matchEnd
1054
+ matchEnd,
1055
+ form
412
1056
  );
413
1057
  const matchedForm = file.source.slice(start, end);
414
1058
  const lowerMatchedForm = matchedForm.toLowerCase();
415
- const overlaps = occurrences.some(
416
- (occurrence) => start < occurrence.end && occurrence.start < end
417
- );
418
1059
  if (
419
- overlaps ||
1060
+ occurrenceOverlaps(occurrences, start, end) ||
420
1061
  knownForms.has(matchedForm) ||
421
- knownForms.has(lowerMatchedForm) ||
1062
+ (plan.caseSensitive !== true && knownForms.has(lowerMatchedForm)) ||
422
1063
  !lowerMatchedForm.includes(lowerFrom)
423
1064
  ) {
424
1065
  continue;
425
1066
  }
426
- const { column, line } = lineColumnForOffset(file.source, start);
427
- pushOccurrence({
428
- absolutePath: file.absolutePath,
429
- column,
430
- context: contextForOffset(file.source, start, end),
431
- end,
432
- form: matchedForm,
433
- line,
434
- path: file.path,
435
- start,
436
- });
1067
+ occurrences.push(
1068
+ deferredOccurrenceFromDraft(
1069
+ file,
1070
+ plan,
1071
+ occurrenceDraftForSpan(file, start, end, matchedForm)
1072
+ )
1073
+ );
437
1074
  }
438
1075
  }
439
1076
 
440
1077
  for (const match of file.source.matchAll(tokenPattern)) {
441
1078
  const [form] = match;
442
1079
  const lower = form.toLowerCase();
443
- if (knownForms.has(form) || knownForms.has(lower)) {
1080
+ if (
1081
+ knownForms.has(form) ||
1082
+ (plan.caseSensitive !== true && knownForms.has(lower))
1083
+ ) {
444
1084
  continue;
445
1085
  }
446
1086
  if (!lower.includes(lowerFrom)) {
@@ -448,23 +1088,16 @@ const deferredOccurrencesForFile = (
448
1088
  }
449
1089
  const start = match.index ?? 0;
450
1090
  const end = start + form.length;
451
- const overlaps = occurrences.some(
452
- (occurrence) => start < occurrence.end && occurrence.start < end
453
- );
454
- if (overlaps) {
1091
+ if (occurrenceOverlaps(occurrences, start, end)) {
455
1092
  continue;
456
1093
  }
457
- const { column, line } = lineColumnForOffset(file.source, start);
458
- pushOccurrence({
459
- absolutePath: file.absolutePath,
460
- column,
461
- context: contextForOffset(file.source, start, end),
462
- end,
463
- form,
464
- line,
465
- path: file.path,
466
- start,
467
- });
1094
+ occurrences.push(
1095
+ deferredOccurrenceFromDraft(
1096
+ file,
1097
+ plan,
1098
+ occurrenceDraftForSpan(file, start, end, form)
1099
+ )
1100
+ );
468
1101
  }
469
1102
  return occurrences.toSorted((left, right) =>
470
1103
  left.path === right.path
@@ -548,22 +1181,37 @@ const applyOccurrenceRewrites = (
548
1181
 
549
1182
  const buildVocabularyEvaluation = (params: {
550
1183
  readonly apply?: boolean;
1184
+ readonly effectivePlan?: VocabularyRegradePlan | undefined;
551
1185
  readonly files: readonly SourceFile[];
552
1186
  readonly plan: VocabularyRegradePlan;
1187
+ readonly preserveInventory?: readonly VocabularyPreserveInventoryEntry[];
553
1188
  readonly root: string;
554
1189
  readonly skipped: readonly SkippedSource[];
555
1190
  }): VocabularyEvaluation => {
556
- const targetForms = targetFormsForPlan(params.plan);
1191
+ const effectivePlan = params.effectivePlan ?? params.plan;
1192
+ const targetForms = targetFormsForPlan(effectivePlan);
557
1193
  const scopedFiles = params.files.filter((file) =>
558
- includedByScope(file.path, params.plan.scope)
1194
+ includedByScope(file.path, effectivePlan.scope)
559
1195
  );
560
1196
  const scopeSkipped: SkippedSource[] = params.files
561
- .filter((file) => !includedByScope(file.path, params.plan.scope))
1197
+ .filter((file) => !includedByScope(file.path, effectivePlan.scope))
562
1198
  .map((file) => ({ path: file.path, reason: 'excluded-by-regrade-scope' }));
563
- const occurrences = scopedFiles.flatMap((file) => [
564
- ...occurrencesForFile(file, params.plan, targetForms),
565
- ...deferredOccurrencesForFile(file, params.plan, targetForms),
566
- ]);
1199
+ const occurrences = scopedFiles.flatMap((file) => {
1200
+ const deferredOccurrences = deferredOccurrencesForFile(
1201
+ file,
1202
+ effectivePlan,
1203
+ targetForms
1204
+ );
1205
+ return [
1206
+ ...occurrencesForFile(
1207
+ file,
1208
+ effectivePlan,
1209
+ targetForms,
1210
+ deferredOccurrences
1211
+ ),
1212
+ ...deferredOccurrences,
1213
+ ];
1214
+ });
567
1215
  const occurrencesByPath = new Map<string, SourceOccurrence[]>();
568
1216
  for (const occurrence of occurrences) {
569
1217
  const existing = occurrencesByPath.get(occurrence.path) ?? [];
@@ -594,6 +1242,10 @@ const buildVocabularyEvaluation = (params: {
594
1242
  const deferredOccurrences = occurrences.filter(
595
1243
  (occurrence) => occurrence.verdict === 'deferred'
596
1244
  );
1245
+ const unresolvedOccurrences = occurrences.filter(
1246
+ (occurrence) =>
1247
+ occurrence.verdict === 'modified' || occurrence.verdict === 'deferred'
1248
+ );
597
1249
  const forms: Record<string, VocabularyVerdict> = {};
598
1250
  for (const occurrence of occurrences) {
599
1251
  const current = forms[occurrence.form];
@@ -620,7 +1272,7 @@ const buildVocabularyEvaluation = (params: {
620
1272
  if (deferredForms.length > 0) {
621
1273
  gateReasons.push('deferred-forms-or-occurrences');
622
1274
  }
623
- const open = modifiedOccurrences.length + deferredOccurrences.length;
1275
+ const open = unresolvedOccurrences.length;
624
1276
 
625
1277
  return {
626
1278
  entries: [
@@ -646,13 +1298,21 @@ const buildVocabularyEvaluation = (params: {
646
1298
  ),
647
1299
  },
648
1300
  plan: params.plan,
1301
+ ...(params.preserveInventory === undefined ||
1302
+ params.preserveInventory.length === 0
1303
+ ? {}
1304
+ : { preserveInventory: params.preserveInventory }),
649
1305
  report: {
650
1306
  applied: params.apply === true ? modifiedOccurrences.length : 0,
651
1307
  deferred: deferredOccurrences.length,
1308
+ dispositions: vocabularyDispositionCounts(occurrences),
652
1309
  filesChanged: params.apply === true ? rewrittenPaths.size : 0,
653
1310
  gate: {
654
1311
  reasons: gateReasons,
655
1312
  remaining: open,
1313
+ remainingByDisposition: vocabularyDispositionCounts(
1314
+ unresolvedOccurrences
1315
+ ),
656
1316
  status: gateReasons.length === 0 ? 'green' : 'open',
657
1317
  },
658
1318
  modified: modifiedOccurrences.length,
@@ -744,56 +1404,109 @@ const applyVocabularyEvaluation = (
744
1404
  });
745
1405
  };
746
1406
 
1407
+ const readVocabularySourceFiles = (
1408
+ collected: NonNullable<ReturnType<typeof collectDownstreamSources>>
1409
+ ): {
1410
+ readonly files: readonly SourceFile[];
1411
+ readonly skipped: readonly SkippedSource[];
1412
+ } => {
1413
+ const files: SourceFile[] = [];
1414
+ const skipped: SkippedSource[] = [...collected.skipped];
1415
+ for (const file of collected.files) {
1416
+ try {
1417
+ files.push({
1418
+ absolutePath: file.absolutePath,
1419
+ path: file.path,
1420
+ source: readFileSync(file.absolutePath, 'utf8'),
1421
+ });
1422
+ } catch {
1423
+ skipped.push({ path: file.path, reason: 'unreadable-file' });
1424
+ }
1425
+ }
1426
+ return { files, skipped };
1427
+ };
1428
+
1429
+ const buildRunVocabularyEvaluation = (params: {
1430
+ readonly apply: boolean;
1431
+ readonly effectivePlan: VocabularyRegradePlan;
1432
+ readonly files: readonly SourceFile[];
1433
+ readonly plan: VocabularyRegradePlan;
1434
+ readonly preserveInventory:
1435
+ | readonly VocabularyPreserveInventoryEntry[]
1436
+ | undefined;
1437
+ readonly root: string;
1438
+ readonly skipped: readonly SkippedSource[];
1439
+ }): VocabularyEvaluation =>
1440
+ buildVocabularyEvaluation({
1441
+ apply: params.apply,
1442
+ effectivePlan: params.effectivePlan,
1443
+ files: params.files,
1444
+ plan: params.plan,
1445
+ ...(params.preserveInventory === undefined
1446
+ ? {}
1447
+ : { preserveInventory: params.preserveInventory }),
1448
+ root: params.root,
1449
+ skipped: params.skipped,
1450
+ });
1451
+
747
1452
  export const runVocabularyRegrade = (params: {
748
1453
  readonly apply?: boolean;
749
1454
  readonly includeEntries?: 'actionable' | 'all';
750
1455
  readonly plan: VocabularyRegradePlan;
1456
+ readonly preserveInventory?: readonly VocabularyPreserveInventoryEntry[];
751
1457
  readonly root: string;
752
1458
  }): Result<RegradeReport | null, InternalError | ValidationError> => {
753
1459
  const planValidation = validateVocabularyPlan(params.plan);
754
1460
  if (planValidation.isErr()) {
755
1461
  return planValidation;
756
1462
  }
1463
+ const inventoryValidation = validatePreserveInventory(
1464
+ params.preserveInventory
1465
+ );
1466
+ if (inventoryValidation.isErr()) {
1467
+ return inventoryValidation;
1468
+ }
1469
+
1470
+ const effectivePlan = effectivePlanForRun(
1471
+ params.plan,
1472
+ params.preserveInventory
1473
+ );
757
1474
 
758
1475
  const collected = collectDownstreamSources(params.root, {
759
- extensions: params.plan.scope?.extensions ?? VOCABULARY_SOURCE_EXTENSIONS,
760
- ...(params.plan.scope?.exclude === undefined
1476
+ extensions: effectivePlan.scope?.extensions ?? VOCABULARY_SOURCE_EXTENSIONS,
1477
+ ...(effectivePlan.scope?.exclude === undefined
761
1478
  ? {}
762
- : { exclude: params.plan.scope.exclude }),
763
- ...(params.plan.scope?.ignoredDirectories === undefined
1479
+ : { exclude: effectivePlan.scope.exclude }),
1480
+ ...(effectivePlan.scope?.include === undefined
764
1481
  ? {}
765
- : { ignoredDirectories: params.plan.scope.ignoredDirectories }),
1482
+ : { include: effectivePlan.scope.include }),
1483
+ ...(effectivePlan.scope?.ignoredDirectories === undefined
1484
+ ? {}
1485
+ : { ignoredDirectories: effectivePlan.scope.ignoredDirectories }),
766
1486
  } satisfies DownstreamCollectionOptions);
767
1487
  if (collected === null) {
768
1488
  return Result.ok(null);
769
1489
  }
770
1490
 
771
- const files: SourceFile[] = [];
772
- const skipped: SkippedSource[] = [...collected.skipped];
773
- for (const file of collected.files) {
774
- try {
775
- files.push({
776
- absolutePath: file.absolutePath,
777
- path: file.path,
778
- source: readFileSync(file.absolutePath, 'utf8'),
779
- });
780
- } catch {
781
- skipped.push({ path: file.path, reason: 'unreadable-file' });
782
- }
783
- }
1491
+ const { files, skipped } = readVocabularySourceFiles(collected);
784
1492
 
785
- const dryRunEvaluation = buildVocabularyEvaluation({
1493
+ const dryRunEffectiveEvaluation = buildRunVocabularyEvaluation({
786
1494
  apply: false,
1495
+ effectivePlan,
787
1496
  files,
788
1497
  plan: params.plan,
1498
+ preserveInventory: params.preserveInventory,
789
1499
  root: params.root,
790
1500
  skipped,
791
1501
  });
792
- let reportEvaluation = dryRunEvaluation;
1502
+ let reportEvaluation = dryRunEffectiveEvaluation;
793
1503
  let applySummary: RegradeApplySummary | undefined;
794
1504
 
795
1505
  if (params.apply === true) {
796
- const applyResult = applyVocabularyEvaluation(files, dryRunEvaluation);
1506
+ const applyResult = applyVocabularyEvaluation(
1507
+ files,
1508
+ dryRunEffectiveEvaluation
1509
+ );
797
1510
  if (applyResult.isErr()) {
798
1511
  return applyResult;
799
1512
  }
@@ -802,10 +1515,12 @@ export const runVocabularyRegrade = (params: {
802
1515
  ...file,
803
1516
  source: readFileSync(file.absolutePath, 'utf8'),
804
1517
  }));
805
- reportEvaluation = buildVocabularyEvaluation({
1518
+ reportEvaluation = buildRunVocabularyEvaluation({
806
1519
  apply: true,
1520
+ effectivePlan,
807
1521
  files: appliedFiles,
808
1522
  plan: params.plan,
1523
+ preserveInventory: params.preserveInventory,
809
1524
  root: params.root,
810
1525
  skipped,
811
1526
  });
@@ -860,6 +1575,14 @@ export const runVocabularyRegrade = (params: {
860
1575
  };
861
1576
 
862
1577
  const vocabularyPreserveRuleSchema = z.object({
1578
+ disposition: z
1579
+ .enum(vocabularyDispositionValues)
1580
+ .optional()
1581
+ .describe('Classification for occurrences preserved by this rule'),
1582
+ forms: z
1583
+ .array(z.string().min(1))
1584
+ .optional()
1585
+ .describe('Matched forms this preserve rule applies to'),
863
1586
  paths: z
864
1587
  .array(z.string())
865
1588
  .optional()
@@ -868,6 +1591,23 @@ const vocabularyPreserveRuleSchema = z.object({
868
1591
  reason: z.string().optional().describe('Why this form is preserved'),
869
1592
  });
870
1593
 
1594
+ const vocabularyPreserveInventoryEntrySchema =
1595
+ vocabularyPreserveRuleSchema.extend({
1596
+ evidence: z
1597
+ .array(z.string().min(1))
1598
+ .describe('Graph or surface facts that justify this derived preserve'),
1599
+ source: z.literal('derived-live-api').describe('Derived inventory source'),
1600
+ });
1601
+
1602
+ const vocabularyDispositionCountSchema = z.object(
1603
+ Object.fromEntries(
1604
+ vocabularyDispositionValues.map((disposition) => [
1605
+ disposition,
1606
+ z.number().optional(),
1607
+ ])
1608
+ ) as Record<VocabularyDisposition, z.ZodOptional<z.ZodNumber>>
1609
+ );
1610
+
871
1611
  const vocabularyRegradeScopeSchema = z.object({
872
1612
  exclude: z
873
1613
  .array(z.string())
@@ -890,6 +1630,14 @@ const vocabularyRegradeScopeSchema = z.object({
890
1630
  });
891
1631
 
892
1632
  export const vocabularyRegradePlanSchema = z.object({
1633
+ caseSensitive: z
1634
+ .boolean()
1635
+ .optional()
1636
+ .describe('Whether source form scanning preserves case exactly'),
1637
+ deferForms: z
1638
+ .array(z.string().min(1))
1639
+ .optional()
1640
+ .describe('Known forms that must be inventoried for review, not rewritten'),
893
1641
  from: z.string().min(1).describe('Source vocabulary term or phrase'),
894
1642
  id: z.string().optional().describe('Stable authored regrade plan id'),
895
1643
  intent: z.string().optional().describe('Human-authored migration intent'),
@@ -920,6 +1668,9 @@ export const vocabularyRegradeRunOutput = z.object({
920
1668
  z.object({
921
1669
  column: z.number().describe('One-based source column'),
922
1670
  context: z.string().describe('Source-line context'),
1671
+ disposition: z
1672
+ .enum(vocabularyDispositionValues)
1673
+ .describe('Occurrence-level classification beside the verdict'),
923
1674
  end: z.number().describe('Source end offset'),
924
1675
  form: z.string().describe('Matched vocabulary form'),
925
1676
  line: z.number().describe('One-based source line'),
@@ -939,15 +1690,27 @@ export const vocabularyRegradeRunOutput = z.object({
939
1690
  })
940
1691
  .describe('Observed run ledger'),
941
1692
  plan: vocabularyRegradePlanSchema.describe('Authored regrade plan'),
1693
+ preserveInventory: z
1694
+ .array(vocabularyPreserveInventoryEntrySchema)
1695
+ .optional()
1696
+ .describe(
1697
+ 'Derived live-API preserve inventory applied at run time without changing the authored plan'
1698
+ ),
942
1699
  report: z
943
1700
  .object({
944
1701
  applied: z.number().describe('Modified occurrences applied to disk'),
945
1702
  deferred: z.number().describe('Deferred occurrence count'),
1703
+ dispositions: z
1704
+ .object(vocabularyDispositionCountSchema.shape)
1705
+ .describe('Occurrence counts grouped by disposition'),
946
1706
  filesChanged: z.number().describe('Distinct files changed on disk'),
947
1707
  gate: z
948
1708
  .object({
949
1709
  reasons: z.array(z.string()).describe('Open-gate reasons'),
950
1710
  remaining: z.number().describe('Unresolved occurrence count'),
1711
+ remainingByDisposition: z
1712
+ .object(vocabularyDispositionCountSchema.shape)
1713
+ .describe('Unresolved occurrence counts grouped by disposition'),
951
1714
  status: z
952
1715
  .enum(['green', 'open'])
953
1716
  .describe('Whether the run is complete'),
@@ -963,3 +1726,282 @@ export const vocabularyRegradeRunOutput = z.object({
963
1726
  })
964
1727
  .describe('Projected run report'),
965
1728
  });
1729
+
1730
+ const vocabularyTransitionRecordEnvironmentSchema = z
1731
+ .object({
1732
+ commitSha: z.string().optional(),
1733
+ engineVersion: z.string().optional(),
1734
+ graphHash: z.string().optional(),
1735
+ root: z.string(),
1736
+ })
1737
+ .strict();
1738
+
1739
+ const vocabularyTransitionRecordReportSchema = z
1740
+ .object({
1741
+ apply: z.unknown().optional(),
1742
+ entries: z.array(z.unknown()),
1743
+ matched: z.number(),
1744
+ review: z.number(),
1745
+ rewritten: z.number(),
1746
+ root: z.string(),
1747
+ run: vocabularyRegradeRunOutput,
1748
+ scan: z.unknown(),
1749
+ scanned: z.number(),
1750
+ selectedClassIds: z.array(z.string()),
1751
+ skipped: z.number(),
1752
+ skipsByReason: z.record(z.string(), z.number()),
1753
+ unknownClassIds: z.array(z.string()),
1754
+ })
1755
+ .strict();
1756
+
1757
+ const normalizeTransitionRecordPath = (path: string): string =>
1758
+ normalize(path).replaceAll('\\', '/');
1759
+
1760
+ const isSafeRootRelativeRecordPath = (path: string): boolean => {
1761
+ const normalized = normalizeTransitionRecordPath(path);
1762
+ return (
1763
+ normalized.length > 0 &&
1764
+ !isAbsolute(normalized) &&
1765
+ normalized !== '..' &&
1766
+ !normalized.startsWith('../')
1767
+ );
1768
+ };
1769
+
1770
+ export const vocabularyTransitionRecordSchema = z
1771
+ .object({
1772
+ environment: vocabularyTransitionRecordEnvironmentSchema,
1773
+ kind: z.literal('vocabulary-transition-record'),
1774
+ recordPath: z.string().refine(isSafeRootRelativeRecordPath),
1775
+ report: vocabularyTransitionRecordReportSchema,
1776
+ schemaVersion: z.literal(VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION),
1777
+ transition: z
1778
+ .object({
1779
+ from: z.string(),
1780
+ id: z.string(),
1781
+ to: z.string(),
1782
+ })
1783
+ .strict(),
1784
+ })
1785
+ .strict();
1786
+
1787
+ const transitionRecordSlug = (run: VocabularyRegradeRun): string =>
1788
+ `${run.plan.from}-to-${run.plan.to}`
1789
+ .toLowerCase()
1790
+ .replaceAll(/[^a-z0-9]+/g, '-')
1791
+ .replaceAll(/^-|-$/g, '');
1792
+
1793
+ const stableJson = (value: unknown): string =>
1794
+ JSON.stringify(value, (_key, nested) => {
1795
+ if (
1796
+ nested === null ||
1797
+ typeof nested !== 'object' ||
1798
+ Array.isArray(nested)
1799
+ ) {
1800
+ return nested as unknown;
1801
+ }
1802
+ return Object.fromEntries(
1803
+ Object.entries(nested as Record<string, unknown>).toSorted(
1804
+ ([left], [right]) => left.localeCompare(right)
1805
+ )
1806
+ );
1807
+ });
1808
+
1809
+ const shortHashForRun = (
1810
+ run: VocabularyRegradeRun,
1811
+ environment?: Partial<VocabularyTransitionRecordEnvironment>
1812
+ ): string => {
1813
+ const explicitHash = environment?.graphHash ?? environment?.commitSha;
1814
+ if (explicitHash !== undefined && explicitHash.length > 0) {
1815
+ return explicitHash.slice(0, 7);
1816
+ }
1817
+ return createHash('sha256')
1818
+ .update(stableJson({ ledger: run.ledger, plan: run.plan }))
1819
+ .digest('hex')
1820
+ .slice(0, 7);
1821
+ };
1822
+
1823
+ export const vocabularyTransitionRecordPath = (params: {
1824
+ readonly environment?: Partial<VocabularyTransitionRecordEnvironment>;
1825
+ readonly root: string;
1826
+ readonly run: VocabularyRegradeRun;
1827
+ }): string =>
1828
+ join(
1829
+ '.trails',
1830
+ 'regrade',
1831
+ 'history',
1832
+ `${transitionRecordSlug(params.run)}-${shortHashForRun(params.run, params.environment)}.json`
1833
+ );
1834
+
1835
+ const reportWithoutRecord = (
1836
+ report: RegradeReport
1837
+ ): Omit<RegradeReport, 'record'> => {
1838
+ const { record: _record, ...rest } = report;
1839
+ return rest;
1840
+ };
1841
+
1842
+ const transitionRecordPathForWrite = (params: {
1843
+ readonly environment: VocabularyTransitionRecordEnvironment;
1844
+ readonly recordPath?: string;
1845
+ readonly report: RegradeReport;
1846
+ readonly root: string;
1847
+ }): Result<string, ValidationError> => {
1848
+ const recordPath =
1849
+ params.recordPath ??
1850
+ vocabularyTransitionRecordPath({
1851
+ environment: params.environment,
1852
+ root: params.root,
1853
+ run: params.report.run as VocabularyRegradeRun,
1854
+ });
1855
+ const normalized = normalizeTransitionRecordPath(
1856
+ isAbsolute(recordPath) ? relative(params.root, recordPath) : recordPath
1857
+ );
1858
+ if (!isSafeRootRelativeRecordPath(normalized)) {
1859
+ return Result.err(
1860
+ new ValidationError(
1861
+ 'Vocabulary transition record path must stay inside the regrade root.',
1862
+ { context: { recordPath } }
1863
+ )
1864
+ );
1865
+ }
1866
+ return Result.ok(normalized);
1867
+ };
1868
+
1869
+ export const buildVocabularyTransitionRecord = (params: {
1870
+ readonly environment?: Partial<VocabularyTransitionRecordEnvironment>;
1871
+ readonly recordPath?: string;
1872
+ readonly report: RegradeReport;
1873
+ readonly root: string;
1874
+ }): Result<VocabularyTransitionRecord, ValidationError> => {
1875
+ if (params.report.run === undefined) {
1876
+ return Result.err(
1877
+ new ValidationError(
1878
+ 'Vocabulary transition records require a vocabulary Regrade report.'
1879
+ )
1880
+ );
1881
+ }
1882
+
1883
+ const environment: VocabularyTransitionRecordEnvironment = {
1884
+ ...(params.environment?.commitSha === undefined
1885
+ ? {}
1886
+ : { commitSha: params.environment.commitSha }),
1887
+ ...(params.environment?.engineVersion === undefined
1888
+ ? {}
1889
+ : { engineVersion: params.environment.engineVersion }),
1890
+ ...(params.environment?.graphHash === undefined
1891
+ ? {}
1892
+ : { graphHash: params.environment.graphHash }),
1893
+ root: params.root,
1894
+ };
1895
+ const recordPathResult = transitionRecordPathForWrite({
1896
+ environment,
1897
+ report: params.report,
1898
+ root: params.root,
1899
+ ...(params.recordPath === undefined
1900
+ ? {}
1901
+ : { recordPath: params.recordPath }),
1902
+ });
1903
+ if (recordPathResult.isErr()) {
1904
+ return recordPathResult;
1905
+ }
1906
+ const recordPath = recordPathResult.value;
1907
+ const record: VocabularyTransitionRecord = {
1908
+ environment,
1909
+ kind: 'vocabulary-transition-record',
1910
+ recordPath,
1911
+ report: reportWithoutRecord(params.report),
1912
+ schemaVersion: VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION,
1913
+ transition: {
1914
+ from: params.report.run.plan.from,
1915
+ id:
1916
+ params.report.run.plan.id ??
1917
+ `vocabulary:${params.report.run.plan.from}->${params.report.run.plan.to}`,
1918
+ to: params.report.run.plan.to,
1919
+ },
1920
+ };
1921
+ const parsed = vocabularyTransitionRecordSchema.safeParse(record);
1922
+ if (!parsed.success) {
1923
+ return Result.err(
1924
+ new ValidationError('Invalid vocabulary transition record.', {
1925
+ context: { issues: parsed.error.issues },
1926
+ })
1927
+ );
1928
+ }
1929
+ return Result.ok(parsed.data as VocabularyTransitionRecord);
1930
+ };
1931
+
1932
+ export const writeVocabularyTransitionRecord = (params: {
1933
+ readonly environment?: Partial<VocabularyTransitionRecordEnvironment>;
1934
+ readonly recordPath?: string;
1935
+ readonly report: RegradeReport;
1936
+ readonly root: string;
1937
+ readonly status: VocabularyTransitionRecordSummary['status'];
1938
+ }): Result<
1939
+ {
1940
+ readonly record: VocabularyTransitionRecord;
1941
+ readonly summary: VocabularyTransitionRecordSummary;
1942
+ },
1943
+ InternalError | ValidationError
1944
+ > => {
1945
+ const recordResult = buildVocabularyTransitionRecord(params);
1946
+ if (recordResult.isErr()) {
1947
+ return recordResult;
1948
+ }
1949
+ const record = recordResult.value;
1950
+ const absolutePath = isAbsolute(record.recordPath)
1951
+ ? record.recordPath
1952
+ : join(params.root, record.recordPath);
1953
+ try {
1954
+ mkdirSync(dirname(absolutePath), { recursive: true });
1955
+ writeFileSync(absolutePath, `${JSON.stringify(record, null, 2)}\n`);
1956
+ } catch (error) {
1957
+ return Result.err(
1958
+ new InternalError('Failed to write vocabulary transition record.', {
1959
+ ...(error instanceof Error ? { cause: error } : {}),
1960
+ context: { path: record.recordPath },
1961
+ })
1962
+ );
1963
+ }
1964
+ return Result.ok({
1965
+ record,
1966
+ summary: {
1967
+ path: record.recordPath,
1968
+ schemaVersion: record.schemaVersion,
1969
+ status: params.status,
1970
+ },
1971
+ });
1972
+ };
1973
+
1974
+ export const readVocabularyTransitionRecord = (
1975
+ path: string
1976
+ ): Result<VocabularyTransitionRecord, InternalError | ValidationError> => {
1977
+ if (!existsSync(path)) {
1978
+ return Result.err(
1979
+ new ValidationError(`Vocabulary transition record "${path}" not found.`)
1980
+ );
1981
+ }
1982
+ let parsedJson: unknown;
1983
+ try {
1984
+ parsedJson = JSON.parse(readFileSync(path, 'utf8'));
1985
+ } catch (error) {
1986
+ return Result.err(
1987
+ new InternalError('Failed to read vocabulary transition record.', {
1988
+ ...(error instanceof Error ? { cause: error } : {}),
1989
+ context: { path },
1990
+ })
1991
+ );
1992
+ }
1993
+ const parsed = vocabularyTransitionRecordSchema.safeParse(parsedJson);
1994
+ if (!parsed.success) {
1995
+ return Result.err(
1996
+ new ValidationError('Invalid vocabulary transition record.', {
1997
+ context: { issues: parsed.error.issues, path },
1998
+ })
1999
+ );
2000
+ }
2001
+ return Result.ok(parsed.data as VocabularyTransitionRecord);
2002
+ };
2003
+
2004
+ export const transitionRecordReportWithSummary = (
2005
+ report: RegradeReport,
2006
+ summary: VocabularyTransitionRecordSummary
2007
+ ): RegradeReport => ({ ...report, record: summary });