@ontrails/regrade 1.0.0-beta.32 → 1.0.0-beta.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,9 @@ import {
5
5
  escapeRegExp,
6
6
  matchesAnyPathGlob,
7
7
  } from '@ontrails/core';
8
- import { readFileSync, writeFileSync } from 'node:fs';
8
+ import { createHash } from 'node:crypto';
9
+ import { dirname, isAbsolute, join, normalize, relative } from 'node:path';
10
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
9
11
  import { z } from 'zod';
10
12
 
11
13
  import { collectDownstreamSources } from './collect.js';
@@ -19,12 +21,36 @@ import { buildRegradeScanSummary } from './scan-summary.js';
19
21
 
20
22
  export type VocabularyVerdict = 'deferred' | 'modified' | 'skipped';
21
23
 
24
+ export const vocabularyDispositionValues = [
25
+ 'code-context-out-of-engine',
26
+ 'docs-only',
27
+ 'explicit-preserve',
28
+ 'forward-pointer',
29
+ 'ignored-by-scope',
30
+ 'in-family-modified',
31
+ 'in-family-unresolved',
32
+ 'out-of-family',
33
+ 'preserve-current-live-api',
34
+ ] as const;
35
+
36
+ export type VocabularyDisposition =
37
+ (typeof vocabularyDispositionValues)[number];
38
+
39
+ const vocabularyDispositions = new Set<string>(vocabularyDispositionValues);
40
+
22
41
  export interface VocabularyPreserveRule {
42
+ readonly disposition?: VocabularyDisposition;
43
+ readonly forms?: readonly string[];
23
44
  readonly pattern: string;
24
45
  readonly reason?: string;
25
46
  readonly paths?: readonly string[];
26
47
  }
27
48
 
49
+ export interface VocabularyPreserveInventoryEntry extends VocabularyPreserveRule {
50
+ readonly evidence: readonly string[];
51
+ readonly source: 'derived-live-api';
52
+ }
53
+
28
54
  export interface VocabularyRegradeScope {
29
55
  readonly exclude?: readonly string[];
30
56
  readonly extensions?: readonly string[];
@@ -38,6 +64,8 @@ export interface VocabularyRegradeScope {
38
64
  }
39
65
 
40
66
  export interface VocabularyRegradePlan {
67
+ readonly caseSensitive?: boolean;
68
+ readonly deferForms?: readonly string[];
41
69
  readonly from: string;
42
70
  readonly id?: string;
43
71
  readonly intent?: string;
@@ -51,6 +79,7 @@ export interface VocabularyRegradePlan {
51
79
  export interface VocabularyOccurrence {
52
80
  readonly column: number;
53
81
  readonly context: string;
82
+ readonly disposition: VocabularyDisposition;
54
83
  readonly end: number;
55
84
  readonly form: string;
56
85
  readonly line: number;
@@ -69,6 +98,9 @@ export interface VocabularyRunLedger {
69
98
 
70
99
  export interface VocabularyRunGate {
71
100
  readonly remaining: number;
101
+ readonly remainingByDisposition: Partial<
102
+ Readonly<Record<VocabularyDisposition, number>>
103
+ >;
72
104
  readonly reasons: readonly string[];
73
105
  readonly status: 'green' | 'open';
74
106
  }
@@ -76,6 +108,9 @@ export interface VocabularyRunGate {
76
108
  export interface VocabularyRunReport {
77
109
  readonly applied: number;
78
110
  readonly deferred: number;
111
+ readonly dispositions: Partial<
112
+ Readonly<Record<VocabularyDisposition, number>>
113
+ >;
79
114
  readonly filesChanged: number;
80
115
  readonly gate: VocabularyRunGate;
81
116
  readonly modified: number;
@@ -86,9 +121,38 @@ export interface VocabularyRunReport {
86
121
  export interface VocabularyRegradeRun {
87
122
  readonly ledger: VocabularyRunLedger;
88
123
  readonly plan: VocabularyRegradePlan;
124
+ readonly preserveInventory?: readonly VocabularyPreserveInventoryEntry[];
89
125
  readonly report: VocabularyRunReport;
90
126
  }
91
127
 
128
+ export const VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION = 1;
129
+
130
+ export interface VocabularyTransitionRecordEnvironment {
131
+ readonly commitSha?: string;
132
+ readonly engineVersion?: string;
133
+ readonly graphHash?: string;
134
+ readonly root: string;
135
+ }
136
+
137
+ export interface VocabularyTransitionRecord {
138
+ readonly environment: VocabularyTransitionRecordEnvironment;
139
+ readonly kind: 'vocabulary-transition-record';
140
+ readonly recordPath: string;
141
+ readonly report: Omit<RegradeReport, 'record'>;
142
+ readonly schemaVersion: typeof VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION;
143
+ readonly transition: {
144
+ readonly from: string;
145
+ readonly id: string;
146
+ readonly to: string;
147
+ };
148
+ }
149
+
150
+ export interface VocabularyTransitionRecordSummary {
151
+ readonly path: string;
152
+ readonly schemaVersion: typeof VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION;
153
+ readonly status: 'candidate' | 'applied' | 'checked';
154
+ }
155
+
92
156
  interface SourceFile {
93
157
  readonly absolutePath: string;
94
158
  readonly path: string;
@@ -99,6 +163,13 @@ interface SourceOccurrence extends VocabularyOccurrence {
99
163
  readonly absolutePath: string;
100
164
  }
101
165
 
166
+ interface SourceOccurrenceDraft extends Omit<
167
+ SourceOccurrence,
168
+ 'disposition' | 'reason' | 'verdict'
169
+ > {
170
+ readonly contextColumn: number;
171
+ }
172
+
102
173
  interface VocabularyEvaluation {
103
174
  readonly entries: readonly RegradeReportEntry[];
104
175
  readonly occurrences: readonly SourceOccurrence[];
@@ -125,6 +196,23 @@ const VOCABULARY_SOURCE_EXTENSIONS = Object.freeze([
125
196
  const uniqueSorted = (values: readonly string[]): readonly string[] =>
126
197
  [...new Set(values)].toSorted((a, b) => a.localeCompare(b));
127
198
 
199
+ const vocabularyDispositionCounts = (
200
+ occurrences: readonly VocabularyOccurrence[]
201
+ ): Partial<Readonly<Record<VocabularyDisposition, number>>> => {
202
+ const counts = new Map<VocabularyDisposition, number>();
203
+ for (const occurrence of occurrences) {
204
+ counts.set(
205
+ occurrence.disposition,
206
+ (counts.get(occurrence.disposition) ?? 0) + 1
207
+ );
208
+ }
209
+ return Object.fromEntries(
210
+ [...counts.entries()].toSorted(([left], [right]) =>
211
+ left.localeCompare(right)
212
+ )
213
+ );
214
+ };
215
+
128
216
  const isVocabularyTokenCharacter = (value: string): boolean =>
129
217
  /[A-Za-z0-9_$-]/.test(value);
130
218
 
@@ -181,15 +269,137 @@ const lineColumnForOffset = (
181
269
  return { column, line };
182
270
  };
183
271
 
184
- const contextForOffset = (
272
+ const contextDetailsForOffset = (
185
273
  source: string,
186
274
  start: number,
187
275
  end: number
188
- ): string => {
276
+ ): { readonly context: string; readonly contextColumn: number } => {
189
277
  const lineStart = source.lastIndexOf('\n', start - 1) + 1;
190
278
  const nextLine = source.indexOf('\n', end);
191
279
  const lineEnd = nextLine === -1 ? source.length : nextLine;
192
- return source.slice(lineStart, lineEnd).trim();
280
+ const rawLine = source.slice(lineStart, lineEnd);
281
+ const leadingTrimmed = rawLine.length - rawLine.trimStart().length;
282
+ return {
283
+ context: rawLine.trim(),
284
+ contextColumn: start - lineStart - leadingTrimmed + 1,
285
+ };
286
+ };
287
+
288
+ const isMarkdownPath = (path: string): boolean =>
289
+ path.endsWith('.md') || path.endsWith('.mdx');
290
+
291
+ const sourceLineBoundsForOffset = (
292
+ source: string,
293
+ start: number,
294
+ end: number
295
+ ): { readonly lineEnd: number; readonly lineStart: number } => {
296
+ const lineStart = source.lastIndexOf('\n', start - 1) + 1;
297
+ const nextLine = source.indexOf('\n', end);
298
+ return { lineEnd: nextLine === -1 ? source.length : nextLine, lineStart };
299
+ };
300
+
301
+ const markdownBacktickRuns = (value: string): readonly RegExpMatchArray[] => [
302
+ ...value.matchAll(/(?<!\\)`+/g),
303
+ ];
304
+
305
+ const isMarkdownInlineCodeContext = (
306
+ source: string,
307
+ start: number,
308
+ end: number
309
+ ): boolean => {
310
+ const { lineEnd, lineStart } = sourceLineBoundsForOffset(source, start, end);
311
+ const line = source.slice(lineStart, lineEnd);
312
+ const relativeStart = start - lineStart;
313
+ const relativeEnd = end - lineStart;
314
+ let openRun: { readonly length: number; readonly start: number } | undefined;
315
+
316
+ for (const run of markdownBacktickRuns(line)) {
317
+ const runStart = run.index ?? 0;
318
+ const [value] = run;
319
+ const runLength = value.length;
320
+ if (openRun === undefined) {
321
+ openRun = { length: runLength, start: runStart };
322
+ continue;
323
+ }
324
+ if (runLength !== openRun.length) {
325
+ continue;
326
+ }
327
+ if (
328
+ openRun.start + openRun.length <= relativeStart &&
329
+ relativeEnd <= runStart
330
+ ) {
331
+ return true;
332
+ }
333
+ openRun = undefined;
334
+ }
335
+
336
+ return false;
337
+ };
338
+
339
+ const markdownFenceLinePattern = /^\s*(?:>\s*){0,8}(```|~~~)/;
340
+
341
+ const isMarkdownFenceContext = (source: string, start: number): boolean => {
342
+ const before = source.slice(0, start);
343
+ let fenced = false;
344
+ for (const line of before.split('\n')) {
345
+ if (markdownFenceLinePattern.test(line)) {
346
+ fenced = !fenced;
347
+ }
348
+ }
349
+ return fenced;
350
+ };
351
+
352
+ const isMarkdownCodeContext = (
353
+ file: SourceFile,
354
+ start: number,
355
+ end: number
356
+ ): boolean =>
357
+ isMarkdownPath(file.path) &&
358
+ (isMarkdownInlineCodeContext(file.source, start, end) ||
359
+ isMarkdownFenceContext(file.source, start));
360
+
361
+ const vocabularyOccurrenceReason = (
362
+ preserveRule: VocabularyPreserveRule | undefined,
363
+ markdownCodeContext: boolean,
364
+ defaultReason: string
365
+ ): string => {
366
+ if (preserveRule !== undefined) {
367
+ return preserveRule.reason ?? 'preserved-by-plan';
368
+ }
369
+ if (markdownCodeContext) {
370
+ return 'markdown-code-context';
371
+ }
372
+ return defaultReason;
373
+ };
374
+
375
+ const capturedVocabularyVerdict = (
376
+ preserveRule: VocabularyPreserveRule | undefined,
377
+ markdownCodeContext: boolean
378
+ ): VocabularyVerdict => {
379
+ if (preserveRule !== undefined) {
380
+ return 'skipped';
381
+ }
382
+ if (markdownCodeContext) {
383
+ return 'deferred';
384
+ }
385
+ return 'modified';
386
+ };
387
+
388
+ const vocabularyOccurrenceDisposition = (
389
+ verdict: VocabularyVerdict,
390
+ preserveRule: VocabularyPreserveRule | undefined,
391
+ markdownCodeContext: boolean
392
+ ): VocabularyDisposition => {
393
+ if (preserveRule !== undefined) {
394
+ return preserveRule.disposition ?? 'explicit-preserve';
395
+ }
396
+ if (markdownCodeContext) {
397
+ return 'code-context-out-of-engine';
398
+ }
399
+ if (verdict === 'modified') {
400
+ return 'in-family-modified';
401
+ }
402
+ return 'in-family-unresolved';
193
403
  };
194
404
 
195
405
  const preserveCase = (sourceForm: string, replacement: string): string => {
@@ -203,10 +413,70 @@ const preserveCase = (sourceForm: string, replacement: string): string => {
203
413
  return replacement;
204
414
  };
205
415
 
206
- const pluralize = (value: string): string =>
207
- value.endsWith('s') || value.endsWith('x') || value.endsWith('ch')
208
- ? `${value}es`
209
- : `${value}s`;
416
+ const isSimpleVocabularyWord = (value: string): boolean =>
417
+ /^[A-Za-z]+$/.test(value);
418
+
419
+ const endsWithConsonantY = (value: string): boolean => {
420
+ const penultimate = value.at(-2);
421
+ return (
422
+ value.endsWith('y') &&
423
+ penultimate !== undefined &&
424
+ !/[aeiou]/.test(penultimate)
425
+ );
426
+ };
427
+
428
+ const pluralize = (value: string): string => {
429
+ const lower = value.toLowerCase();
430
+ let lowerForm: string;
431
+ if (endsWithConsonantY(lower)) {
432
+ lowerForm = `${lower.slice(0, -1)}ies`;
433
+ } else if (
434
+ lower.endsWith('s') ||
435
+ lower.endsWith('x') ||
436
+ lower.endsWith('ch')
437
+ ) {
438
+ lowerForm = `${lower}es`;
439
+ } else {
440
+ lowerForm = `${lower}s`;
441
+ }
442
+ return preserveCase(value, lowerForm);
443
+ };
444
+
445
+ const pastTenseForm = (value: string): string => {
446
+ const lower = value.toLowerCase();
447
+ let lowerForm: string;
448
+ if (endsWithConsonantY(lower)) {
449
+ lowerForm = `${lower.slice(0, -1)}ied`;
450
+ } else if (lower.endsWith('e')) {
451
+ lowerForm = `${lower}d`;
452
+ } else {
453
+ lowerForm = `${lower}ed`;
454
+ }
455
+ return preserveCase(value, lowerForm);
456
+ };
457
+
458
+ const presentParticipleForm = (value: string): string => {
459
+ const lower = value.toLowerCase();
460
+ let lowerForm: string;
461
+ if (lower.endsWith('ie')) {
462
+ lowerForm = `${lower.slice(0, -2)}ying`;
463
+ } else if (lower.endsWith('e') && !lower.endsWith('ee')) {
464
+ lowerForm = `${lower.slice(0, -1)}ing`;
465
+ } else {
466
+ lowerForm = `${lower}ing`;
467
+ }
468
+ return preserveCase(value, lowerForm);
469
+ };
470
+
471
+ const defaultDeferredVocabularyForms = (from: string): readonly string[] => {
472
+ if (!isSimpleVocabularyWord(from)) {
473
+ return [];
474
+ }
475
+ return uniqueSorted([
476
+ pastTenseForm(from),
477
+ presentParticipleForm(from),
478
+ ]).filter((form) => form !== from && form !== pluralize(from));
479
+ };
210
480
 
211
481
  const defaultVocabularyForms = (from: string, to: string) =>
212
482
  new Map<string, string>([
@@ -221,6 +491,11 @@ const normalizedOverrideEntries = (
221
491
  left.localeCompare(right)
222
492
  );
223
493
 
494
+ const formIdentityForPlan = (
495
+ plan: VocabularyRegradePlan,
496
+ form: string
497
+ ): string => (plan.caseSensitive === true ? form : form.toLowerCase());
498
+
224
499
  const targetFormsForPlan = (
225
500
  plan: VocabularyRegradePlan
226
501
  ): Map<string, string> => {
@@ -231,6 +506,20 @@ const targetFormsForPlan = (
231
506
  return forms;
232
507
  };
233
508
 
509
+ const deferFormsForPlan = (plan: VocabularyRegradePlan): readonly string[] => {
510
+ const overrideForms = new Set(
511
+ normalizedOverrideEntries(plan.overrides).map(([form]) =>
512
+ formIdentityForPlan(plan, form)
513
+ )
514
+ );
515
+ return uniqueSorted([
516
+ ...defaultDeferredVocabularyForms(plan.from).filter(
517
+ (form) => !overrideForms.has(formIdentityForPlan(plan, form))
518
+ ),
519
+ ...(plan.deferForms ?? []),
520
+ ]);
521
+ };
522
+
234
523
  const validateVocabularyPlan = (
235
524
  plan: VocabularyRegradePlan
236
525
  ): Result<void, ValidationError> => {
@@ -260,6 +549,15 @@ const validateVocabularyPlan = (
260
549
  );
261
550
  }
262
551
  }
552
+ for (const form of deferFormsForPlan(plan)) {
553
+ if (form.trim().length === 0) {
554
+ return Result.err(
555
+ new ValidationError(
556
+ 'Vocabulary Regrade plan deferForms entries cannot be empty.'
557
+ )
558
+ );
559
+ }
560
+ }
263
561
  for (const rule of plan.preserve ?? []) {
264
562
  if (rule.pattern.trim().length === 0) {
265
563
  return Result.err(
@@ -268,10 +566,83 @@ const validateVocabularyPlan = (
268
566
  )
269
567
  );
270
568
  }
569
+ if (
570
+ rule.disposition !== undefined &&
571
+ !vocabularyDispositions.has(rule.disposition)
572
+ ) {
573
+ return Result.err(
574
+ new ValidationError(
575
+ `Vocabulary Regrade plan preserve disposition "${rule.disposition}" is not supported.`
576
+ )
577
+ );
578
+ }
579
+ if (rule.forms?.some((form) => form.trim().length === 0) === true) {
580
+ return Result.err(
581
+ new ValidationError(
582
+ 'Vocabulary Regrade plan preserve forms cannot be empty.'
583
+ )
584
+ );
585
+ }
586
+ }
587
+ return Result.ok();
588
+ };
589
+
590
+ const validatePreserveInventory = (
591
+ inventory: readonly VocabularyPreserveInventoryEntry[] | undefined
592
+ ): Result<void, ValidationError> => {
593
+ for (const entry of inventory ?? []) {
594
+ if (entry.pattern.trim().length === 0) {
595
+ return Result.err(
596
+ new ValidationError(
597
+ 'Vocabulary Regrade preserve inventory patterns cannot be empty.'
598
+ )
599
+ );
600
+ }
601
+ if (entry.forms?.some((form) => form.trim().length === 0) === true) {
602
+ return Result.err(
603
+ new ValidationError(
604
+ 'Vocabulary Regrade preserve inventory forms cannot be empty.'
605
+ )
606
+ );
607
+ }
608
+ if (
609
+ entry.disposition !== undefined &&
610
+ !vocabularyDispositions.has(entry.disposition)
611
+ ) {
612
+ return Result.err(
613
+ new ValidationError(
614
+ `Vocabulary Regrade preserve inventory disposition "${entry.disposition}" is not supported.`
615
+ )
616
+ );
617
+ }
618
+ if (entry.evidence.length === 0) {
619
+ return Result.err(
620
+ new ValidationError(
621
+ 'Vocabulary Regrade preserve inventory entries need evidence.'
622
+ )
623
+ );
624
+ }
271
625
  }
272
626
  return Result.ok();
273
627
  };
274
628
 
629
+ const effectivePlanForRun = (
630
+ plan: VocabularyRegradePlan,
631
+ preserveInventory: readonly VocabularyPreserveInventoryEntry[] | undefined
632
+ ): VocabularyRegradePlan => {
633
+ if (preserveInventory === undefined || preserveInventory.length === 0) {
634
+ return plan;
635
+ }
636
+
637
+ return {
638
+ ...plan,
639
+ preserve: [...(plan.preserve ?? []), ...preserveInventory],
640
+ };
641
+ };
642
+
643
+ const vocabularyScanFlags = (plan: VocabularyRegradePlan): string =>
644
+ plan.caseSensitive === true ? 'g' : 'gi';
645
+
275
646
  const includedByScope = (
276
647
  path: string,
277
648
  scope: VocabularyRegradeScope | undefined
@@ -289,27 +660,202 @@ const compilePreservePattern = (pattern: string): RegExp => {
289
660
  }
290
661
  };
291
662
 
663
+ const globalPreservePattern = (pattern: RegExp): RegExp => {
664
+ const flags = pattern.flags.includes('g')
665
+ ? pattern.flags
666
+ : `${pattern.flags}g`;
667
+ return new RegExp(pattern.source, flags);
668
+ };
669
+
670
+ const patternOverlapsOccurrence = (
671
+ pattern: RegExp,
672
+ occurrence: SourceOccurrenceDraft
673
+ ): boolean => {
674
+ const occurrenceStart = occurrence.contextColumn - 1;
675
+ const occurrenceEnd = occurrenceStart + occurrence.form.length;
676
+
677
+ for (const match of occurrence.context.matchAll(
678
+ globalPreservePattern(pattern)
679
+ )) {
680
+ const matchStart = match.index ?? 0;
681
+ const matchEnd = matchStart + match[0].length;
682
+ if (
683
+ matchStart !== matchEnd &&
684
+ occurrenceStart < matchEnd &&
685
+ matchStart < occurrenceEnd
686
+ ) {
687
+ return true;
688
+ }
689
+ }
690
+
691
+ return false;
692
+ };
693
+
292
694
  const preserveRuleForOccurrence = (
293
- occurrence: Omit<SourceOccurrence, 'reason' | 'verdict'>,
695
+ occurrence: SourceOccurrenceDraft,
294
696
  plan: VocabularyRegradePlan
295
697
  ): VocabularyPreserveRule | undefined =>
296
698
  plan.preserve?.find((rule) => {
699
+ if (rule.forms !== undefined && !rule.forms.includes(occurrence.form)) {
700
+ return false;
701
+ }
297
702
  if (
298
703
  rule.paths !== undefined &&
299
704
  !matchesAnyPathGlob(occurrence.path, rule.paths)
300
705
  ) {
301
706
  return false;
302
707
  }
303
- return (
304
- compilePreservePattern(rule.pattern).test(occurrence.context) ||
305
- compilePreservePattern(rule.pattern).test(occurrence.form)
306
- );
708
+ const pattern = compilePreservePattern(rule.pattern);
709
+ if (
710
+ pattern.test(occurrence.form) ||
711
+ patternOverlapsOccurrence(pattern, occurrence)
712
+ ) {
713
+ return true;
714
+ }
715
+ return rule.forms === undefined && pattern.test(occurrence.context);
307
716
  });
308
717
 
309
- const occurrencesForFile = (
718
+ const occurrenceOverlaps = (
719
+ occurrences: readonly {
720
+ readonly end: number;
721
+ readonly start: number;
722
+ }[],
723
+ start: number,
724
+ end: number
725
+ ): boolean =>
726
+ occurrences.some(
727
+ (occurrence) => start < occurrence.end && occurrence.start < end
728
+ );
729
+
730
+ const occurrenceDraftForSpan = (
731
+ file: SourceFile,
732
+ start: number,
733
+ end: number,
734
+ form = file.source.slice(start, end)
735
+ ): SourceOccurrenceDraft => {
736
+ const { column, line } = lineColumnForOffset(file.source, start);
737
+ const context = contextDetailsForOffset(file.source, start, end);
738
+ return {
739
+ absolutePath: file.absolutePath,
740
+ column,
741
+ context: context.context,
742
+ contextColumn: context.contextColumn,
743
+ end,
744
+ form,
745
+ line,
746
+ path: file.path,
747
+ start,
748
+ };
749
+ };
750
+
751
+ const deferredOccurrenceFromDraft = (
752
+ file: SourceFile,
753
+ plan: VocabularyRegradePlan,
754
+ baseOccurrence: SourceOccurrenceDraft,
755
+ reason = 'unclassified-neighbor'
756
+ ): SourceOccurrence => {
757
+ const preserveRule = preserveRuleForOccurrence(baseOccurrence, plan);
758
+ const markdownCodeContext = isMarkdownCodeContext(
759
+ file,
760
+ baseOccurrence.start,
761
+ baseOccurrence.end
762
+ );
763
+ const verdict = preserveRule === undefined ? 'deferred' : 'skipped';
764
+ return {
765
+ absolutePath: baseOccurrence.absolutePath,
766
+ column: baseOccurrence.column,
767
+ context: baseOccurrence.context,
768
+ disposition: vocabularyOccurrenceDisposition(
769
+ verdict,
770
+ preserveRule,
771
+ markdownCodeContext
772
+ ),
773
+ end: baseOccurrence.end,
774
+ form: baseOccurrence.form,
775
+ line: baseOccurrence.line,
776
+ path: baseOccurrence.path,
777
+ reason: vocabularyOccurrenceReason(
778
+ preserveRule,
779
+ markdownCodeContext,
780
+ reason
781
+ ),
782
+ start: baseOccurrence.start,
783
+ verdict,
784
+ };
785
+ };
786
+
787
+ const exactDeferredFormOccurrencesForFile = (
788
+ file: SourceFile,
789
+ plan: VocabularyRegradePlan,
790
+ deferForms: readonly string[],
791
+ targetFormSpans: readonly {
792
+ readonly end: number;
793
+ readonly start: number;
794
+ }[]
795
+ ): readonly SourceOccurrence[] => {
796
+ const occurrences: SourceOccurrence[] = [];
797
+ const authoredDeferForms = new Set(
798
+ (plan.deferForms ?? []).map((form) => formIdentityForPlan(plan, form))
799
+ );
800
+ for (const form of deferForms) {
801
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
802
+ for (const match of file.source.matchAll(pattern)) {
803
+ const start = match.index ?? 0;
804
+ const end = start + match[0].length;
805
+ const isAuthoredDefer = authoredDeferForms.has(
806
+ formIdentityForPlan(plan, form)
807
+ );
808
+ if (
809
+ !hasWordBoundary(file.source, start, end) ||
810
+ occurrenceOverlaps(occurrences, start, end) ||
811
+ (!isAuthoredDefer && occurrenceOverlaps(targetFormSpans, start, end))
812
+ ) {
813
+ continue;
814
+ }
815
+ occurrences.push(
816
+ deferredOccurrenceFromDraft(
817
+ file,
818
+ plan,
819
+ occurrenceDraftForSpan(file, start, end, match[0]),
820
+ 'deferred-form'
821
+ )
822
+ );
823
+ }
824
+ }
825
+ return occurrences;
826
+ };
827
+
828
+ const targetFormSpansForFile = (
310
829
  file: SourceFile,
311
830
  plan: VocabularyRegradePlan,
312
831
  targetForms: Map<string, string>
832
+ ): readonly {
833
+ readonly end: number;
834
+ readonly start: number;
835
+ }[] => {
836
+ const spans: { end: number; start: number }[] = [];
837
+ for (const form of targetForms.keys()) {
838
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
839
+ for (const match of file.source.matchAll(pattern)) {
840
+ const start = match.index ?? 0;
841
+ const end = start + match[0].length;
842
+ if (
843
+ !hasWordBoundary(file.source, start, end) ||
844
+ occurrenceOverlaps(spans, start, end)
845
+ ) {
846
+ continue;
847
+ }
848
+ spans.push({ end, start });
849
+ }
850
+ }
851
+ return spans;
852
+ };
853
+
854
+ const occurrencesForFile = (
855
+ file: SourceFile,
856
+ plan: VocabularyRegradePlan,
857
+ targetForms: Map<string, string>,
858
+ deferredOccurrences: readonly SourceOccurrence[]
313
859
  ): readonly SourceOccurrence[] => {
314
860
  const occurrences: SourceOccurrence[] = [];
315
861
  const candidates: SourceOccurrence[] = [];
@@ -318,18 +864,23 @@ const occurrencesForFile = (
318
864
  );
319
865
 
320
866
  for (const [form, replacement] of forms) {
321
- const pattern = new RegExp(escapeRegExp(form), 'gi');
867
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
322
868
  for (const match of file.source.matchAll(pattern)) {
323
869
  const start = match.index ?? 0;
324
870
  const end = start + match[0].length;
325
- if (!hasWordBoundary(file.source, start, end)) {
871
+ if (
872
+ !hasWordBoundary(file.source, start, end) ||
873
+ occurrenceOverlaps(deferredOccurrences, start, end)
874
+ ) {
326
875
  continue;
327
876
  }
328
877
  const { column, line } = lineColumnForOffset(file.source, start);
878
+ const context = contextDetailsForOffset(file.source, start, end);
329
879
  const baseOccurrence = {
330
880
  absolutePath: file.absolutePath,
331
881
  column,
332
- context: contextForOffset(file.source, start, end),
882
+ context: context.context,
883
+ contextColumn: context.contextColumn,
333
884
  end,
334
885
  form: match[0],
335
886
  line,
@@ -337,15 +888,34 @@ const occurrencesForFile = (
337
888
  start,
338
889
  };
339
890
  const preserveRule = preserveRuleForOccurrence(baseOccurrence, plan);
891
+ const markdownCodeContext = isMarkdownCodeContext(file, start, end);
892
+ const verdict = capturedVocabularyVerdict(
893
+ preserveRule,
894
+ markdownCodeContext
895
+ );
340
896
  candidates.push({
341
- ...baseOccurrence,
342
- reason:
343
- preserveRule?.reason ??
344
- (preserveRule === undefined ? 'captured-form' : 'preserved-by-plan'),
345
- ...(preserveRule === undefined
897
+ absolutePath: baseOccurrence.absolutePath,
898
+ column: baseOccurrence.column,
899
+ context: baseOccurrence.context,
900
+ disposition: vocabularyOccurrenceDisposition(
901
+ verdict,
902
+ preserveRule,
903
+ markdownCodeContext
904
+ ),
905
+ end: baseOccurrence.end,
906
+ form: baseOccurrence.form,
907
+ line: baseOccurrence.line,
908
+ path: baseOccurrence.path,
909
+ reason: vocabularyOccurrenceReason(
910
+ preserveRule,
911
+ markdownCodeContext,
912
+ 'captured-form'
913
+ ),
914
+ ...(preserveRule === undefined && !markdownCodeContext
346
915
  ? { replacement: preserveCase(match[0], replacement) }
347
916
  : {}),
348
- verdict: preserveRule === undefined ? 'modified' : 'skipped',
917
+ start: baseOccurrence.start,
918
+ verdict,
349
919
  });
350
920
  }
351
921
  }
@@ -376,29 +946,29 @@ const deferredOccurrencesForFile = (
376
946
  plan: VocabularyRegradePlan,
377
947
  targetForms: Map<string, string>
378
948
  ): readonly SourceOccurrence[] => {
949
+ const deferForms = deferFormsForPlan(plan);
950
+ const targetFormSpans = targetFormSpansForFile(file, plan, targetForms);
379
951
  const knownForms = new Set(
380
- [...targetForms.keys()].flatMap((form) => [form, form.toLowerCase()])
952
+ plan.caseSensitive === true
953
+ ? [...targetForms.keys(), ...deferForms]
954
+ : [...targetForms.keys(), ...deferForms].flatMap((form) => [
955
+ form,
956
+ form.toLowerCase(),
957
+ ])
381
958
  );
382
959
  const lowerFrom = plan.from.toLowerCase();
383
960
  const tokenPattern = /[A-Za-z_$][A-Za-z0-9_$-]*/g;
384
- const occurrences: SourceOccurrence[] = [];
385
- const pushOccurrence = (
386
- baseOccurrence: Omit<SourceOccurrence, 'reason' | 'verdict'>
387
- ): void => {
388
- const preserveRule = preserveRuleForOccurrence(baseOccurrence, plan);
389
- occurrences.push({
390
- ...baseOccurrence,
391
- reason:
392
- preserveRule?.reason ??
393
- (preserveRule === undefined
394
- ? 'unclassified-neighbor'
395
- : 'preserved-by-plan'),
396
- verdict: preserveRule === undefined ? 'deferred' : 'skipped',
397
- });
398
- };
961
+ const occurrences = [
962
+ ...exactDeferredFormOccurrencesForFile(
963
+ file,
964
+ plan,
965
+ deferForms,
966
+ targetFormSpans
967
+ ),
968
+ ];
399
969
 
400
970
  for (const form of targetForms.keys()) {
401
- const pattern = new RegExp(escapeRegExp(form), 'gi');
971
+ const pattern = new RegExp(escapeRegExp(form), vocabularyScanFlags(plan));
402
972
  for (const match of file.source.matchAll(pattern)) {
403
973
  const matchStart = match.index ?? 0;
404
974
  const matchEnd = matchStart + match[0].length;
@@ -412,35 +982,31 @@ const deferredOccurrencesForFile = (
412
982
  );
413
983
  const matchedForm = file.source.slice(start, end);
414
984
  const lowerMatchedForm = matchedForm.toLowerCase();
415
- const overlaps = occurrences.some(
416
- (occurrence) => start < occurrence.end && occurrence.start < end
417
- );
418
985
  if (
419
- overlaps ||
986
+ occurrenceOverlaps(occurrences, start, end) ||
420
987
  knownForms.has(matchedForm) ||
421
- knownForms.has(lowerMatchedForm) ||
988
+ (plan.caseSensitive !== true && knownForms.has(lowerMatchedForm)) ||
422
989
  !lowerMatchedForm.includes(lowerFrom)
423
990
  ) {
424
991
  continue;
425
992
  }
426
- const { column, line } = lineColumnForOffset(file.source, start);
427
- pushOccurrence({
428
- absolutePath: file.absolutePath,
429
- column,
430
- context: contextForOffset(file.source, start, end),
431
- end,
432
- form: matchedForm,
433
- line,
434
- path: file.path,
435
- start,
436
- });
993
+ occurrences.push(
994
+ deferredOccurrenceFromDraft(
995
+ file,
996
+ plan,
997
+ occurrenceDraftForSpan(file, start, end, matchedForm)
998
+ )
999
+ );
437
1000
  }
438
1001
  }
439
1002
 
440
1003
  for (const match of file.source.matchAll(tokenPattern)) {
441
1004
  const [form] = match;
442
1005
  const lower = form.toLowerCase();
443
- if (knownForms.has(form) || knownForms.has(lower)) {
1006
+ if (
1007
+ knownForms.has(form) ||
1008
+ (plan.caseSensitive !== true && knownForms.has(lower))
1009
+ ) {
444
1010
  continue;
445
1011
  }
446
1012
  if (!lower.includes(lowerFrom)) {
@@ -448,23 +1014,16 @@ const deferredOccurrencesForFile = (
448
1014
  }
449
1015
  const start = match.index ?? 0;
450
1016
  const end = start + form.length;
451
- const overlaps = occurrences.some(
452
- (occurrence) => start < occurrence.end && occurrence.start < end
453
- );
454
- if (overlaps) {
1017
+ if (occurrenceOverlaps(occurrences, start, end)) {
455
1018
  continue;
456
1019
  }
457
- const { column, line } = lineColumnForOffset(file.source, start);
458
- pushOccurrence({
459
- absolutePath: file.absolutePath,
460
- column,
461
- context: contextForOffset(file.source, start, end),
462
- end,
463
- form,
464
- line,
465
- path: file.path,
466
- start,
467
- });
1020
+ occurrences.push(
1021
+ deferredOccurrenceFromDraft(
1022
+ file,
1023
+ plan,
1024
+ occurrenceDraftForSpan(file, start, end, form)
1025
+ )
1026
+ );
468
1027
  }
469
1028
  return occurrences.toSorted((left, right) =>
470
1029
  left.path === right.path
@@ -548,22 +1107,37 @@ const applyOccurrenceRewrites = (
548
1107
 
549
1108
  const buildVocabularyEvaluation = (params: {
550
1109
  readonly apply?: boolean;
1110
+ readonly effectivePlan?: VocabularyRegradePlan | undefined;
551
1111
  readonly files: readonly SourceFile[];
552
1112
  readonly plan: VocabularyRegradePlan;
1113
+ readonly preserveInventory?: readonly VocabularyPreserveInventoryEntry[];
553
1114
  readonly root: string;
554
1115
  readonly skipped: readonly SkippedSource[];
555
1116
  }): VocabularyEvaluation => {
556
- const targetForms = targetFormsForPlan(params.plan);
1117
+ const effectivePlan = params.effectivePlan ?? params.plan;
1118
+ const targetForms = targetFormsForPlan(effectivePlan);
557
1119
  const scopedFiles = params.files.filter((file) =>
558
- includedByScope(file.path, params.plan.scope)
1120
+ includedByScope(file.path, effectivePlan.scope)
559
1121
  );
560
1122
  const scopeSkipped: SkippedSource[] = params.files
561
- .filter((file) => !includedByScope(file.path, params.plan.scope))
1123
+ .filter((file) => !includedByScope(file.path, effectivePlan.scope))
562
1124
  .map((file) => ({ path: file.path, reason: 'excluded-by-regrade-scope' }));
563
- const occurrences = scopedFiles.flatMap((file) => [
564
- ...occurrencesForFile(file, params.plan, targetForms),
565
- ...deferredOccurrencesForFile(file, params.plan, targetForms),
566
- ]);
1125
+ const occurrences = scopedFiles.flatMap((file) => {
1126
+ const deferredOccurrences = deferredOccurrencesForFile(
1127
+ file,
1128
+ effectivePlan,
1129
+ targetForms
1130
+ );
1131
+ return [
1132
+ ...occurrencesForFile(
1133
+ file,
1134
+ effectivePlan,
1135
+ targetForms,
1136
+ deferredOccurrences
1137
+ ),
1138
+ ...deferredOccurrences,
1139
+ ];
1140
+ });
567
1141
  const occurrencesByPath = new Map<string, SourceOccurrence[]>();
568
1142
  for (const occurrence of occurrences) {
569
1143
  const existing = occurrencesByPath.get(occurrence.path) ?? [];
@@ -594,6 +1168,10 @@ const buildVocabularyEvaluation = (params: {
594
1168
  const deferredOccurrences = occurrences.filter(
595
1169
  (occurrence) => occurrence.verdict === 'deferred'
596
1170
  );
1171
+ const unresolvedOccurrences = occurrences.filter(
1172
+ (occurrence) =>
1173
+ occurrence.verdict === 'modified' || occurrence.verdict === 'deferred'
1174
+ );
597
1175
  const forms: Record<string, VocabularyVerdict> = {};
598
1176
  for (const occurrence of occurrences) {
599
1177
  const current = forms[occurrence.form];
@@ -620,7 +1198,7 @@ const buildVocabularyEvaluation = (params: {
620
1198
  if (deferredForms.length > 0) {
621
1199
  gateReasons.push('deferred-forms-or-occurrences');
622
1200
  }
623
- const open = modifiedOccurrences.length + deferredOccurrences.length;
1201
+ const open = unresolvedOccurrences.length;
624
1202
 
625
1203
  return {
626
1204
  entries: [
@@ -646,13 +1224,21 @@ const buildVocabularyEvaluation = (params: {
646
1224
  ),
647
1225
  },
648
1226
  plan: params.plan,
1227
+ ...(params.preserveInventory === undefined ||
1228
+ params.preserveInventory.length === 0
1229
+ ? {}
1230
+ : { preserveInventory: params.preserveInventory }),
649
1231
  report: {
650
1232
  applied: params.apply === true ? modifiedOccurrences.length : 0,
651
1233
  deferred: deferredOccurrences.length,
1234
+ dispositions: vocabularyDispositionCounts(occurrences),
652
1235
  filesChanged: params.apply === true ? rewrittenPaths.size : 0,
653
1236
  gate: {
654
1237
  reasons: gateReasons,
655
1238
  remaining: open,
1239
+ remainingByDisposition: vocabularyDispositionCounts(
1240
+ unresolvedOccurrences
1241
+ ),
656
1242
  status: gateReasons.length === 0 ? 'green' : 'open',
657
1243
  },
658
1244
  modified: modifiedOccurrences.length,
@@ -744,56 +1330,109 @@ const applyVocabularyEvaluation = (
744
1330
  });
745
1331
  };
746
1332
 
1333
+ const readVocabularySourceFiles = (
1334
+ collected: NonNullable<ReturnType<typeof collectDownstreamSources>>
1335
+ ): {
1336
+ readonly files: readonly SourceFile[];
1337
+ readonly skipped: readonly SkippedSource[];
1338
+ } => {
1339
+ const files: SourceFile[] = [];
1340
+ const skipped: SkippedSource[] = [...collected.skipped];
1341
+ for (const file of collected.files) {
1342
+ try {
1343
+ files.push({
1344
+ absolutePath: file.absolutePath,
1345
+ path: file.path,
1346
+ source: readFileSync(file.absolutePath, 'utf8'),
1347
+ });
1348
+ } catch {
1349
+ skipped.push({ path: file.path, reason: 'unreadable-file' });
1350
+ }
1351
+ }
1352
+ return { files, skipped };
1353
+ };
1354
+
1355
+ const buildRunVocabularyEvaluation = (params: {
1356
+ readonly apply: boolean;
1357
+ readonly effectivePlan: VocabularyRegradePlan;
1358
+ readonly files: readonly SourceFile[];
1359
+ readonly plan: VocabularyRegradePlan;
1360
+ readonly preserveInventory:
1361
+ | readonly VocabularyPreserveInventoryEntry[]
1362
+ | undefined;
1363
+ readonly root: string;
1364
+ readonly skipped: readonly SkippedSource[];
1365
+ }): VocabularyEvaluation =>
1366
+ buildVocabularyEvaluation({
1367
+ apply: params.apply,
1368
+ effectivePlan: params.effectivePlan,
1369
+ files: params.files,
1370
+ plan: params.plan,
1371
+ ...(params.preserveInventory === undefined
1372
+ ? {}
1373
+ : { preserveInventory: params.preserveInventory }),
1374
+ root: params.root,
1375
+ skipped: params.skipped,
1376
+ });
1377
+
747
1378
  export const runVocabularyRegrade = (params: {
748
1379
  readonly apply?: boolean;
749
1380
  readonly includeEntries?: 'actionable' | 'all';
750
1381
  readonly plan: VocabularyRegradePlan;
1382
+ readonly preserveInventory?: readonly VocabularyPreserveInventoryEntry[];
751
1383
  readonly root: string;
752
1384
  }): Result<RegradeReport | null, InternalError | ValidationError> => {
753
1385
  const planValidation = validateVocabularyPlan(params.plan);
754
1386
  if (planValidation.isErr()) {
755
1387
  return planValidation;
756
1388
  }
1389
+ const inventoryValidation = validatePreserveInventory(
1390
+ params.preserveInventory
1391
+ );
1392
+ if (inventoryValidation.isErr()) {
1393
+ return inventoryValidation;
1394
+ }
1395
+
1396
+ const effectivePlan = effectivePlanForRun(
1397
+ params.plan,
1398
+ params.preserveInventory
1399
+ );
757
1400
 
758
1401
  const collected = collectDownstreamSources(params.root, {
759
- extensions: params.plan.scope?.extensions ?? VOCABULARY_SOURCE_EXTENSIONS,
760
- ...(params.plan.scope?.exclude === undefined
1402
+ extensions: effectivePlan.scope?.extensions ?? VOCABULARY_SOURCE_EXTENSIONS,
1403
+ ...(effectivePlan.scope?.exclude === undefined
761
1404
  ? {}
762
- : { exclude: params.plan.scope.exclude }),
763
- ...(params.plan.scope?.ignoredDirectories === undefined
1405
+ : { exclude: effectivePlan.scope.exclude }),
1406
+ ...(effectivePlan.scope?.include === undefined
764
1407
  ? {}
765
- : { ignoredDirectories: params.plan.scope.ignoredDirectories }),
1408
+ : { include: effectivePlan.scope.include }),
1409
+ ...(effectivePlan.scope?.ignoredDirectories === undefined
1410
+ ? {}
1411
+ : { ignoredDirectories: effectivePlan.scope.ignoredDirectories }),
766
1412
  } satisfies DownstreamCollectionOptions);
767
1413
  if (collected === null) {
768
1414
  return Result.ok(null);
769
1415
  }
770
1416
 
771
- const files: SourceFile[] = [];
772
- const skipped: SkippedSource[] = [...collected.skipped];
773
- for (const file of collected.files) {
774
- try {
775
- files.push({
776
- absolutePath: file.absolutePath,
777
- path: file.path,
778
- source: readFileSync(file.absolutePath, 'utf8'),
779
- });
780
- } catch {
781
- skipped.push({ path: file.path, reason: 'unreadable-file' });
782
- }
783
- }
1417
+ const { files, skipped } = readVocabularySourceFiles(collected);
784
1418
 
785
- const dryRunEvaluation = buildVocabularyEvaluation({
1419
+ const dryRunEffectiveEvaluation = buildRunVocabularyEvaluation({
786
1420
  apply: false,
1421
+ effectivePlan,
787
1422
  files,
788
1423
  plan: params.plan,
1424
+ preserveInventory: params.preserveInventory,
789
1425
  root: params.root,
790
1426
  skipped,
791
1427
  });
792
- let reportEvaluation = dryRunEvaluation;
1428
+ let reportEvaluation = dryRunEffectiveEvaluation;
793
1429
  let applySummary: RegradeApplySummary | undefined;
794
1430
 
795
1431
  if (params.apply === true) {
796
- const applyResult = applyVocabularyEvaluation(files, dryRunEvaluation);
1432
+ const applyResult = applyVocabularyEvaluation(
1433
+ files,
1434
+ dryRunEffectiveEvaluation
1435
+ );
797
1436
  if (applyResult.isErr()) {
798
1437
  return applyResult;
799
1438
  }
@@ -802,10 +1441,12 @@ export const runVocabularyRegrade = (params: {
802
1441
  ...file,
803
1442
  source: readFileSync(file.absolutePath, 'utf8'),
804
1443
  }));
805
- reportEvaluation = buildVocabularyEvaluation({
1444
+ reportEvaluation = buildRunVocabularyEvaluation({
806
1445
  apply: true,
1446
+ effectivePlan,
807
1447
  files: appliedFiles,
808
1448
  plan: params.plan,
1449
+ preserveInventory: params.preserveInventory,
809
1450
  root: params.root,
810
1451
  skipped,
811
1452
  });
@@ -860,6 +1501,14 @@ export const runVocabularyRegrade = (params: {
860
1501
  };
861
1502
 
862
1503
  const vocabularyPreserveRuleSchema = z.object({
1504
+ disposition: z
1505
+ .enum(vocabularyDispositionValues)
1506
+ .optional()
1507
+ .describe('Classification for occurrences preserved by this rule'),
1508
+ forms: z
1509
+ .array(z.string().min(1))
1510
+ .optional()
1511
+ .describe('Matched forms this preserve rule applies to'),
863
1512
  paths: z
864
1513
  .array(z.string())
865
1514
  .optional()
@@ -868,6 +1517,23 @@ const vocabularyPreserveRuleSchema = z.object({
868
1517
  reason: z.string().optional().describe('Why this form is preserved'),
869
1518
  });
870
1519
 
1520
+ const vocabularyPreserveInventoryEntrySchema =
1521
+ vocabularyPreserveRuleSchema.extend({
1522
+ evidence: z
1523
+ .array(z.string().min(1))
1524
+ .describe('Graph or surface facts that justify this derived preserve'),
1525
+ source: z.literal('derived-live-api').describe('Derived inventory source'),
1526
+ });
1527
+
1528
+ const vocabularyDispositionCountSchema = z.object(
1529
+ Object.fromEntries(
1530
+ vocabularyDispositionValues.map((disposition) => [
1531
+ disposition,
1532
+ z.number().optional(),
1533
+ ])
1534
+ ) as Record<VocabularyDisposition, z.ZodOptional<z.ZodNumber>>
1535
+ );
1536
+
871
1537
  const vocabularyRegradeScopeSchema = z.object({
872
1538
  exclude: z
873
1539
  .array(z.string())
@@ -890,6 +1556,14 @@ const vocabularyRegradeScopeSchema = z.object({
890
1556
  });
891
1557
 
892
1558
  export const vocabularyRegradePlanSchema = z.object({
1559
+ caseSensitive: z
1560
+ .boolean()
1561
+ .optional()
1562
+ .describe('Whether source form scanning preserves case exactly'),
1563
+ deferForms: z
1564
+ .array(z.string().min(1))
1565
+ .optional()
1566
+ .describe('Known forms that must be inventoried for review, not rewritten'),
893
1567
  from: z.string().min(1).describe('Source vocabulary term or phrase'),
894
1568
  id: z.string().optional().describe('Stable authored regrade plan id'),
895
1569
  intent: z.string().optional().describe('Human-authored migration intent'),
@@ -920,6 +1594,9 @@ export const vocabularyRegradeRunOutput = z.object({
920
1594
  z.object({
921
1595
  column: z.number().describe('One-based source column'),
922
1596
  context: z.string().describe('Source-line context'),
1597
+ disposition: z
1598
+ .enum(vocabularyDispositionValues)
1599
+ .describe('Occurrence-level classification beside the verdict'),
923
1600
  end: z.number().describe('Source end offset'),
924
1601
  form: z.string().describe('Matched vocabulary form'),
925
1602
  line: z.number().describe('One-based source line'),
@@ -939,15 +1616,27 @@ export const vocabularyRegradeRunOutput = z.object({
939
1616
  })
940
1617
  .describe('Observed run ledger'),
941
1618
  plan: vocabularyRegradePlanSchema.describe('Authored regrade plan'),
1619
+ preserveInventory: z
1620
+ .array(vocabularyPreserveInventoryEntrySchema)
1621
+ .optional()
1622
+ .describe(
1623
+ 'Derived live-API preserve inventory applied at run time without changing the authored plan'
1624
+ ),
942
1625
  report: z
943
1626
  .object({
944
1627
  applied: z.number().describe('Modified occurrences applied to disk'),
945
1628
  deferred: z.number().describe('Deferred occurrence count'),
1629
+ dispositions: z
1630
+ .object(vocabularyDispositionCountSchema.shape)
1631
+ .describe('Occurrence counts grouped by disposition'),
946
1632
  filesChanged: z.number().describe('Distinct files changed on disk'),
947
1633
  gate: z
948
1634
  .object({
949
1635
  reasons: z.array(z.string()).describe('Open-gate reasons'),
950
1636
  remaining: z.number().describe('Unresolved occurrence count'),
1637
+ remainingByDisposition: z
1638
+ .object(vocabularyDispositionCountSchema.shape)
1639
+ .describe('Unresolved occurrence counts grouped by disposition'),
951
1640
  status: z
952
1641
  .enum(['green', 'open'])
953
1642
  .describe('Whether the run is complete'),
@@ -963,3 +1652,282 @@ export const vocabularyRegradeRunOutput = z.object({
963
1652
  })
964
1653
  .describe('Projected run report'),
965
1654
  });
1655
+
1656
+ const vocabularyTransitionRecordEnvironmentSchema = z
1657
+ .object({
1658
+ commitSha: z.string().optional(),
1659
+ engineVersion: z.string().optional(),
1660
+ graphHash: z.string().optional(),
1661
+ root: z.string(),
1662
+ })
1663
+ .strict();
1664
+
1665
+ const vocabularyTransitionRecordReportSchema = z
1666
+ .object({
1667
+ apply: z.unknown().optional(),
1668
+ entries: z.array(z.unknown()),
1669
+ matched: z.number(),
1670
+ review: z.number(),
1671
+ rewritten: z.number(),
1672
+ root: z.string(),
1673
+ run: vocabularyRegradeRunOutput,
1674
+ scan: z.unknown(),
1675
+ scanned: z.number(),
1676
+ selectedClassIds: z.array(z.string()),
1677
+ skipped: z.number(),
1678
+ skipsByReason: z.record(z.string(), z.number()),
1679
+ unknownClassIds: z.array(z.string()),
1680
+ })
1681
+ .strict();
1682
+
1683
+ const normalizeTransitionRecordPath = (path: string): string =>
1684
+ normalize(path).replaceAll('\\', '/');
1685
+
1686
+ const isSafeRootRelativeRecordPath = (path: string): boolean => {
1687
+ const normalized = normalizeTransitionRecordPath(path);
1688
+ return (
1689
+ normalized.length > 0 &&
1690
+ !isAbsolute(normalized) &&
1691
+ normalized !== '..' &&
1692
+ !normalized.startsWith('../')
1693
+ );
1694
+ };
1695
+
1696
+ export const vocabularyTransitionRecordSchema = z
1697
+ .object({
1698
+ environment: vocabularyTransitionRecordEnvironmentSchema,
1699
+ kind: z.literal('vocabulary-transition-record'),
1700
+ recordPath: z.string().refine(isSafeRootRelativeRecordPath),
1701
+ report: vocabularyTransitionRecordReportSchema,
1702
+ schemaVersion: z.literal(VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION),
1703
+ transition: z
1704
+ .object({
1705
+ from: z.string(),
1706
+ id: z.string(),
1707
+ to: z.string(),
1708
+ })
1709
+ .strict(),
1710
+ })
1711
+ .strict();
1712
+
1713
+ const transitionRecordSlug = (run: VocabularyRegradeRun): string =>
1714
+ `${run.plan.from}-to-${run.plan.to}`
1715
+ .toLowerCase()
1716
+ .replaceAll(/[^a-z0-9]+/g, '-')
1717
+ .replaceAll(/^-|-$/g, '');
1718
+
1719
+ const stableJson = (value: unknown): string =>
1720
+ JSON.stringify(value, (_key, nested) => {
1721
+ if (
1722
+ nested === null ||
1723
+ typeof nested !== 'object' ||
1724
+ Array.isArray(nested)
1725
+ ) {
1726
+ return nested as unknown;
1727
+ }
1728
+ return Object.fromEntries(
1729
+ Object.entries(nested as Record<string, unknown>).toSorted(
1730
+ ([left], [right]) => left.localeCompare(right)
1731
+ )
1732
+ );
1733
+ });
1734
+
1735
+ const shortHashForRun = (
1736
+ run: VocabularyRegradeRun,
1737
+ environment?: Partial<VocabularyTransitionRecordEnvironment>
1738
+ ): string => {
1739
+ const explicitHash = environment?.graphHash ?? environment?.commitSha;
1740
+ if (explicitHash !== undefined && explicitHash.length > 0) {
1741
+ return explicitHash.slice(0, 7);
1742
+ }
1743
+ return createHash('sha256')
1744
+ .update(stableJson({ ledger: run.ledger, plan: run.plan }))
1745
+ .digest('hex')
1746
+ .slice(0, 7);
1747
+ };
1748
+
1749
+ export const vocabularyTransitionRecordPath = (params: {
1750
+ readonly environment?: Partial<VocabularyTransitionRecordEnvironment>;
1751
+ readonly root: string;
1752
+ readonly run: VocabularyRegradeRun;
1753
+ }): string =>
1754
+ join(
1755
+ '.trails',
1756
+ 'regrade',
1757
+ 'history',
1758
+ `${transitionRecordSlug(params.run)}-${shortHashForRun(params.run, params.environment)}.json`
1759
+ );
1760
+
1761
+ const reportWithoutRecord = (
1762
+ report: RegradeReport
1763
+ ): Omit<RegradeReport, 'record'> => {
1764
+ const { record: _record, ...rest } = report;
1765
+ return rest;
1766
+ };
1767
+
1768
+ const transitionRecordPathForWrite = (params: {
1769
+ readonly environment: VocabularyTransitionRecordEnvironment;
1770
+ readonly recordPath?: string;
1771
+ readonly report: RegradeReport;
1772
+ readonly root: string;
1773
+ }): Result<string, ValidationError> => {
1774
+ const recordPath =
1775
+ params.recordPath ??
1776
+ vocabularyTransitionRecordPath({
1777
+ environment: params.environment,
1778
+ root: params.root,
1779
+ run: params.report.run as VocabularyRegradeRun,
1780
+ });
1781
+ const normalized = normalizeTransitionRecordPath(
1782
+ isAbsolute(recordPath) ? relative(params.root, recordPath) : recordPath
1783
+ );
1784
+ if (!isSafeRootRelativeRecordPath(normalized)) {
1785
+ return Result.err(
1786
+ new ValidationError(
1787
+ 'Vocabulary transition record path must stay inside the regrade root.',
1788
+ { context: { recordPath } }
1789
+ )
1790
+ );
1791
+ }
1792
+ return Result.ok(normalized);
1793
+ };
1794
+
1795
+ export const buildVocabularyTransitionRecord = (params: {
1796
+ readonly environment?: Partial<VocabularyTransitionRecordEnvironment>;
1797
+ readonly recordPath?: string;
1798
+ readonly report: RegradeReport;
1799
+ readonly root: string;
1800
+ }): Result<VocabularyTransitionRecord, ValidationError> => {
1801
+ if (params.report.run === undefined) {
1802
+ return Result.err(
1803
+ new ValidationError(
1804
+ 'Vocabulary transition records require a vocabulary Regrade report.'
1805
+ )
1806
+ );
1807
+ }
1808
+
1809
+ const environment: VocabularyTransitionRecordEnvironment = {
1810
+ ...(params.environment?.commitSha === undefined
1811
+ ? {}
1812
+ : { commitSha: params.environment.commitSha }),
1813
+ ...(params.environment?.engineVersion === undefined
1814
+ ? {}
1815
+ : { engineVersion: params.environment.engineVersion }),
1816
+ ...(params.environment?.graphHash === undefined
1817
+ ? {}
1818
+ : { graphHash: params.environment.graphHash }),
1819
+ root: params.root,
1820
+ };
1821
+ const recordPathResult = transitionRecordPathForWrite({
1822
+ environment,
1823
+ report: params.report,
1824
+ root: params.root,
1825
+ ...(params.recordPath === undefined
1826
+ ? {}
1827
+ : { recordPath: params.recordPath }),
1828
+ });
1829
+ if (recordPathResult.isErr()) {
1830
+ return recordPathResult;
1831
+ }
1832
+ const recordPath = recordPathResult.value;
1833
+ const record: VocabularyTransitionRecord = {
1834
+ environment,
1835
+ kind: 'vocabulary-transition-record',
1836
+ recordPath,
1837
+ report: reportWithoutRecord(params.report),
1838
+ schemaVersion: VOCABULARY_TRANSITION_RECORD_SCHEMA_VERSION,
1839
+ transition: {
1840
+ from: params.report.run.plan.from,
1841
+ id:
1842
+ params.report.run.plan.id ??
1843
+ `vocabulary:${params.report.run.plan.from}->${params.report.run.plan.to}`,
1844
+ to: params.report.run.plan.to,
1845
+ },
1846
+ };
1847
+ const parsed = vocabularyTransitionRecordSchema.safeParse(record);
1848
+ if (!parsed.success) {
1849
+ return Result.err(
1850
+ new ValidationError('Invalid vocabulary transition record.', {
1851
+ context: { issues: parsed.error.issues },
1852
+ })
1853
+ );
1854
+ }
1855
+ return Result.ok(parsed.data as VocabularyTransitionRecord);
1856
+ };
1857
+
1858
+ export const writeVocabularyTransitionRecord = (params: {
1859
+ readonly environment?: Partial<VocabularyTransitionRecordEnvironment>;
1860
+ readonly recordPath?: string;
1861
+ readonly report: RegradeReport;
1862
+ readonly root: string;
1863
+ readonly status: VocabularyTransitionRecordSummary['status'];
1864
+ }): Result<
1865
+ {
1866
+ readonly record: VocabularyTransitionRecord;
1867
+ readonly summary: VocabularyTransitionRecordSummary;
1868
+ },
1869
+ InternalError | ValidationError
1870
+ > => {
1871
+ const recordResult = buildVocabularyTransitionRecord(params);
1872
+ if (recordResult.isErr()) {
1873
+ return recordResult;
1874
+ }
1875
+ const record = recordResult.value;
1876
+ const absolutePath = isAbsolute(record.recordPath)
1877
+ ? record.recordPath
1878
+ : join(params.root, record.recordPath);
1879
+ try {
1880
+ mkdirSync(dirname(absolutePath), { recursive: true });
1881
+ writeFileSync(absolutePath, `${JSON.stringify(record, null, 2)}\n`);
1882
+ } catch (error) {
1883
+ return Result.err(
1884
+ new InternalError('Failed to write vocabulary transition record.', {
1885
+ ...(error instanceof Error ? { cause: error } : {}),
1886
+ context: { path: record.recordPath },
1887
+ })
1888
+ );
1889
+ }
1890
+ return Result.ok({
1891
+ record,
1892
+ summary: {
1893
+ path: record.recordPath,
1894
+ schemaVersion: record.schemaVersion,
1895
+ status: params.status,
1896
+ },
1897
+ });
1898
+ };
1899
+
1900
+ export const readVocabularyTransitionRecord = (
1901
+ path: string
1902
+ ): Result<VocabularyTransitionRecord, InternalError | ValidationError> => {
1903
+ if (!existsSync(path)) {
1904
+ return Result.err(
1905
+ new ValidationError(`Vocabulary transition record "${path}" not found.`)
1906
+ );
1907
+ }
1908
+ let parsedJson: unknown;
1909
+ try {
1910
+ parsedJson = JSON.parse(readFileSync(path, 'utf8'));
1911
+ } catch (error) {
1912
+ return Result.err(
1913
+ new InternalError('Failed to read vocabulary transition record.', {
1914
+ ...(error instanceof Error ? { cause: error } : {}),
1915
+ context: { path },
1916
+ })
1917
+ );
1918
+ }
1919
+ const parsed = vocabularyTransitionRecordSchema.safeParse(parsedJson);
1920
+ if (!parsed.success) {
1921
+ return Result.err(
1922
+ new ValidationError('Invalid vocabulary transition record.', {
1923
+ context: { issues: parsed.error.issues, path },
1924
+ })
1925
+ );
1926
+ }
1927
+ return Result.ok(parsed.data as VocabularyTransitionRecord);
1928
+ };
1929
+
1930
+ export const transitionRecordReportWithSummary = (
1931
+ report: RegradeReport,
1932
+ summary: VocabularyTransitionRecordSummary
1933
+ ): RegradeReport => ({ ...report, record: summary });