@llman-sdd/core 0.5.0 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@llman-sdd/core",
3
- "version": "0.5.0",
3
+ "version": "0.5.1",
4
4
  "description": "Pure domain logic for llman-sdd (config, gherkin specs, validation, templates)",
5
5
  "license": "MIT",
6
6
  "repository": {
package/src/index.ts CHANGED
@@ -33,8 +33,22 @@ export {
33
33
  localeToGherkinLang,
34
34
  parseCapability,
35
35
  parseFeatureSource,
36
+ sourceDialect,
36
37
  } from './spec/parser.ts';
37
- export { buildReqRegistry, type ReqRegistry, type RegistryDuplicate } from './spec/reqRegistry.ts';
38
+ export {
39
+ BLOCK_KEYWORD_LINE_RE,
40
+ officialKeywords,
41
+ officialKeywordsOrEn,
42
+ stepKeywordToOfficialKind,
43
+ STEP_KIND_BY_KEYWORD,
44
+ type GherkinKeywords,
45
+ } from './spec/keywords.ts';
46
+ export {
47
+ buildReqRegistry,
48
+ type RegistryDuplicate,
49
+ type RegistryOccurrence,
50
+ type ReqRegistry,
51
+ } from './spec/reqRegistry.ts';
38
52
  export {
39
53
  analyzeLegacy,
40
54
  hasNativeRules,
@@ -187,7 +201,12 @@ export { graphData, graphMermaid, type GraphDataIr, type GraphFsIo } from './rep
187
201
  export { parseDeps } from './report/graph.ts';
188
202
  export { renderMachine, type MachineFormat } from './render/machine.ts';
189
203
  export { showChangeJson, type ShowDeps, type ShowFsIo } from './report/show.ts';
190
- export { nextReqId, scaffoldSpec, type SpecHelperIo } from './report/specHelpers.ts';
204
+ export {
205
+ nextReqId,
206
+ scaffoldSpec,
207
+ skeletonContent,
208
+ type SpecHelperIo,
209
+ } from './report/specHelpers.ts';
191
210
  export {
192
211
  collectSpecs,
193
212
  morphologyOf,
@@ -1,3 +1,4 @@
1
+ import { officialKeywords, officialKeywordsOrEn } from '../spec/keywords.ts';
1
2
  import { localeToGherkinLang, parseCapability } from '../spec/parser.ts';
2
3
  import { buildReqRegistry } from '../spec/reqRegistry.ts';
3
4
  /**
@@ -57,22 +58,23 @@ export function nextReqId(io: SpecHelperIo, specsDir: string): string {
57
58
  export function skeletonContent(capability: string, reqId: string, locale: string): string {
58
59
  const zh = localeFallbacks(locale)[0] === 'zh-Hans';
59
60
  // r7: the `# language:` header derives from the locale mapping — never a
60
- // second hardcoded caliber.
61
- const language = localeToGherkinLang(zh ? 'zh-Hans' : 'en');
61
+ // second hardcoded caliber. Table-less locales fall back to en wholesale
62
+ // (header + keywords same-source) so the skeleton stays parseable.
63
+ const mapped = localeToGherkinLang(locale);
64
+ const language = officialKeywords(mapped) !== null ? mapped : 'en';
65
+ const kw = officialKeywordsOrEn(language);
62
66
  const header = zh
63
67
  ? `# language: ${language}\n# capability: ${capability}\n# purpose: TODO: 一句话描述该能力与其目的。\n# scope: llmanspec/`
64
68
  : `# language: ${language}\n# capability: ${capability}\n# purpose: TODO: Describe this capability and its purpose.\n# scope: llmanspec/`;
65
- const feature = zh ? `功能: ${capability}` : `Feature: ${capability}`;
66
- const ruleKw = zh ? '规则' : 'Rule';
67
- const scenarioKw = zh ? '场景' : 'Scenario';
68
- const ruleTitle = 'TODO-rule';
69
69
  const ruleDesc = zh ? 'TODO: 需求描述(自由文本)。' : 'TODO: requirement statement (free text).';
70
- const scTitle = 'TODO-acceptance';
71
- // native v2 skeleton: a `规则:` block (with @req handle) + one nested example
70
+ // Skeleton placeholder prose is not vocabulary — the zh/en copy stays; keywords do not.
71
+ const steps = zh
72
+ ? ` ${kw.given} TODO 前置\n ${kw.when} TODO 动作\n ${kw.thenText} TODO 断言`
73
+ : ` ${kw.given} TODO precondition\n ${kw.when} TODO action\n ${kw.thenText} TODO assertion`;
74
+ // native v2 skeleton: a rule block (with @req handle) + one nested example
72
75
  return (
73
- `${header}\n\n${feature}\n\n @req:${reqId}\n ${ruleKw}: ${ruleTitle}\n` +
74
- ` ${ruleDesc}\n\n ${scenarioKw}: ${scTitle}\n` +
75
- ` 假如 TODO 前置\n 当 TODO 动作\n 那么 TODO 断言\n`
76
+ `${header}\n\n${kw.feature}: ${capability}\n\n @req:${reqId}\n ${kw.rule}: TODO-rule\n` +
77
+ ` ${ruleDesc}\n\n ${kw.scenario}: TODO-acceptance\n${steps}\n`
76
78
  );
77
79
  }
78
80
 
@@ -7,6 +7,9 @@
7
7
 
8
8
  import type { CapabilityDoc, RuleIR } from './ir.ts';
9
9
  import { specIdOf } from './ir.ts';
10
+ import { BLOCK_KEYWORD_LINE_RE, officialKeywordsOrEn } from './keywords.ts';
11
+ import { sourceDialect } from './parser.ts';
12
+ import type { RegistryDuplicate } from './reqRegistry.ts';
10
13
 
11
14
  export class AuthoringError extends Error {}
12
15
 
@@ -30,9 +33,14 @@ interface KeywordSet {
30
33
  }
31
34
 
32
35
  function keywordsOf(content: string): KeywordSet {
33
- return content.includes('功能:')
34
- ? { rule: '规则', scenario: '场景', given: '假如', when: '当', thenText: '那么' }
35
- : { rule: 'Rule', scenario: 'Scenario', given: 'Given', when: 'When', thenText: 'Then' };
36
+ const kw = officialKeywordsOrEn(sourceDialect(content));
37
+ return {
38
+ rule: kw.rule,
39
+ scenario: kw.scenario,
40
+ given: kw.given,
41
+ when: kw.when,
42
+ thenText: kw.thenText,
43
+ };
36
44
  }
37
45
 
38
46
  function findRule(
@@ -137,15 +145,16 @@ export function addScenario(
137
145
  const lines = content.split('\n');
138
146
  const tagIdx = lines.findIndex((l) => l.trim() === `@req:${opts.reqId}`);
139
147
  if (tagIdx === -1) throw new AuthoringError(`req id not found in file: ${opts.reqId}`);
140
- // block end = first line after the tag's own `规则:` header at 2-space top-level
141
- // indent that is a tag line or a `规则:/场景:` keyword line (2-space + keyword
142
- // immediately). The rule header directly following the tag opens the block, so
143
- // the scan starts after it — otherwise the header itself is mistaken for the
144
- // boundary and the scenario is inserted before the `规则:` line (broken output).
148
+ // block end = first line after the tag's own rule header at 2-space top-level
149
+ // indent that is a tag line or an official block keyword line (2-space +
150
+ // keyword immediately, any official dialect). The rule header directly
151
+ // following the tag opens the block, so the scan starts after it — otherwise
152
+ // the header itself is mistaken for the boundary and the scenario is
153
+ // inserted before the rule line (broken output).
145
154
  let end = lines.length;
146
155
  for (let i = tagIdx + 2; i < lines.length; i++) {
147
156
  const l = lines[i] ?? '';
148
- if (l.startsWith(' @') || /^ (规则|Rule|场景|Scenario|功能|Feature):/u.test(l)) {
157
+ if (l.startsWith(' @') || BLOCK_KEYWORD_LINE_RE.test(l)) {
149
158
  end = i;
150
159
  break;
151
160
  }
@@ -187,21 +196,44 @@ export function resolveReq(entries: readonly SpecEntryLike[], reqId: string): Re
187
196
 
188
197
  export interface DedupePlanItem {
189
198
  reqId: string;
199
+ /** File of the kept (first) occurrence. */
190
200
  keepFile: string;
191
201
  remapFile: string;
192
202
  newReqId: string;
203
+ /** 1-based ordinal of the remapped `@req:<id>` occurrence within remapFile's text. */
204
+ occurrenceOrdinal: number;
205
+ }
206
+
207
+ /**
208
+ * Replace the ordinal-th `@req:<id>` tag in raw text. The tag match is
209
+ * boundary-exact (`@req:r1` never matches inside `@req:r10`), and only the
210
+ * target ordinal is rewritten so a kept first occurrence in the same file
211
+ * stays untouched (r43: 按出现顺序首现保留、其余出现逐个重取号).
212
+ */
213
+ function replaceNthReqTag(
214
+ content: string,
215
+ reqId: string,
216
+ newReqId: string,
217
+ ordinal: number,
218
+ ): string {
219
+ const pattern = new RegExp(`@req:${reqId}(?!\\d)`, 'gu');
220
+ let seen = 0;
221
+ return content.replace(pattern, (tag) => {
222
+ seen += 1;
223
+ return seen === ordinal ? `@req:${newReqId}` : tag;
224
+ });
193
225
  }
194
226
 
195
227
  /**
196
- * Plan (and optionally apply) a re-map of globally duplicated rN ids: the
197
- * first file keeps the id, later files get the next free id (r43). `apply:
198
- * false` (`--dry-run`) returns the plan without writing anything.
228
+ * Plan (and optionally apply) a re-map of globally duplicated rN ids
229
+ * (r43: 同文件内共用或跨文件,按出现顺序首现保留、其余出现逐个重取号).
230
+ * `apply: false` (`--dry-run`) returns the plan without writing anything.
199
231
  */
200
232
  export function planDedupe(
201
233
  entries: readonly SpecEntryLike[],
202
234
  io: WriteIo,
203
235
  specsRoot: string,
204
- duplicates: readonly { reqId: string; files: string[] }[],
236
+ duplicates: readonly RegistryDuplicate[],
205
237
  opts: { apply?: boolean } = {},
206
238
  ): DedupePlanItem[] {
207
239
  const used = ruleReqIds(entries);
@@ -213,12 +245,25 @@ export function planDedupe(
213
245
  };
214
246
  const plan: DedupePlanItem[] = [];
215
247
  for (const dup of duplicates) {
216
- const [keep, ...rest] = dup.files;
217
- for (const remapFile of rest) {
218
- const newReqId = fresh();
219
- if (keep !== undefined) {
220
- plan.push({ reqId: dup.reqId, keepFile: keep, remapFile, newReqId });
248
+ // occurrences are in scan order (sorted files, in-file rule order); the
249
+ // per-file ordinal counts every occurrence so kept ones hold their slot
250
+ // in the text and remapped ones target the exact `@req:` tag to rewrite.
251
+ const perFileOrdinal = new Map<string, number>();
252
+ let keepFile: string | null = null;
253
+ for (const occ of dup.occurrences) {
254
+ const ordinal = (perFileOrdinal.get(occ.fileName) ?? 0) + 1;
255
+ perFileOrdinal.set(occ.fileName, ordinal);
256
+ if (keepFile === null) {
257
+ keepFile = occ.fileName;
258
+ continue;
221
259
  }
260
+ plan.push({
261
+ reqId: dup.reqId,
262
+ keepFile,
263
+ remapFile: occ.fileName,
264
+ newReqId: fresh(),
265
+ occurrenceOrdinal: ordinal,
266
+ });
222
267
  }
223
268
  }
224
269
  if (opts.apply === false) return plan;
@@ -228,7 +273,10 @@ export function planDedupe(
228
273
  ? item.remapFile
229
274
  : `${specsRoot}/${item.remapFile}`;
230
275
  const content = io.readText(path);
231
- io.writeText(path, content.replaceAll(`@req:${item.reqId}`, `@req:${item.newReqId}`));
276
+ io.writeText(
277
+ path,
278
+ replaceNthReqTag(content, item.reqId, item.newReqId, item.occurrenceOrdinal),
279
+ );
232
280
  }
233
281
  return plan;
234
282
  }
@@ -0,0 +1,147 @@
1
+ /**
2
+ * Official Gherkin dialect keyword lookup — the vocabulary SSOT is the
3
+ * official table shipped with @cucumber/gherkin (`dialects`, i.e.
4
+ * gherkin-languages.json, 80 dialects). Emitting keywords from anything
5
+ * else risks mixed-dialect output the official parser rejects.
6
+ *
7
+ * Selection policy (deterministic): the official tables list synonyms in an
8
+ * order that reproduces no contract (en `scenario: ["Example", "Scenario"]`,
9
+ * zh-CN `rule: ["Rule", "规则"]`), so position heuristics cannot work. The
10
+ * en and zh-CN contract keywords (r88 — 不得漂移) are pinned explicitly and
11
+ * MUST be members of their official table (runtime-checked, tests pin the
12
+ * bytes); every other dialect takes the first non-star synonym.
13
+ */
14
+ import { dialects, type Dialect } from '@cucumber/gherkin';
15
+
16
+ export interface GherkinKeywords {
17
+ feature: string;
18
+ rule: string;
19
+ scenario: string;
20
+ given: string;
21
+ when: string;
22
+ thenText: string;
23
+ }
24
+
25
+ /**
26
+ * GherkinKeywords keys ↔ official Dialect field keys (thenText ↔ then).
27
+ */
28
+ const TABLE_KEY_PAIRS = [
29
+ ['feature', 'feature'],
30
+ ['rule', 'rule'],
31
+ ['scenario', 'scenario'],
32
+ ['given', 'given'],
33
+ ['when', 'when'],
34
+ ['then', 'thenText'],
35
+ ] as const;
36
+
37
+ /** Contract-locked keyword bytes (r88) — membership-checked against the table. */
38
+ const LOCKED_KEYWORDS: Record<string, GherkinKeywords> = {
39
+ en: {
40
+ feature: 'Feature',
41
+ rule: 'Rule',
42
+ scenario: 'Scenario',
43
+ given: 'Given',
44
+ when: 'When',
45
+ thenText: 'Then',
46
+ },
47
+ 'zh-CN': {
48
+ feature: '功能',
49
+ rule: '规则',
50
+ scenario: '场景',
51
+ given: '假如',
52
+ when: '当',
53
+ thenText: '那么',
54
+ },
55
+ };
56
+
57
+ function pickKeyword(entries: readonly string[]): string {
58
+ return entries.map((k) => k.trim()).find((k) => k !== '' && k !== '*') ?? '';
59
+ }
60
+
61
+ function keywordsFromTable(table: Dialect): GherkinKeywords | null {
62
+ const kw: GherkinKeywords = {
63
+ feature: pickKeyword(table.feature),
64
+ rule: pickKeyword(table.rule),
65
+ scenario: pickKeyword(table.scenario),
66
+ given: pickKeyword(table.given),
67
+ when: pickKeyword(table.when),
68
+ thenText: pickKeyword(table.then),
69
+ };
70
+ if (Object.values(kw).some((v) => v === '')) return null;
71
+ return kw;
72
+ }
73
+
74
+ /** Official keywords for a gherkin dialect, or null when the language has no table. */
75
+ export function officialKeywords(language: string): GherkinKeywords | null {
76
+ const table = dialects[language];
77
+ if (!table) return null;
78
+ const locked = LOCKED_KEYWORDS[language];
79
+ if (locked === undefined) return keywordsFromTable(table);
80
+ // A locked byte that left the official table would silently drift the
81
+ // dialect — fall through to the table instead of emitting it.
82
+ const fromTable = keywordsFromTable(table);
83
+ if (fromTable === null) return null;
84
+ return TABLE_KEY_PAIRS.every(([tableKey, kwKey]) =>
85
+ table[tableKey].map((k) => k.trim()).includes(locked[kwKey]),
86
+ )
87
+ ? locked
88
+ : fromTable;
89
+ }
90
+
91
+ /** Official keywords, falling back to the en table for table-less languages. */
92
+ export function officialKeywordsOrEn(language: string): GherkinKeywords {
93
+ return officialKeywords(language) ?? officialKeywords('en') ?? LOCKED_KEYWORDS['en']!;
94
+ }
95
+
96
+ /**
97
+ * Official step keyword → kind across every dialect (star-filtered,
98
+ * trimmed). And/But/`*` are absent by construction — callers decide their
99
+ * inheritance. en/zh-CN resolve to the same kinds as the former hand-rolled
100
+ * regexes; other official dialects (fr Soit/Quand/Alors, …) now classify
101
+ * correctly instead of falling into a default.
102
+ */
103
+ export const STEP_KIND_BY_KEYWORD: ReadonlyMap<string, 'given' | 'when' | 'then'> = (() => {
104
+ const map = new Map<string, 'given' | 'when' | 'then'>();
105
+ for (const table of Object.values(dialects)) {
106
+ for (const kind of ['given', 'when', 'then'] as const) {
107
+ for (const raw of table[kind]) {
108
+ const k = raw.trim();
109
+ if (k !== '' && k !== '*' && !map.has(k)) map.set(k, kind);
110
+ }
111
+ }
112
+ }
113
+ return map;
114
+ })();
115
+
116
+ export function stepKeywordToOfficialKind(keyword: string): 'given' | 'when' | 'then' | null {
117
+ return STEP_KIND_BY_KEYWORD.get(keyword.trim()) ?? null;
118
+ }
119
+
120
+ function escapeRegExp(k: string): string {
121
+ return k.replaceAll(/[.*+?^${}()|[\]\\]/gu, '\\$&');
122
+ }
123
+
124
+ /**
125
+ * Top-level block keywords across every official dialect (2-space indent +
126
+ * keyword + `:`) — next-block boundary scanning must recognize a block end
127
+ * in any language, not just the four hardcoded ones.
128
+ */
129
+ export const BLOCK_KEYWORD_LINE_RE: RegExp = new RegExp(
130
+ `^ (?:${[
131
+ ...new Set(
132
+ Object.values(dialects)
133
+ .flatMap((table) => [
134
+ ...table.feature,
135
+ ...table.rule,
136
+ ...table.scenario,
137
+ ...table.scenarioOutline,
138
+ ...table.background,
139
+ ])
140
+ .map((k) => k.trim())
141
+ .filter((k) => k !== '' && k !== '*'),
142
+ ),
143
+ ]
144
+ .map(escapeRegExp)
145
+ .join('|')}):`,
146
+ 'u',
147
+ );
@@ -9,7 +9,8 @@
9
9
  * derived from the tag set the official parser exposes per scenario.
10
10
  */
11
11
 
12
- import { parseFeatureSource } from './parser.ts';
12
+ import { officialKeywordsOrEn } from './keywords.ts';
13
+ import { parseFeatureSource, sourceDialect } from './parser.ts';
13
14
 
14
15
  export interface MigrateBlock {
15
16
  reqIds: string[];
@@ -22,6 +23,8 @@ export interface MigrateBlock {
22
23
  }
23
24
 
24
25
  export interface MigrateAnalysis {
26
+ /** Gherkin dialect the source resolved to (en start, zh-CN fallback). */
27
+ language: string;
25
28
  blocks: MigrateBlock[];
26
29
  }
27
30
 
@@ -86,7 +89,7 @@ export function analyzeLegacy(source: string): MigrateAnalysis | { ok: false; me
86
89
  steps: sc.steps.map((s) => ({ keyword: s.keyword.trim(), text: s.text.trim() })),
87
90
  });
88
91
  }
89
- return { blocks };
92
+ return { language: sourceDialect(source), blocks };
90
93
  }
91
94
 
92
95
  /** Strip `- ` list markers from a rule-statement line (legacy prose residue). */
@@ -106,11 +109,20 @@ export function migrateNativeSource(source: string): MigrateResult {
106
109
  const analysis = analyzeLegacy(source);
107
110
  if ('ok' in analysis) return { ok: false, message: analysis.message };
108
111
 
109
- const { blocks } = analysis as MigrateAnalysis;
112
+ const { language, blocks } = analysis as MigrateAnalysis;
110
113
  if (blocks.length === 0) {
111
114
  return { ok: false, message: 'no legacy scenarios found (already native?)' };
112
115
  }
113
116
 
117
+ // Keywords come from the official gherkin dialect table so the output
118
+ // stays parseable in one language — the preamble (`# language:` header,
119
+ // `Feature:`/`功能:` line) is kept verbatim and already matches it. The
120
+ // trailing parse self-check guards the rest.
121
+ const kw = officialKeywordsOrEn(language);
122
+ // The auto-nested acceptance title is a synthesized name, not a keyword —
123
+ // no official source exists, so it keeps its localized map.
124
+ const autoAcceptance = language === 'zh-CN' ? '验收示例' : 'Acceptance example';
125
+
114
126
  // preamble: everything before the first top-level tag line (headers,
115
127
  // feature line, comments) — cut textually, preserved verbatim.
116
128
  const lines = source.split('\n');
@@ -130,11 +142,21 @@ export function migrateNativeSource(source: string): MigrateResult {
130
142
  if (!b.isRule) continue;
131
143
  rules++;
132
144
  out.push(` @req:${b.reqIds[0] ?? ''}`);
133
- out.push(` 规则: ${b.title}`);
145
+ out.push(` ${kw.rule}: ${b.title}`);
134
146
  for (const line of b.descriptionLines) {
135
147
  const text = stripBullet(line);
136
148
  if (text !== '') out.push(` ${text}`);
137
149
  }
150
+ // A legacy rule scenario may carry its own acceptance steps inline
151
+ // (description + steps in one block). They become the rule's first
152
+ // nested scenario — file-order semantics, never dropped.
153
+ if (b.steps.length > 0) {
154
+ out.push('');
155
+ if (b.skip) out.push(' @skip');
156
+ out.push(` ${kw.scenario}: ${autoAcceptance}`);
157
+ for (const s of b.steps) out.push(` ${s.keyword} ${s.text}`);
158
+ scenarios++;
159
+ }
138
160
  for (const [ai, a] of blocks.entries()) {
139
161
  if (a.isRule || consumed.has(ai)) continue;
140
162
  if (!a.reqIds.some((rid) => b.reqIds.includes(rid))) continue;
@@ -142,7 +164,7 @@ export function migrateNativeSource(source: string): MigrateResult {
142
164
  scenarios++;
143
165
  out.push('');
144
166
  if (a.skip) out.push(' @skip');
145
- out.push(` 场景: ${a.title}`);
167
+ out.push(` ${kw.scenario}: ${a.title}`);
146
168
  for (const s of a.steps) out.push(` ${s.keyword} ${s.text}`);
147
169
  }
148
170
  }
@@ -156,12 +178,24 @@ export function migrateNativeSource(source: string): MigrateResult {
156
178
  scenarios++;
157
179
  out.push('');
158
180
  if (a.skip) out.push(' @skip');
159
- out.push(` 场景: ${a.title}`);
181
+ out.push(` ${kw.scenario}: ${a.title}`);
160
182
  for (const s of a.steps) out.push(` ${s.keyword} ${s.text}`);
161
183
  }
162
184
 
163
185
  if (rules === 0) {
164
186
  return { ok: false, message: 'no legacy rule scenarios found (@req + @rule/@human)' };
165
187
  }
166
- return { ok: true, content: `${out.join('\n')}\n`, rules, scenarios };
188
+
189
+ const content = `${out.join('\n')}\n`;
190
+ // Fail closed: never hand back content the official parser rejects — a
191
+ // dialect mismatch would otherwise land on disk as silent corruption.
192
+ try {
193
+ parseFeatureSource(content);
194
+ } catch (error) {
195
+ return {
196
+ ok: false,
197
+ message: `migrated output failed parse self-check: ${error instanceof Error ? error.message : String(error)}`,
198
+ };
199
+ }
200
+ return { ok: true, content, rules, scenarios };
167
201
  }
@@ -19,6 +19,7 @@ import {
19
19
  type ScenarioStepKind,
20
20
  type SpecStructuralError,
21
21
  } from './ir.ts';
22
+ import { stepKeywordToOfficialKind } from './keywords.ts';
22
23
 
23
24
  export class SpecParseError extends Error {}
24
25
 
@@ -52,15 +53,30 @@ export function parseFeatureSource(source: string): { doc: GherkinDocument; lang
52
53
  );
53
54
  }
54
55
 
56
+ /**
57
+ * Unified source-dialect policy (r41/r88): an explicit per-file
58
+ * `# language:` header wins; headerless content is auto-discovered through
59
+ * the official matcher chain (en start, zh-CN fallback); anything still
60
+ * undiscoverable falls back to en. Unknown header names are returned
61
+ * as-is — callers pick vocabulary via officialKeywordsOrEn, and the
62
+ * migration parse self-check fail-closes genuinely broken headers.
63
+ */
64
+ export function sourceDialect(source: string): string {
65
+ const firstLine = source.split('\n').find((l) => l.trim() !== '');
66
+ const header = firstLine?.match(/^#\s*language:\s*(\S+)\s*$/u)?.[1];
67
+ if (header !== undefined) return header;
68
+ try {
69
+ return parseFeatureSource(source).language;
70
+ } catch {
71
+ return 'en';
72
+ }
73
+ }
74
+
55
75
  const REQ_TAG_RE = /^@?req:(r\d+)$/u;
56
76
 
57
- /** Gherkin keyword (zh-CN + en) → step kind; And/But/* inherit via fallback. */
77
+ /** Gherkin keyword → step kind, from the official dialect tables; And/But/* inherit via fallback. */
58
78
  function stepKeywordToKind(keyword: string): ScenarioStepKind {
59
- const kw = keyword.trim();
60
- if (/^(假如|Given)/iu.test(kw)) return 'given';
61
- if (/^(当|When)/iu.test(kw)) return 'when';
62
- if (/^(那么|Then)/iu.test(kw)) return 'then';
63
- return 'given';
79
+ return stepKeywordToOfficialKind(keyword) ?? 'given';
64
80
  }
65
81
 
66
82
  const HEADER_RE = /^#\s*(capability|purpose|scope):\s*(.*)$/u;
@@ -1,36 +1,53 @@
1
1
  /**
2
2
  * Global rN registry (spec-parsing capability): @req:rN ids on `规则:` blocks
3
- * form a single global namespace across all capability specs; duplicates are
4
- * reported with the conflicting file pairs.
3
+ * form a single global namespace across all capability specs. A duplicate is
4
+ * any id carried by more than one rule — within one file or across files
5
+ * (r10: 按携带该 id 的规则条数 > 1 判定).
5
6
  */
6
7
  import type { CapabilityDoc } from './ir.ts';
7
8
 
9
+ export interface RegistryOccurrence {
10
+ fileName: string;
11
+ /** 0-based index of the rule within its doc (in-file locating). */
12
+ ruleIndex: number;
13
+ title: string;
14
+ }
15
+
8
16
  export interface RegistryDuplicate {
9
17
  reqId: string;
18
+ /** Unique files referencing the id (sorted, display surface). */
10
19
  files: string[];
20
+ /** Every rule occurrence referencing the id, in scan order. */
21
+ occurrences: RegistryOccurrence[];
11
22
  }
12
23
 
13
24
  export interface ReqRegistry {
14
- /** reqId → files referencing it (sorted). */
15
- byId: Map<string, string[]>;
25
+ /** reqId → every rule occurrence referencing it (scan order). */
26
+ byId: Map<string, RegistryOccurrence[]>;
16
27
  duplicates: RegistryDuplicate[];
17
28
  }
18
29
 
19
30
  export function buildReqRegistry(
20
31
  docs: readonly { fileName: string; doc: CapabilityDoc }[],
21
32
  ): ReqRegistry {
22
- const byId = new Map<string, string[]>();
33
+ const byId = new Map<string, RegistryOccurrence[]>();
23
34
  for (const { fileName, doc } of docs) {
24
- for (const rule of doc.rules) {
25
- if (rule.reqId === '') continue;
26
- const files = byId.get(rule.reqId) ?? [];
27
- if (!files.includes(fileName)) files.push(fileName);
28
- byId.set(rule.reqId, files);
29
- }
35
+ doc.rules.forEach((rule, ruleIndex) => {
36
+ if (rule.reqId === '') return;
37
+ const occurrences = byId.get(rule.reqId) ?? [];
38
+ occurrences.push({ fileName, ruleIndex, title: rule.title });
39
+ byId.set(rule.reqId, occurrences);
40
+ });
30
41
  }
31
42
  const duplicates: RegistryDuplicate[] = [];
32
- for (const [reqId, files] of byId) {
33
- if (files.length > 1) duplicates.push({ reqId, files: files.toSorted() });
43
+ for (const [reqId, occurrences] of byId) {
44
+ if (occurrences.length > 1) {
45
+ duplicates.push({
46
+ reqId,
47
+ files: [...new Set(occurrences.map((o) => o.fileName))].toSorted(),
48
+ occurrences,
49
+ });
50
+ }
34
51
  }
35
52
  duplicates.sort((a, b) => a.reqId.localeCompare(b.reqId));
36
53
  return { byId, duplicates };