@llman-sdd/core 0.3.1 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/package.json +2 -1
  2. package/src/archive/freeze.ts +86 -18
  3. package/src/archive/frozenCard.ts +105 -0
  4. package/src/archive/sevenzip.ts +15 -13
  5. package/src/change/closeOutHarness.ts +29 -0
  6. package/src/change/collect.ts +140 -0
  7. package/src/change/frontmatter.ts +48 -6
  8. package/src/change/id.ts +2 -6
  9. package/src/change/lifecycle.ts +285 -86
  10. package/src/change/nextId.ts +63 -2
  11. package/src/change/resolve.ts +2 -2
  12. package/src/change/tasks.ts +59 -0
  13. package/src/config/changeId.ts +14 -12
  14. package/src/config/load.ts +14 -0
  15. package/src/config/schema.ts +4 -41
  16. package/src/config/surface.ts +6 -36
  17. package/src/context/indexStore.ts +7 -3
  18. package/src/context/retrieve.ts +8 -10
  19. package/src/context/tree.ts +28 -24
  20. package/src/git/spawnGit.ts +90 -2
  21. package/src/index.ts +81 -54
  22. package/src/init/defaultConfig.ts +1 -5
  23. package/src/init/init.ts +19 -4
  24. package/src/ports.ts +1 -7
  25. package/src/project/migrateNotes.ts +104 -0
  26. package/src/render/machine.ts +30 -0
  27. package/src/report/collect.ts +11 -127
  28. package/src/report/graph/analysis.ts +152 -0
  29. package/src/report/graph/deps.ts +30 -0
  30. package/src/report/graph/graphData.ts +53 -0
  31. package/src/report/graph/nodes.ts +130 -0
  32. package/src/report/graph/render.ts +83 -0
  33. package/src/report/graph/types.ts +47 -0
  34. package/src/report/graph.ts +9 -381
  35. package/src/report/show.ts +20 -22
  36. package/src/report/specHelpers.ts +45 -22
  37. package/src/report/specs.ts +23 -25
  38. package/src/review/review.ts +45 -30
  39. package/src/spec/authoring.ts +147 -71
  40. package/src/spec/ir.ts +43 -15
  41. package/src/spec/keywords.ts +147 -0
  42. package/src/spec/migrateNative.ts +201 -0
  43. package/src/spec/parser.ts +95 -83
  44. package/src/spec/reqRegistry.ts +31 -15
  45. package/src/templates/embedded.ts +10 -16
  46. package/src/templates/engine.ts +10 -5
  47. package/src/templates/locale.ts +1 -1
  48. package/src/templates/skills.ts +4 -5
  49. package/src/validation/changeCheck.ts +128 -105
  50. package/src/validation/harness.ts +161 -0
  51. package/src/validation/staleness.ts +9 -5
  52. package/src/validation/validate.ts +60 -88
  53. package/templates/en/skills/llman-sdd-apply-cycle.md +20 -28
  54. package/templates/en/skills/llman-sdd-apply.md +58 -76
  55. package/templates/en/skills/llman-sdd-arch-review.md +12 -19
  56. package/templates/en/skills/llman-sdd-archive.md +27 -42
  57. package/templates/en/skills/llman-sdd-continue.md +17 -24
  58. package/templates/en/skills/llman-sdd-draft.md +17 -28
  59. package/templates/en/skills/llman-sdd-explore.md +29 -43
  60. package/templates/en/skills/llman-sdd-ff.md +12 -17
  61. package/templates/en/skills/llman-sdd-graph.md +14 -32
  62. package/templates/en/skills/llman-sdd-propose.md +48 -63
  63. package/templates/en/skills/llman-sdd-quick.md +12 -27
  64. package/templates/en/skills/llman-sdd-research.md +13 -24
  65. package/templates/en/skills/llman-sdd-specs-compact.md +14 -39
  66. package/templates/en/skills/llman-sdd-validate.md +11 -15
  67. package/templates/en/skills/llman-sdd-verify.md +23 -44
  68. package/templates/en/skills/llman-sdd-wayfinder.md +18 -22
  69. package/templates/en/units/skills/cli-footer.md +2 -0
  70. package/templates/en/units/skills/git-native-flow-brief.md +7 -6
  71. package/templates/en/units/skills/git-native-flow.md +21 -11
  72. package/templates/en/units/skills/human-readable-summary.md +2 -3
  73. package/templates/en/units/skills/stage-guard.md +7 -7
  74. package/templates/en/units/skills/structured-protocol.md +5 -8
  75. package/templates/en/units/skills/validation-hints.md +10 -14
  76. package/templates/en/units/spec/feature-contract.md +27 -16
  77. package/templates/en/units/workflow/archive-freeze-guidance.md +6 -3
  78. package/templates/zh-Hans/skills/llman-sdd-apply-cycle.md +23 -31
  79. package/templates/zh-Hans/skills/llman-sdd-apply.md +63 -81
  80. package/templates/zh-Hans/skills/llman-sdd-arch-review.md +21 -28
  81. package/templates/zh-Hans/skills/llman-sdd-archive.md +29 -44
  82. package/templates/zh-Hans/skills/llman-sdd-continue.md +17 -24
  83. package/templates/zh-Hans/skills/llman-sdd-draft.md +18 -29
  84. package/templates/zh-Hans/skills/llman-sdd-explore.md +34 -48
  85. package/templates/zh-Hans/skills/llman-sdd-ff.md +13 -18
  86. package/templates/zh-Hans/skills/llman-sdd-graph.md +16 -34
  87. package/templates/zh-Hans/skills/llman-sdd-propose.md +51 -65
  88. package/templates/zh-Hans/skills/llman-sdd-quick.md +15 -30
  89. package/templates/zh-Hans/skills/llman-sdd-research.md +17 -28
  90. package/templates/zh-Hans/skills/llman-sdd-specs-compact.md +15 -40
  91. package/templates/zh-Hans/skills/llman-sdd-validate.md +11 -15
  92. package/templates/zh-Hans/skills/llman-sdd-verify.md +26 -47
  93. package/templates/zh-Hans/skills/llman-sdd-wayfinder.md +25 -29
  94. package/templates/zh-Hans/units/skills/cli-footer.md +2 -0
  95. package/templates/zh-Hans/units/skills/git-native-flow-brief.md +7 -6
  96. package/templates/zh-Hans/units/skills/git-native-flow.md +22 -12
  97. package/templates/zh-Hans/units/skills/human-readable-summary.md +4 -5
  98. package/templates/zh-Hans/units/skills/stage-guard.md +9 -9
  99. package/templates/zh-Hans/units/skills/structured-protocol.md +5 -8
  100. package/templates/zh-Hans/units/skills/validation-hints.md +10 -14
  101. package/templates/zh-Hans/units/spec/feature-contract.md +25 -16
  102. package/templates/zh-Hans/units/workflow/archive-freeze-guidance.md +6 -2
  103. package/templates/en/skills/llman-sdd-onboard.md +0 -34
  104. package/templates/en/skills/llman-sdd-show.md +0 -24
  105. package/templates/en/units/migrate-prompt.md +0 -28
  106. package/templates/zh-Hans/skills/llman-sdd-onboard.md +0 -34
  107. package/templates/zh-Hans/skills/llman-sdd-show.md +0 -24
  108. package/templates/zh-Hans/units/migrate-prompt.md +0 -28
package/src/spec/ir.ts CHANGED
@@ -2,6 +2,12 @@
2
2
  * Spec IR (spec-parsing capability): pure data shapes produced by
3
3
  * parseCapability() and consumed by validation (Phase 3) and downstream
4
4
  * commands. No IO of any kind lives here.
5
+ *
6
+ * Native model (v2): a capability feature is `功能:` → `规则:` blocks (the
7
+ * requirement: title + free-form description + `@req:<id>` handle on the block
8
+ * header) → nested `场景:` (executable GWT examples). Top-level scenarios that
9
+ * are not under any rule are orphans. Legacy tags (@human/@rule/@executable)
10
+ * are inert to parsing; migrate with `spec migrate-native`.
5
11
  */
6
12
 
7
13
  export interface CapabilityHeader {
@@ -10,24 +16,40 @@ export interface CapabilityHeader {
10
16
  scope: string | null;
11
17
  }
12
18
 
13
- export type ScenarioClassification = 'human' | 'executable' | 'unclassified';
19
+ export type ScenarioStepKind = 'given' | 'when' | 'then';
20
+
21
+ export interface ScenarioStep {
22
+ kind: ScenarioStepKind;
23
+ text: string;
24
+ }
14
25
 
15
26
  export interface ScenarioIR {
16
27
  name: string;
17
- /** Tag names without the leading `@`. */
28
+ /** Non-`@req` tags on this scenario (e.g. `@skip`, `@experimental`). */
18
29
  tags: string[];
19
- /** `@req:rN` links, normalized to `rN`. */
20
- reqIds: string[];
21
- classification: ScenarioClassification;
22
- /** Rule statement (description lines, trimmed) + step texts for executables. */
23
- statement: string;
30
+ /** Whether the runner should execute this scenario (false for @skip etc.). */
31
+ runnable: boolean;
24
32
  stepCount: number;
25
- /** Executable-scenario steps with their keyword kinds (context-index tree). */
26
- steps: { kind: 'given' | 'when' | 'then'; text: string }[];
33
+ steps: ScenarioStep[];
34
+ /** Step-texts joined (retrieval/context surface); empty for a stepless scenario. */
35
+ statement: string;
36
+ }
37
+
38
+ export interface RuleIR {
39
+ /** The `@req:<id>` handle on the rule block header (global-registry key). */
40
+ reqId: string;
41
+ /** The `规则:` block title. */
42
+ title: string;
43
+ /** Free-form requirement statement (block description lines, as authored). */
44
+ description: string;
45
+ /** Nested executable examples belonging to this rule. */
46
+ scenarios: ScenarioIR[];
47
+ /** Non-`@req` tags on the rule block header (legacy @human etc. — inert). */
48
+ tags: string[];
27
49
  }
28
50
 
29
51
  export interface SpecStructuralError {
30
- /** Machine-ish anchor, e.g. `missing-header:purpose` or `scenario:规则样例`. */
52
+ /** Machine-ish anchor, e.g. `missing-header:purpose` or `scenario:样例`. */
31
53
  code: string;
32
54
  message: string;
33
55
  }
@@ -37,13 +59,19 @@ export interface CapabilityDoc {
37
59
  header: CapabilityHeader;
38
60
  featureName: string;
39
61
  language: string;
40
- scenarios: ScenarioIR[];
62
+ /** Requirement blocks (`规则:`); the canonical rule set. */
63
+ rules: RuleIR[];
64
+ /** Top-level scenarios not enclosed by any rule (orphans). */
65
+ orphans: ScenarioIR[];
41
66
  errors: SpecStructuralError[];
42
67
  }
43
68
 
44
69
  /**
45
- * v1 wording: constraint statements must contain one of these tokens.
46
- * Shared by the parser (structural error) and validation (verdict gate) so
47
- * the two MUST-word checks cannot drift apart.
70
+ * Single spec-id caliber (r25) for every consumer that labels a discovered
71
+ * spec entry: the `# capability:` header wins, else the fileName minus the
72
+ * `.feature` suffix. Unifies the former dual caliber (bare `fileName` vs
73
+ * stripped stem) shared by review/context-tree/specs-report/validate paths.
48
74
  */
49
- export const MUST_WORD_RE = /\bMUST\b|\bSHALL\b|必须|不得|禁止/u;
75
+ export function specIdOf(entry: { fileName: string; doc: CapabilityDoc }): string {
76
+ return entry.doc.header.capability ?? entry.fileName.replace(/\.feature$/u, '');
77
+ }
@@ -0,0 +1,147 @@
1
+ /**
2
+ * Official Gherkin dialect keyword lookup — the vocabulary SSOT is the
3
+ * official table shipped with @cucumber/gherkin (`dialects`, i.e.
4
+ * gherkin-languages.json, 80 dialects). Emitting keywords from anything
5
+ * else risks mixed-dialect output the official parser rejects.
6
+ *
7
+ * Selection policy (deterministic): the official tables list synonyms in an
8
+ * order that reproduces no contract (en `scenario: ["Example", "Scenario"]`,
9
+ * zh-CN `rule: ["Rule", "规则"]`), so position heuristics cannot work. The
10
+ * en and zh-CN contract keywords (r88 — 不得漂移) are pinned explicitly and
11
+ * MUST be members of their official table (runtime-checked, tests pin the
12
+ * bytes); every other dialect takes the first non-star synonym.
13
+ */
14
+ import { dialects, type Dialect } from '@cucumber/gherkin';
15
+
16
+ export interface GherkinKeywords {
17
+ feature: string;
18
+ rule: string;
19
+ scenario: string;
20
+ given: string;
21
+ when: string;
22
+ thenText: string;
23
+ }
24
+
25
+ /**
26
+ * GherkinKeywords keys ↔ official Dialect field keys (thenText ↔ then).
27
+ */
28
+ const TABLE_KEY_PAIRS = [
29
+ ['feature', 'feature'],
30
+ ['rule', 'rule'],
31
+ ['scenario', 'scenario'],
32
+ ['given', 'given'],
33
+ ['when', 'when'],
34
+ ['then', 'thenText'],
35
+ ] as const;
36
+
37
+ /** Contract-locked keyword bytes (r88) — membership-checked against the table. */
38
+ const LOCKED_KEYWORDS: Record<string, GherkinKeywords> = {
39
+ en: {
40
+ feature: 'Feature',
41
+ rule: 'Rule',
42
+ scenario: 'Scenario',
43
+ given: 'Given',
44
+ when: 'When',
45
+ thenText: 'Then',
46
+ },
47
+ 'zh-CN': {
48
+ feature: '功能',
49
+ rule: '规则',
50
+ scenario: '场景',
51
+ given: '假如',
52
+ when: '当',
53
+ thenText: '那么',
54
+ },
55
+ };
56
+
57
+ function pickKeyword(entries: readonly string[]): string {
58
+ return entries.map((k) => k.trim()).find((k) => k !== '' && k !== '*') ?? '';
59
+ }
60
+
61
+ function keywordsFromTable(table: Dialect): GherkinKeywords | null {
62
+ const kw: GherkinKeywords = {
63
+ feature: pickKeyword(table.feature),
64
+ rule: pickKeyword(table.rule),
65
+ scenario: pickKeyword(table.scenario),
66
+ given: pickKeyword(table.given),
67
+ when: pickKeyword(table.when),
68
+ thenText: pickKeyword(table.then),
69
+ };
70
+ if (Object.values(kw).some((v) => v === '')) return null;
71
+ return kw;
72
+ }
73
+
74
+ /** Official keywords for a gherkin dialect, or null when the language has no table. */
75
+ export function officialKeywords(language: string): GherkinKeywords | null {
76
+ const table = dialects[language];
77
+ if (!table) return null;
78
+ const locked = LOCKED_KEYWORDS[language];
79
+ if (locked === undefined) return keywordsFromTable(table);
80
+ // A locked byte that left the official table would silently drift the
81
+ // dialect — fall through to the table instead of emitting it.
82
+ const fromTable = keywordsFromTable(table);
83
+ if (fromTable === null) return null;
84
+ return TABLE_KEY_PAIRS.every(([tableKey, kwKey]) =>
85
+ table[tableKey].map((k) => k.trim()).includes(locked[kwKey]),
86
+ )
87
+ ? locked
88
+ : fromTable;
89
+ }
90
+
91
+ /** Official keywords, falling back to the en table for table-less languages. */
92
+ export function officialKeywordsOrEn(language: string): GherkinKeywords {
93
+ return officialKeywords(language) ?? officialKeywords('en') ?? LOCKED_KEYWORDS['en']!;
94
+ }
95
+
96
+ /**
97
+ * Official step keyword → kind across every dialect (star-filtered,
98
+ * trimmed). And/But/`*` are absent by construction — callers decide their
99
+ * inheritance. en/zh-CN resolve to the same kinds as the former hand-rolled
100
+ * regexes; other official dialects (fr Soit/Quand/Alors, …) now classify
101
+ * correctly instead of falling into a default.
102
+ */
103
+ export const STEP_KIND_BY_KEYWORD: ReadonlyMap<string, 'given' | 'when' | 'then'> = (() => {
104
+ const map = new Map<string, 'given' | 'when' | 'then'>();
105
+ for (const table of Object.values(dialects)) {
106
+ for (const kind of ['given', 'when', 'then'] as const) {
107
+ for (const raw of table[kind]) {
108
+ const k = raw.trim();
109
+ if (k !== '' && k !== '*' && !map.has(k)) map.set(k, kind);
110
+ }
111
+ }
112
+ }
113
+ return map;
114
+ })();
115
+
116
+ export function stepKeywordToOfficialKind(keyword: string): 'given' | 'when' | 'then' | null {
117
+ return STEP_KIND_BY_KEYWORD.get(keyword.trim()) ?? null;
118
+ }
119
+
120
+ function escapeRegExp(k: string): string {
121
+ return k.replaceAll(/[.*+?^${}()|[\]\\]/gu, '\\$&');
122
+ }
123
+
124
+ /**
125
+ * Top-level block keywords across every official dialect (2-space indent +
126
+ * keyword + `:`) — next-block boundary scanning must recognize a block end
127
+ * in any language, not just the four hardcoded ones.
128
+ */
129
+ export const BLOCK_KEYWORD_LINE_RE: RegExp = new RegExp(
130
+ `^ (?:${[
131
+ ...new Set(
132
+ Object.values(dialects)
133
+ .flatMap((table) => [
134
+ ...table.feature,
135
+ ...table.rule,
136
+ ...table.scenario,
137
+ ...table.scenarioOutline,
138
+ ...table.background,
139
+ ])
140
+ .map((k) => k.trim())
141
+ .filter((k) => k !== '' && k !== '*'),
142
+ ),
143
+ ]
144
+ .map(escapeRegExp)
145
+ .join('|')}):`,
146
+ 'u',
147
+ );
@@ -0,0 +1,201 @@
1
+ /**
2
+ * Legacy → native spec migration (v1 tag-based flat layout → native `规则:`
3
+ * blocks with nested `场景:`). Pure text transform with IO injected by the
4
+ * caller. Used by `spec migrate-native` (interactive + --dry-run) and the
5
+ * in-repo one-shot migration.
6
+ *
7
+ * Parsing ALWAYS goes through the official @cucumber/gherkin parser
8
+ * (parseFeatureSource) — no hand-rolled line scanning; the legacy ROLE is
9
+ * derived from the tag set the official parser exposes per scenario.
10
+ */
11
+
12
+ import { officialKeywordsOrEn } from './keywords.ts';
13
+ import { parseFeatureSource, sourceDialect } from './parser.ts';
14
+
15
+ export interface MigrateBlock {
16
+ reqIds: string[];
17
+ isRule: boolean;
18
+ skip: boolean;
19
+ title: string;
20
+ /** Statement description lines for rules; step [keyword,text] pairs for acceptances. */
21
+ descriptionLines: string[];
22
+ steps: { keyword: string; text: string }[];
23
+ }
24
+
25
+ export interface MigrateAnalysis {
26
+ /** Gherkin dialect the source resolved to (en start, zh-CN fallback). */
27
+ language: string;
28
+ blocks: MigrateBlock[];
29
+ }
30
+
31
+ export type MigrateResult =
32
+ | { ok: true; content: string; rules: number; scenarios: number }
33
+ | { ok: false; message: string };
34
+
35
+ const REQ_TAG_RE = /^@?req:(r\d+)$/u;
36
+
37
+ function tagsOf(tags: readonly { name: string }[]): {
38
+ reqIds: string[];
39
+ isRule: boolean;
40
+ skip: boolean;
41
+ } {
42
+ const names = tags.map((t) => t.name);
43
+ const reqIds = names
44
+ .map((n) => n.match(REQ_TAG_RE)?.[1])
45
+ .filter((v): v is string => v !== undefined);
46
+ const isRule = names.some((n) => n === '@rule' || n === '@human');
47
+ const skip = names.some((n) => n === '@skip' || n === '@experimental');
48
+ return { reqIds, isRule, skip };
49
+ }
50
+
51
+ /**
52
+ * True when the source already uses the native layout (has at least one
53
+ * top-level `规则:` block). Uses the official parser.
54
+ */
55
+ export function hasNativeRules(source: string): boolean {
56
+ try {
57
+ const { doc } = parseFeatureSource(source);
58
+ return (doc.feature?.children ?? []).some((c) => c.rule !== undefined);
59
+ } catch {
60
+ return false;
61
+ }
62
+ }
63
+
64
+ /**
65
+ * Analyze a legacy source through the official parser: every top-level
66
+ * scenario becomes a block; its role comes from the tags, its body from the
67
+ * official description/step fields.
68
+ */
69
+ export function analyzeLegacy(source: string): MigrateAnalysis | { ok: false; message: string } {
70
+ const { doc } = parseFeatureSource(source);
71
+ const blocks: MigrateBlock[] = [];
72
+ for (const child of doc.feature?.children ?? []) {
73
+ if (child.rule) {
74
+ // native source — nothing to migrate here; caller should skip
75
+ continue;
76
+ }
77
+ const sc = child.scenario;
78
+ if (!sc) continue;
79
+ const { reqIds, isRule, skip } = tagsOf(sc.tags);
80
+ blocks.push({
81
+ reqIds,
82
+ isRule,
83
+ skip,
84
+ title: sc.name,
85
+ descriptionLines: (sc.description ?? '')
86
+ .split('\n')
87
+ .map((l) => l.trim())
88
+ .filter((l) => l !== ''),
89
+ steps: sc.steps.map((s) => ({ keyword: s.keyword.trim(), text: s.text.trim() })),
90
+ });
91
+ }
92
+ return { language: sourceDialect(source), blocks };
93
+ }
94
+
95
+ /** Strip `- ` list markers from a rule-statement line (legacy prose residue). */
96
+ function stripBullet(line: string): string {
97
+ return line.replace(/^-\s+/u, '');
98
+ }
99
+
100
+ /**
101
+ * Migrate one legacy feature source to the native layout. Structural rules:
102
+ * rules keep file order; acceptances nest under the rule carrying their
103
+ * @req id (file order); acceptances without a matching rule stay top-level
104
+ * orphans. Non-scenario lines (headers/feature/comments) are NOT touched —
105
+ * the native output only emits tag/rule/scenario lines, so the preamble is
106
+ * reconstructed by the caller from the original file.
107
+ */
108
+ export function migrateNativeSource(source: string): MigrateResult {
109
+ const analysis = analyzeLegacy(source);
110
+ if ('ok' in analysis) return { ok: false, message: analysis.message };
111
+
112
+ const { language, blocks } = analysis as MigrateAnalysis;
113
+ if (blocks.length === 0) {
114
+ return { ok: false, message: 'no legacy scenarios found (already native?)' };
115
+ }
116
+
117
+ // Keywords come from the official gherkin dialect table so the output
118
+ // stays parseable in one language — the preamble (`# language:` header,
119
+ // `Feature:`/`功能:` line) is kept verbatim and already matches it. The
120
+ // trailing parse self-check guards the rest.
121
+ const kw = officialKeywordsOrEn(language);
122
+ // The auto-nested acceptance title is a synthesized name, not a keyword —
123
+ // no official source exists, so it keeps its localized map.
124
+ const autoAcceptance = language === 'zh-CN' ? '验收示例' : 'Acceptance example';
125
+
126
+ // preamble: everything before the first top-level tag line (headers,
127
+ // feature line, comments) — cut textually, preserved verbatim.
128
+ const lines = source.split('\n');
129
+ let preEnd = 0;
130
+ for (let i = 0; i < lines.length; i++) {
131
+ if ((lines[i] ?? '').startsWith(' @')) break;
132
+ preEnd = i + 1;
133
+ }
134
+ const preamble = lines.slice(0, preEnd);
135
+
136
+ const out: string[] = [...preamble];
137
+ let rules = 0;
138
+ let scenarios = 0;
139
+ const consumed = new Set<number>();
140
+
141
+ for (const b of blocks) {
142
+ if (!b.isRule) continue;
143
+ rules++;
144
+ out.push(` @req:${b.reqIds[0] ?? ''}`);
145
+ out.push(` ${kw.rule}: ${b.title}`);
146
+ for (const line of b.descriptionLines) {
147
+ const text = stripBullet(line);
148
+ if (text !== '') out.push(` ${text}`);
149
+ }
150
+ // A legacy rule scenario may carry its own acceptance steps inline
151
+ // (description + steps in one block). They become the rule's first
152
+ // nested scenario — file-order semantics, never dropped.
153
+ if (b.steps.length > 0) {
154
+ out.push('');
155
+ if (b.skip) out.push(' @skip');
156
+ out.push(` ${kw.scenario}: ${autoAcceptance}`);
157
+ for (const s of b.steps) out.push(` ${s.keyword} ${s.text}`);
158
+ scenarios++;
159
+ }
160
+ for (const [ai, a] of blocks.entries()) {
161
+ if (a.isRule || consumed.has(ai)) continue;
162
+ if (!a.reqIds.some((rid) => b.reqIds.includes(rid))) continue;
163
+ consumed.add(ai);
164
+ scenarios++;
165
+ out.push('');
166
+ if (a.skip) out.push(' @skip');
167
+ out.push(` ${kw.scenario}: ${a.title}`);
168
+ for (const s of a.steps) out.push(` ${s.keyword} ${s.text}`);
169
+ }
170
+ }
171
+ // Unbound acceptance scenarios (no matching rule) are emitted last at
172
+ // top level — native feature-level examples with no requirement handle.
173
+ // Under official Gherkin parsing they are absorbed into the preceding rule,
174
+ // which is their natural functional home; no special signal exists for them.
175
+ for (const [ai, a] of blocks.entries()) {
176
+ if (a.isRule || consumed.has(ai)) continue;
177
+ consumed.add(ai);
178
+ scenarios++;
179
+ out.push('');
180
+ if (a.skip) out.push(' @skip');
181
+ out.push(` ${kw.scenario}: ${a.title}`);
182
+ for (const s of a.steps) out.push(` ${s.keyword} ${s.text}`);
183
+ }
184
+
185
+ if (rules === 0) {
186
+ return { ok: false, message: 'no legacy rule scenarios found (@req + @rule/@human)' };
187
+ }
188
+
189
+ const content = `${out.join('\n')}\n`;
190
+ // Fail closed: never hand back content the official parser rejects — a
191
+ // dialect mismatch would otherwise land on disk as silent corruption.
192
+ try {
193
+ parseFeatureSource(content);
194
+ } catch (error) {
195
+ return {
196
+ ok: false,
197
+ message: `migrated output failed parse self-check: ${error instanceof Error ? error.message : String(error)}`,
198
+ };
199
+ }
200
+ return { ok: true, content, rules, scenarios };
201
+ }
@@ -1,21 +1,25 @@
1
1
  /**
2
2
  * Single-track capability .feature parsing (spec-parsing capability).
3
3
  *
4
- * Language fallback chain (r7): start with the `en` matcher — a
5
- * `# language:` header switches dialect automatically mid-scan; on failure
6
- * retry with the zh-CN matcher (Chinese keywords without a header); only then
7
- * surface a parse error. locale zh-Hans maps to gherkin zh-CN.
4
+ * Native v2 model: `功能:` → `规则:` blocks (requirement: title + free-form
5
+ * description + `@req:<id>` handle) → nested `场景:` (executable examples).
6
+ * Top-level `场景:` outside any rule become orphans. Legacy role tags
7
+ * (@human/@rule/@executable/@manual) are inert — the parser only reads
8
+ * `@req` (handle) and `@skip`/`@experimental` (runner opt-out). Files in the
9
+ * legacy flat layout must be migrated with `spec migrate-native`.
8
10
  */
9
11
  import { AstBuilder, GherkinClassicTokenMatcher, Parser } from '@cucumber/gherkin';
10
12
  import type { GherkinDocument } from '@cucumber/messages';
11
13
 
12
14
  import {
13
- MUST_WORD_RE,
14
15
  type CapabilityDoc,
15
16
  type CapabilityHeader,
17
+ type RuleIR,
16
18
  type ScenarioIR,
19
+ type ScenarioStepKind,
17
20
  type SpecStructuralError,
18
21
  } from './ir.ts';
22
+ import { stepKeywordToOfficialKind } from './keywords.ts';
19
23
 
20
24
  export class SpecParseError extends Error {}
21
25
 
@@ -49,16 +53,32 @@ export function parseFeatureSource(source: string): { doc: GherkinDocument; lang
49
53
  );
50
54
  }
51
55
 
56
+ /**
57
+ * Unified source-dialect policy (r41/r88): an explicit per-file
58
+ * `# language:` header wins; headerless content is auto-discovered through
59
+ * the official matcher chain (en start, zh-CN fallback); anything still
60
+ * undiscoverable falls back to en. Unknown header names are returned
61
+ * as-is — callers pick vocabulary via officialKeywordsOrEn, and the
62
+ * migration parse self-check fail-closes genuinely broken headers.
63
+ */
64
+ export function sourceDialect(source: string): string {
65
+ const firstLine = source.split('\n').find((l) => l.trim() !== '');
66
+ const header = firstLine?.match(/^#\s*language:\s*(\S+)\s*$/u)?.[1];
67
+ if (header !== undefined) return header;
68
+ try {
69
+ return parseFeatureSource(source).language;
70
+ } catch {
71
+ return 'en';
72
+ }
73
+ }
74
+
52
75
  const REQ_TAG_RE = /^@?req:(r\d+)$/u;
53
76
 
54
- /** Gherkin keyword (zh-CN + en) → step kind; And/But/* inherit via fallback. */
55
- function stepKeywordToKind(keyword: string): 'given' | 'when' | 'then' {
56
- const kw = keyword.trim();
57
- if (/^(假如|Given)/iu.test(kw)) return 'given';
58
- if (/^(当|When)/iu.test(kw)) return 'when';
59
- if (/^(那么|Then)/iu.test(kw)) return 'then';
60
- return 'given';
77
+ /** Gherkin keyword → step kind, from the official dialect tables; And/But/* inherit via fallback. */
78
+ function stepKeywordToKind(keyword: string): ScenarioStepKind {
79
+ return stepKeywordToOfficialKind(keyword) ?? 'given';
61
80
  }
81
+
62
82
  const HEADER_RE = /^#\s*(capability|purpose|scope):\s*(.*)$/u;
63
83
 
64
84
  function extractHeader(source: string): CapabilityHeader {
@@ -75,37 +95,47 @@ function extractHeader(source: string): CapabilityHeader {
75
95
  return header;
76
96
  }
77
97
 
78
- function classify(tags: string[]): {
79
- classification: ScenarioIR['classification'];
80
- errors: SpecStructuralError[];
81
- } {
82
- const errors: SpecStructuralError[] = [];
83
- const has = (t: string): boolean => tags.includes(t);
84
- const human = has('human');
85
- const executable = has('executable');
86
- const label = tags.join(',');
87
-
88
- if (has('manual')) {
89
- errors.push({
90
- code: 'tag:manual-removed',
91
- message: `@manual was removed in 0.3.0 — drop the tag (@human already carries the human-judgement semantics) (tags: ${label})`,
92
- });
93
- }
94
- if (human && executable) {
95
- errors.push({
96
- code: 'tag:mutually-exclusive',
97
- message: `@human 与 @executable 互斥(tags: ${label})`,
98
- });
99
- }
100
- const classification: ScenarioIR['classification'] = human
101
- ? 'human'
102
- : executable
103
- ? 'executable'
104
- : 'unclassified';
105
- return { classification, errors };
98
+ interface RawTags {
99
+ names: string[]; // without leading '@'
100
+ reqId: string;
101
+ }
102
+
103
+ function parseTags(tags: readonly { name: string }[]): RawTags {
104
+ const names = tags.map((t) => t.name.replace(/^@/u, ''));
105
+ const req = tags.map((t) => t.name.match(REQ_TAG_RE)?.[1]).find((v): v is string => !!v);
106
+ return { names, reqId: req ?? '' };
106
107
  }
107
108
 
108
- /** Parse one capability .feature source into the single-track IR. */
109
+ function collectScenario(sc: {
110
+ tags: readonly { name: string }[];
111
+ name: string;
112
+ steps: readonly { keyword: string; text: string }[];
113
+ }): ScenarioIR {
114
+ const { names, reqId: _ignored } = parseTags(sc.tags);
115
+ const skipTokens = new Set(['skip', 'experimental']);
116
+ const runnable = !names.some((n) => skipTokens.has(n));
117
+ const steps = sc.steps.map((s) => ({
118
+ kind: stepKeywordToKind(s.keyword),
119
+ text: s.text.trim(),
120
+ }));
121
+ return {
122
+ name: sc.name,
123
+ tags: names.filter(
124
+ (n) =>
125
+ !n.startsWith('req') &&
126
+ !skipTokens.has(n) &&
127
+ !n.startsWith('rule') &&
128
+ !n.startsWith('human') &&
129
+ !n.startsWith('executable'),
130
+ ),
131
+ runnable,
132
+ stepCount: steps.length,
133
+ steps,
134
+ statement: steps.map((s) => s.text).join('\n'),
135
+ };
136
+ }
137
+
138
+ /** Parse one capability .feature source into the native single-track IR. */
109
139
  export function parseCapability(source: string, fileName = '<inline>'): CapabilityDoc {
110
140
  const errors: SpecStructuralError[] = [];
111
141
  const header = extractHeader(source);
@@ -118,56 +148,38 @@ export function parseCapability(source: string, fileName = '<inline>'): Capabili
118
148
  const { doc, language } = parseFeatureSource(source);
119
149
  const feature = doc.feature;
120
150
  const featureName = feature?.name ?? '';
121
- const scenarios: ScenarioIR[] = [];
151
+ const rules: RuleIR[] = [];
152
+ const orphans: ScenarioIR[] = [];
122
153
 
123
154
  for (const child of feature?.children ?? []) {
124
155
  const rule = child.rule;
125
- if (rule && rule.children.some((c) => c.scenario)) {
126
- errors.push({
127
- code: 'rule:nested-scenario',
128
- message: `Rule 块内嵌场景被拒绝(rule: ${rule.name})`,
156
+ if (rule) {
157
+ const { names, reqId } = parseTags(rule.tags);
158
+ const description = (rule.description ?? '')
159
+ .split('\n')
160
+ .map((l) => l.trim())
161
+ .filter((l) => l !== '')
162
+ .join('\n');
163
+ const scenarios: ScenarioIR[] = [];
164
+ for (const rc of rule.children) {
165
+ if (rc.scenario) scenarios.push(collectScenario(rc.scenario));
166
+ }
167
+ rules.push({
168
+ reqId,
169
+ title: rule.name,
170
+ description,
171
+ scenarios,
172
+ tags: names.filter(
173
+ (n) => !n.startsWith('req') && !n.startsWith('rule') && !n.startsWith('human'),
174
+ ),
129
175
  });
130
176
  continue;
131
177
  }
132
178
  const scenario = child.scenario;
133
- if (!scenario) continue;
134
-
135
- const tags = scenario.tags.map((t) => t.name.replace(/^@/u, ''));
136
- const reqIds = scenario.tags
137
- .map((t) => t.name.match(REQ_TAG_RE)?.[1])
138
- .filter((v): v is string => v !== undefined);
139
-
140
- const kind = classify(tags);
141
- errors.push(...kind.errors);
142
-
143
- const description = (scenario.description ?? '')
144
- .split('\n')
145
- .map((l) => l.trim())
146
- .filter((l) => l !== '');
147
- const stepTexts = scenario.steps.map((s) => s.text.trim());
148
- const statement = [...description, ...stepTexts].join('\n');
149
- const steps = scenario.steps.map((s) => ({
150
- kind: stepKeywordToKind(s.keyword),
151
- text: s.text.trim(),
152
- }));
153
-
154
- if (kind.classification === 'human' && !MUST_WORD_RE.test(statement)) {
155
- errors.push({
156
- code: 'rule:missing-must-word',
157
- message: `@human 规则场景描述必须含 MUST/SHALL(scenario: ${scenario.name})`,
158
- });
179
+ if (scenario) {
180
+ orphans.push(collectScenario(scenario));
159
181
  }
160
-
161
- scenarios.push({
162
- name: scenario.name,
163
- tags,
164
- reqIds,
165
- classification: kind.classification,
166
- statement,
167
- stepCount: stepTexts.length,
168
- steps,
169
- });
170
182
  }
171
183
 
172
- return { fileName, header, featureName, language, scenarios, errors };
184
+ return { fileName, header, featureName, language, rules, orphans, errors };
173
185
  }