@sun-asterisk/sungen 3.2.16-beta.1 → 3.2.16-beta.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/dist/cli/commands/audit.d.ts.map +1 -1
  2. package/dist/cli/commands/audit.js +6 -0
  3. package/dist/cli/commands/audit.js.map +1 -1
  4. package/dist/cli/commands/delivery.d.ts.map +1 -1
  5. package/dist/cli/commands/delivery.js +67 -20
  6. package/dist/cli/commands/delivery.js.map +1 -1
  7. package/dist/exporters/feature-parser.d.ts +16 -1
  8. package/dist/exporters/feature-parser.d.ts.map +1 -1
  9. package/dist/exporters/feature-parser.js +21 -2
  10. package/dist/exporters/feature-parser.js.map +1 -1
  11. package/dist/exporters/matrix/build.d.ts +27 -2
  12. package/dist/exporters/matrix/build.d.ts.map +1 -1
  13. package/dist/exporters/matrix/build.js +277 -47
  14. package/dist/exporters/matrix/build.js.map +1 -1
  15. package/dist/exporters/matrix/export.d.ts +7 -4
  16. package/dist/exporters/matrix/export.d.ts.map +1 -1
  17. package/dist/exporters/matrix/export.js +22 -6
  18. package/dist/exporters/matrix/export.js.map +1 -1
  19. package/dist/exporters/matrix/gates.d.ts.map +1 -1
  20. package/dist/exporters/matrix/gates.js +112 -3
  21. package/dist/exporters/matrix/gates.js.map +1 -1
  22. package/dist/exporters/matrix/map-loader.d.ts.map +1 -1
  23. package/dist/exporters/matrix/map-loader.js +20 -0
  24. package/dist/exporters/matrix/map-loader.js.map +1 -1
  25. package/dist/exporters/matrix/render-csv.d.ts +3 -2
  26. package/dist/exporters/matrix/render-csv.d.ts.map +1 -1
  27. package/dist/exporters/matrix/render-csv.js +53 -30
  28. package/dist/exporters/matrix/render-csv.js.map +1 -1
  29. package/dist/exporters/matrix/render-xlsx.d.ts +32 -8
  30. package/dist/exporters/matrix/render-xlsx.d.ts.map +1 -1
  31. package/dist/exporters/matrix/render-xlsx.js +215 -83
  32. package/dist/exporters/matrix/render-xlsx.js.map +1 -1
  33. package/dist/exporters/matrix/types.d.ts +53 -7
  34. package/dist/exporters/matrix/types.d.ts.map +1 -1
  35. package/dist/exporters/matrix/types.js +2 -2
  36. package/dist/exporters/matrix/types.js.map +1 -1
  37. package/dist/exporters/matrix/wording.d.ts +61 -0
  38. package/dist/exporters/matrix/wording.d.ts.map +1 -0
  39. package/dist/exporters/matrix/wording.js +221 -0
  40. package/dist/exporters/matrix/wording.js.map +1 -0
  41. package/dist/exporters/scenario-merger.d.ts.map +1 -1
  42. package/dist/exporters/scenario-merger.js +2 -1
  43. package/dist/exporters/scenario-merger.js.map +1 -1
  44. package/dist/exporters/spec-parser.d.ts.map +1 -1
  45. package/dist/exporters/spec-parser.js +2 -1
  46. package/dist/exporters/spec-parser.js.map +1 -1
  47. package/dist/harness/audit.d.ts +6 -0
  48. package/dist/harness/audit.d.ts.map +1 -1
  49. package/dist/harness/audit.js +80 -15
  50. package/dist/harness/audit.js.map +1 -1
  51. package/dist/harness/blindspot.d.ts.map +1 -1
  52. package/dist/harness/blindspot.js +2 -1
  53. package/dist/harness/blindspot.js.map +1 -1
  54. package/dist/harness/capability-plan.d.ts.map +1 -1
  55. package/dist/harness/capability-plan.js +3 -2
  56. package/dist/harness/capability-plan.js.map +1 -1
  57. package/dist/harness/feedback.d.ts.map +1 -1
  58. package/dist/harness/feedback.js +3 -2
  59. package/dist/harness/feedback.js.map +1 -1
  60. package/dist/harness/flow-check.d.ts.map +1 -1
  61. package/dist/harness/flow-check.js +2 -1
  62. package/dist/harness/flow-check.js.map +1 -1
  63. package/dist/harness/flow-plan.d.ts.map +1 -1
  64. package/dist/harness/flow-plan.js +3 -2
  65. package/dist/harness/flow-plan.js.map +1 -1
  66. package/dist/harness/intent.d.ts.map +1 -1
  67. package/dist/harness/intent.js +2 -1
  68. package/dist/harness/intent.js.map +1 -1
  69. package/dist/harness/journey.d.ts.map +1 -1
  70. package/dist/harness/journey.js +3 -2
  71. package/dist/harness/journey.js.map +1 -1
  72. package/dist/harness/ledger.d.ts.map +1 -1
  73. package/dist/harness/ledger.js +3 -2
  74. package/dist/harness/ledger.js.map +1 -1
  75. package/dist/harness/manifest.d.ts.map +1 -1
  76. package/dist/harness/manifest.js +4 -3
  77. package/dist/harness/manifest.js.map +1 -1
  78. package/dist/harness/parse.d.ts.map +1 -1
  79. package/dist/harness/parse.js +16 -3
  80. package/dist/harness/parse.js.map +1 -1
  81. package/dist/harness/quality-gates.d.ts.map +1 -1
  82. package/dist/harness/quality-gates.js +2 -1
  83. package/dist/harness/quality-gates.js.map +1 -1
  84. package/dist/harness/read-text.d.ts +22 -0
  85. package/dist/harness/read-text.d.ts.map +1 -0
  86. package/dist/harness/read-text.js +64 -0
  87. package/dist/harness/read-text.js.map +1 -0
  88. package/dist/harness/script-check.d.ts.map +1 -1
  89. package/dist/harness/script-check.js +3 -2
  90. package/dist/harness/script-check.js.map +1 -1
  91. package/dist/harness/sensors.d.ts +13 -1
  92. package/dist/harness/sensors.d.ts.map +1 -1
  93. package/dist/harness/sensors.js +63 -20
  94. package/dist/harness/sensors.js.map +1 -1
  95. package/dist/harness/spec-coverage.d.ts +5 -0
  96. package/dist/harness/spec-coverage.d.ts.map +1 -1
  97. package/dist/harness/spec-coverage.js +17 -7
  98. package/dist/harness/spec-coverage.js.map +1 -1
  99. package/dist/harness/trace.d.ts.map +1 -1
  100. package/dist/harness/trace.js +4 -3
  101. package/dist/harness/trace.js.map +1 -1
  102. package/dist/harness/viewpoint-ledger.d.ts.map +1 -1
  103. package/dist/harness/viewpoint-ledger.js +2 -1
  104. package/dist/harness/viewpoint-ledger.js.map +1 -1
  105. package/dist/orchestrator/templates/ai-src/commands/create-test.md +9 -0
  106. package/dist/orchestrator/templates/ai-src/commands/delivery.md +103 -18
  107. package/dist/orchestrator/templates/ai-src/skills/sungen-delivery/SKILL.md +77 -11
  108. package/dist/orchestrator/templates/ai-src/skills/sungen-gherkin-syntax/SKILL.md +1 -0
  109. package/dist/orchestrator/templates/ai-src/skills/sungen-tc-generation/SKILL.md +22 -0
  110. package/package.json +4 -4
  111. package/src/cli/commands/audit.ts +5 -0
  112. package/src/cli/commands/delivery.ts +68 -22
  113. package/src/exporters/feature-parser.ts +21 -2
  114. package/src/exporters/matrix/build.ts +281 -43
  115. package/src/exporters/matrix/export.ts +31 -6
  116. package/src/exporters/matrix/gates.ts +119 -3
  117. package/src/exporters/matrix/map-loader.ts +22 -1
  118. package/src/exporters/matrix/render-csv.ts +53 -30
  119. package/src/exporters/matrix/render-xlsx.ts +216 -85
  120. package/src/exporters/matrix/types.ts +58 -8
  121. package/src/exporters/matrix/wording.ts +221 -0
  122. package/src/exporters/scenario-merger.ts +2 -1
  123. package/src/exporters/spec-parser.ts +2 -1
  124. package/src/harness/audit.ts +84 -16
  125. package/src/harness/blindspot.ts +2 -1
  126. package/src/harness/capability-plan.ts +3 -2
  127. package/src/harness/feedback.ts +3 -2
  128. package/src/harness/flow-check.ts +2 -1
  129. package/src/harness/flow-plan.ts +3 -2
  130. package/src/harness/intent.ts +2 -1
  131. package/src/harness/journey.ts +3 -2
  132. package/src/harness/ledger.ts +3 -2
  133. package/src/harness/manifest.ts +4 -3
  134. package/src/harness/parse.ts +17 -3
  135. package/src/harness/quality-gates.ts +2 -1
  136. package/src/harness/read-text.ts +28 -0
  137. package/src/harness/script-check.ts +3 -2
  138. package/src/harness/sensors.ts +55 -8
  139. package/src/harness/spec-coverage.ts +22 -7
  140. package/src/harness/trace.ts +4 -3
  141. package/src/harness/viewpoint-ledger.ts +2 -1
  142. package/src/orchestrator/templates/ai-src/commands/create-test.md +9 -0
  143. package/src/orchestrator/templates/ai-src/commands/delivery.md +103 -18
  144. package/src/orchestrator/templates/ai-src/skills/sungen-delivery/SKILL.md +77 -11
  145. package/src/orchestrator/templates/ai-src/skills/sungen-gherkin-syntax/SKILL.md +1 -0
  146. package/src/orchestrator/templates/ai-src/skills/sungen-tc-generation/SKILL.md +22 -0
@@ -7,6 +7,7 @@
7
7
  */
8
8
  import * as fs from 'fs';
9
9
  import { GherkinParser, ParsedScenario, ParsedStep } from '../generators/gherkin-parser';
10
+ import { readTextFile } from './read-text';
10
11
 
11
12
  export type Priority = 'high' | 'normal' | 'low' | 'unknown';
12
13
 
@@ -51,9 +52,22 @@ export function idPrefix(id: string): string {
51
52
 
52
53
  // ---------- test-viewpoint.md ----------
53
54
 
55
+ /**
56
+ * A viewpoint OVERVIEW id may be a bare category (`VP-LOGIC`, `VP-SEC`) — that is
57
+ * the scheme sungen itself prescribes, with the sequence number living on the
58
+ * scenario (`VP-LOGIC-013`). `isIdLike` demands a digit to avoid picking prose out
59
+ * of tables, which silently rejected every category id: a project declaring
60
+ * VP-LOGIC/VP-VAL/VP-SEC/VP-UI/VP-NAV/VP-PERF/VP-I18N parsed as ONE viewpoint
61
+ * (only VP-I18N, for the "18"), traceability collapsed to ~0%, and the audit told
62
+ * QA to re-tag scenarios that were already correct.
63
+ */
64
+ function isViewpointId(s: string): boolean {
65
+ return isIdLike(s) || /^VP-[A-Z][A-Z0-9]*$/i.test(s);
66
+ }
67
+
54
68
  export function parseViewpointOverview(filePath: string): ViewpointEntry[] {
55
69
  if (!fs.existsSync(filePath)) return [];
56
- const text = fs.readFileSync(filePath, 'utf-8');
70
+ const text = readTextFile(filePath);
57
71
  const lines = text.split('\n');
58
72
 
59
73
  const entries = new Map<string, ViewpointEntry>();
@@ -68,7 +82,7 @@ export function parseViewpointOverview(filePath: string): ViewpointEntry[] {
68
82
  const cells = line.split('|').map((c) => c.trim()).filter((_, i, a) => i > 0 && i < a.length - 1);
69
83
  if (cells.length >= 3) {
70
84
  const id = cells[0];
71
- if (isIdLike(id) && !/^-+$/.test(cells[1])) {
85
+ if (isViewpointId(id) && !/^-+$/.test(cells[1])) {
72
86
  const pr = /high/i.test(cells[1]) ? 'High' : /medium/i.test(cells[1]) ? 'Medium' : /low/i.test(cells[1]) ? 'Low' : 'Unknown';
73
87
  entries.set(id.toUpperCase(), { id: id.toUpperCase(), priority: pr as any, reason: cells[2] });
74
88
  }
@@ -85,7 +99,7 @@ export function parseViewpointOverview(filePath: string): ViewpointEntry[] {
85
99
  if (/^##\s/.test(line)) { group = undefined; }
86
100
  if (group) {
87
101
  const m = line.match(/^[-*+]\s+([A-Za-z][A-Za-z0-9.-]*)/);
88
- if (m && isIdLike(m[1])) {
102
+ if (m && isViewpointId(m[1])) {
89
103
  const id = m[1].toUpperCase();
90
104
  const existing = entries.get(id);
91
105
  if (existing) existing.group = group;
@@ -7,6 +7,7 @@ import * as fs from 'fs';
7
7
  import * as path from 'path';
8
8
  import { ScenarioInfo, loadScenarios, idPrefix } from './parse';
9
9
  import { parseManualComments } from '../exporters/scenario-merger';
10
+ import { readTextFile } from './read-text';
10
11
 
11
12
  // ---------- #2 Downstream-scope ----------
12
13
 
@@ -202,5 +203,5 @@ export function crossArtifactOwnership(screenDir: string, scenarios: ScenarioInf
202
203
 
203
204
  // convenience reader
204
205
  export function readText(p: string): string {
205
- return fs.existsSync(p) ? fs.readFileSync(p, 'utf-8') : '';
206
+ return fs.existsSync(p) ? readTextFile(p) : '';
206
207
  }
@@ -0,0 +1,28 @@
1
+ import * as fs from 'fs';
2
+
3
+ /**
4
+ * Read a user-authored text file with line endings NORMALISED to `\n`.
5
+ *
6
+ * Why this exists: the harness parses spec.md / test-viewpoint.md / .feature with
7
+ * line-anchored regexes such as `/^\s*Scenario:\s*(.+)$/`. In JavaScript `.` does
8
+ * NOT match `\r`, so on a CRLF file `(.+)$` can never reach the end of the line
9
+ * and the match silently fails. The parser then reports "nothing found" rather
10
+ * than an error, which is far worse than crashing:
11
+ *
12
+ * - `spec-coverage` found 0 requirements → the FR sensor scored a vacuous
13
+ * 100% and never emitted SPEC-UNCOVERED / SPEC-TRACE-IMPLICIT.
14
+ * - `parseViewpointOverview` found 1 of N viewpoints → traceability collapsed
15
+ * to ~0% and the audit told QA to "re-tag" scenarios that were already
16
+ * correct.
17
+ *
18
+ * A Windows-authored spec is completely normal, so every harness parser that
19
+ * reads a user file must go through here.
20
+ */
21
+ export function readTextFile(filePath: string): string {
22
+ return fs.readFileSync(filePath, 'utf-8').replace(/\r\n?/g, '\n');
23
+ }
24
+
25
+ /** `readTextFile` split into lines — the common case for line-anchored parsers. */
26
+ export function readTextLines(filePath: string): string[] {
27
+ return readTextFile(filePath).split('\n');
28
+ }
@@ -17,6 +17,7 @@ import * as path from 'path';
17
17
  import * as os from 'os';
18
18
  import { loadScenarios, ScenarioInfo } from './parse';
19
19
  import { featureBasename } from './unit-paths';
20
+ import { readTextFile } from './read-text';
20
21
 
21
22
  export interface ScriptCheckResult {
22
23
  screen: string;
@@ -153,7 +154,7 @@ export async function runScriptCheck(screenDir: string, screenName: string, kind
153
154
  let specTitles: string[] = [];
154
155
  let specSrc = '';
155
156
  if (committedSpec) {
156
- specSrc = fs.readFileSync(committedSpec, 'utf-8');
157
+ specSrc = readTextFile(committedSpec);
157
158
  specTitles = extractTestTitles(specSrc);
158
159
  } else {
159
160
  findings.push('No generated spec found under specs/generated/ — run `sungen generate` / `/sungen:run-test` first.');
@@ -195,7 +196,7 @@ export async function runScriptCheck(screenDir: string, screenName: string, kind
195
196
  const fresh = findSpec(tmp, screenName, kind);
196
197
  if (fresh) {
197
198
  const a = normalize(specSrc);
198
- const b = normalize(fs.readFileSync(fresh, 'utf-8'));
199
+ const b = normalize(readTextFile(fresh));
199
200
  if (a !== b) {
200
201
  drift = 'drift';
201
202
  // collect a few differing lines
@@ -10,6 +10,7 @@ import * as fs from 'fs';
10
10
  import * as path from 'path';
11
11
  import { parse as parseYaml } from 'yaml';
12
12
  import { ScenarioInfo, ViewpointEntry, idPrefix } from './parse';
13
+ import { readTextFile } from './read-text';
13
14
 
14
15
  // Business-critical category keywords (matched by CONTAINMENT against the VP category, so a
15
16
  // compound category like LIST-DISPLAY / ADD-TO-CART / PRODUCT-DISCOVERY classifies correctly).
@@ -56,7 +57,7 @@ export interface Catalog {
56
57
 
57
58
  export function loadCatalog(): Catalog {
58
59
  const p = path.join(__dirname, 'catalog', 'universal-viewpoints.yaml');
59
- return parseYaml(fs.readFileSync(p, 'utf-8')) as Catalog;
60
+ return parseYaml(readTextFile(p)) as Catalog;
60
61
  }
61
62
 
62
63
  // Word-aware keyword match: `\b<kw>(s|es|ed|ing)?\b`, so a keyword matches whole words (plus
@@ -76,6 +77,10 @@ const has = (haystacks: string[], kw: string) => {
76
77
 
77
78
  export interface GateResult {
78
79
  pageType: string | null;
80
+ /** How the page type was decided — a declaration is authoritative, a guess is not. */
81
+ pageTypeSource?: 'declared' | 'detected' | 'undetermined';
82
+ /** Keyword evidence behind a detected type (best hits vs the runner-up). */
83
+ pageTypeEvidence?: { hits: number; runnerUp: number };
79
84
  themesTotal: number;
80
85
  themesCovered: number; // deeply covered (has a data assertion)
81
86
  coverageRatio: number;
@@ -83,7 +88,18 @@ export interface GateResult {
83
88
  universalGaps: string[];
84
89
  }
85
90
 
86
- export function viewpointGate(scenarios: ScenarioInfo[], viewpoints: ViewpointEntry[], catalog: Catalog, isMobile = false): GateResult {
91
+ /**
92
+ * An explicit `page-type: <id>` line in test-viewpoint.md — the project stating what
93
+ * kind of screen this is. A declaration always beats keyword detection.
94
+ */
95
+ export function declaredPageType(viewpointText: string, catalog: Catalog): string | null {
96
+ const m = viewpointText.match(/^\s*(?:[-*]\s*)?\**page[- ]?type\**\s*[:=]\s*`?([a-z0-9-]+)`?/im);
97
+ if (!m) return null;
98
+ const id = m[1].toLowerCase();
99
+ return Object.keys(catalog.page_types).includes(id) ? id : null;
100
+ }
101
+
102
+ export function viewpointGate(scenarios: ScenarioInfo[], viewpoints: ViewpointEntry[], catalog: Catalog, isMobile = false, viewpointText = ''): GateResult {
87
103
  const haystacks = [
88
104
  ...scenarios.map((s) => s.haystack),
89
105
  ...viewpoints.map((v) => `${v.id} ${v.reason}`.toLowerCase()),
@@ -94,14 +110,35 @@ export function viewpointGate(scenarios: ScenarioInfo[], viewpoints: ViewpointEn
94
110
  // too, so a mobile commerce/form screen is selected as ecommerce-list/form by keyword fit —
95
111
  // `mobile-home` does not crowd out a better-fitting type. With no type matching → pageType stays
96
112
  // null → no themes required → the gate passes leniently rather than false-FAILing.
113
+ // Detection needs EVIDENCE, not a single coincidental word. A knowledge-base search
114
+ // screen was classified `ecommerce-list` on ONE keyword hit and then judged against
115
+ // cart / add-to-cart / brand-filter themes it can never have — 40% of the quality
116
+ // score decided by one word, in both directions (an earlier misread scored it `auth`
117
+ // and handed it 3/3). Below the threshold the page type is UNDETERMINED: no themes
118
+ // are demanded (the audit says so instead of inventing a gap), and the project can
119
+ // settle it for good by declaring `page-type:` in test-viewpoint.md.
120
+ const MIN_HITS = 2; // at least two distinct keywords
121
+ const MIN_MARGIN = 1; // and a clear lead over the runner-up
97
122
  let pageType: string | null = null;
123
+ let source: GateResult['pageTypeSource'] = 'undetermined';
98
124
  let best = 0;
99
- for (const [pt, def] of Object.entries(catalog.page_types)) {
100
- if (pt.startsWith('mobile-') && !isMobile) continue;
101
- const hits = def.detect_keywords.filter((k) => has(haystacks, k)).length;
102
- if (hits > best) { best = hits; pageType = pt; }
125
+ let runnerUp = 0;
126
+ const declared = declaredPageType(viewpointText, catalog);
127
+ if (declared) {
128
+ pageType = declared;
129
+ source = 'declared';
130
+ } else {
131
+ for (const [pt, def] of Object.entries(catalog.page_types)) {
132
+ if (pt.startsWith('mobile-') && !isMobile) continue;
133
+ const hits = def.detect_keywords.filter((k) => has(haystacks, k)).length;
134
+ if (hits > best) { runnerUp = best; best = hits; pageType = pt; }
135
+ else if (hits > runnerUp) { runnerUp = hits; }
136
+ }
137
+ if (pageType && best >= MIN_HITS && best - runnerUp >= MIN_MARGIN) source = 'detected';
138
+ else { pageType = null; source = 'undetermined'; }
103
139
  }
104
140
 
141
+ const evidence = { hits: best, runnerUp };
105
142
  const gaps: GateResult['gaps'] = [];
106
143
  let total = 0, covered = 0;
107
144
  if (pageType) {
@@ -127,8 +164,13 @@ export function viewpointGate(scenarios: ScenarioInfo[], viewpoints: ViewpointEn
127
164
 
128
165
  return {
129
166
  pageType,
167
+ pageTypeSource: source,
168
+ pageTypeEvidence: evidence,
130
169
  themesTotal: total,
131
170
  themesCovered: covered,
171
+ // No page type → no themes → the ratio is NOT a free 1.0: the score treats this
172
+ // axis as not-applicable and renormalises, so breadth can neither be faked nor
173
+ // punished by a guess.
132
174
  coverageRatio: total ? covered / total : 1,
133
175
  gaps,
134
176
  universalGaps,
@@ -376,8 +418,13 @@ export function coverageBalance(scenarios: ScenarioInfo[]): BalanceResult {
376
418
  byBucket[bucketForCategory(s.category)]++;
377
419
  }
378
420
 
379
- const core = byBucket['business-core'];
380
- const secondary = byBucket['presentation'] + byBucket['validation-security'];
421
+ // `behavior` (LOGIC / TRANSITION / WORKFLOW) IS business behaviour — it is the
422
+ // main category of sungen's own prescribed taxonomy. Counting only the commerce/API
423
+ // vocabulary (CART, PRODUCT, CHECKOUT, ENDPOINT…) as core meant a standard suite of
424
+ // 37 VP-LOGIC scenarios scored core=0 → a balance axis of 0% on a perfectly
425
+ // well-balanced suite. Navigation joins the secondary side, where it belongs.
426
+ const core = byBucket['business-core'] + byBucket['behavior'];
427
+ const secondary = byBucket['presentation'] + byBucket['validation-security'] + byBucket['navigation'];
381
428
  const imbalanced = secondary > core * 1.5 && core > 0;
382
429
  const unclassifiedRatio = scenarios.length ? byBucket['other'] / scenarios.length : 0;
383
430
  // A high `other` share means the VP taxonomy drifted from the catalog — the balance axis is then
@@ -10,6 +10,7 @@
10
10
  */
11
11
  import * as fs from 'fs';
12
12
  import { ScenarioInfo } from './parse';
13
+ import { readTextFile } from './read-text';
13
14
 
14
15
  export type Modality = 'MUST' | 'SHOULD' | 'MAY';
15
16
 
@@ -20,6 +21,11 @@ export interface SpecCoverageResult {
20
21
  frTotal: number;
21
22
  frCovered: number;
22
23
  uncoveredMust: { id: string; text: string }[];
24
+ /** Requirements counted as covered ONLY by keyword inference — no scenario
25
+ * cites the id. The audit accepts that (it asks "did you think about this
26
+ * requirement?"), but the link is not machine-checkable: delivery's
27
+ * requirement table and any later refactor cannot follow it. */
28
+ inferredOnly: string[];
23
29
  triggerGaps: TriggerGap[]; // per-constraint trigger matrix gaps
24
30
  verdict: 'pass' | 'warn' | 'fail';
25
31
  }
@@ -40,8 +46,12 @@ const ACTION_TRIGGER: { trigger: string; re: RegExp }[] = [
40
46
  ];
41
47
 
42
48
  function modalityOf(text: string): Modality {
43
- if (/\bMUST\b/.test(text)) return 'MUST';
44
- if (/\bSHOULD\b/.test(text)) return 'SHOULD';
49
+ // Specs are written in the project's language — a Vietnamese spec states
50
+ // obligation with PHẢI / KHÔNG ĐƯỢC and recommendation with NÊN. Reading only
51
+ // English keywords silently demoted every requirement to MAY, so the
52
+ // "uncovered MUST" finding could never fire on those projects.
53
+ if (/\bMUST\b/i.test(text) || /\bPHẢI\b/i.test(text) || /KHÔNG ĐƯỢC/i.test(text)) return 'MUST';
54
+ if (/\bSHOULD\b/i.test(text) || /\bNÊN\b/i.test(text)) return 'SHOULD';
45
55
  return 'MAY';
46
56
  }
47
57
 
@@ -54,7 +64,7 @@ function actionTriggersIn(text: string): string[] {
54
64
 
55
65
  export function parseSpecClauses(specPath: string): { frs: FrClause[]; valRows: ValRow[] } {
56
66
  if (!fs.existsSync(specPath)) return { frs: [], valRows: [] };
57
- const lines = fs.readFileSync(specPath, 'utf-8').split('\n');
67
+ const lines = readTextFile(specPath).split('\n');
58
68
 
59
69
  const frs: FrClause[] = [];
60
70
  for (const line of lines) {
@@ -97,12 +107,13 @@ function scenarioBlocks(featureText: string): string[] {
97
107
  export function specCoverage(specPath: string, scenarios: ScenarioInfo[], featureText: string): SpecCoverageResult {
98
108
  const { frs, valRows } = parseSpecClauses(specPath);
99
109
  if (!fs.existsSync(specPath) || (frs.length === 0 && valRows.length === 0)) {
100
- return { hasSpec: fs.existsSync(specPath), frTotal: 0, frCovered: 0, uncoveredMust: [], triggerGaps: [], verdict: 'pass' };
110
+ return { hasSpec: fs.existsSync(specPath), frTotal: 0, frCovered: 0, uncoveredMust: [], inferredOnly: [], triggerGaps: [], verdict: 'pass' };
101
111
  }
102
112
  const featLower = featureText.toLowerCase();
103
113
 
104
114
  // FR coverage: explicit @spec:FR / literal FR-id citation, else keyword fallback.
105
115
  const uncoveredMust: { id: string; text: string }[] = [];
116
+ const inferredOnly: string[] = [];
106
117
  let frCovered = 0;
107
118
  for (const fr of frs) {
108
119
  const idLower = fr.id.toLowerCase();
@@ -110,8 +121,12 @@ export function specCoverage(specPath: string, scenarios: ScenarioInfo[], featur
110
121
  const words = [...new Set((fr.text.toLowerCase().match(/[a-z][a-z-]{4,}/g) || []))]
111
122
  .filter((w) => !/must|should|system|screen|users?|value|input|field/.test(w));
112
123
  const kwHit = words.length > 0 && scenarios.some((s) => words.filter((w) => s.haystack.includes(w)).length >= Math.min(2, words.length));
113
- if (cited || kwHit) frCovered++;
114
- else if (fr.modality === 'MUST') uncoveredMust.push({ id: fr.id, text: fr.text.slice(0, 90) });
124
+ if (cited || kwHit) {
125
+ frCovered++;
126
+ if (!cited) inferredOnly.push(fr.id); // covered, but nothing links it explicitly
127
+ } else if (fr.modality === 'MUST') {
128
+ uncoveredMust.push({ id: fr.id, text: fr.text.slice(0, 90) });
129
+ }
115
130
  }
116
131
 
117
132
  // Per-constraint trigger coverage — the matrix-collapse catch.
@@ -135,5 +150,5 @@ export function specCoverage(specPath: string, scenarios: ScenarioInfo[], featur
135
150
  const verdict: SpecCoverageResult['verdict'] =
136
151
  uncoveredMust.length > 0 || triggerGaps.length > 0 ? 'fail' : 'pass';
137
152
 
138
- return { hasSpec: true, frTotal: frs.length, frCovered, uncoveredMust, triggerGaps, verdict };
153
+ return { hasSpec: true, frTotal: frs.length, frCovered, uncoveredMust, inferredOnly, triggerGaps, verdict };
139
154
  }
@@ -15,23 +15,24 @@ import * as fs from 'fs';
15
15
  import * as path from 'path';
16
16
  import { reportSlug } from './unit-paths';
17
17
  import { segmentRuns, latestRunEvents, LedgerEvent } from './ledger';
18
+ import { readTextFile } from './read-text';
18
19
 
19
20
  interface ManualItem { scenario: string; reason: string }
20
21
 
21
22
  function readJson(p: string): any | null {
22
- try { return fs.existsSync(p) ? JSON.parse(fs.readFileSync(p, 'utf-8')) : null; } catch { return null; }
23
+ try { return fs.existsSync(p) ? JSON.parse(readTextFile(p)) : null; } catch { return null; }
23
24
  }
24
25
 
25
26
  function readLedger(screen: string): any[] {
26
27
  const p = path.join(process.cwd(), '.sungen', 'ledger', `${reportSlug(screen)}.jsonl`);
27
28
  if (!fs.existsSync(p)) return [];
28
- return fs.readFileSync(p, 'utf-8').split('\n').filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
29
+ return readTextFile(p).split('\n').filter(Boolean).map((l) => { try { return JSON.parse(l); } catch { return null; } }).filter(Boolean);
29
30
  }
30
31
 
31
32
  /** Parse @manual scenarios + the explanatory comment line above each. */
32
33
  function parseManual(featurePath: string): ManualItem[] {
33
34
  if (!fs.existsSync(featurePath)) return [];
34
- const lines = fs.readFileSync(featurePath, 'utf-8').split('\n');
35
+ const lines = readTextFile(featurePath).split('\n');
35
36
  const out: ManualItem[] = [];
36
37
  for (let i = 0; i < lines.length; i++) {
37
38
  const m = lines[i].match(/^\s*Scenario:\s*(.+)$/);
@@ -10,6 +10,7 @@
10
10
  */
11
11
  import * as fs from 'fs';
12
12
  import { ScenarioInfo } from './parse';
13
+ import { readTextFile } from './read-text';
13
14
 
14
15
  export interface LedgerItem { id?: string; text: string; covered: boolean }
15
16
 
@@ -27,7 +28,7 @@ const GENERIC = new Set(['display', 'shown', 'value', 'field', 'input', 'page',
27
28
  /** Extract atomic checklist items from a viewpoint file (format-tolerant). */
28
29
  export function parseViewpointItems(viewpointPath: string): { id?: string; text: string }[] {
29
30
  if (!fs.existsSync(viewpointPath)) return [];
30
- const lines = fs.readFileSync(viewpointPath, 'utf-8').split('\n');
31
+ const lines = readTextFile(viewpointPath).split('\n');
31
32
  const items: { id?: string; text: string }[] = [];
32
33
  let inFence = false;
33
34
  for (const raw of lines) {
@@ -68,6 +68,15 @@ If the unit is **api-first** (`qa/api/<name>/` or `qa/api/flows/<name>/`), the d
68
68
  If the unit is **api-first** (`qa/api/<name>/` or `qa/api/flows/<name>/`), the design loop differs — **no visual capture, no selectors**; the contract is the named-endpoint catalog. **Follow the `sungen-api-design` + `sungen-api-coverage-model` skills end-to-end** instead of the screen/flow steps: `sungen context --area <name>` (discover endpoints + `fields:`) → **enumerate the Tier-1 case list per endpoint from the coverage model** (contract + not-found/required-matrix + auth + idempotency, expanded mechanically from `fields:`) → generate `@api`/`@cases`/flow/`@concurrent`/`@query` scenarios with **strict assertions** (prove the effect, never status-only) → **`sungen audit --area <name>` gate + reviewer + repair loop to businessDepth ≥ 0.7** → record + trace. Then recommend `/sungen-run-test <name>`. The capture / viewpoint-group / selector steps do **not** apply.
69
69
  {{/cap}}
70
70
 
71
+ ## Requirement traceability (before you finish)
72
+
73
+ Cross-check `requirements/spec.md` against the scenarios you wrote: every `FR-`/`TR-`/`NFR-` id
74
+ must either carry a `@spec:<id>` tag on the scenario that proves it, or be a conscious
75
+ out-of-scope decision you state in the summary. `sungen audit` reports `SPEC-TRACE-IMPLICIT` for
76
+ requirements it could only match by keyword — treat that list as a to-do: add the tag to the
77
+ proving scenario, do not leave the link to inference. Delivery's requirement table follows the
78
+ tag only, so an untagged requirement is later reported as an uncovered gap.
79
+
71
80
  ## Steps
72
81
 
73
82
  {{#cap parallel-subagents}}
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: delivery
3
3
  description: "Export the Test Case & Coverage Matrix (review/manual/customer deliverable) from Gherkin + Playwright results. --legacy exports the classic per-scenario CSV/XLSX."
4
- argument-hint: "[name...] [--env <locale>] [--legacy] (omit names for all; --env for locale-specific export)"
4
+ argument-hint: "[name...] [--format csv] [--env <locale>] [--legacy] (omit names for all)"
5
5
  order: 50
6
6
  claude-tools: "Bash, Read, Write, AskUserQuestion"
7
7
  copilot-tools: "[read, execute, edit, vscode/askQuestions]"
@@ -26,7 +26,14 @@ review → approve → official render.
26
26
  ## Parameters
27
27
 
28
28
  Parse from `$ARGUMENTS`:
29
- - **names** — zero or more screen/flow/api names. Empty → all targets.
29
+ - **names** — zero or more screen/flow/api names. Empty → **a sweep of every unit**.
30
+ Strictness follows the scope: a **named** unit that is not ready aborts the run (the user asked
31
+ for that unit); a **sweep** exports every ready unit and skips the rest with an accounted
32
+ "Not exported" list, succeeding as long as something was written. So `/sungen:delivery` with no
33
+ argument is safe in a work-in-progress project — scaffolded units (added but `create-test` not
34
+ run) no longer block the units that are ready.
35
+ - **`--format <xlsx|csv|both>`** — output format. **Default `xlsx` — one artifact only.** Pass
36
+ `--format csv` when the user wants the flat CSV (pipelines/diffing), `--format both` for both.
30
37
  - **`--env <locale>`** — sets `SUNGEN_ENV=<locale>` for the run (accept `--locale` as alias).
31
38
  - **`--legacy` / `--full`** — pass through to the CLI and skip the map flow entirely.
32
39
 
@@ -37,14 +44,19 @@ Parse from `$ARGUMENTS`:
37
44
  ```bash
38
45
  [ -x ./bin/sungen.js ] && ./bin/sungen.js delivery <names> || npx sungen delivery <names>
39
46
  ```
40
- (prepend `SUNGEN_ENV=<locale>` when `--env` was given; append `--legacy` when requested — then skip
41
- to step 5.)
47
+ (prepend `SUNGEN_ENV=<locale>` when `--env` was given; append `--format <fmt>` when the user asked
48
+ for a non-default format; append `--legacy` when requested — then skip to step 5.)
42
49
 
43
50
  Three outcomes per target:
44
51
  - **Rendered** → done, go to step 5.
45
52
  - **"no delivery map"** → go to step 2 (propose it).
46
53
  - **Gate findings / "not approved"** → go to step 4 (review & approve).
47
54
 
55
+ On a sweep, read the **"Not exported"** list at the end: units marked *not authored yet* need
56
+ `/sungen:create-test` first, the others each printed their own reason above (missing map, gate
57
+ errors, or awaiting approval). Handle them one unit at a time — do not re-run the sweep expecting
58
+ a different result.
59
+
48
60
  ### 2. Propose the Delivery Map (only when missing or the user asks to regroup)
49
61
 
50
62
  Read the unit's `.feature` (and `requirements/spec.md` for target naming). Write
@@ -58,27 +70,97 @@ groups:
58
70
  target: login.email # ONE target: field/component dot-path, flow phrase, or METHOD /path
59
71
  intent: <one behavior/rule this item verifies>
60
72
  oracle: <the shared observable Pass/Fail statement>
61
- category: normal | abnormal | security | nfr
73
+ dimensions: violated rule — required ×3 · format ×10 · full-width ×2 # see below
74
+ category: normal | abnormal | security | nfr # see the rule below
62
75
  review: proposed # ALWAYS proposed — only QA approval flips it
63
- variants: [VP-VAL-001-B, VP-VAL-001-S] # VP-ids; bare id on a @cases scenario = all its rows
76
+ variants: [VP-VAL-001-B, VP-VAL-001-S] # the scenario ids of THIS project (VP-…, SEC-123, MS-HP-001 —
77
+ # whatever the titles use); a bare id on a @cases scenario = all its rows
64
78
  dispositions: # scenarios intentionally NOT delivered as test cases
65
79
  VP-DATA-000: { as: excluded, reason: data-setup checklist }
66
80
  # as: excluded | blocked | covered_elsewhere | accepted_risk
67
81
  ```
68
82
 
69
- **Grouping rules (the aggregation signature):**
70
- - One group = **one target + one intent + one oracle family**. When unsure, keep items separate —
71
- the gates and QA decide, never guess-merge.
72
- - MAY share a group (become coverage dimensions): equivalence partitions, boundary values,
73
- different data (`@cases` rows), a different trigger with the same oracle (blur vs submit),
74
- locales.
75
- - MUST split: different target, intent, oracle family, category, execution mode (`@manual` vs
76
- auto), test layer (`@api`/`@query`), or priority tag; sequence-sensitive flows (re-Given/When
77
- after a Then) stay solo.
83
+ **Grouping rules (the aggregation signature) — group COMPACTLY.** The matrix exists to be
84
+ substantially shorter than the scenario list, so a reviewer can see missing viewpoints at a
85
+ glance. Merge whenever the cases share ALL of: target · test intent/business rule ·
86
+ precondition/condition · trigger or procedure shape · **the way the expected result is
87
+ determined** (its oracle *family*, not its exact message).
88
+
89
+ - **Oracle family = the determination method, parameterized.** All validation branches of ONE
90
+ field belong to ONE item — required, format, length, character-class are *expected branches*
91
+ (parameters) of "the field shows the validation message defined for the violated rule", shown
92
+ per variant, never separate items.
93
+ - MAY vary inside one item (coverage dimensions, visible on the sub-rows): data values, boundary
94
+ points, **account states** (a seeded/locked/deleted account next to a wrong-password case),
95
+ provider/browser/locale, `@cases` rows, a different trigger with the same oracle (blur vs
96
+ submit), **execution mode** (auto + manual mix — the parent shows `Auto n · Manual m`), and
97
+ **priority** (the item takes the highest; per-variant priorities stay visible).
98
+ - MUST split: different target, different intent/business rule, different way of determining the
99
+ expected result (a field-error family ≠ a session-established family), different test layer
100
+ (`@api`/`@query`), materially different precondition, or a different procedure shape —
101
+ sequence-sensitive flows (re-Given/When after a Then) stay solo. **Different risk classes never
102
+ merge**: XSS and SQL injection are separate items (different risk and determination), even on
103
+ the same field.
104
+ - **Look for these families before settling on a grouping** — they are where under-merging happens:
105
+ | Family | Merge into one item |
106
+ |---|---|
107
+ | Static render | every "element X is visible/has its content on load" scenario of the screen — title, instructions, progress step, buttons present, header/footer |
108
+ | Field validation branches | all rules of ONE field (required · format · length · character class · full-width) |
109
+ | Account/entity states | wrong-credentials · locked · deleted · unverified for the same rejection oracle |
110
+ | Provider / surface sets | the 3 OAuth providers, header+footer, the a11y surfaces of one behaviour |
111
+ | Lifetime / mode pairs | checked vs unchecked, mobile vs desktop, when the oracle is one rule with two branches |
112
+ On a 36-scenario screen, four separate one-variant "renders on load" items should have been one.
113
+ - When unsure, keep items separate — the gates and QA decide, never guess-merge.
78
114
  - Every scenario must land in exactly one group **or** one disposition (Gate B enforces 100%
79
115
  disposition). Data-setup blocks (`@manual:data-setup`) → `excluded`; SPEC-GAP placeholders →
80
116
  `blocked`.
81
117
 
118
+ **`dimensions:` — the compact coverage digest (required for items with >3 variants).**
119
+ This one short line is what the collapsed parent row shows instead of listing every variant, so a
120
+ reviewer sees *which dimensions* the item covers without expanding it. Name the dimension, then the
121
+ branches with counts:
122
+ - `violated rule — required ×3 · format ×10 · full-width ×2`
123
+ - `account state — wrong password · unregistered · locked · soft-deleted`
124
+ - `submission method — Login button · Enter in Password · Enter in Email`
125
+
126
+ Keep it ≤120 chars (Gate W warns). **YAML caveat:** a bare `: ` inside the value breaks the parse —
127
+ use ` — ` as the label separator (as above) or quote the whole string.
128
+
129
+ **`category` is not free choice for two classes (Gate K checks it):** a group whose variants are
130
+ `VP-SEC-*` MUST be `category: security`, and `VP-NFR-*` MUST be `nfr` — otherwise the Coverage
131
+ sheet's security/nfr column renders empty and the grid reports a gap the unit does not have while
132
+ hiding the work it does have. `normal` vs `abnormal` stays your judgement.
133
+
134
+ **Wording rules for `intent`/`oracle` (customer-facing — Gate W lints these):**
135
+ - Plain product language, present simple, ~10–20 words, one behavior:
136
+ "A user can sign in with valid credentials and is redirected to the Jobs page."
137
+ - Oracle = the observable outcome as a definite assertion ("The Jobs page is displayed and the
138
+ Logout link is visible.") — no `should`, no tester actions.
139
+ - NEVER: `{{tokens}}`, `[Selector]` references, DSL phrasing (`User fill/click/see`), generator
140
+ labels (`Setup:`/`Observable:`/`Oracle:`), or vague verbs (`handles`, `surfaces`) when a precise
141
+ behavior exists. Use the visible UI label (the Login button, the Email field).
142
+ - **Preserve the source meaning exactly** — never strengthen, weaken, or reinterpret an oracle
143
+ (a security assertion especially: if the source says "the password appears ONLY in the HTTPS
144
+ POST body", do not write "no plaintext password on the network").
145
+
146
+ **Requirement coverage (`requirements:` section, optional):** `sungen delivery` scans
147
+ `requirements/spec.md` for FR-/TR-/NFR- ids; ids traced by `@spec:` tags are `covered`, the rest
148
+ are `gap` (Gate R warning). Record the reviewed status for genuine non-gaps:
149
+
150
+ ```yaml
151
+ requirements:
152
+ TR-007: { status: planned, note: Performance needs Lighthouse-style tooling }
153
+ TR-004: { status: partially_covered, note: client-side covered by VP-SEC-003; hashing needs DB verify }
154
+ # status: covered | partially_covered | covered_elsewhere | planned | gap | not_applicable
155
+ ```
156
+
157
+ **Never write `status: covered` for a requirement no variant traces to** (Gate R flags it). A note
158
+ saying "proven by DI-SEC-CSRF" is prose — nothing detects it when that scenario later changes. If a
159
+ scenario in THIS feature proves the requirement, **add `@spec:<id>` to that scenario** so the trace
160
+ is real, then drop the override (it derives as `covered` on its own). Use `covered_elsewhere` only
161
+ when another suite proves it, and name that suite; `not_applicable` when the spec itself excludes
162
+ the requirement.
163
+
82
164
  Then validate and fix any ERROR findings:
83
165
 
84
166
  ```bash
@@ -115,6 +197,7 @@ counted in variants, never items**). Then `AskUserQuestion`:
115
197
  - **Open the workbook** — inspect `qa/deliverables/<unit>-testcases.xlsx` (Testcases sheet:
116
198
  collapse outline level 1 for the customer view; Coverage sheet: target × category grid + gaps).
117
199
  - **Run tests to refresh results** — `/sungen:run-test <unit>`, then re-run delivery.
200
+ - **Also export CSV** — `sungen delivery <unit> --format csv` (flat `item`/`variant` rows for pipelines).
118
201
  - **Export the legacy workbook too** — `sungen delivery <unit> --legacy`.
119
202
  - **Done**
120
203
 
@@ -132,7 +215,8 @@ counted in variants, never items**). Then `AskUserQuestion`:
132
215
  ## CLI reference
133
216
 
134
217
  ```
135
- sungen delivery [names...] # matrix (default; needs the map)
218
+ sungen delivery [names...] # matrix (default; needs the map) → XLSX only
219
+ --format <xlsx|csv|both> # output format; default xlsx (one artifact)
136
220
  --check # gates only — validate the map, write nothing
137
221
  --approve [DI-a,DI-b] # flip proposed→approved (+ stamp fingerprints); all groups when bare
138
222
  --preview # render despite review findings (DRAFT watermark)
@@ -140,5 +224,6 @@ sungen delivery [names...] # matrix (default; needs the map)
140
224
  --skip-preflight | --continue-on-missing | --env <env> # as before
141
225
  ```
142
226
 
143
- Outputs: `qa/deliverables/<unit>-testcases.xlsx` (Testcases + Coverage sheets) + `.csv`
144
- (flat, `Level` column `item|variant`). Legacy mode writes the classic files instead.
227
+ Outputs: `qa/deliverables/<unit>-testcases.xlsx` (Testcases + Coverage sheets) by default;
228
+ `--format csv` writes `<unit>-testcases.csv` instead (flat, `Level` column `item|variant`),
229
+ `--format both` writes both. Legacy mode writes the classic files instead.