@cspeach/cli 1.1.14 → 1.1.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,802 @@
1
+ /**
2
+ * An ATC result export, read in as a measurement (D8).
3
+ *
4
+ * For a team with no connection from CSPeach to SAP: somebody exports the ATC
5
+ * result list from the system, and this turns it into the same measurement
6
+ * record a live run writes — marked `basis: 'file'`, carrying the file's name
7
+ * and checksum, and always shown as "measured from a file". CSPeach cannot see
8
+ * how the file was made, so it never claims more than "this file said so".
9
+ *
10
+ * ONE rule decides everything below: a file proves "clean" only for a row
11
+ * CSPeach understood COMPLETELY. A column it does not know, a line cut short, a
12
+ * finding it cannot tie to an object, a type this team project does not have —
13
+ * each of those ends as "not measured", is counted, and is named. Doubt spreads
14
+ * to the object: if a row that could not be read may belong to ZCL_FI_TAX, then
15
+ * nothing in this file may write ZCL_FI_TAX clean, not even the person saying
16
+ * the export covers the whole scope.
17
+ *
18
+ * The file is untrusted input. It is read once, as text, from the one path a
19
+ * person typed: no path inside it is ever followed, nothing in it is executed,
20
+ * and every piece of its text that reaches the terminal goes through `shown()`.
21
+ *
22
+ * Program code only. No model tool reaches this file.
23
+ */
24
+ import { createHash } from 'node:crypto';
25
+ import fs from 'node:fs';
26
+ import path from 'node:path';
27
+ import { objectKey } from '@cspeach/register-core';
28
+ import { findingsToRows, scopeLabel, scopeObjects, } from './measure.js';
29
+ import { writeRecord } from './records.js';
30
+ import { RegisterError } from './store.js';
31
+ export class AtcFileError extends RegisterError {
32
+ }
33
+ /*
34
+ * The three tables below decide what a column means, and with it what "clean"
35
+ * can mean in this file.
36
+ *
37
+ * CONFIRMED on 2026-09-20 against `sap-client/test/fixtures/atc-export-real/
38
+ * atc-findings-en-unconverted.txt` — an English logon, SAP GUI's classic
39
+ * "unconverted" list, run series over package ZTEST_LAS with an S/4HANA
40
+ * readiness variant, default layout, 26 columns:
41
+ *
42
+ * read Priority · Check Title · Check Message · Object name · Obj. ·
43
+ * Note · RObT · Ref. Obj. · Count · Msg. Code
44
+ * finding Exemption State · Short text · Component · Category (twice) ·
45
+ * Text · Marked Bsl · App. Comp. · Identity · Check Class
46
+ * neutral Contact · Package · 1st Found · Obj. Resp. · Changed by · Changed On
47
+ *
48
+ * Why each of the judged ones landed where it did:
49
+ * · `Obj.` is the object TYPE (`PROG`), not its name — the name has a column
50
+ * of its own. `Obj. Resp.` is a person, so it is neutral.
51
+ * · `Msg. Code` is the message id (`SELECT`, `LOOP_WRITE`) — the half of a
52
+ * rule's name that a live run calls `messageId`. `Check Title` is the other
53
+ * half. Together they give a file finding the SAME rule id a live finding
54
+ * gets, which is the whole point of reading the file.
55
+ * · `Check Message` is the message title with no instance text in it.
56
+ * · `RObT` / `Ref. Obj.` name the object the finding POINTS AT (`TABL BSEG`),
57
+ * not the object that was checked. Read for the display message only.
58
+ * · `Short text` is the cited SAP Note's title, not the finding's long text,
59
+ * so it is not read into `messageText`.
60
+ * · `Category` holds `I` / `T` beside a `Text` column reading `Technical`:
61
+ * both describe the finding, so both are finding-side. Neither is read —
62
+ * `Check Title` is what names a rule — which is why two columns called
63
+ * `Category` can stand side by side without CSPeach having to guess.
64
+ * · `Identity` is the finding's own number (`671,599,554-`), `Check Class`
65
+ * the check's implementing class, `Marked Bsl` whether it sits in an ATC
66
+ * baseline: all three belong to a finding, none is read.
67
+ * · `Component` and `App. Comp.` are the cited note's component, which only a
68
+ * finding has. The NEVER-neutral list from Task 7 stands: `Text`, `Status`,
69
+ * `Note` and their kin are finding-side, whatever else they could mean.
70
+ *
71
+ * STILL JUDGED, with no capture behind them: every German title, every other
72
+ * layout ("all columns"), the spreadsheet export of the findings grid, and an
73
+ * Excel round-trip. A column nobody has seen is an unknown column, and an
74
+ * unknown column means this file proves no clean row — which is the safe way
75
+ * round and is where those stay until somebody exports one.
76
+ *
77
+ * A title is compared WHOLE, after `titleKey`: lower case, diacritics folded,
78
+ * and the separators between whole words turned into single spaces. Nothing is
79
+ * deleted from inside a word, so `D-a-t-u-m` is not `Datum`.
80
+ */
81
+ /** Columns CSPeach reads. First match wins, in the order written here. */
82
+ const HEADERS = {
83
+ type: ['object type', 'objecttype', 'obj type', 'objtype', 'objtyp', 'type', 'objekttyp', 'objekt typ', 'obj'],
84
+ name: ['object name', 'objectname', 'obj name', 'objname', 'object', 'name', 'objektname', 'objekt name'],
85
+ checkId: ['check id', 'checkid', 'check', 'test id', 'testid', 'test', 'prufung', 'prufungen'],
86
+ messageId: ['message id', 'messageid', 'msg id', 'msgid', 'message code', 'messagecode', 'code',
87
+ 'message number', 'messagenumber', 'meldungsid', 'meldungs id', 'meldungsnummer', 'msg code', 'msgcode'],
88
+ title: ['message title', 'messagetitle', 'meldungstitel', 'meldung', 'prufmeldung',
89
+ 'check message', 'checkmessage', 'message', 'message text', 'messagetext', 'meldungstext',
90
+ 'short text', 'shorttext', 'kurztext', 'description', 'beschreibung', 'title'],
91
+ text: ['message long text', 'messagelongtext', 'long text', 'longtext', 'langtext',
92
+ 'message text', 'messagetext', 'meldungstext'],
93
+ category: ['check title', 'checktitle', 'pruftitel', 'category', 'kategorie', 'check category', 'checkcategory'],
94
+ priority: ['priority', 'prio', 'prioritat', 'prioritaet', 'prioritt', 'severity'],
95
+ line: ['line', 'lineno', 'line no', 'line number', 'linenumber', 'source line', 'sourceline', 'row', 'zeile', 'zeilennummer'],
96
+ // These four are read last, so a column an older table already claims keeps it.
97
+ refType: ['robt', 'ref obj type', 'refobjtype', 'referenced object type', 'referencedobjecttype'],
98
+ refObject: ['ref obj', 'refobj', 'ref obj name', 'refobjname', 'referenced object name', 'referencedobjectname', 'referenced object'],
99
+ note: ['note', 'hinweis'],
100
+ count: ['count', 'anzahl'],
101
+ };
102
+ /**
103
+ * Columns CSPeach knows to be part of a FINDING but does not read. It never
104
+ * takes their text into a record; a cell with anything in one of them is why a
105
+ * row is a finding and not a clean object.
106
+ */
107
+ export const FINDING_HEADERS = [
108
+ // what inside the object the finding sits in, and what it points at
109
+ 'sub object type', 'subobjecttype', 'sub object name', 'subobjectname', 'sub object', 'subobject',
110
+ 'teilobjekttyp', 'teilobjektname', 'teilobjekt',
111
+ 'referenced object type', 'referencedobjecttype', 'referenced object name', 'referencedobjectname',
112
+ 'referenced object', 'ref obj type', 'refobjtype', 'ref obj name', 'refobjname', 'referenziertes objekt',
113
+ 'include', 'program', 'programm', 'column', 'spalte',
114
+ // what SAP says about it
115
+ 'sap note', 'sapnote', 'note', 'hinweis', 'sap hinweis',
116
+ 'additional information', 'additionalinformation', 'zusatzinformation', 'zusatzinformationen',
117
+ 'quality of finding', 'qualityoffinding', 'availability', 'verfugbarkeit',
118
+ 'quick fix', 'quickfix', 'quick fixes', 'quickfixes',
119
+ // how it was dealt with
120
+ 'exemption', 'exemption status', 'exemptionstatus', 'exemption state', 'exemption id', 'ausnahme', 'ausnahmestatus',
121
+ 'state', 'status', 'result', 'ergebnis',
122
+ 'text', 'info', 'information', 'comment', 'kommentar', 'bemerkung',
123
+ // Confirmed by the 2026-09-20 export: each of these is a property of a
124
+ // FINDING, so a cell under one is why a row is a finding and not clean.
125
+ 'short text', 'shorttext', 'kurztext', 'component', 'komponente', 'app comp', 'category', 'kategorie',
126
+ 'marked bsl', 'marked baseline', 'baseline', 'identity', 'check class', 'checkclass', 'prufklasse',
127
+ ];
128
+ const FINDING_ONLY = new Set(FINDING_HEADERS);
129
+ /**
130
+ * Columns that say something about WHERE a row came from, never what was found
131
+ * in it. A row is "checked and clean" only when every cell outside the object
132
+ * type, the object name and this list is empty — so a column nobody put on one
133
+ * of these three lists can never be mistaken for an empty one.
134
+ */
135
+ export const NEUTRAL_HEADERS = [
136
+ // where the object lives
137
+ 'package', 'devclass', 'development class', 'paket', 'paketname',
138
+ 'software component', 'softwarecomponent', 'softwarekomponente',
139
+ 'application component', 'applicationcomponent', 'anwendungskomponente',
140
+ 'namespace', 'namensraum', 'transport request', 'transportrequest', 'transport', 'transportauftrag',
141
+ // whose it is
142
+ 'author', 'autor', 'owner', 'responsible', 'person responsible', 'personresponsible', 'verantwortlicher',
143
+ 'obj resp', 'object responsible', 'objectresponsible',
144
+ 'contact', 'contact person', 'contactperson', 'ansprechpartner',
145
+ 'user', 'user name', 'username', 'benutzer',
146
+ // when
147
+ 'created on', 'createdon', 'created by', 'createdby', 'angelegt am', 'angelegt von', 'erstellt am',
148
+ 'changed on', 'changedon', 'changed by', 'changedby', 'geandert am', 'geandert von', 'letzter anderer',
149
+ 'last changed on', 'lastchangedon', 'last changed by', 'lastchangedby',
150
+ 'detected on', 'detectedon', 'date', 'datum', 'time', 'uhrzeit', 'timestamp', 'zeitstempel',
151
+ // When the run series first reported this object. A date, never a finding.
152
+ '1st found', 'first found', 'firstfound', 'erstmals gefunden',
153
+ // which system
154
+ 'system', 'sid', 'client', 'mandant', 'release',
155
+ // which run
156
+ 'run id', 'runid', 'run series', 'runseries', 'series', 'series id', 'seriesid',
157
+ 'worklist', 'worklist id', 'worklistid', 'result id', 'resultid', 'display id', 'displayid',
158
+ 'variant', 'check variant', 'checkvariant', 'variante', 'prufvariante',
159
+ ];
160
+ const NEUTRAL = new Set(NEUTRAL_HEADERS);
161
+ /**
162
+ * A plain row counter is neutral only while every cell under it is digits. The
163
+ * same short titles are what a person writes over a column of their own notes.
164
+ */
165
+ export const COUNTER_HEADERS = ['no', 'nr', 'row no', 'rowno', 'row number', 'rownumber', 'lfd nr', 'lfdnr', 'laufende nummer'];
166
+ const COUNTER = new Set(COUNTER_HEADERS);
167
+ /** The cells that make a row a FINDING. A row with none of them names an object the export checked. */
168
+ const FINDING_FIELDS = ['checkId', 'messageId', 'category', 'title', 'text', 'priority', 'line',
169
+ 'refType', 'refObject', 'note', 'count', 'other'];
170
+ const PRIORITY_WORDS = { error: 1, warning: 2, info: 3, information: 3 };
171
+ /**
172
+ * How many findings one row stands for. SAP's result list carries a `Count`,
173
+ * and a row that says 5 is five findings in that object, not one.
174
+ *
175
+ * Anything that is not a whole number above zero is ONE finding, silently: a
176
+ * count CSPeach cannot read is not a reason to drop a finding, and a row can
177
+ * never be worth none. The cap is there so a file claiming a billion cannot
178
+ * fill the laptop's memory; no real result row comes near it, and erring high
179
+ * is the safe direction — a finding too many never makes an object look clean.
180
+ */
181
+ const MAX_COUNT = 1000;
182
+ const THOUSANDS = /^\d{1,3}(?:[.,  ]\d{3})+$/;
183
+ export function countOf(cell) {
184
+ const t = cell.trim();
185
+ if (!t)
186
+ return undefined;
187
+ const digits = THOUSANDS.test(t) ? t.replace(/[.,  ]/g, '') : t;
188
+ const n = /^\d+$/.test(digits) ? Number.parseInt(digits, 10) : 1;
189
+ return Math.min(Math.max(n, 1), MAX_COUNT);
190
+ }
191
+ /**
192
+ * What a person reads under a finding: the message, and then what it points at.
193
+ * `DB Operation JOIN found (TABL BSEG, note 2431747)`.
194
+ *
195
+ * DISPLAY ONLY. It never names a rule: the referenced object changes from one
196
+ * finding of a rule to the next, and an id built from it would split one rule
197
+ * into as many rules as it has instances — the trap `messageTitle` is kept out
198
+ * of an id for (README answer 5, F1).
199
+ */
200
+ export function fileMessage(row) {
201
+ const base = row.title ?? '';
202
+ const ref = [row.refType, row.refObject].filter(Boolean).join(' ');
203
+ const tail = [ref, row.note ? `note ${row.note}` : ''].filter(Boolean).join(', ');
204
+ if (!tail)
205
+ return base;
206
+ return base ? `${base} (${tail})` : `(${tail})`;
207
+ }
208
+ /**
209
+ * Characters that are in the text and not on the screen: a zero-width space in
210
+ * an object name reads as another object, and the eye cannot tell them apart.
211
+ * They come off type and name before anything is matched, and out of every
212
+ * string that is printed.
213
+ *
214
+ * Each one is written as an escape, and so is the combining-mark range below:
215
+ * a character class whose payload is the characters themselves is a class no
216
+ * reviewer can read — the line shows a dash, two gaps and a trust exercise.
217
+ * Both are exported so the set can be pinned code point by code point
218
+ * (`atc-file.test.ts`), which is the only way to prove that rewriting a line
219
+ * nobody can see changed nothing.
220
+ */
221
+ export const ZERO_WIDTH = /[\u00AD\u200B-\u200F\u202A-\u202E\u2060-\u2064\uFEFF]/g;
222
+ export const INVISIBLE = /[\u0000-\u001F\u007F-\u009F\u00AD\u200B-\u200F\u202A-\u202E\u2060-\u2064\uFEFF]/g;
223
+ /** A header title, whole: lower case, diacritics folded, separators between whole words made single spaces. */
224
+ export function titleKey(header) {
225
+ const folded = String(header).replace(INVISIBLE, '')
226
+ .normalize('NFD').replace(/[\u0300-\u036F]/g, '')
227
+ .replace(/ß/g, 'ss').replace(/Ø/gi, 'o')
228
+ .toLowerCase().trim().replace(/\s+/g, ' ');
229
+ const words = folded.split(/[\s_\-./]+/).filter(Boolean);
230
+ if (!words.length)
231
+ return '';
232
+ // One letter on its own is not a word: `D-a-t-u-m` must never become `Datum`.
233
+ if (words.some((w) => w.length < 2))
234
+ return folded;
235
+ return words.join(' ');
236
+ }
237
+ /** A sheet is not a file this reads. Said as what to do, not as what is wrong. */
238
+ const EXCEL = 'This is an Excel workbook. Open it, save the sheet as CSV, and run this again with the CSV file.';
239
+ const NOT_UTF8 = 'This file is not UTF-8 text. In Excel choose "CSV UTF-8" when you save it.';
240
+ /** A result list is small. Anything this size is a download that went wrong, and reading it would hang the terminal. */
241
+ const MAX_BYTES = 50 * 1024 * 1024;
242
+ /** How far into the file the delimiter and the encoding are judged from. */
243
+ const SNIFF = 4096;
244
+ /** A cell inside quotation marks that runs on for this many lines is a quote nobody closed. */
245
+ const MAX_QUOTED_LINES = 20;
246
+ /** How many lines at the top may be a title above the header. */
247
+ const HEADER_LINES = 5;
248
+ /** Off an object type or name before anything is matched: what cannot be seen must not decide anything. */
249
+ const clean = (cell) => cell.replace(INVISIBLE, '').trim();
250
+ const quoteNeverClosed = (line) => `A quotation mark in line ${line} is never closed, so CSPeach cannot read this file safely.`;
251
+ /**
252
+ * Text that came out of the file, made safe to print. A customer's export can
253
+ * carry an escape sequence that repaints the terminal, a NUL, or a megabyte of
254
+ * binary on one line — none of which may reach a person through CSPeach.
255
+ */
256
+ export function shown(text) {
257
+ const flat = String(text)
258
+ // An OSC sequence runs to BEL or ESC-backslash and can retitle the window.
259
+ .replace(/\x1b\][^\x07\x1b]*(\x07|\x1b\\)?/g, ' ')
260
+ // A CSI sequence repaints, clears or moves the cursor.
261
+ .replace(/\x1b\[[0-9;?]*[ -/]*[@-~]?/g, ' ')
262
+ .replace(/\x1b[@-_]?/g, ' ')
263
+ // Zero-width and bidi characters: invisible on screen, and one of them can
264
+ // turn a printed name back to front. They go, leaving no gap.
265
+ .replace(ZERO_WIDTH, '')
266
+ // A control character stood between two words: it leaves a space behind, so
267
+ // the words do not run into one.
268
+ .replace(/[\x00-\x1f\x7f-\x9f]+/g, ' ')
269
+ .replace(/\s+/g, ' ')
270
+ .trim();
271
+ return flat.length > 60 ? `${flat.slice(0, 60)}…` : flat;
272
+ }
273
+ /** Five names and the count of the rest: a list nobody can read is a list nobody acts on. */
274
+ export const upToFive = (items) => (items.length > 5 ? `${items.slice(0, 5).join(', ')} and ${items.length - 5} more` : items.join(', '));
275
+ function splitCsv(text, delimiter) {
276
+ const rows = [];
277
+ let cells = [];
278
+ let cell = '';
279
+ let quoted = false;
280
+ let quoteAt = 0;
281
+ let atLine = 1; // where the scanner is
282
+ let rowAt = 1; // where the current row started
283
+ for (let i = 0; i < text.length; i += 1) {
284
+ const c = text[i];
285
+ if (quoted) {
286
+ if (c === '\n') {
287
+ atLine += 1;
288
+ if (atLine - quoteAt > MAX_QUOTED_LINES)
289
+ throw new AtcFileError(quoteNeverClosed(quoteAt));
290
+ cell += c;
291
+ }
292
+ else if (c !== '"')
293
+ cell += c;
294
+ else if (text[i + 1] === '"') {
295
+ cell += '"';
296
+ i += 1;
297
+ }
298
+ else
299
+ quoted = false;
300
+ continue;
301
+ }
302
+ if (c === '"' && cell === '') {
303
+ quoted = true;
304
+ quoteAt = atLine;
305
+ continue;
306
+ }
307
+ if (c === delimiter) {
308
+ cells.push(cell);
309
+ cell = '';
310
+ continue;
311
+ }
312
+ if (c === '\n' || c === '\r') {
313
+ if (c === '\r' && text[i + 1] === '\n')
314
+ i += 1;
315
+ atLine += 1;
316
+ cells.push(cell);
317
+ cell = '';
318
+ rows.push({ cells, line: rowAt });
319
+ cells = [];
320
+ rowAt = atLine;
321
+ continue;
322
+ }
323
+ cell += c;
324
+ }
325
+ // A quotation mark nobody closed swallows every row after it. Silently
326
+ // reading three rows as one is how a file comes to prove something it does
327
+ // not say, so this is refused rather than guessed at.
328
+ if (quoted)
329
+ throw new AtcFileError(quoteNeverClosed(quoteAt));
330
+ if (cell !== '' || cells.length) {
331
+ cells.push(cell);
332
+ rows.push({ cells, line: rowAt });
333
+ }
334
+ return rows.filter((r) => r.cells.some((v) => v.trim() !== ''));
335
+ }
336
+ /* ── SAP GUI's classic "unconverted" list ───────────────────────────────────
337
+ *
338
+ * Confirmed 2026-09-20. Not a CSV at all: ASCII, CRLF, a rule of `-` above the
339
+ * header and below it, every data line fenced with `|`, every cell padded with
340
+ * spaces to its column's width and numbers pushed to the right. A closing rule
341
+ * may end the list.
342
+ *
343
+ * A `|` INSIDE a cell cannot be escaped in this format, so nothing here tries
344
+ * to guess one back together: a row that splits into a different number of
345
+ * cells than the header is a row of the wrong length, and the short/long rule
346
+ * below counts it, names its line and never calls it clean.
347
+ */
348
+ /** A line that draws a rule: dashes or equals signs, and nothing that carries text. */
349
+ const RULE_LINE = /^[-=+|\s]*[-=][-=+|\s]*$/;
350
+ /** Three bars fence two cells: fewer than that is a sentence with a bar in it. */
351
+ const MIN_BARS = 3;
352
+ /** How far in the unconverted shape is judged from. */
353
+ const UNCONVERTED_LINES = 20;
354
+ const isFenced = (line) => line.length > 2 && line.startsWith('|') && line.endsWith('|')
355
+ && line.split('|').length - 1 >= MIN_BARS;
356
+ function looksUnconverted(text) {
357
+ let fenced = 0;
358
+ let rules = 0;
359
+ for (const raw of text.split('\n').slice(0, UNCONVERTED_LINES)) {
360
+ const line = raw.replace(/\r$/, '').trim();
361
+ if (!line)
362
+ continue;
363
+ if (RULE_LINE.test(line)) {
364
+ rules += 1;
365
+ continue;
366
+ }
367
+ if (isFenced(line))
368
+ fenced += 1;
369
+ }
370
+ return fenced >= 2 || (fenced >= 1 && rules >= 1);
371
+ }
372
+ function splitUnconverted(text) {
373
+ const rows = [];
374
+ const lines = text.split('\n');
375
+ for (let i = 0; i < lines.length; i += 1) {
376
+ const line = lines[i].replace(/\r$/, '').trim();
377
+ if (!line || RULE_LINE.test(line))
378
+ continue;
379
+ // A line this format does not have — a title, a page footer — becomes one
380
+ // cell, which is not a header and not a row CSPeach can read whole.
381
+ const cells = isFenced(line) ? line.slice(1, -1).split('|').map((c) => c.trim()) : [line];
382
+ rows.push({ cells, line: i + 1 });
383
+ }
384
+ return rows.filter((r) => r.cells.some((v) => v.trim() !== ''));
385
+ }
386
+ /**
387
+ * The ATC RUN MONITOR: the list of runs, not the list of findings. It is one
388
+ * menu away from the right screen and it holds no finding at all, so a customer
389
+ * hands it over by mistake. Said as what to do next, not as what is wrong.
390
+ */
391
+ const RUN_LIST_TITLES = new Set(['run series', 'execution id', 'restarts', 'expires on', 'started by',
392
+ 'config access factory', 'duration seconds']);
393
+ const RUN_LIST = 'This is the list of ATC runs, not the list of findings. '
394
+ + 'In SAP GUI, select your run, press Result, and export the findings list.';
395
+ /** Two of its own column titles: one alone can stand in a findings export too. */
396
+ const RUN_LIST_MARKS = 2;
397
+ const looksLikeRunList = (rows) => rows.some((r) => r.cells.filter((c) => RUN_LIST_TITLES.has(titleKey(c))).length >= RUN_LIST_MARKS);
398
+ export function parseAtcCsv(input) {
399
+ const text = input.charCodeAt(0) === 0xfeff ? input.slice(1) : input;
400
+ // 'PK' + bytes 3 and 4: the first bytes of every .xlsx (it is a zip). Not a bare 'PK': a header can start with "PKG".
401
+ if (text.startsWith(`PK${String.fromCharCode(3, 4)}`))
402
+ throw new AtcFileError(EXCEL);
403
+ // UTF-16 read as UTF-8 is full of NULs. `squash` would strip them and match a
404
+ // header that is not there, so this is refused before anything is normalised.
405
+ if (text.slice(0, SNIFF).includes('\u0000') || /^[\uFFFE\uFEFF]/.test(text))
406
+ throw new AtcFileError(NOT_UTF8);
407
+ // SAP GUI's unconverted list first: its `Identity` cells hold thousands
408
+ // separators, so a comma sniff would read it as a comma-separated file and
409
+ // shred every row. It is told apart by its shape, not by its punctuation.
410
+ const unconverted = looksUnconverted(text);
411
+ // The delimiter is judged from the first few lines, not from the whole file:
412
+ // a title line above the header holds none of them.
413
+ const head = text.slice(0, SNIFF).split(/\r?\n/).slice(0, HEADER_LINES + 1);
414
+ const count = (d) => Math.max(...head.map((l) => l.split(d).length - 1), 0);
415
+ const delimiter = unconverted
416
+ ? '|'
417
+ : [';', '\t', ','].reduce((best, d) => (count(d) > count(best) ? d : best), ',');
418
+ const all = unconverted ? splitUnconverted(text) : splitCsv(text, delimiter);
419
+ // A real export can carry a title line or two above the header.
420
+ let headerAt = -1;
421
+ let at = {};
422
+ let columns = {};
423
+ for (let i = 0; i < Math.min(HEADER_LINES, all.length); i += 1) {
424
+ const mapped = mapHeader(all[i].cells);
425
+ if (mapped) {
426
+ headerAt = i;
427
+ at = mapped.at;
428
+ columns = mapped.columns;
429
+ break;
430
+ }
431
+ }
432
+ if (headerAt < 0) {
433
+ // The wrong screen's export, named before the generic refusal: it has no
434
+ // object column and never will, and "which columns it needs" is no help.
435
+ if (looksLikeRunList(all.slice(0, HEADER_LINES)))
436
+ throw new AtcFileError(RUN_LIST);
437
+ const first = all[0]?.cells ?? [];
438
+ throw new AtcFileError('This file needs a column for the object type, one for the object name, and one for the check or message. '
439
+ + `It has: ${first.map((h) => shown(h)).filter(Boolean).join(', ') || 'no header row'}`);
440
+ }
441
+ const header = all[headerAt].cells;
442
+ const data = all.slice(headerAt + 1);
443
+ const taken = new Set(Object.values(at));
444
+ /*
445
+ * Two columns with one name.
446
+ *
447
+ * When CSPeach READS one of them it has to pick, and the one it did not pick
448
+ * becomes a cell it treats as empty — and an empty cell is what "clean" is
449
+ * made of. It cannot know which is true, so it refuses the file.
450
+ *
451
+ * When it reads NEITHER, nothing is hidden: both stand where they are and
452
+ * both are watched, finding-side or neutral, by position. The real export has
453
+ * exactly that — `Category` twice, confirmed 2026-09-20 — and a file CSPeach
454
+ * can account for whole is a file it may read. The second one is shown as
455
+ * `Category (2)`, so a column named on screen can still be found by eye.
456
+ */
457
+ const firstAt = new Map();
458
+ const seenTitles = new Map();
459
+ const label = header.map((h) => (h ?? '').trim());
460
+ for (let n = 0; n < header.length; n += 1) {
461
+ const w = titleKey(header[n] ?? '');
462
+ if (!w)
463
+ continue;
464
+ const before = seenTitles.get(w) ?? 0;
465
+ seenTitles.set(w, before + 1);
466
+ if (!before) {
467
+ firstAt.set(w, n);
468
+ continue;
469
+ }
470
+ const first = firstAt.get(w);
471
+ if (taken.has(n) || taken.has(first)) {
472
+ throw new AtcFileError(`This file has the column "${shown(label[first])}" more than once, so CSPeach cannot tell its columns apart. Nothing was read.`);
473
+ }
474
+ label[n] = `${label[n]} (${before + 1})`;
475
+ }
476
+ // A row that repeats the header is the header again, not an object called
477
+ // "OBJECT NAME". It is dropped whole: it says nothing either way.
478
+ const isHeaderRow = (cells) => cells.length === header.length && cells.every((c, n) => titleKey(c) === titleKey(header[n] ?? ''));
479
+ const body = data.filter((r) => !isHeaderRow(r.cells));
480
+ const emptyColumn = (n) => body.every((r) => (r.cells[n] ?? '').trim() === '');
481
+ const digitsColumn = (n) => body.every((r) => /^\d*$/.test((r.cells[n] ?? '').trim()));
482
+ const unknownColumns = [];
483
+ const findingAt = [...taken].filter((n) => n !== at.type && n !== at.name);
484
+ header.forEach((h, n) => {
485
+ if (taken.has(n))
486
+ return;
487
+ const key = titleKey(h);
488
+ if (FINDING_ONLY.has(key)) {
489
+ findingAt.push(n);
490
+ return;
491
+ }
492
+ if (NEUTRAL.has(key))
493
+ return;
494
+ if (COUNTER.has(key) && digitsColumn(n))
495
+ return;
496
+ // A column with no title CSPeach can read hides whatever is under it — so it
497
+ // is unknown unless there is nothing under it at all.
498
+ if (!key) {
499
+ if (!emptyColumn(n))
500
+ unknownColumns.push(`column ${n + 1}`);
501
+ return;
502
+ }
503
+ unknownColumns.push(shown(label[n] ?? h));
504
+ });
505
+ const fieldAt = new Map(Object.entries(at).map(([f, n]) => [n, f]));
506
+ const rows = [];
507
+ const unreadable = [];
508
+ let orphanFindings = 0;
509
+ for (const raw of body) {
510
+ const cellOf = (field) => (at[field] === undefined ? '' : (raw.cells[at[field]] ?? '').trim());
511
+ const type = clean(cellOf('type')).toUpperCase();
512
+ const name = clean(cellOf('name')).toUpperCase();
513
+ const prio = cellOf('priority');
514
+ const priority = PRIORITY_WORDS[prio.toLowerCase()] ?? (Number.parseInt(prio, 10) || undefined);
515
+ const line = Number.parseInt(cellOf('line'), 10) || undefined;
516
+ const row = {
517
+ type, name,
518
+ ...(cellOf('checkId') ? { checkId: cellOf('checkId') } : {}),
519
+ ...(cellOf('messageId') ? { messageId: cellOf('messageId') } : {}),
520
+ ...(cellOf('category') ? { category: cellOf('category') } : {}),
521
+ ...(cellOf('title') ? { title: cellOf('title') } : {}),
522
+ ...(cellOf('text') ? { text: cellOf('text') } : {}),
523
+ ...(priority ? { priority } : {}),
524
+ ...(line ? { line } : {}),
525
+ ...(cellOf('refType') ? { refType: cellOf('refType') } : {}),
526
+ ...(cellOf('refObject') ? { refObject: cellOf('refObject') } : {}),
527
+ ...(cellOf('note') ? { note: cellOf('note') } : {}),
528
+ ...(countOf(cellOf('count')) ? { count: countOf(cellOf('count')) } : {}),
529
+ };
530
+ // A finding is judged by the CELL, never by what CSPeach could make of it:
531
+ // `Very High` in the priority column is a finding, not an empty cell.
532
+ for (const n of findingAt) {
533
+ const cell = (raw.cells[n] ?? '').trim();
534
+ if (!cell)
535
+ continue;
536
+ const field = fieldAt.get(n);
537
+ if (field && row[field] !== undefined)
538
+ continue;
539
+ row.other = row.other ?? cell;
540
+ }
541
+ const hadFinding = isFindingRow(row);
542
+ if (!type || !name) {
543
+ unreadable.push({ line: raw.line, hadFinding, why: name ? 'no-type' : 'no-name', ...(name ? { name } : {}) });
544
+ // A finding that names no object may belong to ANY object in this file.
545
+ if (!name && hadFinding)
546
+ orphanFindings += 1;
547
+ continue;
548
+ }
549
+ // A line cut short is missing cells CSPeach would otherwise read as empty;
550
+ // a line with more cells than the header has slipped a column. With a
551
+ // finding in what is left it is still a finding; with nothing in it, it is
552
+ // not a clean object, it is half a line.
553
+ if (raw.cells.length !== header.length && !hadFinding) {
554
+ unreadable.push({ line: raw.line, name, hadFinding, why: raw.cells.length < header.length ? 'short' : 'long' });
555
+ continue;
556
+ }
557
+ rows.push(row);
558
+ }
559
+ return {
560
+ rows, columns, skipped: unreadable.length, delimiter, unknownColumns, unreadable,
561
+ canProveClean: unknownColumns.length === 0 && orphanFindings === 0,
562
+ };
563
+ }
564
+ /** The read columns of a header row, or null when this row is not a header at all. */
565
+ function mapHeader(cells) {
566
+ const at = {};
567
+ const columns = {};
568
+ for (const field of Object.keys(HEADERS)) {
569
+ for (const word of HEADERS[field]) {
570
+ const i = cells.findIndex((h, n) => titleKey(h) === word && !Object.values(at).includes(n));
571
+ if (i >= 0) {
572
+ at[field] = i;
573
+ columns[field] = cells[i].trim();
574
+ break;
575
+ }
576
+ }
577
+ }
578
+ const named = at.checkId !== undefined || at.messageId !== undefined || at.title !== undefined;
579
+ return at.type !== undefined && at.name !== undefined && named ? { at, columns } : null;
580
+ }
581
+ /** True when the row names a finding; false when it only names an object the export checked. */
582
+ export const isFindingRow = (row) => FINDING_FIELDS.some((f) => row[f] !== undefined);
583
+ /** `R3TR PROG` → `PROG`, `CLAS/OC` → `CLAS`. The last word, before any slash. */
584
+ const typeOf = (cell) => {
585
+ const t = cell.trim().toUpperCase().split(/\s+/).pop() ?? '';
586
+ return t.split('/')[0] ?? t;
587
+ };
588
+ /** `ZCL_FI_TAX=>CALC`, `ZCL_FI_TAX=CM001`, `ZIF_X~METH` all point at `ZCL_FI_TAX`. */
589
+ const CLASS_INCLUDE = /=+C\w+$/;
590
+ export function doubtCandidates(name) {
591
+ const whole = name.trim().toUpperCase();
592
+ const head = whole.split(/=>|\s+|~/)[0] ?? whole;
593
+ return [...new Set([whole, head, head.replace(CLASS_INCLUDE, ''), whole.replace(CLASS_INCLUDE, '')])].filter(Boolean);
594
+ }
595
+ export function measureFromFile(input) {
596
+ const bytes = readFileBytes(input.filePath);
597
+ const parsed = parseAtcCsv(bytes.toString('utf8'));
598
+ if (!parsed.rows.length && !parsed.unreadable.length) {
599
+ throw new AtcFileError('The file holds no result rows. An empty file is not proof that anything is clean, so nothing was written.');
600
+ }
601
+ // One object, one row, however often the baseline or the file names it.
602
+ const seen = new Set();
603
+ const baseline = input.baseline.filter((o) => { const k = objectKey(o.type, o.name); return seen.has(k) ? false : (seen.add(k), true); });
604
+ const byName = new Map();
605
+ for (const o of baseline) {
606
+ const n = o.name.trim().toUpperCase();
607
+ byName.set(n, [...(byName.get(n) ?? []), objectKey(o.type, o.name)]);
608
+ }
609
+ const inScope = input.scope ? scopeObjects(baseline, input.lanes, input.scope) : null;
610
+ const scopeKeys = inScope ? new Set(inScope.map((o) => objectKey(o.type, o.name))) : null;
611
+ const findings = [];
612
+ const unknown = new Set();
613
+ const ambiguous = new Set();
614
+ const disagreed = new Set();
615
+ const doubted = new Set();
616
+ const knownInFile = new Set();
617
+ const cleanRows = new Set();
618
+ const named = new Set();
619
+ /**
620
+ * Findings this file holds that CSPeach could not give to an object of this
621
+ * team project. Every one of them may belong to an object the file also calls
622
+ * clean, so ONE of them is enough: the file vouches for nothing.
623
+ */
624
+ const unplaced = [];
625
+ let outsideScope = 0;
626
+ /** A row nobody could read casts doubt on every baseline object its name could mean. */
627
+ const doubt = (name) => {
628
+ let hit = false;
629
+ for (const c of doubtCandidates(name))
630
+ if (byName.has(c)) {
631
+ doubted.add(c);
632
+ hit = true;
633
+ }
634
+ return hit;
635
+ };
636
+ for (const u of parsed.unreadable) {
637
+ if (u.name)
638
+ doubt(u.name);
639
+ if (u.hadFinding)
640
+ unplaced.push(shown(u.name ?? `line ${u.line}`));
641
+ }
642
+ for (const r of parsed.rows) {
643
+ // The TYPE cell decides. A name this team project holds under another type
644
+ // is never taken instead: that is how a finding in ZCL_FI_TAX comes to be
645
+ // written against the report of the same name.
646
+ const key = objectKey(typeOf(r.type), r.name);
647
+ if (!seen.has(key)) {
648
+ const sameName = byName.get(r.name.trim().toUpperCase()) ?? [];
649
+ if (sameName.length > 1)
650
+ ambiguous.add(shown(key));
651
+ else if (sameName.length === 1)
652
+ disagreed.add(shown(key));
653
+ // A method, a function module or an include names its object in its own
654
+ // name. It is not another system's row, and it is not this object's proof.
655
+ if (!doubt(r.name))
656
+ unknown.add(shown(key));
657
+ if (isFindingRow(r))
658
+ unplaced.push(shown(r.name));
659
+ continue;
660
+ }
661
+ knownInFile.add(key);
662
+ if (scopeKeys && !scopeKeys.has(key)) {
663
+ outsideScope += 1;
664
+ continue;
665
+ }
666
+ if (isFindingRow(r)) {
667
+ named.add(key);
668
+ const display = fileMessage(r);
669
+ const finding = {
670
+ key, ...(r.checkId ? { checkId: r.checkId } : {}), ...(r.messageId ? { messageId: r.messageId } : {}),
671
+ ...(r.category ? { category: r.category } : {}), ...(r.title ? { messageTitle: r.title } : {}),
672
+ ...(r.text ? { messageText: r.text } : {}),
673
+ ...(r.priority ? { priority: r.priority } : {}), ...(r.line ? { line: r.line } : {}),
674
+ ...(display && display !== r.title ? { displayMessage: display } : {}),
675
+ };
676
+ // A row that says it was found 5 times is five findings in that object.
677
+ // The same finding five times over is what the export means, and counting
678
+ // it once would make the object look better than SAP says it is.
679
+ for (let n = 0; n < (r.count ?? 1); n += 1)
680
+ findings.push(finding);
681
+ continue;
682
+ }
683
+ cleanRows.add(key);
684
+ }
685
+ if (unknown.size > knownInFile.size) {
686
+ throw new AtcFileError(`${unknown.size} of the ${unknown.size + knownInFile.size} objects in this file are not in this team project `
687
+ + `(${upToFive([...unknown])}). It looks like the export of another system or another scope, so nothing was written.`);
688
+ }
689
+ /**
690
+ * Whether this file may be taken at its word about ANY object being clean.
691
+ * One finding it could not place, one column it does not know, one finding
692
+ * with no object: each of those could be the finding that belongs to the
693
+ * object the file calls clean.
694
+ */
695
+ const vouches = parsed.canProveClean && unplaced.length === 0;
696
+ if (vouches)
697
+ for (const key of cleanRows)
698
+ named.add(key);
699
+ else
700
+ for (const key of cleanRows)
701
+ doubt(key.slice(key.indexOf('|') + 1));
702
+ const hasFinding = new Set(findings.map((f) => f.key));
703
+ const orphaned = parsed.unreadable.filter((u) => u.why === 'no-name' && u.hadFinding).length;
704
+ const canBeClean = (o) => vouches && !doubted.has(o.name.trim().toUpperCase());
705
+ // Row order follows the baseline, so two imports of the same scope read the same way.
706
+ const pool = inScope ?? baseline;
707
+ const rowRefs = pool
708
+ .filter((o) => {
709
+ const key = objectKey(o.type, o.name);
710
+ if (hasFinding.has(key))
711
+ return true;
712
+ if (!canBeClean(o))
713
+ return false;
714
+ return input.scope && input.absentMeansClean ? true : named.has(key);
715
+ })
716
+ .map((o) => ({ type: o.type, name: o.name }));
717
+ const rowKeys = new Set(rowRefs.map((o) => objectKey(o.type, o.name)));
718
+ // Named by this file, left as not measured: said one by one, never as a number
719
+ // alone. An object the file never mentions is a different sentence.
720
+ const touched = (o) => knownInFile.has(objectKey(o.type, o.name)) || doubted.has(o.name.trim().toUpperCase());
721
+ const leftOut = pool.filter((o) => !rowKeys.has(objectKey(o.type, o.name)) && touched(o))
722
+ .map((o) => objectKey(o.type, o.name));
723
+ const label = input.scope ? scopeLabel(input.scope) : 'file';
724
+ const asked = inScope ? inScope.length : rowRefs.length + leftOut.length;
725
+ // Every object of the scope that got no row, named. `doubted` above is the
726
+ // subset the file DID mention and could not be read; these are all of them.
727
+ const absentNames = pool.map((o) => objectKey(o.type, o.name)).filter((k) => !rowKeys.has(k)).slice(0, 2000);
728
+ const counts = {
729
+ scope: label, asked, measured: rowRefs.length, absent: asked - rowRefs.length,
730
+ failed: 0, unsupported: 0, unsupportedNames: [], absentNames, notRun: 0,
731
+ refusedChunks: [], fileRows: parsed.rows.length + parsed.unreadable.length,
732
+ unknown: [...unknown].sort(), ambiguous: [...ambiguous].sort(), disagreed: [...disagreed].sort(),
733
+ unplaced, vouches, doubted: leftOut.sort(), unknownColumns: parsed.unknownColumns,
734
+ shortRows: parsed.unreadable.filter((u) => u.why === 'short').map((u) => u.line),
735
+ longRows: parsed.unreadable.filter((u) => u.why === 'long').map((u) => u.line),
736
+ noObjectRows: parsed.unreadable.filter((u) => u.why === 'no-name').length,
737
+ orphanFindings: orphaned,
738
+ outsideScope, keys: [...rowKeys],
739
+ };
740
+ if (!rowRefs.length)
741
+ return { ...counts, file: null, withFindings: 0, findings: 0, refusedRows: [] };
742
+ // A clean the FILE listed and a clean the PERSON asserted are two different
743
+ // claims, and a page showing them side by side has to be able to tell them
744
+ // apart. So every zero-finding row says which it is: `listed` when the file
745
+ // carries a checked-and-clean row for that object, `absent` when it is there
746
+ // only because somebody passed --absent-means-clean. A row WITH findings says
747
+ // nothing — the findings are its evidence.
748
+ const rows = findingsToRows(rowRefs, findings, input.project.type)
749
+ .map((r) => (r.findings.length ? r : { ...r, evidence: cleanRows.has(r.key) ? 'listed' : 'absent' }));
750
+ const res = writeRecord(input.joined, 'measurement', rows, input.by, {
751
+ source: { file: path.basename(input.filePath), sha256: createHash('sha256').update(bytes).digest('hex') },
752
+ ...(input.now ? { now: input.now } : {}), ...(input.rand ? { rand: input.rand } : {}),
753
+ extra: {
754
+ // A file carries no reliable variant: the name is written only when the
755
+ // person named one, and then it is their word, not the file's.
756
+ ...(input.variant ? { atcVariant: input.variant } : {}),
757
+ scope: label, basis: 'file',
758
+ // Only when it was really used: a flag CSPeach turned off must not be in
759
+ // the record as though it had counted.
760
+ ...(input.absentMeansClean && vouches ? { absentMeansClean: true } : {}),
761
+ coverage: { asked, measured: rowRefs.length, absent: asked - rowRefs.length, failed: 0, unsupported: 0, notRun: 0 },
762
+ },
763
+ });
764
+ // A row the baseline on disk no longer holds was never measured here either.
765
+ const refusedRows = res.refused;
766
+ const kept = rows.filter((r) => !refusedRows.includes(r.key));
767
+ return {
768
+ ...counts, measured: kept.length, failed: refusedRows.length, absent: asked - kept.length - refusedRows.length,
769
+ keys: kept.map((r) => r.key),
770
+ file: res.file, refusedRows,
771
+ withFindings: kept.filter((r) => r.findings.length > 0).length,
772
+ findings: kept.reduce((n, r) => n + r.findings.length, 0),
773
+ };
774
+ }
775
+ /** The bytes of the file the person named — once, with a size the terminal can hold. */
776
+ function readFileBytes(filePath) {
777
+ if (/\.xls[xmb]?$/i.test(filePath))
778
+ throw new AtcFileError(EXCEL);
779
+ let size;
780
+ try {
781
+ size = fs.statSync(filePath).size;
782
+ }
783
+ catch {
784
+ throw new AtcFileError(`Cannot read ${filePath}`);
785
+ }
786
+ if (size > MAX_BYTES) {
787
+ throw new AtcFileError('This file is larger than 50 MB, so it was not read. An ATC result export is a list of findings, not a database dump.');
788
+ }
789
+ let bytes;
790
+ try {
791
+ bytes = fs.readFileSync(filePath);
792
+ }
793
+ catch {
794
+ throw new AtcFileError(`Cannot read ${filePath}`);
795
+ }
796
+ // UTF-16, before it is decoded into something that looks like a header row.
797
+ const head = bytes.subarray(0, SNIFF);
798
+ if ((bytes[0] === 0xff && bytes[1] === 0xfe) || (bytes[0] === 0xfe && bytes[1] === 0xff) || head.includes(0)) {
799
+ throw new AtcFileError(NOT_UTF8);
800
+ }
801
+ return bytes;
802
+ }