@portll/cobolwork 0.0.1 → 0.2.75

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/LICENSE +661 -0
  2. package/LICENSING.md +93 -0
  3. package/NOTICE +9 -0
  4. package/README.md +323 -3
  5. package/THIRD-PARTY-NOTICES.md +118 -0
  6. package/bin/cobolwork.mjs +354 -0
  7. package/lib/advisories.mjs +133 -0
  8. package/lib/baseline.mjs +154 -0
  9. package/lib/bms.mjs +453 -0
  10. package/lib/build.mjs +402 -0
  11. package/lib/capabilities.mjs +79 -0
  12. package/lib/cics-commands.mjs +281 -0
  13. package/lib/compliance.mjs +81 -0
  14. package/lib/consequence.mjs +139 -0
  15. package/lib/control.mjs +1515 -0
  16. package/lib/csd.mjs +77 -0
  17. package/lib/dataflow.mjs +1506 -0
  18. package/lib/diff.mjs +344 -0
  19. package/lib/explain.mjs +145 -0
  20. package/lib/gate.mjs +383 -0
  21. package/lib/index.mjs +6 -0
  22. package/lib/inventory.mjs +79 -0
  23. package/lib/jcl.mjs +478 -0
  24. package/lib/kernel/findings.mjs +94 -0
  25. package/lib/kernel/identity.mjs +216 -0
  26. package/lib/kernel/memory.mjs +217 -0
  27. package/lib/kernel/printable.mjs +6 -0
  28. package/lib/kernel/registry.mjs +79 -0
  29. package/lib/kernel/ruleset.mjs +72 -0
  30. package/lib/kernel/source-tree.mjs +159 -0
  31. package/lib/kev.mjs +27 -0
  32. package/lib/options.mjs +512 -0
  33. package/lib/packs.mjs +148 -0
  34. package/lib/parser.mjs +2055 -0
  35. package/lib/policy.mjs +163 -0
  36. package/lib/precompile-cics.mjs +169 -0
  37. package/lib/precompile.mjs +544 -0
  38. package/lib/reach.mjs +122 -0
  39. package/lib/revision.json +1 -0
  40. package/lib/revision.mjs +89 -0
  41. package/lib/sarif.mjs +222 -0
  42. package/lib/scan.mjs +272 -0
  43. package/lib/sets/build.mjs +234 -0
  44. package/lib/sets/cics.mjs +306 -0
  45. package/lib/sets/compile.mjs +187 -0
  46. package/lib/sets/copybook.mjs +174 -0
  47. package/lib/sets/flow.mjs +487 -0
  48. package/lib/sets/hidden.mjs +216 -0
  49. package/lib/sets/jcl.mjs +440 -0
  50. package/lib/sets/log.mjs +406 -0
  51. package/lib/sets/opaque.mjs +102 -0
  52. package/lib/sets/priv.mjs +322 -0
  53. package/lib/sets/recon.mjs +267 -0
  54. package/lib/sets/vendor.mjs +117 -0
  55. package/lib/sets/web.mjs +327 -0
  56. package/lib/site.mjs +164 -0
  57. package/lib/sources.mjs +156 -0
  58. package/lib/tui/app.mjs +325 -0
  59. package/lib/tui/keys.mjs +39 -0
  60. package/lib/tui/model.mjs +96 -0
  61. package/lib/tui/run.mjs +38 -0
  62. package/lib/tui/screen.mjs +59 -0
  63. package/lib/tui/terminal.mjs +46 -0
  64. package/lib/utilities.mjs +296 -0
  65. package/lib/version.mjs +15 -0
  66. package/lib/words.mjs +318 -0
  67. package/package.json +45 -6
  68. package/rules/advisories.json +264 -0
  69. package/rules/compliance-dora.json +2151 -0
  70. package/rules/compliance-ffiec.json +2134 -0
  71. package/rules/compliance-nist80053.json +2134 -0
  72. package/rules/gitleaks-mainframe.toml +57 -0
  73. package/rules/kev-ids.json +1729 -0
  74. package/rules/packs/broadcom.json +124 -0
  75. package/rules/packs/connectdirect.json +116 -0
  76. package/rules/packs/controlm.json +114 -0
  77. package/rules/system-layouts.json +28 -0
  78. package/schema/cobolwork-coverage.schema.json +65 -0
  79. package/schema/cobolwork.policy.schema.json +54 -0
@@ -0,0 +1,216 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // What a finding is, as distinct from where it is printed today.
3
+ //
4
+ // A report that cannot say a finding is the one it reported yesterday cannot say fixed, new or
5
+ // regressed, and a baseline has nothing to accept. The line number is the obvious key and the wrong
6
+ // one: code added above a finding moves it, and commitwork measured a line-keyed identity reporting
7
+ // eight findings fixed that had only shifted. So identity is built from names the source gives
8
+ // itself: the program, then the section and paragraph a statement sits in, or the record a data
9
+ // entry belongs to; the job, step and DD for JCL; and the statement's own text, with the sequence
10
+ // and identification areas removed because renumbering rewrites them.
11
+ //
12
+ // No position enters it anywhere, including as a tiebreak. An ordinal among look-alikes would be a
13
+ // line number under another name, unstable in exactly the cases it exists for. Two findings that
14
+ // agree on rule, scope and text are the same statement flagged twice, and they share a fingerprint;
15
+ // the report counts how many did, so the collapse is visible rather than silent.
16
+ import { createHash } from 'node:crypto';
17
+ import { join } from 'node:path';
18
+ import { detectFormat, normalize, tokenize, VERBS, NOT_LABELS, SCOPE_TERMINATORS } from '../parser.mjs';
19
+ import { parseJcl } from '../jcl.mjs';
20
+ import { isJcl, isProgram, isCopybook, readSource } from '../sources.mjs';
21
+
22
+ export const FINGERPRINT_VERSION = 'cobolwork/v1';
23
+
24
+ const LEVEL_RECORD = /^(0?1|77)$/;
25
+
26
+ // The named scope of every line of a COBOL source: program, then section.paragraph in the
27
+ // procedure division, or the 01 or 77 record elsewhere. Read from tokens, so a comment or a
28
+ // literal cannot pose as a label. A copybook has no divisions, so there a label and a record are
29
+ // both read, whichever the copybook turns out to hold.
30
+ export function cobolScopes(src, file = '') {
31
+ const norm = normalize(src, detectFormat(src), new Map());
32
+ const { tokens } = tokenize(norm, file);
33
+ const at = new Array(norm.entries.length + 1).fill(null);
34
+ let program = null, division = null, section = null, paragraph = null, record = null;
35
+ let sentenceStart = true;
36
+ const inProcedure = () => division === 'PROCEDURE' || (division === null && program === null);
37
+ const current = () => {
38
+ const where = division === 'PROCEDURE' || ((section || paragraph) && division === null)
39
+ ? [section, paragraph].filter(Boolean).join('.')
40
+ : record || division || '';
41
+ return program ? `${program}/${where}` : where;
42
+ };
43
+ for (let i = 0; i < tokens.length; i++) {
44
+ const t = tokens[i];
45
+ const next = tokens[i + 1];
46
+ if (t.t === 'period') { sentenceStart = true; continue; }
47
+ if (t.t === 'word') {
48
+ if (t.u === 'PROGRAM-ID' || t.u === 'FUNCTION-ID') {
49
+ let j = i + 1;
50
+ if (tokens[j] && tokens[j].t === 'period') j++;
51
+ if (tokens[j]) program = String(tokens[j].v).toUpperCase();
52
+ division = 'IDENTIFICATION'; section = paragraph = record = null;
53
+ } else if (next && next.t === 'word' && next.u === 'DIVISION') {
54
+ division = t.u === 'ID' ? 'IDENTIFICATION' : t.u;
55
+ section = paragraph = record = null;
56
+ } else if (sentenceStart && inProcedure() && next && next.t === 'word' && next.u === 'SECTION' && !NOT_LABELS.has(t.u)) {
57
+ section = t.u; paragraph = null;
58
+ } else if (sentenceStart && inProcedure() && next && next.t === 'period' && !NOT_LABELS.has(t.u) && !VERBS.has(t.u) && !SCOPE_TERMINATORS.has(t.u)) {
59
+ paragraph = t.u;
60
+ } else if (sentenceStart && division !== 'PROCEDURE' && (t.u === 'FD' || t.u === 'SD') && next && next.t === 'word') {
61
+ record = next.u;
62
+ } else if (sentenceStart && division !== 'PROCEDURE' && LEVEL_RECORD.test(t.v) && next && next.t === 'word') {
63
+ record = next.u;
64
+ }
65
+ }
66
+ sentenceStart = false;
67
+ if (at[t.line] == null) at[t.line] = current();
68
+ }
69
+ // A line holding no token of its own - a comment, a continuation, a line inside EXEC - is in
70
+ // whatever scope was in force above it.
71
+ let last = '';
72
+ for (let l = 1; l < at.length; l++) { if (at[l] == null) at[l] = last; else last = at[l]; }
73
+ return at;
74
+ }
75
+
76
+ // Job, step and DD for every line of a job. In-stream data belongs to the DD that opens it.
77
+ export function jclScopes(src, file = '') {
78
+ const lines = String(src).split(/\r?\n/).length;
79
+ const at = new Array(lines + 1).fill('');
80
+ let job;
81
+ try { job = parseJcl(src, file); } catch { return at; }
82
+ const events = [
83
+ ...job.jobs.map((j) => ({ line: j.line, kind: 'job', name: j.name })),
84
+ ...job.steps.map((s) => ({ line: s.line, kind: 'step', name: s.name || `(${s.pgm || s.proc || 'step'})` })),
85
+ ...job.dds.map((d) => ({ line: d.line, kind: 'dd', name: d.name || '(concatenated)' })),
86
+ ].sort((a, b) => a.line - b.line);
87
+ let jobName = '', stepName = '', ddName = '';
88
+ let e = 0;
89
+ for (let l = 1; l <= lines; l++) {
90
+ while (e < events.length && events[e].line <= l) {
91
+ const ev = events[e++];
92
+ if (ev.kind === 'job') { jobName = ev.name || ''; stepName = ''; ddName = ''; }
93
+ else if (ev.kind === 'step') { stepName = ev.name; ddName = ''; }
94
+ else ddName = ev.name;
95
+ }
96
+ at[l] = [jobName, stepName, ddName].filter(Boolean).join('/');
97
+ }
98
+ return at;
99
+ }
100
+
101
+ // A fingerprint can be tested against guesses, so its line has credentials masked or is left out.
102
+ const SECRET_LINE = new Set(['jcl-instream-credential', 'cd-signon-password', 'cd-snode-credentials']);
103
+ // Every repetition is bounded or cannot overlap its neighbour, so a crafted line costs linear time.
104
+ const maskSecrets = (s) => {
105
+ const out = s
106
+ .replace(/\b(PASSWORD|PASSWRD|PASSPHRASE|PHRASE|PWD|PASS)(\s*[=(]\s*)('[^']{0,256}'|"[^"]{0,256}"|[^\s,)'"]{1,256})/gi, '$1$2*')
107
+ .replace(/\b(USERID|SNODEID|PNODEID)(\s*=\s*\([^,)]{0,256},)[^)]{0,256}\)/gi, '$1$2*)');
108
+ const named = out.search(/PASS|PWD|PSWD/i);
109
+ return named < 0 ? out : out.replace(/\bVALUE(\s+)('[^']{0,256}'|"[^"]{0,256}")/gi,
110
+ (m, space, literal, at) => (at > named ? `${m.slice(0, m.length - literal.length)}'*'` : m));
111
+ };
112
+ // Strips trailing blanks and periods in linear time; a regular expression for it is quadratic on a long run.
113
+ const trimEnd = (s) => {
114
+ let end = s.length;
115
+ while (end > 0 && /[\s.]/.test(s[end - 1])) end--;
116
+ return s.slice(0, end);
117
+ };
118
+
119
+ // A line as the compiler or the reader sees it: without the sequence area, without columns 73 to
120
+ // 80, with its spacing collapsed. Case is folded for COBOL and JCL, which do not distinguish it, and
121
+ // a COBOL line loses the period that ends its sentence, which moves when a statement is added after it.
122
+ function codeText(line, kind, format) {
123
+ // A line past this is not code anyone reads, and every step below is linear in what is left.
124
+ let s = (line || '').slice(0, 4096);
125
+ if (kind === 'cobol') {
126
+ if (format === 'fixed') s = s.slice(7, 72);
127
+ else if (format === 'variable') s = s.slice(7, 250);
128
+ else if (format === 'terminal') s = s.slice(1);
129
+ s = trimEnd(s.replace(/\*>.*$/, '').toUpperCase());
130
+ } else if (kind === 'jcl') s = s.slice(0, 72).toUpperCase();
131
+ return maskSecrets(s).replace(/\s+/g, ' ').trim();
132
+ }
133
+
134
+ // Reads each file once, however many findings it holds.
135
+ function fileFacts(root, path, cache) {
136
+ let facts = cache.get(path);
137
+ if (facts) return facts;
138
+ facts = { kind: 'other', lines: [], scopes: null, format: null };
139
+ cache.set(path, facts);
140
+ let src;
141
+ try { src = readSource(join(root, path)).text; } catch { return facts; }
142
+ facts.lines = src.split(/\r?\n/);
143
+ if (isJcl(join(root, path))) {
144
+ facts.kind = 'jcl';
145
+ facts.scopes = jclScopes(src, path);
146
+ } else if (isProgram(join(root, path)) || isCopybook(join(root, path))) {
147
+ facts.kind = 'cobol';
148
+ facts.format = detectFormat(src);
149
+ try { facts.scopes = cobolScopes(src, path); } catch { facts.scopes = null; }
150
+ }
151
+ return facts;
152
+ }
153
+
154
+ // What a finding's fingerprint is made of.
155
+ function partsOf(f, root, cache) {
156
+ const facts = f.path ? fileFacts(root, f.path, cache) : { kind: 'other', lines: [], scopes: null };
157
+ const line = Number(f.line) || 0;
158
+ const scoped = facts.scopes && line > 0 && line < facts.scopes.length ? facts.scopes[line] : '';
159
+ // A program-level scope names the program, which survives the file being moved or renamed.
160
+ // Anything without one - a copybook's data, a Dockerfile, a site file - is known by its path.
161
+ const scope = facts.kind === 'cobol' && /\//.test(scoped) && !isCopybook(join(root, f.path)) ? scoped : `${f.path || ''}#${scoped}`;
162
+ // Line 1 is where a set reports a whole file or a whole program; its text is not what the
163
+ // finding is about.
164
+ const text = line > 1 && !SECRET_LINE.has(f.rule) ? codeText(facts.lines[line - 1], facts.kind, facts.format) : '';
165
+ const subject = [f.program, f.name, f.step].filter((x) => x != null && x !== '').join('|');
166
+ return { scope, subject, text };
167
+ }
168
+
169
+ const digest = (parts) => createHash('sha256').update(parts.join('\u0000')).digest('hex').slice(0, 32);
170
+
171
+ // Stamps `fingerprint` on every finding and returns how many shared one with an earlier finding.
172
+ // `repo` names the repository when one report holds several, where the same program id in two
173
+ // repositories is two programs.
174
+ export function stampFingerprints(findings, { root, repo = '' } = {}) {
175
+ const cache = new Map();
176
+ const seen = new Set();
177
+ let shared = 0;
178
+ for (const f of findings) {
179
+ const { scope, subject, text } = partsOf(f, root, cache);
180
+ const fp = digest([FINGERPRINT_VERSION, f.rule, repo, scope, subject, text]);
181
+ if (seen.has(fp)) shared++;
182
+ seen.add(fp);
183
+ f.fingerprint = fp;
184
+ }
185
+ return { version: FINGERPRINT_VERSION, shared };
186
+ }
187
+
188
+ // Looser keys for one finding seen in two trees whose line was edited between them: `scope` is the
189
+ // fingerprint without the line's text, and `route` names a data-flow finding by the statement its
190
+ // trace starts at. Null where a finding has no route.
191
+ export function pairingKeys(findings, { root, repo = '' } = {}) {
192
+ const cache = new Map();
193
+ return findings.map((f) => {
194
+ const { scope, subject } = partsOf(f, root, cache);
195
+ const src = f.evidence === 'path' && f.related && f.related[0];
196
+ let route = null;
197
+ if (src && src.path) {
198
+ const facts = fileFacts(root, src.path, cache);
199
+ const text = codeText(facts.lines[(Number(src.line) || 0) - 1], facts.kind, facts.format);
200
+ route = digest([FINGERPRINT_VERSION, 'route', f.rule, repo, f.program || '', src.path, text]);
201
+ }
202
+ return { scope: digest([FINGERPRINT_VERSION, 'scope', f.rule, repo, scope, subject]), route };
203
+ });
204
+ }
205
+
206
+ // Every line of a file as the fingerprint reads it, led by a fixed-format line's indicator column so
207
+ // that a line commented out reads as changed. Null when the file cannot be read.
208
+ export function codeLines(root, path) {
209
+ const full = join(root, path);
210
+ let src;
211
+ try { src = readSource(full).text; } catch { return null; }
212
+ const kind = isJcl(full) ? 'jcl' : isProgram(full) || isCopybook(full) ? 'cobol' : 'other';
213
+ const format = kind === 'cobol' ? detectFormat(src) : null;
214
+ const indicator = format === 'fixed';
215
+ return src.split(/\r?\n/).map((l) => (indicator ? l.charAt(6) : '') + codeText(l, kind, format));
216
+ }
@@ -0,0 +1,217 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // One place that decides how much more this process can afford to read, and one loop that every
3
+ // rule set uses to stay inside it.
4
+ //
5
+ // Two different failures put a scan on the floor, and a budget in bytes of source only answers the
6
+ // first:
7
+ //
8
+ // volume a repository holds more source than the heap can take. A byte budget answers
9
+ // this: stop reading at N bytes, report the rest as unread.
10
+ //
11
+ // accumulation a rule set holds a structure whose size is not proportional to the source it has
12
+ // read - a parse tree per program, a graph node per data item. The byte budget is
13
+ // useless here, because the thing that grows is not the bytes. A 100,000-program
14
+ // repository exhausted an 8GB heap with a 32MB source budget in force, because the
15
+ // budget bounded the input and nothing bounded the structure.
16
+ //
17
+ // So this module measures the heap itself rather than trusting a proxy for it. `getAvailableMemory`
18
+ // asks V8 what is left and asks the operating system whether it agrees; `watchMemoryBuffer` turns
19
+ // that into a decision with hysteresis, so a scan does not stop on one unlucky sample taken before
20
+ // a collection; and `eachWithinMemory` is the loop, so no rule set has to get this right on its own.
21
+ //
22
+ // What is never done is stopping quietly. Every function here reports what it did not read, and the
23
+ // caller is expected to put that in `coverageIncomplete` - a finding count over files nobody opened
24
+ // is not a clean result, and running out of memory is not an excuse to pretend otherwise.
25
+ import { freemem } from 'node:os';
26
+ import { getHeapStatistics } from 'node:v8';
27
+
28
+ const MB = 1024 * 1024;
29
+
30
+ // A floor below which we stop regardless of fractions. Enough to finish the statement in progress,
31
+ // build a report and serialise it.
32
+ export const MEMORY_FLOOR = 192 * MB;
33
+
34
+ // How much of the heap to leave unspent. V8 needs headroom to collect at all: a mark-compact that
35
+ // cannot find room is the "ineffective mark-compacts near heap limit" failure, which is a crash
36
+ // rather than a slowdown.
37
+ export const MEMORY_RESERVE = 0.2;
38
+
39
+ let override = null;
40
+
41
+ // The two readings this module takes from outside itself, and the only two. Both are injectable,
42
+ // because a test that asks the real machine is a test whose verdict is a property of the machine.
43
+ //
44
+ // This was not hypothetical. Two sessions ran the same commit: one saw 206 of 206 pass with 502 MB
45
+ // free, the other saw 17 fail with 330 MB free. The failing assertion was that a watcher reports
46
+ // 'ok' when there is room, and whether there was room depended on what else the box happened to be
47
+ // doing - other test workers, an editor, a second agent. `node --test` runs a process per file, so
48
+ // the heaps were never shared; freemem() was, because it is the whole machine.
49
+ //
50
+ // Note which of the two is the dangerous one. `available` is the lesser of the heap's headroom and
51
+ // the machine's, and on both boxes the machine term won - by 8x on one and 13x on the other. So
52
+ // injecting a heap reader alone would leave the term that actually decides still wired to the
53
+ // operating system, and the tests would look fixed while remaining a measurement of the hardware.
54
+
55
+ // An operator's statement about the machine, in megabytes, for a process whose share of it is not
56
+ // what `freemem` reports: a container with a memory limit, a CI runner sharing a host, a scan run
57
+ // beside other work. `freemem` is the whole box, so without this the scan's coverage is decided by
58
+ // what else the box happens to be doing - which is a real result, and usually not the one wanted.
59
+ // It is honoured by a child process too, which is how a spawned scan inherits the same statement.
60
+ export const FREE_MEMORY_ENV = 'COBOLWORK_FREE_MEMORY_MB';
61
+
62
+ function envFree() {
63
+ const mb = Number(process.env[FREE_MEMORY_ENV]);
64
+ return Number.isFinite(mb) && mb > 0 ? () => mb * MB : null;
65
+ }
66
+
67
+ const defaultFree = () => (envFree() || freemem)();
68
+
69
+ let readHeap = getHeapStatistics;
70
+ let readFree = defaultFree;
71
+
72
+ // Replace either reading. Called with nothing, or with a missing key, it restores the default one -
73
+ // so a test can hand back the machine as easily as it took it. The default free reader honours
74
+ // COBOLWORK_FREE_MEMORY_MB, so restoring does not discard an operator's statement.
75
+ export function setMemoryReaders({ heap = null, free = null } = {}) {
76
+ readHeap = heap || getHeapStatistics;
77
+ readFree = free || defaultFree;
78
+ }
79
+
80
+ // An explicit ceiling, in bytes, for callers that know better than the heap does - a CI box sharing
81
+ // a machine, a test that wants deterministic behaviour, an operator who has measured. Pass null to
82
+ // go back to asking V8.
83
+ export function setAvailableMemory(bytes) {
84
+ override = bytes == null ? null : Math.max(0, Number(bytes));
85
+ }
86
+
87
+ export function memoryStatus() {
88
+ const h = readHeap();
89
+ const used = h.used_heap_size;
90
+ const limit = override == null ? h.heap_size_limit : override;
91
+ const heapHeadroom = Math.max(0, limit - used);
92
+ // The heap limit is a promise V8 makes about itself, not about the machine. If the operating
93
+ // system has less free than V8 thinks it may take, the operating system wins.
94
+ const osHeadroom = readFree();
95
+ return {
96
+ limit, used, heapHeadroom, osHeadroom,
97
+ available: Math.max(0, Math.min(heapHeadroom, osHeadroom)),
98
+ fractionUsed: limit > 0 ? used / limit : 1,
99
+ overridden: override != null,
100
+ };
101
+ }
102
+
103
+ export const getAvailableMemory = () => memoryStatus().available;
104
+
105
+ // Is there room to keep going? Returns 'ok', 'tight' (stop taking on new work) or 'exhausted'.
106
+ //
107
+ // Hysteresis matters: heapUsed sawtooths, and a single sample taken just before a collection would
108
+ // stop a scan that had plenty of room. A watcher reports 'tight' only after `patience` consecutive
109
+ // samples agree, so one unlucky reading is not a verdict.
110
+ export function watchMemoryBuffer({ reserve = MEMORY_RESERVE, floor = MEMORY_FLOOR, patience = 3 } = {}) {
111
+ let consecutive = 0;
112
+ let peak = 0;
113
+ let samples = 0;
114
+
115
+ return {
116
+ check() {
117
+ const s = memoryStatus();
118
+ samples++;
119
+ peak = Math.max(peak, s.used);
120
+ const tight = s.available < floor || s.fractionUsed > 1 - reserve;
121
+ consecutive = tight ? consecutive + 1 : 0;
122
+ if (s.available < floor / 4) return 'exhausted'; // one sample is enough to believe this
123
+ return consecutive >= patience ? 'tight' : 'ok';
124
+ },
125
+ // A collection, if the process was started with --expose-gc. Never required: the loop is
126
+ // correct without it, and this only buys back headroom that was already garbage.
127
+ //
128
+ // It resets the patience counter because the readings before a collection describe a heap that
129
+ // no longer exists. That reset is why `stillTight` exists and why the caller must use it: on
130
+ // its own it would hand back a clean slate every time, and a loop that collects whenever it is
131
+ // told it is tight would never accumulate the consecutive readings needed to stop.
132
+ collect() {
133
+ if (typeof global.gc === 'function') { global.gc(); consecutive = 0; return true; }
134
+ return false;
135
+ },
136
+ // Is it still tight, right now, with no hysteresis?
137
+ //
138
+ // Hysteresis exists so that one unlucky sample taken just before a collection cannot stop a
139
+ // scan that had room. Immediately after a collection there is nothing left to be unlucky
140
+ // about, so the question is no longer "have several readings agreed" but the simpler one:
141
+ // did the collection actually help. Asking `check()` here instead is what defeated the guard -
142
+ // the reset inside `collect` guaranteed it answered 'ok', so a run started with --expose-gc
143
+ // ran on through sustained pressure until V8 aborted with ineffective mark-compacts. That is
144
+ // not hypothetical: it is how a 100,000-program repository killed a corpus run, on the guard
145
+ // written to stop exactly that.
146
+ stillTight() {
147
+ const s = memoryStatus();
148
+ return s.available < floor || s.fractionUsed > 1 - reserve;
149
+ },
150
+ get peak() { return peak; },
151
+ get samples() { return samples; },
152
+ };
153
+ }
154
+
155
+ const STOPPED_BECAUSE = {
156
+ 'byte budget': 'the scan reached its source byte budget',
157
+ 'memory reserve': 'the scan reached its memory reserve',
158
+ 'memory exhausted': 'memory ran out',
159
+ };
160
+ export const stoppedBecause = (stoppedBy) => STOPPED_BECAUSE[stoppedBy] || `the scan stopped (${stoppedBy})`;
161
+
162
+ // The loop every rule set uses instead of `for (const f of files)`.
163
+ //
164
+ // It stops when memory runs short or when a byte budget is spent, and it says which, and it names
165
+ // everything it did not get to. `work` returns the number of bytes it took on, or nothing; a rule
166
+ // set that reads a file should return its length so the byte budget means something.
167
+ export function eachWithinMemory(items, work, {
168
+ maxBytes = Infinity,
169
+ every = 8, // sample the heap this often, not on every item
170
+ watcher = null,
171
+ label = 'scan',
172
+ } = {}) {
173
+ const w = watcher || watchMemoryBuffer();
174
+ const skipped = [];
175
+ let processed = 0;
176
+ let bytes = 0;
177
+ let stoppedBy = null;
178
+
179
+ for (let i = 0; i < items.length; i++) {
180
+ if (stoppedBy) { skipped.push(items[i]); continue; }
181
+
182
+ if (bytes >= maxBytes) { stoppedBy = 'byte budget'; skipped.push(items[i]); continue; }
183
+
184
+ // Checking costs a syscall, so it is sampled rather than continuous - except once the heap is
185
+ // already past the reserve, when every item is checked because the next one may be the last.
186
+ if (i % every === 0 || w.check === undefined || memoryStatus().fractionUsed > 1 - MEMORY_RESERVE) {
187
+ const state = w.check();
188
+ if (state === 'exhausted' || state === 'tight') {
189
+ // Collect, and carry on only if the collection genuinely helped. Asking `check()` here
190
+ // instead asked a counter that `collect` had just reset, so the answer was always 'ok'.
191
+ if (!w.collect() || w.stillTight()) {
192
+ stoppedBy = state === 'exhausted' ? 'memory exhausted' : 'memory reserve';
193
+ skipped.push(items[i]);
194
+ continue;
195
+ }
196
+ }
197
+ }
198
+
199
+ const took = work(items[i], i);
200
+ if (typeof took === 'number') bytes += took;
201
+ processed++;
202
+ }
203
+
204
+ return {
205
+ label,
206
+ processed,
207
+ bytes,
208
+ skipped,
209
+ complete: skipped.length === 0,
210
+ stoppedBy,
211
+ peakHeapBytes: w.peak,
212
+ // The sentence a report should carry. Written here so every rule set says it the same way.
213
+ note: skipped.length === 0 ? null
214
+ : `${label}: ${skipped.length} of ${items.length} files were not read because ${stoppedBecause(stoppedBy)}. `
215
+ + 'The findings below are over what was read, and are not a result for this repository as a whole.',
216
+ };
217
+ }
@@ -0,0 +1,6 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // Text bound for a report, terminal or log: control and bidirectional characters replaced, length capped.
3
+ export const printable = (s, max = 200) => {
4
+ const t = String(s ?? '').replace(/[\u0000-\u001f\u007f-\u009f\u200e\u200f\u202a-\u202e\u2066-\u2069]/g, '?');
5
+ return t.length > max ? `${t.slice(0, max)}...` : t;
6
+ };
@@ -0,0 +1,79 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // Every rule set this tool has, in one list.
3
+ //
4
+ // Adding the opaque set cost nine edits across four files: an import, a spread into ALL_RULES, an
5
+ // entry in RULE_SETS, a branch in scanAll, a line in the tool-name map, two lines in the feed
6
+ // catalogue, the CLI usage string, and the key list test/cli.test.mjs enumerates by hand. Eight of
7
+ // those nine are this file's job, and none of them is a decision - they are the same fact written
8
+ // down nine times, which is eight chances to write it down wrong.
9
+ //
10
+ // One of those chances was taken. The flow set was renamed to `cobolwork-flow` while the map from
11
+ // tool name to report key still said `cobolwork`, so that set's file counts went out under the
12
+ // literal key `undefined` in every report for several commits. lib/scan.mjs grew an ESETNAME throw
13
+ // to make the next rename fail loudly instead. The throw stays - see `reportKey` below - but the
14
+ // map it guarded is gone, because a name derived from one place cannot disagree with itself.
15
+ import { scan as scanFlow, RULES as FLOW_RULES } from '../sets/flow.mjs';
16
+ import { scanCics, CICS_RULES } from '../sets/cics.mjs';
17
+ import { scanHidden, HIDDEN_RULES } from '../sets/hidden.mjs';
18
+ import { scanCopybooks, COPYBOOK_RULES } from '../sets/copybook.mjs';
19
+ import { scanJcl, JCL_RULES } from '../sets/jcl.mjs';
20
+ import { scanBuild, BUILD_RULES } from '../sets/build.mjs';
21
+ import { scanRecon, RECON_RULES } from '../sets/recon.mjs';
22
+ import { scanVendor, VENDOR_RULES } from '../sets/vendor.mjs';
23
+ import { scanOpaque, OPAQUE_RULES } from '../sets/opaque.mjs';
24
+ import { scanWeb, WEB_RULES } from '../sets/web.mjs';
25
+ import { scanPriv, PRIV_RULES } from '../sets/priv.mjs';
26
+ import { scanLog, LOG_RULES } from '../sets/log.mjs';
27
+ import { scanCompile, COMPILE_RULES } from '../sets/compile.mjs';
28
+ import { toolName } from './ruleset.mjs';
29
+
30
+ // The order is the order a scan runs them and the order --only lists them. It is not significant
31
+ // to correctness; it is significant to a report being comparable with yesterday's.
32
+ export const REGISTRY = [
33
+ { name: 'flow', scan: scanFlow, rules: FLOW_RULES },
34
+ { name: 'cics', scan: scanCics, rules: CICS_RULES },
35
+ { name: 'hidden', scan: scanHidden, rules: HIDDEN_RULES },
36
+ { name: 'copybook', scan: scanCopybooks, rules: COPYBOOK_RULES },
37
+ { name: 'jcl', scan: scanJcl, rules: JCL_RULES },
38
+ { name: 'build', scan: scanBuild, rules: BUILD_RULES },
39
+ { name: 'recon', scan: scanRecon, rules: RECON_RULES },
40
+ { name: 'vendor', scan: scanVendor, rules: VENDOR_RULES },
41
+ { name: 'opaque', scan: scanOpaque, rules: OPAQUE_RULES },
42
+ { name: 'web', scan: scanWeb, rules: WEB_RULES },
43
+ { name: 'compile', scan: scanCompile, rules: COMPILE_RULES },
44
+ { name: 'priv', scan: scanPriv, rules: PRIV_RULES },
45
+ { name: 'log', scan: scanLog, rules: LOG_RULES },
46
+ ];
47
+
48
+ // Re-exported from the kernel's ruleset module, which is where it has to live: this file imports
49
+ // every rule set, and every rule set imports that one.
50
+ export { toolName };
51
+
52
+ export const RULE_SETS = REGISTRY.map((s) => s.name);
53
+
54
+ export const ALL_RULES = Object.assign({}, ...REGISTRY.map((s) => s.rules));
55
+
56
+ const BY_TOOL = new Map(REGISTRY.map((s) => [toolName(s.name), s.name]));
57
+
58
+ // What a part's counts are filed under. An unregistered set still fails loudly rather than filing
59
+ // under `undefined`, which is the property the ESETNAME throw was added for and the reason this
60
+ // function exists instead of a bare lookup.
61
+ export function reportKey(tool) {
62
+ const name = BY_TOOL.get(tool);
63
+ if (!name) throw Object.assign(new Error(`rule set ${tool} has no report key`), { code: 'ESETNAME', tool });
64
+ return name;
65
+ }
66
+
67
+ // Rule ids must be unique across sets: ALL_RULES is one flat object, so a duplicate would silently
68
+ // take the last definition and a finding would carry another set's severity.
69
+ export function duplicateRuleIds() {
70
+ const seen = new Map();
71
+ const dupes = [];
72
+ for (const s of REGISTRY) {
73
+ for (const id of Object.keys(s.rules)) {
74
+ if (seen.has(id)) dupes.push(`${id} is declared by both ${seen.get(id)} and ${s.name}`);
75
+ else seen.set(id, s.name);
76
+ }
77
+ }
78
+ return dupes;
79
+ }
@@ -0,0 +1,72 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // What a rule set says about itself once it has finished looking.
3
+ //
4
+ // Eight sets wrote the same closing five lines: take the guarded loop's account of what it did not
5
+ // reach, stamp the findings, tally them, and assemble a summary. The assembly varied, and none of
6
+ // the variation was a decision - one set set coverageIncomplete inside its stats, another passed it
7
+ // in the returned object, a third ORed it with a flag from the JCL parser, and one had a second
8
+ // return site that skipped the whole thing and shipped a different shape.
9
+ //
10
+ // That last one matters more than tidiness. SECURITY.md treats a report that overstates coverage as
11
+ // a security bug rather than a defect: a file the tool failed to read while still reporting
12
+ // coverageIncomplete: false is the failure this project exists to prevent. Eight hand-written
13
+ // copies of the code that makes that claim is eight places for it to be wrong, and the one that was
14
+ // wrong was found by a test rather than by reading.
15
+ //
16
+ // WHAT THIS IS NOT. The specification proposed a `defineRuleSet({ perFile, finish })` harness that
17
+ // would own each set's control flow and refuse a perFile that returned a parse tree. That is not
18
+ // what this is, and the difference is deliberate. By the time this was written the traversal, the
19
+ // reading, the parse configuration, the sort, the severity stamping and the tally had all moved
20
+ // into lib/kernel/ on their own - so the inversion would have restructured nine working sets to
21
+ // take ownership of what it already had. The ETREELEAK guard went with it: it only guards a
22
+ // perFile that does not exist, and a set that accumulates parse trees in its own closure is
23
+ // something no harness can prevent. What remains is the closing assembly, which is the part that
24
+ // was still duplicated and the part that makes a security claim.
25
+ import { finish } from './findings.mjs';
26
+
27
+ // A set's tool name is its report key with the tool's name in front. It lives here rather than in
28
+ // the registry because the registry imports every rule set, and every rule set imports this: taking
29
+ // the name from there would close the loop, and the error a circular import gives - "cannot access
30
+ // X_RULES before initialization" - names the rule table rather than the cycle that broke it.
31
+ export const toolName = (name) => `cobolwork-${name}`;
32
+
33
+ // `run` is what eachWithinMemory returned, or null for a set that did not traverse - a set refusing
34
+ // to run for want of configuration still has to report in the same shape as one that ran.
35
+ export function report(name, { rules, findings, stats = {}, run = null }) {
36
+ const skipped = run ? run.skipped.length : 0;
37
+
38
+ // The guarded loop's account of what it did not get to. Written here so every set says it the
39
+ // same way, which is the same reason lib/memory.mjs composes the sentence rather than each
40
+ // caller writing its own.
41
+ if (run) {
42
+ if (skipped) stats.filesNotRead = skipped;
43
+ if (run.stoppedBy) stats.stoppedBy = run.stoppedBy;
44
+ if (run.peakHeapBytes) stats.peakHeapBytes = run.peakHeapBytes;
45
+ if (run.note) stats.notRead = run.note;
46
+ }
47
+
48
+ const { byRule } = finish(rules, findings, name);
49
+
50
+ return {
51
+ tool: toolName(name),
52
+ summary: {
53
+ findings: findings.length,
54
+ byRule,
55
+ ...stats,
56
+ // ORed, never overwritten. A set may already know its reading was short for a reason the
57
+ // loop knows nothing about - the JCL parser reporting an INCLUDE it could not resolve, for
58
+ // one - and that claim must survive being combined with this one.
59
+ //
60
+ // A file opened and not understood is a file that did not contribute, exactly like one the
61
+ // loop never reached: the set has no findings from it and cannot say there were none. Only
62
+ // the compile set had joined those up, by hand, and the other nine reported complete coverage
63
+ // over source they could not parse - which is the I4 failure SECURITY.md calls a security bug
64
+ // rather than a defect. lib/sarif.mjs already counted all three as a shortfall, so the two
65
+ // halves of the tool disagreed about what a clean result means.
66
+ coverageIncomplete: stats.coverageIncomplete === true || skipped > 0
67
+ || (stats.filesUnparsed || 0) > 0 || (stats.filesUnreadable || 0) > 0,
68
+ nosrc: (stats.filesScanned || 0) === 0,
69
+ },
70
+ findings,
71
+ };
72
+ }