@portll/cobolwork 0.0.1 → 0.2.76
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/LICENSING.md +93 -0
- package/NOTICE +9 -0
- package/README.md +325 -3
- package/THIRD-PARTY-NOTICES.md +118 -0
- package/bin/cobolwork.mjs +354 -0
- package/lib/advisories.mjs +133 -0
- package/lib/baseline.mjs +154 -0
- package/lib/bms.mjs +453 -0
- package/lib/build.mjs +402 -0
- package/lib/capabilities.mjs +79 -0
- package/lib/cics-commands.mjs +281 -0
- package/lib/compliance.mjs +81 -0
- package/lib/consequence.mjs +139 -0
- package/lib/control.mjs +1515 -0
- package/lib/csd.mjs +77 -0
- package/lib/dataflow.mjs +1506 -0
- package/lib/diff.mjs +344 -0
- package/lib/explain.mjs +145 -0
- package/lib/gate.mjs +383 -0
- package/lib/index.mjs +6 -0
- package/lib/inventory.mjs +79 -0
- package/lib/jcl.mjs +478 -0
- package/lib/kernel/findings.mjs +94 -0
- package/lib/kernel/identity.mjs +216 -0
- package/lib/kernel/memory.mjs +217 -0
- package/lib/kernel/printable.mjs +6 -0
- package/lib/kernel/registry.mjs +79 -0
- package/lib/kernel/ruleset.mjs +72 -0
- package/lib/kernel/source-tree.mjs +159 -0
- package/lib/kev.mjs +27 -0
- package/lib/options.mjs +512 -0
- package/lib/packs.mjs +148 -0
- package/lib/parser.mjs +2055 -0
- package/lib/policy.mjs +163 -0
- package/lib/precompile-cics.mjs +169 -0
- package/lib/precompile.mjs +544 -0
- package/lib/reach.mjs +122 -0
- package/lib/revision.json +1 -0
- package/lib/revision.mjs +89 -0
- package/lib/sarif.mjs +222 -0
- package/lib/scan.mjs +272 -0
- package/lib/sets/build.mjs +234 -0
- package/lib/sets/cics.mjs +306 -0
- package/lib/sets/compile.mjs +187 -0
- package/lib/sets/copybook.mjs +174 -0
- package/lib/sets/flow.mjs +487 -0
- package/lib/sets/hidden.mjs +216 -0
- package/lib/sets/jcl.mjs +440 -0
- package/lib/sets/log.mjs +406 -0
- package/lib/sets/opaque.mjs +102 -0
- package/lib/sets/priv.mjs +322 -0
- package/lib/sets/recon.mjs +267 -0
- package/lib/sets/vendor.mjs +117 -0
- package/lib/sets/web.mjs +327 -0
- package/lib/site.mjs +164 -0
- package/lib/sources.mjs +156 -0
- package/lib/tui/app.mjs +325 -0
- package/lib/tui/keys.mjs +39 -0
- package/lib/tui/model.mjs +96 -0
- package/lib/tui/run.mjs +38 -0
- package/lib/tui/screen.mjs +59 -0
- package/lib/tui/terminal.mjs +46 -0
- package/lib/utilities.mjs +296 -0
- package/lib/version.mjs +15 -0
- package/lib/words.mjs +318 -0
- package/package.json +45 -6
- package/rules/advisories.json +264 -0
- package/rules/compliance-dora.json +2151 -0
- package/rules/compliance-ffiec.json +2134 -0
- package/rules/compliance-nist80053.json +2134 -0
- package/rules/gitleaks-mainframe.toml +57 -0
- package/rules/kev-ids.json +1729 -0
- package/rules/packs/broadcom.json +124 -0
- package/rules/packs/connectdirect.json +116 -0
- package/rules/packs/controlm.json +114 -0
- package/rules/system-layouts.json +28 -0
- package/schema/cobolwork-coverage.schema.json +65 -0
- package/schema/cobolwork.policy.schema.json +54 -0
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// What a finding is, as distinct from where it is printed today.
|
|
3
|
+
//
|
|
4
|
+
// A report that cannot say a finding is the one it reported yesterday cannot say fixed, new or
|
|
5
|
+
// regressed, and a baseline has nothing to accept. The line number is the obvious key and the wrong
|
|
6
|
+
// one: code added above a finding moves it, and commitwork measured a line-keyed identity reporting
|
|
7
|
+
// eight findings fixed that had only shifted. So identity is built from names the source gives
|
|
8
|
+
// itself: the program, then the section and paragraph a statement sits in, or the record a data
|
|
9
|
+
// entry belongs to; the job, step and DD for JCL; and the statement's own text, with the sequence
|
|
10
|
+
// and identification areas removed because renumbering rewrites them.
|
|
11
|
+
//
|
|
12
|
+
// No position enters it anywhere, including as a tiebreak. An ordinal among look-alikes would be a
|
|
13
|
+
// line number under another name, unstable in exactly the cases it exists for. Two findings that
|
|
14
|
+
// agree on rule, scope and text are the same statement flagged twice, and they share a fingerprint;
|
|
15
|
+
// the report counts how many did, so the collapse is visible rather than silent.
|
|
16
|
+
import { createHash } from 'node:crypto';
|
|
17
|
+
import { join } from 'node:path';
|
|
18
|
+
import { detectFormat, normalize, tokenize, VERBS, NOT_LABELS, SCOPE_TERMINATORS } from '../parser.mjs';
|
|
19
|
+
import { parseJcl } from '../jcl.mjs';
|
|
20
|
+
import { isJcl, isProgram, isCopybook, readSource } from '../sources.mjs';
|
|
21
|
+
|
|
22
|
+
export const FINGERPRINT_VERSION = 'cobolwork/v1';
|
|
23
|
+
|
|
24
|
+
const LEVEL_RECORD = /^(0?1|77)$/;
|
|
25
|
+
|
|
26
|
+
// The named scope of every line of a COBOL source: program, then section.paragraph in the
|
|
27
|
+
// procedure division, or the 01 or 77 record elsewhere. Read from tokens, so a comment or a
|
|
28
|
+
// literal cannot pose as a label. A copybook has no divisions, so there a label and a record are
|
|
29
|
+
// both read, whichever the copybook turns out to hold.
|
|
30
|
+
export function cobolScopes(src, file = '') {
|
|
31
|
+
const norm = normalize(src, detectFormat(src), new Map());
|
|
32
|
+
const { tokens } = tokenize(norm, file);
|
|
33
|
+
const at = new Array(norm.entries.length + 1).fill(null);
|
|
34
|
+
let program = null, division = null, section = null, paragraph = null, record = null;
|
|
35
|
+
let sentenceStart = true;
|
|
36
|
+
const inProcedure = () => division === 'PROCEDURE' || (division === null && program === null);
|
|
37
|
+
const current = () => {
|
|
38
|
+
const where = division === 'PROCEDURE' || ((section || paragraph) && division === null)
|
|
39
|
+
? [section, paragraph].filter(Boolean).join('.')
|
|
40
|
+
: record || division || '';
|
|
41
|
+
return program ? `${program}/${where}` : where;
|
|
42
|
+
};
|
|
43
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
44
|
+
const t = tokens[i];
|
|
45
|
+
const next = tokens[i + 1];
|
|
46
|
+
if (t.t === 'period') { sentenceStart = true; continue; }
|
|
47
|
+
if (t.t === 'word') {
|
|
48
|
+
if (t.u === 'PROGRAM-ID' || t.u === 'FUNCTION-ID') {
|
|
49
|
+
let j = i + 1;
|
|
50
|
+
if (tokens[j] && tokens[j].t === 'period') j++;
|
|
51
|
+
if (tokens[j]) program = String(tokens[j].v).toUpperCase();
|
|
52
|
+
division = 'IDENTIFICATION'; section = paragraph = record = null;
|
|
53
|
+
} else if (next && next.t === 'word' && next.u === 'DIVISION') {
|
|
54
|
+
division = t.u === 'ID' ? 'IDENTIFICATION' : t.u;
|
|
55
|
+
section = paragraph = record = null;
|
|
56
|
+
} else if (sentenceStart && inProcedure() && next && next.t === 'word' && next.u === 'SECTION' && !NOT_LABELS.has(t.u)) {
|
|
57
|
+
section = t.u; paragraph = null;
|
|
58
|
+
} else if (sentenceStart && inProcedure() && next && next.t === 'period' && !NOT_LABELS.has(t.u) && !VERBS.has(t.u) && !SCOPE_TERMINATORS.has(t.u)) {
|
|
59
|
+
paragraph = t.u;
|
|
60
|
+
} else if (sentenceStart && division !== 'PROCEDURE' && (t.u === 'FD' || t.u === 'SD') && next && next.t === 'word') {
|
|
61
|
+
record = next.u;
|
|
62
|
+
} else if (sentenceStart && division !== 'PROCEDURE' && LEVEL_RECORD.test(t.v) && next && next.t === 'word') {
|
|
63
|
+
record = next.u;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
sentenceStart = false;
|
|
67
|
+
if (at[t.line] == null) at[t.line] = current();
|
|
68
|
+
}
|
|
69
|
+
// A line holding no token of its own - a comment, a continuation, a line inside EXEC - is in
|
|
70
|
+
// whatever scope was in force above it.
|
|
71
|
+
let last = '';
|
|
72
|
+
for (let l = 1; l < at.length; l++) { if (at[l] == null) at[l] = last; else last = at[l]; }
|
|
73
|
+
return at;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// Job, step and DD for every line of a job. In-stream data belongs to the DD that opens it.
|
|
77
|
+
export function jclScopes(src, file = '') {
|
|
78
|
+
const lines = String(src).split(/\r?\n/).length;
|
|
79
|
+
const at = new Array(lines + 1).fill('');
|
|
80
|
+
let job;
|
|
81
|
+
try { job = parseJcl(src, file); } catch { return at; }
|
|
82
|
+
const events = [
|
|
83
|
+
...job.jobs.map((j) => ({ line: j.line, kind: 'job', name: j.name })),
|
|
84
|
+
...job.steps.map((s) => ({ line: s.line, kind: 'step', name: s.name || `(${s.pgm || s.proc || 'step'})` })),
|
|
85
|
+
...job.dds.map((d) => ({ line: d.line, kind: 'dd', name: d.name || '(concatenated)' })),
|
|
86
|
+
].sort((a, b) => a.line - b.line);
|
|
87
|
+
let jobName = '', stepName = '', ddName = '';
|
|
88
|
+
let e = 0;
|
|
89
|
+
for (let l = 1; l <= lines; l++) {
|
|
90
|
+
while (e < events.length && events[e].line <= l) {
|
|
91
|
+
const ev = events[e++];
|
|
92
|
+
if (ev.kind === 'job') { jobName = ev.name || ''; stepName = ''; ddName = ''; }
|
|
93
|
+
else if (ev.kind === 'step') { stepName = ev.name; ddName = ''; }
|
|
94
|
+
else ddName = ev.name;
|
|
95
|
+
}
|
|
96
|
+
at[l] = [jobName, stepName, ddName].filter(Boolean).join('/');
|
|
97
|
+
}
|
|
98
|
+
return at;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// A fingerprint can be tested against guesses, so its line has credentials masked or is left out.
|
|
102
|
+
const SECRET_LINE = new Set(['jcl-instream-credential', 'cd-signon-password', 'cd-snode-credentials']);
|
|
103
|
+
// Every repetition is bounded or cannot overlap its neighbour, so a crafted line costs linear time.
|
|
104
|
+
const maskSecrets = (s) => {
|
|
105
|
+
const out = s
|
|
106
|
+
.replace(/\b(PASSWORD|PASSWRD|PASSPHRASE|PHRASE|PWD|PASS)(\s*[=(]\s*)('[^']{0,256}'|"[^"]{0,256}"|[^\s,)'"]{1,256})/gi, '$1$2*')
|
|
107
|
+
.replace(/\b(USERID|SNODEID|PNODEID)(\s*=\s*\([^,)]{0,256},)[^)]{0,256}\)/gi, '$1$2*)');
|
|
108
|
+
const named = out.search(/PASS|PWD|PSWD/i);
|
|
109
|
+
return named < 0 ? out : out.replace(/\bVALUE(\s+)('[^']{0,256}'|"[^"]{0,256}")/gi,
|
|
110
|
+
(m, space, literal, at) => (at > named ? `${m.slice(0, m.length - literal.length)}'*'` : m));
|
|
111
|
+
};
|
|
112
|
+
// Strips trailing blanks and periods in linear time; a regular expression for it is quadratic on a long run.
|
|
113
|
+
const trimEnd = (s) => {
|
|
114
|
+
let end = s.length;
|
|
115
|
+
while (end > 0 && /[\s.]/.test(s[end - 1])) end--;
|
|
116
|
+
return s.slice(0, end);
|
|
117
|
+
};
|
|
118
|
+
|
|
119
|
+
// A line as the compiler or the reader sees it: without the sequence area, without columns 73 to
|
|
120
|
+
// 80, with its spacing collapsed. Case is folded for COBOL and JCL, which do not distinguish it, and
|
|
121
|
+
// a COBOL line loses the period that ends its sentence, which moves when a statement is added after it.
|
|
122
|
+
function codeText(line, kind, format) {
|
|
123
|
+
// A line past this is not code anyone reads, and every step below is linear in what is left.
|
|
124
|
+
let s = (line || '').slice(0, 4096);
|
|
125
|
+
if (kind === 'cobol') {
|
|
126
|
+
if (format === 'fixed') s = s.slice(7, 72);
|
|
127
|
+
else if (format === 'variable') s = s.slice(7, 250);
|
|
128
|
+
else if (format === 'terminal') s = s.slice(1);
|
|
129
|
+
s = trimEnd(s.replace(/\*>.*$/, '').toUpperCase());
|
|
130
|
+
} else if (kind === 'jcl') s = s.slice(0, 72).toUpperCase();
|
|
131
|
+
return maskSecrets(s).replace(/\s+/g, ' ').trim();
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// Reads each file once, however many findings it holds.
|
|
135
|
+
function fileFacts(root, path, cache) {
|
|
136
|
+
let facts = cache.get(path);
|
|
137
|
+
if (facts) return facts;
|
|
138
|
+
facts = { kind: 'other', lines: [], scopes: null, format: null };
|
|
139
|
+
cache.set(path, facts);
|
|
140
|
+
let src;
|
|
141
|
+
try { src = readSource(join(root, path)).text; } catch { return facts; }
|
|
142
|
+
facts.lines = src.split(/\r?\n/);
|
|
143
|
+
if (isJcl(join(root, path))) {
|
|
144
|
+
facts.kind = 'jcl';
|
|
145
|
+
facts.scopes = jclScopes(src, path);
|
|
146
|
+
} else if (isProgram(join(root, path)) || isCopybook(join(root, path))) {
|
|
147
|
+
facts.kind = 'cobol';
|
|
148
|
+
facts.format = detectFormat(src);
|
|
149
|
+
try { facts.scopes = cobolScopes(src, path); } catch { facts.scopes = null; }
|
|
150
|
+
}
|
|
151
|
+
return facts;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// What a finding's fingerprint is made of.
|
|
155
|
+
function partsOf(f, root, cache) {
|
|
156
|
+
const facts = f.path ? fileFacts(root, f.path, cache) : { kind: 'other', lines: [], scopes: null };
|
|
157
|
+
const line = Number(f.line) || 0;
|
|
158
|
+
const scoped = facts.scopes && line > 0 && line < facts.scopes.length ? facts.scopes[line] : '';
|
|
159
|
+
// A program-level scope names the program, which survives the file being moved or renamed.
|
|
160
|
+
// Anything without one - a copybook's data, a Dockerfile, a site file - is known by its path.
|
|
161
|
+
const scope = facts.kind === 'cobol' && /\//.test(scoped) && !isCopybook(join(root, f.path)) ? scoped : `${f.path || ''}#${scoped}`;
|
|
162
|
+
// Line 1 is where a set reports a whole file or a whole program; its text is not what the
|
|
163
|
+
// finding is about.
|
|
164
|
+
const text = line > 1 && !SECRET_LINE.has(f.rule) ? codeText(facts.lines[line - 1], facts.kind, facts.format) : '';
|
|
165
|
+
const subject = [f.program, f.name, f.step].filter((x) => x != null && x !== '').join('|');
|
|
166
|
+
return { scope, subject, text };
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
const digest = (parts) => createHash('sha256').update(parts.join('\u0000')).digest('hex').slice(0, 32);
|
|
170
|
+
|
|
171
|
+
// Stamps `fingerprint` on every finding and returns how many shared one with an earlier finding.
|
|
172
|
+
// `repo` names the repository when one report holds several, where the same program id in two
|
|
173
|
+
// repositories is two programs.
|
|
174
|
+
export function stampFingerprints(findings, { root, repo = '' } = {}) {
|
|
175
|
+
const cache = new Map();
|
|
176
|
+
const seen = new Set();
|
|
177
|
+
let shared = 0;
|
|
178
|
+
for (const f of findings) {
|
|
179
|
+
const { scope, subject, text } = partsOf(f, root, cache);
|
|
180
|
+
const fp = digest([FINGERPRINT_VERSION, f.rule, repo, scope, subject, text]);
|
|
181
|
+
if (seen.has(fp)) shared++;
|
|
182
|
+
seen.add(fp);
|
|
183
|
+
f.fingerprint = fp;
|
|
184
|
+
}
|
|
185
|
+
return { version: FINGERPRINT_VERSION, shared };
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// Looser keys for one finding seen in two trees whose line was edited between them: `scope` is the
|
|
189
|
+
// fingerprint without the line's text, and `route` names a data-flow finding by the statement its
|
|
190
|
+
// trace starts at. Null where a finding has no route.
|
|
191
|
+
export function pairingKeys(findings, { root, repo = '' } = {}) {
|
|
192
|
+
const cache = new Map();
|
|
193
|
+
return findings.map((f) => {
|
|
194
|
+
const { scope, subject } = partsOf(f, root, cache);
|
|
195
|
+
const src = f.evidence === 'path' && f.related && f.related[0];
|
|
196
|
+
let route = null;
|
|
197
|
+
if (src && src.path) {
|
|
198
|
+
const facts = fileFacts(root, src.path, cache);
|
|
199
|
+
const text = codeText(facts.lines[(Number(src.line) || 0) - 1], facts.kind, facts.format);
|
|
200
|
+
route = digest([FINGERPRINT_VERSION, 'route', f.rule, repo, f.program || '', src.path, text]);
|
|
201
|
+
}
|
|
202
|
+
return { scope: digest([FINGERPRINT_VERSION, 'scope', f.rule, repo, scope, subject]), route };
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
// Every line of a file as the fingerprint reads it, led by a fixed-format line's indicator column so
|
|
207
|
+
// that a line commented out reads as changed. Null when the file cannot be read.
|
|
208
|
+
export function codeLines(root, path) {
|
|
209
|
+
const full = join(root, path);
|
|
210
|
+
let src;
|
|
211
|
+
try { src = readSource(full).text; } catch { return null; }
|
|
212
|
+
const kind = isJcl(full) ? 'jcl' : isProgram(full) || isCopybook(full) ? 'cobol' : 'other';
|
|
213
|
+
const format = kind === 'cobol' ? detectFormat(src) : null;
|
|
214
|
+
const indicator = format === 'fixed';
|
|
215
|
+
return src.split(/\r?\n/).map((l) => (indicator ? l.charAt(6) : '') + codeText(l, kind, format));
|
|
216
|
+
}
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// One place that decides how much more this process can afford to read, and one loop that every
|
|
3
|
+
// rule set uses to stay inside it.
|
|
4
|
+
//
|
|
5
|
+
// Two different failures put a scan on the floor, and a budget in bytes of source only answers the
|
|
6
|
+
// first:
|
|
7
|
+
//
|
|
8
|
+
// volume a repository holds more source than the heap can take. A byte budget answers
|
|
9
|
+
// this: stop reading at N bytes, report the rest as unread.
|
|
10
|
+
//
|
|
11
|
+
// accumulation a rule set holds a structure whose size is not proportional to the source it has
|
|
12
|
+
// read - a parse tree per program, a graph node per data item. The byte budget is
|
|
13
|
+
// useless here, because the thing that grows is not the bytes. A 100,000-program
|
|
14
|
+
// repository exhausted an 8GB heap with a 32MB source budget in force, because the
|
|
15
|
+
// budget bounded the input and nothing bounded the structure.
|
|
16
|
+
//
|
|
17
|
+
// So this module measures the heap itself rather than trusting a proxy for it. `getAvailableMemory`
|
|
18
|
+
// asks V8 what is left and asks the operating system whether it agrees; `watchMemoryBuffer` turns
|
|
19
|
+
// that into a decision with hysteresis, so a scan does not stop on one unlucky sample taken before
|
|
20
|
+
// a collection; and `eachWithinMemory` is the loop, so no rule set has to get this right on its own.
|
|
21
|
+
//
|
|
22
|
+
// What is never done is stopping quietly. Every function here reports what it did not read, and the
|
|
23
|
+
// caller is expected to put that in `coverageIncomplete` - a finding count over files nobody opened
|
|
24
|
+
// is not a clean result, and running out of memory is not an excuse to pretend otherwise.
|
|
25
|
+
import { freemem } from 'node:os';
|
|
26
|
+
import { getHeapStatistics } from 'node:v8';
|
|
27
|
+
|
|
28
|
+
const MB = 1024 * 1024;
|
|
29
|
+
|
|
30
|
+
// A floor below which we stop regardless of fractions. Enough to finish the statement in progress,
|
|
31
|
+
// build a report and serialise it.
|
|
32
|
+
export const MEMORY_FLOOR = 192 * MB;
|
|
33
|
+
|
|
34
|
+
// How much of the heap to leave unspent. V8 needs headroom to collect at all: a mark-compact that
|
|
35
|
+
// cannot find room is the "ineffective mark-compacts near heap limit" failure, which is a crash
|
|
36
|
+
// rather than a slowdown.
|
|
37
|
+
export const MEMORY_RESERVE = 0.2;
|
|
38
|
+
|
|
39
|
+
let override = null;
|
|
40
|
+
|
|
41
|
+
// The two readings this module takes from outside itself, and the only two. Both are injectable,
|
|
42
|
+
// because a test that asks the real machine is a test whose verdict is a property of the machine.
|
|
43
|
+
//
|
|
44
|
+
// This was not hypothetical. Two sessions ran the same commit: one saw 206 of 206 pass with 502 MB
|
|
45
|
+
// free, the other saw 17 fail with 330 MB free. The failing assertion was that a watcher reports
|
|
46
|
+
// 'ok' when there is room, and whether there was room depended on what else the box happened to be
|
|
47
|
+
// doing - other test workers, an editor, a second agent. `node --test` runs a process per file, so
|
|
48
|
+
// the heaps were never shared; freemem() was, because it is the whole machine.
|
|
49
|
+
//
|
|
50
|
+
// Note which of the two is the dangerous one. `available` is the lesser of the heap's headroom and
|
|
51
|
+
// the machine's, and on both boxes the machine term won - by 8x on one and 13x on the other. So
|
|
52
|
+
// injecting a heap reader alone would leave the term that actually decides still wired to the
|
|
53
|
+
// operating system, and the tests would look fixed while remaining a measurement of the hardware.
|
|
54
|
+
|
|
55
|
+
// An operator's statement about the machine, in megabytes, for a process whose share of it is not
|
|
56
|
+
// what `freemem` reports: a container with a memory limit, a CI runner sharing a host, a scan run
|
|
57
|
+
// beside other work. `freemem` is the whole box, so without this the scan's coverage is decided by
|
|
58
|
+
// what else the box happens to be doing - which is a real result, and usually not the one wanted.
|
|
59
|
+
// It is honoured by a child process too, which is how a spawned scan inherits the same statement.
|
|
60
|
+
export const FREE_MEMORY_ENV = 'COBOLWORK_FREE_MEMORY_MB';
|
|
61
|
+
|
|
62
|
+
function envFree() {
|
|
63
|
+
const mb = Number(process.env[FREE_MEMORY_ENV]);
|
|
64
|
+
return Number.isFinite(mb) && mb > 0 ? () => mb * MB : null;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const defaultFree = () => (envFree() || freemem)();
|
|
68
|
+
|
|
69
|
+
let readHeap = getHeapStatistics;
|
|
70
|
+
let readFree = defaultFree;
|
|
71
|
+
|
|
72
|
+
// Replace either reading. Called with nothing, or with a missing key, it restores the default one -
|
|
73
|
+
// so a test can hand back the machine as easily as it took it. The default free reader honours
|
|
74
|
+
// COBOLWORK_FREE_MEMORY_MB, so restoring does not discard an operator's statement.
|
|
75
|
+
export function setMemoryReaders({ heap = null, free = null } = {}) {
|
|
76
|
+
readHeap = heap || getHeapStatistics;
|
|
77
|
+
readFree = free || defaultFree;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// An explicit ceiling, in bytes, for callers that know better than the heap does - a CI box sharing
|
|
81
|
+
// a machine, a test that wants deterministic behaviour, an operator who has measured. Pass null to
|
|
82
|
+
// go back to asking V8.
|
|
83
|
+
export function setAvailableMemory(bytes) {
|
|
84
|
+
override = bytes == null ? null : Math.max(0, Number(bytes));
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export function memoryStatus() {
|
|
88
|
+
const h = readHeap();
|
|
89
|
+
const used = h.used_heap_size;
|
|
90
|
+
const limit = override == null ? h.heap_size_limit : override;
|
|
91
|
+
const heapHeadroom = Math.max(0, limit - used);
|
|
92
|
+
// The heap limit is a promise V8 makes about itself, not about the machine. If the operating
|
|
93
|
+
// system has less free than V8 thinks it may take, the operating system wins.
|
|
94
|
+
const osHeadroom = readFree();
|
|
95
|
+
return {
|
|
96
|
+
limit, used, heapHeadroom, osHeadroom,
|
|
97
|
+
available: Math.max(0, Math.min(heapHeadroom, osHeadroom)),
|
|
98
|
+
fractionUsed: limit > 0 ? used / limit : 1,
|
|
99
|
+
overridden: override != null,
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export const getAvailableMemory = () => memoryStatus().available;
|
|
104
|
+
|
|
105
|
+
// Is there room to keep going? Returns 'ok', 'tight' (stop taking on new work) or 'exhausted'.
|
|
106
|
+
//
|
|
107
|
+
// Hysteresis matters: heapUsed sawtooths, and a single sample taken just before a collection would
|
|
108
|
+
// stop a scan that had plenty of room. A watcher reports 'tight' only after `patience` consecutive
|
|
109
|
+
// samples agree, so one unlucky reading is not a verdict.
|
|
110
|
+
export function watchMemoryBuffer({ reserve = MEMORY_RESERVE, floor = MEMORY_FLOOR, patience = 3 } = {}) {
|
|
111
|
+
let consecutive = 0;
|
|
112
|
+
let peak = 0;
|
|
113
|
+
let samples = 0;
|
|
114
|
+
|
|
115
|
+
return {
|
|
116
|
+
check() {
|
|
117
|
+
const s = memoryStatus();
|
|
118
|
+
samples++;
|
|
119
|
+
peak = Math.max(peak, s.used);
|
|
120
|
+
const tight = s.available < floor || s.fractionUsed > 1 - reserve;
|
|
121
|
+
consecutive = tight ? consecutive + 1 : 0;
|
|
122
|
+
if (s.available < floor / 4) return 'exhausted'; // one sample is enough to believe this
|
|
123
|
+
return consecutive >= patience ? 'tight' : 'ok';
|
|
124
|
+
},
|
|
125
|
+
// A collection, if the process was started with --expose-gc. Never required: the loop is
|
|
126
|
+
// correct without it, and this only buys back headroom that was already garbage.
|
|
127
|
+
//
|
|
128
|
+
// It resets the patience counter because the readings before a collection describe a heap that
|
|
129
|
+
// no longer exists. That reset is why `stillTight` exists and why the caller must use it: on
|
|
130
|
+
// its own it would hand back a clean slate every time, and a loop that collects whenever it is
|
|
131
|
+
// told it is tight would never accumulate the consecutive readings needed to stop.
|
|
132
|
+
collect() {
|
|
133
|
+
if (typeof global.gc === 'function') { global.gc(); consecutive = 0; return true; }
|
|
134
|
+
return false;
|
|
135
|
+
},
|
|
136
|
+
// Is it still tight, right now, with no hysteresis?
|
|
137
|
+
//
|
|
138
|
+
// Hysteresis exists so that one unlucky sample taken just before a collection cannot stop a
|
|
139
|
+
// scan that had room. Immediately after a collection there is nothing left to be unlucky
|
|
140
|
+
// about, so the question is no longer "have several readings agreed" but the simpler one:
|
|
141
|
+
// did the collection actually help. Asking `check()` here instead is what defeated the guard -
|
|
142
|
+
// the reset inside `collect` guaranteed it answered 'ok', so a run started with --expose-gc
|
|
143
|
+
// ran on through sustained pressure until V8 aborted with ineffective mark-compacts. That is
|
|
144
|
+
// not hypothetical: it is how a 100,000-program repository killed a corpus run, on the guard
|
|
145
|
+
// written to stop exactly that.
|
|
146
|
+
stillTight() {
|
|
147
|
+
const s = memoryStatus();
|
|
148
|
+
return s.available < floor || s.fractionUsed > 1 - reserve;
|
|
149
|
+
},
|
|
150
|
+
get peak() { return peak; },
|
|
151
|
+
get samples() { return samples; },
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const STOPPED_BECAUSE = {
|
|
156
|
+
'byte budget': 'the scan reached its source byte budget',
|
|
157
|
+
'memory reserve': 'the scan reached its memory reserve',
|
|
158
|
+
'memory exhausted': 'memory ran out',
|
|
159
|
+
};
|
|
160
|
+
export const stoppedBecause = (stoppedBy) => STOPPED_BECAUSE[stoppedBy] || `the scan stopped (${stoppedBy})`;
|
|
161
|
+
|
|
162
|
+
// The loop every rule set uses instead of `for (const f of files)`.
|
|
163
|
+
//
|
|
164
|
+
// It stops when memory runs short or when a byte budget is spent, and it says which, and it names
|
|
165
|
+
// everything it did not get to. `work` returns the number of bytes it took on, or nothing; a rule
|
|
166
|
+
// set that reads a file should return its length so the byte budget means something.
|
|
167
|
+
export function eachWithinMemory(items, work, {
|
|
168
|
+
maxBytes = Infinity,
|
|
169
|
+
every = 8, // sample the heap this often, not on every item
|
|
170
|
+
watcher = null,
|
|
171
|
+
label = 'scan',
|
|
172
|
+
} = {}) {
|
|
173
|
+
const w = watcher || watchMemoryBuffer();
|
|
174
|
+
const skipped = [];
|
|
175
|
+
let processed = 0;
|
|
176
|
+
let bytes = 0;
|
|
177
|
+
let stoppedBy = null;
|
|
178
|
+
|
|
179
|
+
for (let i = 0; i < items.length; i++) {
|
|
180
|
+
if (stoppedBy) { skipped.push(items[i]); continue; }
|
|
181
|
+
|
|
182
|
+
if (bytes >= maxBytes) { stoppedBy = 'byte budget'; skipped.push(items[i]); continue; }
|
|
183
|
+
|
|
184
|
+
// Checking costs a syscall, so it is sampled rather than continuous - except once the heap is
|
|
185
|
+
// already past the reserve, when every item is checked because the next one may be the last.
|
|
186
|
+
if (i % every === 0 || w.check === undefined || memoryStatus().fractionUsed > 1 - MEMORY_RESERVE) {
|
|
187
|
+
const state = w.check();
|
|
188
|
+
if (state === 'exhausted' || state === 'tight') {
|
|
189
|
+
// Collect, and carry on only if the collection genuinely helped. Asking `check()` here
|
|
190
|
+
// instead asked a counter that `collect` had just reset, so the answer was always 'ok'.
|
|
191
|
+
if (!w.collect() || w.stillTight()) {
|
|
192
|
+
stoppedBy = state === 'exhausted' ? 'memory exhausted' : 'memory reserve';
|
|
193
|
+
skipped.push(items[i]);
|
|
194
|
+
continue;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
const took = work(items[i], i);
|
|
200
|
+
if (typeof took === 'number') bytes += took;
|
|
201
|
+
processed++;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
return {
|
|
205
|
+
label,
|
|
206
|
+
processed,
|
|
207
|
+
bytes,
|
|
208
|
+
skipped,
|
|
209
|
+
complete: skipped.length === 0,
|
|
210
|
+
stoppedBy,
|
|
211
|
+
peakHeapBytes: w.peak,
|
|
212
|
+
// The sentence a report should carry. Written here so every rule set says it the same way.
|
|
213
|
+
note: skipped.length === 0 ? null
|
|
214
|
+
: `${label}: ${skipped.length} of ${items.length} files were not read because ${stoppedBecause(stoppedBy)}. `
|
|
215
|
+
+ 'The findings below are over what was read, and are not a result for this repository as a whole.',
|
|
216
|
+
};
|
|
217
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// Text bound for a report, terminal or log: control and bidirectional characters replaced, length capped.
|
|
3
|
+
export const printable = (s, max = 200) => {
|
|
4
|
+
const t = String(s ?? '').replace(/[\u0000-\u001f\u007f-\u009f\u200e\u200f\u202a-\u202e\u2066-\u2069]/g, '?');
|
|
5
|
+
return t.length > max ? `${t.slice(0, max)}...` : t;
|
|
6
|
+
};
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// Every rule set this tool has, in one list.
|
|
3
|
+
//
|
|
4
|
+
// Adding the opaque set cost nine edits across four files: an import, a spread into ALL_RULES, an
|
|
5
|
+
// entry in RULE_SETS, a branch in scanAll, a line in the tool-name map, two lines in the feed
|
|
6
|
+
// catalogue, the CLI usage string, and the key list test/cli.test.mjs enumerates by hand. Eight of
|
|
7
|
+
// those nine are this file's job, and none of them is a decision - they are the same fact written
|
|
8
|
+
// down nine times, which is eight chances to write it down wrong.
|
|
9
|
+
//
|
|
10
|
+
// One of those chances was taken. The flow set was renamed to `cobolwork-flow` while the map from
|
|
11
|
+
// tool name to report key still said `cobolwork`, so that set's file counts went out under the
|
|
12
|
+
// literal key `undefined` in every report for several commits. lib/scan.mjs grew an ESETNAME throw
|
|
13
|
+
// to make the next rename fail loudly instead. The throw stays - see `reportKey` below - but the
|
|
14
|
+
// map it guarded is gone, because a name derived from one place cannot disagree with itself.
|
|
15
|
+
import { scan as scanFlow, RULES as FLOW_RULES } from '../sets/flow.mjs';
|
|
16
|
+
import { scanCics, CICS_RULES } from '../sets/cics.mjs';
|
|
17
|
+
import { scanHidden, HIDDEN_RULES } from '../sets/hidden.mjs';
|
|
18
|
+
import { scanCopybooks, COPYBOOK_RULES } from '../sets/copybook.mjs';
|
|
19
|
+
import { scanJcl, JCL_RULES } from '../sets/jcl.mjs';
|
|
20
|
+
import { scanBuild, BUILD_RULES } from '../sets/build.mjs';
|
|
21
|
+
import { scanRecon, RECON_RULES } from '../sets/recon.mjs';
|
|
22
|
+
import { scanVendor, VENDOR_RULES } from '../sets/vendor.mjs';
|
|
23
|
+
import { scanOpaque, OPAQUE_RULES } from '../sets/opaque.mjs';
|
|
24
|
+
import { scanWeb, WEB_RULES } from '../sets/web.mjs';
|
|
25
|
+
import { scanPriv, PRIV_RULES } from '../sets/priv.mjs';
|
|
26
|
+
import { scanLog, LOG_RULES } from '../sets/log.mjs';
|
|
27
|
+
import { scanCompile, COMPILE_RULES } from '../sets/compile.mjs';
|
|
28
|
+
import { toolName } from './ruleset.mjs';
|
|
29
|
+
|
|
30
|
+
// The order is the order a scan runs them and the order --only lists them. It is not significant
|
|
31
|
+
// to correctness; it is significant to a report being comparable with yesterday's.
|
|
32
|
+
export const REGISTRY = [
|
|
33
|
+
{ name: 'flow', scan: scanFlow, rules: FLOW_RULES },
|
|
34
|
+
{ name: 'cics', scan: scanCics, rules: CICS_RULES },
|
|
35
|
+
{ name: 'hidden', scan: scanHidden, rules: HIDDEN_RULES },
|
|
36
|
+
{ name: 'copybook', scan: scanCopybooks, rules: COPYBOOK_RULES },
|
|
37
|
+
{ name: 'jcl', scan: scanJcl, rules: JCL_RULES },
|
|
38
|
+
{ name: 'build', scan: scanBuild, rules: BUILD_RULES },
|
|
39
|
+
{ name: 'recon', scan: scanRecon, rules: RECON_RULES },
|
|
40
|
+
{ name: 'vendor', scan: scanVendor, rules: VENDOR_RULES },
|
|
41
|
+
{ name: 'opaque', scan: scanOpaque, rules: OPAQUE_RULES },
|
|
42
|
+
{ name: 'web', scan: scanWeb, rules: WEB_RULES },
|
|
43
|
+
{ name: 'compile', scan: scanCompile, rules: COMPILE_RULES },
|
|
44
|
+
{ name: 'priv', scan: scanPriv, rules: PRIV_RULES },
|
|
45
|
+
{ name: 'log', scan: scanLog, rules: LOG_RULES },
|
|
46
|
+
];
|
|
47
|
+
|
|
48
|
+
// Re-exported from the kernel's ruleset module, which is where it has to live: this file imports
|
|
49
|
+
// every rule set, and every rule set imports that one.
|
|
50
|
+
export { toolName };
|
|
51
|
+
|
|
52
|
+
export const RULE_SETS = REGISTRY.map((s) => s.name);
|
|
53
|
+
|
|
54
|
+
export const ALL_RULES = Object.assign({}, ...REGISTRY.map((s) => s.rules));
|
|
55
|
+
|
|
56
|
+
const BY_TOOL = new Map(REGISTRY.map((s) => [toolName(s.name), s.name]));
|
|
57
|
+
|
|
58
|
+
// What a part's counts are filed under. An unregistered set still fails loudly rather than filing
|
|
59
|
+
// under `undefined`, which is the property the ESETNAME throw was added for and the reason this
|
|
60
|
+
// function exists instead of a bare lookup.
|
|
61
|
+
export function reportKey(tool) {
|
|
62
|
+
const name = BY_TOOL.get(tool);
|
|
63
|
+
if (!name) throw Object.assign(new Error(`rule set ${tool} has no report key`), { code: 'ESETNAME', tool });
|
|
64
|
+
return name;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Rule ids must be unique across sets: ALL_RULES is one flat object, so a duplicate would silently
|
|
68
|
+
// take the last definition and a finding would carry another set's severity.
|
|
69
|
+
export function duplicateRuleIds() {
|
|
70
|
+
const seen = new Map();
|
|
71
|
+
const dupes = [];
|
|
72
|
+
for (const s of REGISTRY) {
|
|
73
|
+
for (const id of Object.keys(s.rules)) {
|
|
74
|
+
if (seen.has(id)) dupes.push(`${id} is declared by both ${seen.get(id)} and ${s.name}`);
|
|
75
|
+
else seen.set(id, s.name);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
return dupes;
|
|
79
|
+
}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// What a rule set says about itself once it has finished looking.
|
|
3
|
+
//
|
|
4
|
+
// Eight sets wrote the same closing five lines: take the guarded loop's account of what it did not
|
|
5
|
+
// reach, stamp the findings, tally them, and assemble a summary. The assembly varied, and none of
|
|
6
|
+
// the variation was a decision - one set set coverageIncomplete inside its stats, another passed it
|
|
7
|
+
// in the returned object, a third ORed it with a flag from the JCL parser, and one had a second
|
|
8
|
+
// return site that skipped the whole thing and shipped a different shape.
|
|
9
|
+
//
|
|
10
|
+
// That last one matters more than tidiness. SECURITY.md treats a report that overstates coverage as
|
|
11
|
+
// a security bug rather than a defect: a file the tool failed to read while still reporting
|
|
12
|
+
// coverageIncomplete: false is the failure this project exists to prevent. Eight hand-written
|
|
13
|
+
// copies of the code that makes that claim is eight places for it to be wrong, and the one that was
|
|
14
|
+
// wrong was found by a test rather than by reading.
|
|
15
|
+
//
|
|
16
|
+
// WHAT THIS IS NOT. The specification proposed a `defineRuleSet({ perFile, finish })` harness that
|
|
17
|
+
// would own each set's control flow and refuse a perFile that returned a parse tree. That is not
|
|
18
|
+
// what this is, and the difference is deliberate. By the time this was written the traversal, the
|
|
19
|
+
// reading, the parse configuration, the sort, the severity stamping and the tally had all moved
|
|
20
|
+
// into lib/kernel/ on their own - so the inversion would have restructured nine working sets to
|
|
21
|
+
// take ownership of what it already had. The ETREELEAK guard went with it: it only guards a
|
|
22
|
+
// perFile that does not exist, and a set that accumulates parse trees in its own closure is
|
|
23
|
+
// something no harness can prevent. What remains is the closing assembly, which is the part that
|
|
24
|
+
// was still duplicated and the part that makes a security claim.
|
|
25
|
+
import { finish } from './findings.mjs';
|
|
26
|
+
|
|
27
|
+
// A set's tool name is its report key with the tool's name in front. It lives here rather than in
|
|
28
|
+
// the registry because the registry imports every rule set, and every rule set imports this: taking
|
|
29
|
+
// the name from there would close the loop, and the error a circular import gives - "cannot access
|
|
30
|
+
// X_RULES before initialization" - names the rule table rather than the cycle that broke it.
|
|
31
|
+
export const toolName = (name) => `cobolwork-${name}`;
|
|
32
|
+
|
|
33
|
+
// `run` is what eachWithinMemory returned, or null for a set that did not traverse - a set refusing
|
|
34
|
+
// to run for want of configuration still has to report in the same shape as one that ran.
|
|
35
|
+
export function report(name, { rules, findings, stats = {}, run = null }) {
|
|
36
|
+
const skipped = run ? run.skipped.length : 0;
|
|
37
|
+
|
|
38
|
+
// The guarded loop's account of what it did not get to. Written here so every set says it the
|
|
39
|
+
// same way, which is the same reason lib/memory.mjs composes the sentence rather than each
|
|
40
|
+
// caller writing its own.
|
|
41
|
+
if (run) {
|
|
42
|
+
if (skipped) stats.filesNotRead = skipped;
|
|
43
|
+
if (run.stoppedBy) stats.stoppedBy = run.stoppedBy;
|
|
44
|
+
if (run.peakHeapBytes) stats.peakHeapBytes = run.peakHeapBytes;
|
|
45
|
+
if (run.note) stats.notRead = run.note;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const { byRule } = finish(rules, findings, name);
|
|
49
|
+
|
|
50
|
+
return {
|
|
51
|
+
tool: toolName(name),
|
|
52
|
+
summary: {
|
|
53
|
+
findings: findings.length,
|
|
54
|
+
byRule,
|
|
55
|
+
...stats,
|
|
56
|
+
// ORed, never overwritten. A set may already know its reading was short for a reason the
|
|
57
|
+
// loop knows nothing about - the JCL parser reporting an INCLUDE it could not resolve, for
|
|
58
|
+
// one - and that claim must survive being combined with this one.
|
|
59
|
+
//
|
|
60
|
+
// A file opened and not understood is a file that did not contribute, exactly like one the
|
|
61
|
+
// loop never reached: the set has no findings from it and cannot say there were none. Only
|
|
62
|
+
// the compile set had joined those up, by hand, and the other nine reported complete coverage
|
|
63
|
+
// over source they could not parse - which is the I4 failure SECURITY.md calls a security bug
|
|
64
|
+
// rather than a defect. lib/sarif.mjs already counted all three as a shortfall, so the two
|
|
65
|
+
// halves of the tool disagreed about what a clean result means.
|
|
66
|
+
coverageIncomplete: stats.coverageIncomplete === true || skipped > 0
|
|
67
|
+
|| (stats.filesUnparsed || 0) > 0 || (stats.filesUnreadable || 0) > 0,
|
|
68
|
+
nosrc: (stats.filesScanned || 0) === 0,
|
|
69
|
+
},
|
|
70
|
+
findings,
|
|
71
|
+
};
|
|
72
|
+
}
|