@portll/cobolwork 0.0.1 → 0.2.76
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/LICENSING.md +93 -0
- package/NOTICE +9 -0
- package/README.md +325 -3
- package/THIRD-PARTY-NOTICES.md +118 -0
- package/bin/cobolwork.mjs +354 -0
- package/lib/advisories.mjs +133 -0
- package/lib/baseline.mjs +154 -0
- package/lib/bms.mjs +453 -0
- package/lib/build.mjs +402 -0
- package/lib/capabilities.mjs +79 -0
- package/lib/cics-commands.mjs +281 -0
- package/lib/compliance.mjs +81 -0
- package/lib/consequence.mjs +139 -0
- package/lib/control.mjs +1515 -0
- package/lib/csd.mjs +77 -0
- package/lib/dataflow.mjs +1506 -0
- package/lib/diff.mjs +344 -0
- package/lib/explain.mjs +145 -0
- package/lib/gate.mjs +383 -0
- package/lib/index.mjs +6 -0
- package/lib/inventory.mjs +79 -0
- package/lib/jcl.mjs +478 -0
- package/lib/kernel/findings.mjs +94 -0
- package/lib/kernel/identity.mjs +216 -0
- package/lib/kernel/memory.mjs +217 -0
- package/lib/kernel/printable.mjs +6 -0
- package/lib/kernel/registry.mjs +79 -0
- package/lib/kernel/ruleset.mjs +72 -0
- package/lib/kernel/source-tree.mjs +159 -0
- package/lib/kev.mjs +27 -0
- package/lib/options.mjs +512 -0
- package/lib/packs.mjs +148 -0
- package/lib/parser.mjs +2055 -0
- package/lib/policy.mjs +163 -0
- package/lib/precompile-cics.mjs +169 -0
- package/lib/precompile.mjs +544 -0
- package/lib/reach.mjs +122 -0
- package/lib/revision.json +1 -0
- package/lib/revision.mjs +89 -0
- package/lib/sarif.mjs +222 -0
- package/lib/scan.mjs +272 -0
- package/lib/sets/build.mjs +234 -0
- package/lib/sets/cics.mjs +306 -0
- package/lib/sets/compile.mjs +187 -0
- package/lib/sets/copybook.mjs +174 -0
- package/lib/sets/flow.mjs +487 -0
- package/lib/sets/hidden.mjs +216 -0
- package/lib/sets/jcl.mjs +440 -0
- package/lib/sets/log.mjs +406 -0
- package/lib/sets/opaque.mjs +102 -0
- package/lib/sets/priv.mjs +322 -0
- package/lib/sets/recon.mjs +267 -0
- package/lib/sets/vendor.mjs +117 -0
- package/lib/sets/web.mjs +327 -0
- package/lib/site.mjs +164 -0
- package/lib/sources.mjs +156 -0
- package/lib/tui/app.mjs +325 -0
- package/lib/tui/keys.mjs +39 -0
- package/lib/tui/model.mjs +96 -0
- package/lib/tui/run.mjs +38 -0
- package/lib/tui/screen.mjs +59 -0
- package/lib/tui/terminal.mjs +46 -0
- package/lib/utilities.mjs +296 -0
- package/lib/version.mjs +15 -0
- package/lib/words.mjs +318 -0
- package/package.json +45 -6
- package/rules/advisories.json +264 -0
- package/rules/compliance-dora.json +2151 -0
- package/rules/compliance-ffiec.json +2134 -0
- package/rules/compliance-nist80053.json +2134 -0
- package/rules/gitleaks-mainframe.toml +57 -0
- package/rules/kev-ids.json +1729 -0
- package/rules/packs/broadcom.json +124 -0
- package/rules/packs/connectdirect.json +116 -0
- package/rules/packs/controlm.json +114 -0
- package/rules/system-layouts.json +28 -0
- package/schema/cobolwork-coverage.schema.json +65 -0
- package/schema/cobolwork.policy.schema.json +54 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// Whether a program in the tree is one a compiler would accept: every name it uses is declared by
|
|
3
|
+
// the program, by a copybook it includes, or by the compiler or a translator. Enterprise COBOL stops
|
|
4
|
+
// a program that uses anything else with IGYPS2121-S. Source that cannot compile is not the source
|
|
5
|
+
// of anything that runs, and a change carrying it was merged without a build, which is the mark of
|
|
6
|
+
// generated "modernisation" code nobody compiled.
|
|
7
|
+
//
|
|
8
|
+
// A name is only undefined when everything the program copies was found. With a copybook missing
|
|
9
|
+
// the name is most likely declared in it, so the program is counted as undecided rather than
|
|
10
|
+
// reported: the missing member is the thing to act on, and the inventory already names it. The
|
|
11
|
+
// same holds wherever the parse is not the whole program, and the count says why for each.
|
|
12
|
+
import { basename, extname } from 'node:path';
|
|
13
|
+
import { inScope, isProgram, relPath, readSource } from '../sources.mjs';
|
|
14
|
+
import { report } from '../kernel/ruleset.mjs';
|
|
15
|
+
import { treeFor, noteUnread, noteUnparsed } from '../kernel/source-tree.mjs';
|
|
16
|
+
import { eachWithinMemory } from '../kernel/memory.mjs';
|
|
17
|
+
|
|
18
|
+
// construct: no input has to reach it. med: certain once every copybook is found, and it means the
|
|
19
|
+
// tree does not hold what runs, but it is not exploitable in itself. CWE-1127: MITRE discourages
|
|
20
|
+
// mapping to CWE-710, a pillar, and of its children this is the build that lets errors through, of
|
|
21
|
+
// which source nobody compiled is the limit. CWE-1164 is code that runs to no effect; this cannot run.
|
|
22
|
+
export const COMPILE_RULES = {
|
|
23
|
+
'compile-undefined-name': {
|
|
24
|
+
sev: 'med', evidence: 'construct', cwe: 'CWE-1127',
|
|
25
|
+
text: 'A program uses names that nothing it declares or copies defines, so it cannot compile',
|
|
26
|
+
impact: 'The program uses a name nothing it declares or copies defines, so Enterprise COBOL stops it with IGYPS2121-S: the source in the tree is not the source of anything that runs',
|
|
27
|
+
remedy: 'Define the missing name, or add the copybook that declares it, so the program compiles; if it is generated code, compile it before relying on it',
|
|
28
|
+
},
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
// These declare only DFH names or the SQLCA's own, which the parser already takes as supplied.
|
|
32
|
+
const COVERED_SYSTEM = /^(DFHAID|DFHBMSCA|DFHEIBLK|SQLCA)$/i;
|
|
33
|
+
const PARSE_TROUBLE = new Set(['unterminated-exec', 'unterminated-literal', 'copy-without-period']);
|
|
34
|
+
// An Enterprise COBOL listing kept under a program's extension, whose page headers read as code.
|
|
35
|
+
const LISTING = /^[01 -]?PP \d{4}-[A-Z0-9]{3} IBM /m;
|
|
36
|
+
const SHOWN = 5;
|
|
37
|
+
|
|
38
|
+
const nameOf = (p) => basename(p, extname(p)).toUpperCase();
|
|
39
|
+
|
|
40
|
+
function whyUndecided(r, src) {
|
|
41
|
+
if (r.copies.some((c) => c.status === 'expansion-limit')) return 'its COPY or REPLACE statements expand past what one parse reads';
|
|
42
|
+
if (r.copies.some((c) => c.status !== 'resolved' && !(c.status === 'system' && COVERED_SYSTEM.test(c.name)))) return 'a copybook it includes is not in the tree';
|
|
43
|
+
// Panvalet's ++INCLUDE and Librarian's -INC, which a library manager expands before the compiler
|
|
44
|
+
// runs, and an INCLUDE or COPY somewhere the parser does not read one as a statement.
|
|
45
|
+
if (/^.{0,6}(?:-INC|\+\+INCLUDE)\s/im.test(src) || r.programs.some((p) => p.refs.some((x) => x.tok.u === 'INCLUDE' || x.tok.u === 'COPY'))) return 'it includes source the parser does not expand';
|
|
46
|
+
if (r.diags.some((d) => PARSE_TROUBLE.has(d.kind))) return 'the parse did not read it cleanly';
|
|
47
|
+
// A communication description declares its queue and status names in clauses the parser skips.
|
|
48
|
+
if (r.programs.some((p) => p.diags.some((d) => d.kind === 'unrecognised-data-sentence') || (p.fds || []).some((fd) => fd.kind === 'CD'))) return 'the parse did not recognise all of its data division';
|
|
49
|
+
return null;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// Where a text declares a name: after a level number, FD, SD, RD, CD, SELECT, INDEXED BY or a
|
|
53
|
+
// directive's CONSTANT, or as a paragraph or section header in Area A. And the words of a COPY
|
|
54
|
+
// REPLACING or a REPLACE, which is where a name comes from when the parser did not replace.
|
|
55
|
+
const DECLARATION = /(?:^|[\s.])(?:0?[1-9]|[1-4][0-9]|66|77|78|88|FD|SD|RD|CD|SELECT(?:\s+OPTIONAL)?|INDEXED\s+BY|CONSTANT)\s+([A-Za-z0-9$#@_][A-Za-z0-9$#@_-]*)/gim;
|
|
56
|
+
const HEADER = /^(?:.{0,6}[ \t]{1,4}|[ \t]{0,4})([A-Za-z0-9][A-Za-z0-9-]*)(?:[ \t]+SECTION(?:[ \t]+\d+)?)?[ \t]*\.[ \t]*\r?$/gm;
|
|
57
|
+
const COPY_STATEMENT = /\bCOPY\b([\s\S]*?)(?:\.\s|\.$)/gim;
|
|
58
|
+
const REPLACE_STATEMENT = /\bREPLACE\b([\s\S]*?)(?:\.\s|\.$)/gim;
|
|
59
|
+
|
|
60
|
+
// A header declares a paragraph only where the sentence before it has ended: after a line of code
|
|
61
|
+
// ending in a period, or with no code before it. Without the period the name is read as part of the
|
|
62
|
+
// statement before, and the compiler reports it undefined.
|
|
63
|
+
function sentenceEndsBefore(text, at) {
|
|
64
|
+
const lines = text.slice(0, at).split(/\r?\n/);
|
|
65
|
+
lines.pop();
|
|
66
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
67
|
+
const l = lines[i];
|
|
68
|
+
if (/^.{6}[*/]/.test(l) || /^\s*\*>/.test(l)) continue;
|
|
69
|
+
const code = l.slice(0, 72).replace(/\*>.*$/, '').replace(/^[0-9 ]{6}(?=[ \-D])/, '').trim();
|
|
70
|
+
if (code) return code.endsWith('.');
|
|
71
|
+
}
|
|
72
|
+
return true;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function declaredIn(text, names, into) {
|
|
76
|
+
const note = (w) => { const u = w.toUpperCase(); if (names.has(u)) into.add(u); };
|
|
77
|
+
for (const re of [DECLARATION, HEADER]) {
|
|
78
|
+
re.lastIndex = 0;
|
|
79
|
+
let m;
|
|
80
|
+
while ((m = re.exec(text))) if (re !== HEADER || sentenceEndsBefore(text, m.index)) note(m[1]);
|
|
81
|
+
}
|
|
82
|
+
for (const re of [COPY_STATEMENT, REPLACE_STATEMENT]) {
|
|
83
|
+
re.lastIndex = 0;
|
|
84
|
+
let m;
|
|
85
|
+
while ((m = re.exec(text))) {
|
|
86
|
+
if (re === COPY_STATEMENT && !/\bREPLACING\b/i.test(m[1])) continue;
|
|
87
|
+
for (const w of m[1].match(/[A-Za-z0-9][A-Za-z0-9-]*/g) || []) note(w);
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
return into;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// A reference cut at column 72 names a word the source does not hold.
|
|
94
|
+
function truncated(ref, lines) {
|
|
95
|
+
const line = (lines.get(ref.file) || [])[ref.line - 1] || '';
|
|
96
|
+
const re = new RegExp(`(?<![A-Za-z0-9-])${ref.name.replace(/[$#@]/g, '\\$&')}`, 'gi');
|
|
97
|
+
let whole = false;
|
|
98
|
+
let any = false;
|
|
99
|
+
let m;
|
|
100
|
+
while ((m = re.exec(line))) { any = true; if (!/[A-Za-z0-9-]/.test(line[m.index + ref.name.length] || '')) whole = true; }
|
|
101
|
+
return any && !whole;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function scanCompile(root, opts = {}) {
|
|
105
|
+
const tree = treeFor(root, opts);
|
|
106
|
+
const files = tree.list().filter(isProgram).filter(inScope(opts));
|
|
107
|
+
const findings = [];
|
|
108
|
+
const stats = { filesScanned: 0, filesUnreadable: 0, programsChecked: 0, programsUndecided: 0 };
|
|
109
|
+
const because = {};
|
|
110
|
+
const undecided = [];
|
|
111
|
+
const undecide = (f, id, why, refs) => {
|
|
112
|
+
stats.programsUndecided++;
|
|
113
|
+
because[why] = (because[why] || 0) + 1;
|
|
114
|
+
undecided.push(`${tree.rel(f)} (${id}): ${why}: ${[...new Set(refs.map((x) => x.name))].slice(0, SHOWN).join(', ')}`);
|
|
115
|
+
};
|
|
116
|
+
const textOf = (p) => (tree.contains(p) ? tree.text(p) : readSource(p)).text;
|
|
117
|
+
const candidates = [];
|
|
118
|
+
|
|
119
|
+
const run = eachWithinMemory(files, (f) => {
|
|
120
|
+
let src;
|
|
121
|
+
try { src = tree.text(f).text; } catch (e) { noteUnread(stats, tree, f, e); return 0; }
|
|
122
|
+
if (LISTING.test(src)) { stats.filesScanned++; stats.listingsSkipped = (stats.listingsSkipped || 0) + 1; return src.length; }
|
|
123
|
+
let r;
|
|
124
|
+
try { r = tree.parse(f, src); } catch (e) { noteUnparsed(stats, tree, f, e); return src.length; }
|
|
125
|
+
stats.filesScanned++;
|
|
126
|
+
const blocked = whyUndecided(r, src);
|
|
127
|
+
const copied = [...new Set(r.copies.filter((c) => c.path).map((c) => c.path))];
|
|
128
|
+
for (const p of r.programs) {
|
|
129
|
+
// A class or a method has no PROGRAM-ID, and the data its methods use belongs to no program here.
|
|
130
|
+
if (!p.id || p.diags.some((d) => d.kind === 'no-procedure-division')) continue;
|
|
131
|
+
stats.programsChecked++;
|
|
132
|
+
if (!p.unresolvedRefs.length) continue;
|
|
133
|
+
if (blocked) { undecide(f, p.id, blocked, p.unresolvedRefs); continue; }
|
|
134
|
+
const names = new Set(p.unresolvedRefs.map((x) => x.name));
|
|
135
|
+
const texts = new Map([[f, src]]);
|
|
136
|
+
for (const q of copied) { try { texts.set(q, textOf(q)); } catch { /* the parse read it once; a second failure changes nothing */ } }
|
|
137
|
+
const seen = new Set();
|
|
138
|
+
for (const text of texts.values()) declaredIn(text, names, seen);
|
|
139
|
+
const lines = new Map([...texts].map(([q, t]) => [q, t.split(/\r?\n/)]));
|
|
140
|
+
const left = p.unresolvedRefs.filter((x) => !seen.has(x.name) && !truncated(x, lines));
|
|
141
|
+
if (!left.length) { undecide(f, p.id, 'the parse missed a declaration the text holds', p.unresolvedRefs); continue; }
|
|
142
|
+
candidates.push({ file: f, id: p.id, line: p.line, refs: left, copied: new Set(r.copies.map((c) => nameOf(c.name))) });
|
|
143
|
+
}
|
|
144
|
+
r = null;
|
|
145
|
+
return src.length;
|
|
146
|
+
}, { label: 'compile', maxBytes: opts.maxSourceBytes ?? Infinity });
|
|
147
|
+
|
|
148
|
+
// Another file answering to a name the program copies may be the one its build finds.
|
|
149
|
+
const wanted = new Set(candidates.flatMap((c) => [...c.copied]));
|
|
150
|
+
const answering = new Map();
|
|
151
|
+
if (wanted.size) {
|
|
152
|
+
for (const p of tree.list()) {
|
|
153
|
+
const n = nameOf(p);
|
|
154
|
+
if (!wanted.has(n)) continue;
|
|
155
|
+
if (!answering.has(n)) answering.set(n, []);
|
|
156
|
+
answering.get(n).push(p);
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
for (const c of candidates) {
|
|
160
|
+
const names = new Set(c.refs.map((x) => x.name));
|
|
161
|
+
const seen = new Set();
|
|
162
|
+
for (const n of c.copied) for (const p of answering.get(n) || []) { try { declaredIn(tree.text(p).text, names, seen); } catch { /* unread declares nothing */ } }
|
|
163
|
+
const left = c.refs.filter((x) => !seen.has(x.name));
|
|
164
|
+
if (!left.length) { undecide(c.file, c.id, 'another copybook of a name it copies declares them', c.refs); continue; }
|
|
165
|
+
const firsts = new Map();
|
|
166
|
+
for (const x of left) if (!firsts.has(x.name)) firsts.set(x.name, x);
|
|
167
|
+
const own = left.find((x) => x.file === c.file);
|
|
168
|
+
const at = (x) => (x.file === c.file ? `line ${x.line}` : `${relPath(root, x.file)}:${x.line}`);
|
|
169
|
+
const shown = [...firsts.values()].slice(0, SHOWN).map((x) => `${x.name} (${at(x)})`).join(', ');
|
|
170
|
+
const more = firsts.size > SHOWN ? `, and ${firsts.size - SHOWN} more` : '';
|
|
171
|
+
findings.push({
|
|
172
|
+
rule: 'compile-undefined-name', path: relPath(root, c.file), line: own ? own.line : c.line || 1, program: c.id,
|
|
173
|
+
names: [...firsts.keys()],
|
|
174
|
+
detail: `${c.id} uses ${firsts.size} name(s) that nothing it declares or copies defines, so it cannot compile: ${shown}${more}`,
|
|
175
|
+
related: [...firsts.values()].map((x) => ({ path: relPath(root, x.file), line: x.line, detail: `${x.name} is used here and declared nowhere the program can reach` })),
|
|
176
|
+
});
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
if (stats.programsUndecided) {
|
|
180
|
+
stats.undecidedBecause = because;
|
|
181
|
+
stats.undecided = undecided;
|
|
182
|
+
stats.coverageIncomplete = true;
|
|
183
|
+
stats.readInPart = `${stats.programsUndecided} program(s) use names no declaration the parse read defines, and the names may be declared where it did not reach: ${Object.entries(because).map(([k, v]) => `${v} because ${k}`).join('; ')}`;
|
|
184
|
+
}
|
|
185
|
+
if (stats.filesUnparsed) stats.coverageIncomplete = true;
|
|
186
|
+
return report('compile', { rules: COMPILE_RULES, findings, stats, run });
|
|
187
|
+
}
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
import { readFileSync } from 'node:fs';
|
|
3
|
+
import { basename, dirname, extname } from 'node:path';
|
|
4
|
+
import { detectFormat, parseSource } from '../parser.mjs';
|
|
5
|
+
import { inScope, isCopybook, isProgram, relPath } from '../sources.mjs';
|
|
6
|
+
import { finish } from '../kernel/findings.mjs';
|
|
7
|
+
import { report } from '../kernel/ruleset.mjs';
|
|
8
|
+
import { treeFor, noteUnread, noteUnparsed } from '../kernel/source-tree.mjs';
|
|
9
|
+
import { eachWithinMemory } from '../kernel/memory.mjs';
|
|
10
|
+
|
|
11
|
+
// A copybook is resolved by name along a search path: the program's own directory first, then the
|
|
12
|
+
// include directories, then the system library. Two files answering to one name mean the record
|
|
13
|
+
// layout a program compiles against depends on where the program sits, and a copybook named like a
|
|
14
|
+
// system one is found before the system's own. Both are how a single added file changes the layout
|
|
15
|
+
// of every program that COPYs the name without touching any of them.
|
|
16
|
+
|
|
17
|
+
export const COPYBOOK_RULES = {
|
|
18
|
+
'copybook-shadows-system': {
|
|
19
|
+
sev: 'med', evidence: 'tampering', cwe: 'CWE-427',
|
|
20
|
+
text: 'A copybook in the repository has the name of a system copybook, and is found before it',
|
|
21
|
+
impact: "A repository copybook answers to the name of a system copybook (SQLCA, DFHAID, ...), so the search path finds it before the system's own and a program compiles against the wrong layout",
|
|
22
|
+
remedy: 'Rename or remove the repository copy; let the system copybook resolve from its own library',
|
|
23
|
+
},
|
|
24
|
+
'copybook-shadowed': {
|
|
25
|
+
sev: 'med', evidence: 'tampering', cwe: 'CWE-427',
|
|
26
|
+
text: 'Two copybooks with different contents answer to the same name, and programs resolve it to different files',
|
|
27
|
+
impact: 'Two copybooks answer to one name with different layouts, so which one a program gets depends on its search path, not its source',
|
|
28
|
+
remedy: "Give the two copybooks distinct names, or make every program's search path resolve the intended one unambiguously",
|
|
29
|
+
},
|
|
30
|
+
};
|
|
31
|
+
|
|
32
|
+
const SYSTEM_NAME = /^(DFHAID|DFHBMSCA|DFHEIBLK|DFHCOMMAREA|DFHRESP|SQLCA|SQLDA|DSNHLI|CEEIGZCT|CMQV|CMQODV|CMQMDV|CMQPMOV|CMQGMOV|EZACICSO)$/i;
|
|
33
|
+
|
|
34
|
+
// Layout text, not bytes. In fixed format the sequence and identification areas are not part of the
|
|
35
|
+
// layout, and two copies differing only in change tags are the same copybook. In free format they
|
|
36
|
+
// are ordinary text, and cutting at column 72 hid a PIC clause that sat past it.
|
|
37
|
+
function layout(src) {
|
|
38
|
+
const fixed = detectFormat(src) === 'fixed';
|
|
39
|
+
return src.split(/\r?\n/)
|
|
40
|
+
.map(l => (fixed ? l.slice(6, 72) : l).trimEnd())
|
|
41
|
+
.filter(l => l.trim() && !/^[*/]/.test(l.trimStart()))
|
|
42
|
+
.join('\n')
|
|
43
|
+
.toUpperCase();
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
const nameOf = (p) => basename(p, extname(p)).toUpperCase();
|
|
47
|
+
|
|
48
|
+
const SYSTEM_LAYOUTS = JSON.parse(readFileSync(new URL('../../rules/system-layouts.json', import.meta.url), 'utf8'));
|
|
49
|
+
|
|
50
|
+
// Where a repository copy of a system copybook first departs from the system's layout, compared
|
|
51
|
+
// field by elementary field; null when every field sits where the system puts it. Undefined when no
|
|
52
|
+
// layout is held for the name or the copy does not parse.
|
|
53
|
+
export function systemLayoutDeparture(name, text) {
|
|
54
|
+
const want = SYSTEM_LAYOUTS[String(name).toUpperCase()];
|
|
55
|
+
if (!want || !want.fields) return undefined;
|
|
56
|
+
let items;
|
|
57
|
+
try {
|
|
58
|
+
const wrapped = [' IDENTIFICATION DIVISION.', ' PROGRAM-ID. LAYOUT.', ' DATA DIVISION.', ' WORKING-STORAGE SECTION.', text, ' PROCEDURE DIVISION.', ' GOBACK.'].join('\n');
|
|
59
|
+
items = parseSource(wrapped, `${name}.cpy`, { format: 'auto' }).programs[0]?.items;
|
|
60
|
+
} catch { return undefined; }
|
|
61
|
+
if (!items || !items.length) return undefined;
|
|
62
|
+
const byName = new Map(items.map((it) => [String(it.name).toUpperCase(), it]));
|
|
63
|
+
for (const [field, offset, size, occurs] of want.fields) {
|
|
64
|
+
const it = byName.get(field);
|
|
65
|
+
if (!it) return `${field} is missing`;
|
|
66
|
+
if (it.offset !== offset || it.size !== size || (it.occurs || 1) !== occurs) return `${field} is ${it.size} byte(s) at offset ${it.offset}${(it.occurs || 1) > 1 ? ` times ${it.occurs}` : ''}, where the system's is ${size} at ${offset}${occurs > 1 ? ` times ${occurs}` : ''}`;
|
|
67
|
+
}
|
|
68
|
+
const top = items.find((it) => it.level === 1);
|
|
69
|
+
if (top && top.size !== want.length) return `the record is ${top.size} bytes, where the system's is ${want.length}`;
|
|
70
|
+
return null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export function scanCopybooks(root, opts = {}) {
|
|
74
|
+
const tree = treeFor(root, opts);
|
|
75
|
+
const all = tree.list().filter(inScope(opts));
|
|
76
|
+
const findings = [];
|
|
77
|
+
const stats = { filesScanned: 0, copybookFiles: 0, namesShadowed: 0, filesUnreadable: 0 };
|
|
78
|
+
|
|
79
|
+
const copies = new Map();
|
|
80
|
+
for (const p of all) {
|
|
81
|
+
if (!isCopybook(p)) continue;
|
|
82
|
+
stats.copybookFiles++;
|
|
83
|
+
stats.filesScanned++;
|
|
84
|
+
const n = nameOf(p);
|
|
85
|
+
if (!copies.has(n)) copies.set(n, []);
|
|
86
|
+
copies.get(n).push(p);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Reading the COPY statements out of the text found only what a program names directly, counted a
|
|
90
|
+
// commented-out line as a use, and never saw a copybook reached through another copybook. The
|
|
91
|
+
// parser resolves all of that, so the parse is the answer — paid for once, and only when this
|
|
92
|
+
// tree holds something worth asking about.
|
|
93
|
+
const candidates = new Set([...copies].filter(([n, f]) => f.length > 1 || SYSTEM_NAME.test(n)).map(([n]) => n));
|
|
94
|
+
const referrers = new Map();
|
|
95
|
+
const resolved = new Map();
|
|
96
|
+
// This is the expensive half: it parses every program in the tree, and a parse tree is the kind
|
|
97
|
+
// of structure a byte budget does not bound. Only what each parse yields is kept - a name, a
|
|
98
|
+
// referrer, a resolved path - and the tree is released, so the working set stays flat. The guard
|
|
99
|
+
// is what stops the reading before the heap does.
|
|
100
|
+
let run = { skipped: [], stoppedBy: null, peakHeapBytes: 0, note: null };
|
|
101
|
+
if (candidates.size) {
|
|
102
|
+
run = eachWithinMemory(all.filter(isProgram), (p) => {
|
|
103
|
+
let src;
|
|
104
|
+
try { src = tree.text(p).text; } catch (e) { noteUnread(stats, tree, p, e); return 0; }
|
|
105
|
+
let r;
|
|
106
|
+
try {
|
|
107
|
+
r = tree.parse(p, src);
|
|
108
|
+
} catch (e) { noteUnparsed(stats, tree, p, e); return src.length; }
|
|
109
|
+
stats.filesScanned++;
|
|
110
|
+
for (const c of r.copies) {
|
|
111
|
+
const n = nameOf(c.name);
|
|
112
|
+
if (!candidates.has(n)) continue;
|
|
113
|
+
if (!referrers.has(n)) referrers.set(n, new Set());
|
|
114
|
+
referrers.get(n).add(p);
|
|
115
|
+
if (c.path) {
|
|
116
|
+
if (!resolved.has(n)) resolved.set(n, new Map());
|
|
117
|
+
resolved.get(n).set(p, c.path);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
r = null; // the tree is not needed past this point
|
|
121
|
+
return src.length;
|
|
122
|
+
}, { label: 'copybook', maxBytes: opts.maxSourceBytes ?? Infinity });
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
for (const [n, files] of copies) {
|
|
126
|
+
if (SYSTEM_NAME.test(n)) {
|
|
127
|
+
for (const f of files) {
|
|
128
|
+
const users = [...(referrers.get(n) || [])].map(p => relPath(root, p));
|
|
129
|
+
let departure;
|
|
130
|
+
try { departure = systemLayoutDeparture(n, tree.text(f).text); } catch (e) { noteUnread(stats, tree, f, e); }
|
|
131
|
+
const layout = departure === null ? `; its layout is the system's, field for field`
|
|
132
|
+
: departure ? `; its layout departs from the system's: ${departure}` : '';
|
|
133
|
+
findings.push({ rule: 'copybook-shadows-system', path: relPath(root, f), line: 1, name: n, users: users.length,
|
|
134
|
+
...(departure === null ? { layoutMatchesSystem: true } : {}),
|
|
135
|
+
detail: `${n} is a system copybook name; this file is found before the system library by ${users.length} program(s)${users.length ? `: ${users.slice(0, 5).join(', ')}${users.length > 5 ? ', …' : ''}` : ''}${layout}` });
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
if (files.length < 2) continue;
|
|
139
|
+
const layouts = new Map();
|
|
140
|
+
for (const f of files) {
|
|
141
|
+
let text;
|
|
142
|
+
try { text = layout(tree.text(f).text); } catch (e) { noteUnread(stats, tree, f, e); continue; }
|
|
143
|
+
if (!layouts.has(text)) layouts.set(text, []);
|
|
144
|
+
layouts.get(text).push(f);
|
|
145
|
+
}
|
|
146
|
+
if (layouts.size < 2) continue;
|
|
147
|
+
|
|
148
|
+
const resolvedBy = resolved.get(n) || new Map();
|
|
149
|
+
const targets = new Set(resolvedBy.values());
|
|
150
|
+
stats.namesShadowed++;
|
|
151
|
+
// Differing copies that every program resolves to one file are a latent trap, not a live one:
|
|
152
|
+
// reported, but the detail says which. Programs split across files are the live case.
|
|
153
|
+
const split = targets.size > 1;
|
|
154
|
+
const perFile = files.map(f => `${relPath(root, f)} (${[...resolvedBy.values()].filter(v => v === f).length} program(s))`);
|
|
155
|
+
findings.push({ rule: 'copybook-shadowed', path: relPath(root, files[0]), line: 1, name: n, split,
|
|
156
|
+
detail: `${files.length} files named ${n} with ${layouts.size} different layouts; ${split ? 'programs resolve it to different files' : 'every program that COPYs it resolves to the same file today'}: ${perFile.join(', ')}`,
|
|
157
|
+
related: files.slice(1).map(f => ({ path: relPath(root, f), line: 1, detail: `another copybook named ${n}` })) });
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// This set decides one severity for itself, so it sets it before the shared stamping runs, which
|
|
161
|
+
// fills only what a finding does not already carry.
|
|
162
|
+
//
|
|
163
|
+
// Vendoring DFHAID or SQLCA so a program compiles off the mainframe is common and legitimate;
|
|
164
|
+
// measured on the 200-repository corpus, every system-named copy found was one. The finding
|
|
165
|
+
// stays, because the vendored copy is what those programs are built against, but it is not high.
|
|
166
|
+
for (const f of findings) {
|
|
167
|
+
const latent = (f.rule === 'copybook-shadowed' && !f.split) || (f.rule === 'copybook-shadows-system' && (!f.users || f.layoutMatchesSystem));
|
|
168
|
+
if (latent) f.sev = 'low';
|
|
169
|
+
}
|
|
170
|
+
const { byRule } = finish(COPYBOOK_RULES, findings, 'copybook');
|
|
171
|
+
// A program the scan never parsed is a program whose COPY statements nobody read, so a name
|
|
172
|
+
// that looks unshadowed may only look that way.
|
|
173
|
+
return report('copybook', { rules: COPYBOOK_RULES, findings, stats, run });
|
|
174
|
+
}
|