@portll/cobolwork 0.2.140 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -2
- package/THIRD-PARTY-NOTICES.md +3 -2
- package/bin/cobolwork.mjs +16 -1
- package/lib/baseline.mjs +8 -6
- package/lib/bms.mjs +1 -1
- package/lib/capabilities.mjs +2 -1
- package/lib/cards.mjs +66 -0
- package/lib/compliance.mjs +6 -5
- package/lib/control.mjs +402 -19
- package/lib/dataflow.mjs +33 -9
- package/lib/diff.mjs +65 -89
- package/lib/equivalence.mjs +30 -12
- package/lib/evidence/record.mjs +3 -1
- package/lib/evidence/slsa.mjs +17 -2
- package/lib/execution.mjs +77 -0
- package/lib/explain.mjs +1 -1
- package/lib/ftp.mjs +1 -1
- package/lib/inventory.mjs +8 -9
- package/lib/jcl.mjs +3 -48
- package/lib/kernel/git.mjs +60 -0
- package/lib/kernel/identity.mjs +7 -6
- package/lib/kernel/source-tree.mjs +204 -34
- package/lib/parser.mjs +31 -15
- package/lib/revision.json +1 -1
- package/lib/sbom.mjs +33 -9
- package/lib/scan.mjs +11 -4
- package/lib/sets/build.mjs +1 -1
- package/lib/sets/cics.mjs +27 -3
- package/lib/sets/copybook.mjs +1 -1
- package/lib/sets/hidden.mjs +1 -1
- package/lib/sets/jcl.mjs +1 -1
- package/lib/sets/priv.mjs +1 -1
- package/lib/sets/recon.mjs +2 -2
- package/lib/sets/semantics.mjs +1 -1
- package/lib/sets/vendor.mjs +1 -1
- package/lib/site.mjs +17 -7
- package/lib/sources.mjs +82 -38
- package/lib/utilities.mjs +1 -1
- package/package.json +2 -2
- package/rules/compliance-cobit2019.json +2438 -0
- package/lib/card.mjs +0 -18
package/lib/inventory.mjs
CHANGED
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
4
|
-
import { inScope, isCopybook, isJcl, isProgram, readSource, relPath } from './sources.mjs';
|
|
2
|
+
import { detectFormat } from './parser.mjs';
|
|
3
|
+
import { inScope, isCopybook, isJcl, isProgram, relPath } from './sources.mjs';
|
|
5
4
|
import { SCHEMA_VERSION, TOOL_VERSION } from './version.mjs';
|
|
6
5
|
import { treeFor } from './kernel/source-tree.mjs';
|
|
7
6
|
|
|
@@ -36,19 +35,19 @@ export function inventory(root, opts = {}) {
|
|
|
36
35
|
out.summary.programFiles++;
|
|
37
36
|
let src;
|
|
38
37
|
try {
|
|
39
|
-
const s =
|
|
38
|
+
const s = tree.text(f);
|
|
40
39
|
src = s.text;
|
|
41
|
-
if (s.encoding === 'ebcdic') { out.summary.filesEbcdic++; out.ebcdic.push(
|
|
42
|
-
} catch (e) { out.summary.filesUnreadable++; out.unreadable.push(`${
|
|
40
|
+
if (s.encoding === 'ebcdic') { out.summary.filesEbcdic++; out.ebcdic.push(tree.rel(f)); }
|
|
41
|
+
} catch (e) { out.summary.filesUnreadable++; out.unreadable.push(`${tree.rel(f)}: ${e.code || e.name}`); continue; }
|
|
43
42
|
// A file with a program extension and no program in it is usually a copybook named .cbl; it is
|
|
44
43
|
// counted so the difference between program files and programs read is accounted for.
|
|
45
44
|
if (!/PROCEDURE\s+DIVISION|PROGRAM-ID/i.test(src)) { out.summary.notPrograms++; continue; }
|
|
46
45
|
let res;
|
|
47
46
|
try {
|
|
48
|
-
res =
|
|
47
|
+
res = tree.parse(f, src);
|
|
49
48
|
} catch (e) {
|
|
50
49
|
out.summary.filesUnreadable++;
|
|
51
|
-
out.unreadable.push(`${
|
|
50
|
+
out.unreadable.push(`${tree.rel(f)}: ${e.code || e.name}`);
|
|
52
51
|
continue;
|
|
53
52
|
}
|
|
54
53
|
out.summary.filesScanned++;
|
|
@@ -60,7 +59,7 @@ export function inventory(root, opts = {}) {
|
|
|
60
59
|
if (/EXEC\s+DLI/i.test(src)) out.summary.withDli++;
|
|
61
60
|
for (const c of res.copies) {
|
|
62
61
|
if (c.status === 'resolved') out.summary.copiesResolved++;
|
|
63
|
-
else if (String(c.status).startsWith('refused')) { out.summary.copiesRefused++; out.refusedCopies.push({ name: c.name, from:
|
|
62
|
+
else if (String(c.status).startsWith('refused')) { out.summary.copiesRefused++; out.refusedCopies.push({ name: c.name, from: tree.rel(c.file), line: c.line, why: c.status }); }
|
|
64
63
|
// Before the system names, which a copybook in the tree is free to take.
|
|
65
64
|
else if (c.status === 'expansion-limit') out.summary.copiesOverLimit = (out.summary.copiesOverLimit || 0) + 1;
|
|
66
65
|
else if (c.status === 'system' || SYSTEM_COPY.test(c.name)) out.summary.copiesSystem++;
|
package/lib/jcl.mjs
CHANGED
|
@@ -13,7 +13,9 @@
|
|
|
13
13
|
// there.
|
|
14
14
|
import { readSource } from './sources.mjs';
|
|
15
15
|
import { copiesOf, tsoCommands } from './utilities.mjs';
|
|
16
|
-
import { statementCard } from './
|
|
16
|
+
import { statementCard, parseOperands } from './cards.mjs';
|
|
17
|
+
|
|
18
|
+
export { splitOperands, parseOperands } from './cards.mjs';
|
|
17
19
|
|
|
18
20
|
// The operations a statement may carry. Anything else in the operation field is a statement this
|
|
19
21
|
// reader does not know, which is a diagnostic rather than a silent skip.
|
|
@@ -38,53 +40,6 @@ const SYSTEM_SYMBOLS = new Set(`SYSUID SYSALVL SYSCLONE SYSNAME SYSOSLVL SYSPLEX
|
|
|
38
40
|
YYMMDD LYYMMDD HHMMSS LHHMMSS DAY HR MIN SEC JDAY MON YR2 YR4 WDAY
|
|
39
41
|
LDATE LDAY LHR LMIN LSEC LJDAY LMON LYR2 LYR4 LWDAY LTIME JOBNAME DS SEQ DATE TIME`.split(/\s+/));
|
|
40
42
|
|
|
41
|
-
// Splits an operand field on commas that are not inside parentheses or quotes. JCL nests both, and
|
|
42
|
-
// a naive split on comma turns DISP=(NEW,CATLG,DELETE) into three operands.
|
|
43
|
-
export function splitOperands(s) {
|
|
44
|
-
const out = [];
|
|
45
|
-
let depth = 0, quoted = false, start = 0;
|
|
46
|
-
for (let i = 0; i < s.length; i++) {
|
|
47
|
-
const c = s[i];
|
|
48
|
-
if (quoted) { if (c === "'") { if (s[i + 1] === "'") i++; else quoted = false; } continue; }
|
|
49
|
-
if (c === "'") quoted = true;
|
|
50
|
-
else if (c === '(') depth++;
|
|
51
|
-
else if (c === ')') depth = Math.max(0, depth - 1);
|
|
52
|
-
else if (c === ',' && depth === 0) { out.push(s.slice(start, i)); start = i + 1; }
|
|
53
|
-
}
|
|
54
|
-
out.push(s.slice(start));
|
|
55
|
-
return out.filter((x, i, a) => x.length || i < a.length - 1);
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
// Operands are positional until the first KEY=VALUE, and keyword after it. Both forms matter:
|
|
59
|
-
// EXEC takes its procedure name positionally, and everything interesting on DD is a keyword.
|
|
60
|
-
export function parseOperands(field) {
|
|
61
|
-
const positional = [];
|
|
62
|
-
const keywords = new Map();
|
|
63
|
-
for (const raw of splitOperands(field)) {
|
|
64
|
-
const part = raw.trim();
|
|
65
|
-
if (!part) continue;
|
|
66
|
-
const eq = keywordSplit(part);
|
|
67
|
-
if (eq < 0) positional.push(part);
|
|
68
|
-
else keywords.set(part.slice(0, eq).toUpperCase(), part.slice(eq + 1));
|
|
69
|
-
}
|
|
70
|
-
return { positional, keywords };
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
// The first '=' that is not inside parentheses or quotes. DCB=(RECFM=FB,LRECL=80) is one keyword
|
|
74
|
-
// whose value happens to contain more of them.
|
|
75
|
-
function keywordSplit(part) {
|
|
76
|
-
let depth = 0, quoted = false;
|
|
77
|
-
for (let i = 0; i < part.length; i++) {
|
|
78
|
-
const c = part[i];
|
|
79
|
-
if (quoted) { if (c === "'") { if (part[i + 1] === "'") i++; else quoted = false; } continue; }
|
|
80
|
-
if (c === "'") quoted = true;
|
|
81
|
-
else if (c === '(') depth++;
|
|
82
|
-
else if (c === ')') depth = Math.max(0, depth - 1);
|
|
83
|
-
else if (c === '=' && depth === 0) return i;
|
|
84
|
-
}
|
|
85
|
-
return -1;
|
|
86
|
-
}
|
|
87
|
-
|
|
88
43
|
// Symbolic substitution. &NAME ends at a non-name character, and a trailing dot is a separator
|
|
89
44
|
// that is consumed rather than kept: &PREFIX..DATA resolves to <prefix>.DATA.
|
|
90
45
|
export function substitute(text, symbols) {
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// Reading a revision out of a repository with git plumbing only.
|
|
3
|
+
import { spawnSync } from 'node:child_process';
|
|
4
|
+
|
|
5
|
+
// A reviewed repository's .git/config can name commands; with fsmonitor off and plumbing only, git runs none.
|
|
6
|
+
const GIT_ENV = { ...process.env, GIT_OPTIONAL_LOCKS: '0', GIT_TERMINAL_PROMPT: '0' };
|
|
7
|
+
export const git = (repo, args, opts = {}) => spawnSync('git', ['-c', 'core.fsmonitor=false', '-C', repo, ...args], { env: GIT_ENV, maxBuffer: 256 * 1024 * 1024, ...opts });
|
|
8
|
+
export const refused = (what, ref, why) => Object.assign(new Error(`${what} ${ref} failed: ${why}`), { code: 'EDIFFREF' });
|
|
9
|
+
const BATCH_BYTES = 64 * 1024 * 1024;
|
|
10
|
+
|
|
11
|
+
export function treeOf(repo, ref) {
|
|
12
|
+
if (!ref || String(ref).startsWith('-')) throw refused('git rev-parse', ref, 'a revision cannot be empty or start with "-"');
|
|
13
|
+
const r = git(repo, ['rev-parse', '--verify', '--quiet', '--end-of-options', `${ref}^{tree}`], { encoding: 'utf8' });
|
|
14
|
+
const oid = String(r.stdout || '').trim();
|
|
15
|
+
if (r.status !== 0 || !/^[0-9a-f]{40}([0-9a-f]{24})?$/.test(oid)) throw refused('git rev-parse', ref, String(r.stderr || '').trim() || 'not a revision');
|
|
16
|
+
return oid;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
// The parts of a tree path, or null if a part is empty, climbs, names a drive or stream, or is .git.
|
|
20
|
+
export function treePathParts(path) {
|
|
21
|
+
const parts = path.split('/');
|
|
22
|
+
return parts.some((p) => !p || p === '.' || p === '..' || /[\\:\0]/.test(p) || p.toLowerCase() === '.git') ? null : parts;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
// Every file in the tree `oid` as committed, as [{ path, bytes }] in tree order: blobs as stored,
|
|
26
|
+
// with no filter or line-ending conversion. Links and submodules are left out and counted.
|
|
27
|
+
export function revisionBlobs(repo, oid, ref = oid) {
|
|
28
|
+
const listed = git(repo, ['ls-tree', '-r', '-z', '-l', '--full-tree', oid]);
|
|
29
|
+
if (listed.status !== 0) throw refused('git ls-tree', ref, String(listed.stderr || '').trim() || `exit ${listed.status}`);
|
|
30
|
+
const blobs = [];
|
|
31
|
+
let links = 0;
|
|
32
|
+
for (const entry of listed.stdout.toString('utf8').split('\0')) {
|
|
33
|
+
const m = /^(\d{6}) blob ([0-9a-f]+) +(\d+)\t(.+)$/s.exec(entry);
|
|
34
|
+
if (!m) continue;
|
|
35
|
+
if (m[1] === '120000') { links++; continue; }
|
|
36
|
+
blobs.push({ oid: m[2], size: Number(m[3]), path: m[4] });
|
|
37
|
+
}
|
|
38
|
+
// Batched by bytes as well as count, and each batch's buffer sized to hold it, so one large blob
|
|
39
|
+
// cannot overflow a buffer sized for the others.
|
|
40
|
+
const batches = [];
|
|
41
|
+
for (const b of blobs) {
|
|
42
|
+
const last = batches[batches.length - 1];
|
|
43
|
+
if (!last || last.length >= 500 || last.bytes + b.size > BATCH_BYTES) batches.push(Object.assign([b], { bytes: b.size }));
|
|
44
|
+
else { last.push(b); last.bytes += b.size; }
|
|
45
|
+
}
|
|
46
|
+
for (const batch of batches) {
|
|
47
|
+
const r = git(repo, ['cat-file', '--batch'], { input: batch.map((b) => b.oid).join('\n') + '\n', maxBuffer: batch.bytes + batch.length * 128 + 4096 });
|
|
48
|
+
if (r.status !== 0) throw refused('git cat-file', ref, String(r.stderr || '').trim() || (r.error ? r.error.code || r.error.message : `exit ${r.status}`));
|
|
49
|
+
let at = 0;
|
|
50
|
+
for (const b of batch) {
|
|
51
|
+
const nl = r.stdout.indexOf(0x0a, at);
|
|
52
|
+
const head = r.stdout.toString('utf8', at, nl).split(' ');
|
|
53
|
+
if (head[1] !== 'blob') throw refused('git cat-file', ref, `${b.oid} is ${head[1] || 'missing'}`);
|
|
54
|
+
const size = Number(head[2]);
|
|
55
|
+
b.bytes = r.stdout.subarray(nl + 1, nl + 1 + size);
|
|
56
|
+
at = nl + 1 + size + 1;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
return { blobs, links };
|
|
60
|
+
}
|
package/lib/kernel/identity.mjs
CHANGED
|
@@ -131,14 +131,15 @@ function codeText(line, kind, format) {
|
|
|
131
131
|
return maskSecrets(s).replace(/\s+/g, ' ').trim();
|
|
132
132
|
}
|
|
133
133
|
|
|
134
|
-
// Reads each file once, however many findings it holds
|
|
134
|
+
// Reads each file once, however many findings it holds, from disk or through the reader the
|
|
135
|
+
// caller's source tree supplies.
|
|
135
136
|
function fileFacts(root, path, cache) {
|
|
136
137
|
let facts = cache.get(path);
|
|
137
138
|
if (facts) return facts;
|
|
138
139
|
facts = { kind: 'other', lines: [], scopes: null, format: null };
|
|
139
140
|
cache.set(path, facts);
|
|
140
141
|
let src;
|
|
141
|
-
try { src = readSource(join(root, path)).text; } catch { return facts; }
|
|
142
|
+
try { src = cache.read ? cache.read(path) : readSource(join(root, path)).text; } catch { return facts; }
|
|
142
143
|
facts.lines = src.split(/\r?\n/);
|
|
143
144
|
if (isJcl(join(root, path))) {
|
|
144
145
|
facts.kind = 'jcl';
|
|
@@ -171,8 +172,8 @@ const digest = (parts) => createHash('sha256').update(parts.join('\u0000')).dige
|
|
|
171
172
|
// Stamps `fingerprint` on every finding and returns how many shared one with an earlier finding.
|
|
172
173
|
// `repo` names the repository when one report holds several, where the same program id in two
|
|
173
174
|
// repositories is two programs.
|
|
174
|
-
export function stampFingerprints(findings, { root, repo = '' } = {}) {
|
|
175
|
-
const cache = new Map();
|
|
175
|
+
export function stampFingerprints(findings, { root, repo = '', read = null } = {}) {
|
|
176
|
+
const cache = Object.assign(new Map(), { read });
|
|
176
177
|
const seen = new Set();
|
|
177
178
|
let shared = 0;
|
|
178
179
|
for (const f of findings) {
|
|
@@ -188,8 +189,8 @@ export function stampFingerprints(findings, { root, repo = '' } = {}) {
|
|
|
188
189
|
// Looser keys for one finding seen in two trees whose line was edited between them: `scope` is the
|
|
189
190
|
// fingerprint without the line's text, and `route` names a data-flow finding by the statement its
|
|
190
191
|
// trace starts at. Null where a finding has no route.
|
|
191
|
-
export function pairingKeys(findings, { root, repo = '' } = {}) {
|
|
192
|
-
const cache = new Map();
|
|
192
|
+
export function pairingKeys(findings, { root, repo = '', read = null } = {}) {
|
|
193
|
+
const cache = Object.assign(new Map(), { read });
|
|
193
194
|
return findings.map((f) => {
|
|
194
195
|
const { scope, subject } = partsOf(f, root, cache);
|
|
195
196
|
const src = f.evidence === 'path' && f.related && f.related[0];
|
|
@@ -26,13 +26,16 @@
|
|
|
26
26
|
//
|
|
27
27
|
// An adapter that does not have a filesystem underneath it has to make that decision itself, and
|
|
28
28
|
// `contains` is where it makes it. For a tree held in memory it is key membership. For a git
|
|
29
|
-
// revision it is membership of the tree object, which cannot name a path outside itself.
|
|
30
|
-
//
|
|
31
|
-
//
|
|
32
|
-
|
|
33
|
-
|
|
29
|
+
// revision it is membership of the tree object, which cannot name a path outside itself. For a PDS
|
|
30
|
+
// export it is membership of the members the walk found, each read with no link followed. Anything
|
|
31
|
+
// cleverer than those deserves the six containment tests pointed at it before it ships:
|
|
32
|
+
// test/sources.test.mjs:44, :63, :88, :101, :144 and test/review.test.mjs:114, as
|
|
33
|
+
// test/git-tree.test.mjs and test/pds-export.test.mjs do.
|
|
34
|
+
import { closeSync, constants, fstatSync, openSync, readFileSync, readSync, realpathSync } from 'node:fs';
|
|
35
|
+
import { basename, dirname, relative, resolve, sep } from 'node:path';
|
|
34
36
|
import { buildFileIndex, parseSource } from '../parser.mjs';
|
|
35
|
-
import { readSource, relPath } from '../sources.mjs';
|
|
37
|
+
import { classifyUnder, decodeSource, diskClassifier, heldClassifier, kindOfBytes, looksEbcdic, readSource, relPath } from '../sources.mjs';
|
|
38
|
+
import { revisionBlobs, treeOf, treePathParts } from './git.mjs';
|
|
36
39
|
|
|
37
40
|
// The shape every adapter answers to. Checked rather than documented, because an adapter missing a
|
|
38
41
|
// method fails at the first rule set that happens to call it rather than at the boundary.
|
|
@@ -64,10 +67,14 @@ export function directoryTree(root, opts = {}) {
|
|
|
64
67
|
const idx = buildFileIndex(root);
|
|
65
68
|
const systemDirs = opts.systemDirs || [];
|
|
66
69
|
const top = resolve(root);
|
|
70
|
+
const kindOf = diskClassifier();
|
|
71
|
+
classifyUnder(root, kindOf);
|
|
72
|
+
if (top !== root) classifyUnder(top, kindOf);
|
|
67
73
|
|
|
68
74
|
return {
|
|
69
75
|
kind: 'directory',
|
|
70
76
|
root,
|
|
77
|
+
kindOf,
|
|
71
78
|
// The index itself, for the two callers that need more than a list: the copybook set reads
|
|
72
79
|
// copyDirs, and the inventory reports on unreadable directories and symlinks.
|
|
73
80
|
index: idx,
|
|
@@ -92,40 +99,203 @@ export function directoryTree(root, opts = {}) {
|
|
|
92
99
|
};
|
|
93
100
|
}
|
|
94
101
|
|
|
95
|
-
// A tree
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
102
|
+
// A tree whose files are held rather than on disk. The parser finds copybooks in the index built
|
|
103
|
+
// here and reads them through `readText`, so a COPY resolves to a held file or to a system copy
|
|
104
|
+
// library outside the tree, never to a file on disk that happens to sit under this tree's root.
|
|
105
|
+
function heldTree({ kind, root, store, rel, systemDirs = [], copyDirs = null, classify = heldClassifier,
|
|
106
|
+
symlinks = { followed: 0, outside: 0, broken: 0 }, unreadableDirs = [] }) {
|
|
107
|
+
const index = new Map();
|
|
108
|
+
const dirs = new Set(copyDirs || []);
|
|
109
|
+
for (const p of store.keys()) {
|
|
110
|
+
index.set(resolve(root, p).toLowerCase(), p);
|
|
111
|
+
if (!copyDirs && /\.(cpy|copy|inc|cbl|cob)$/i.test(p)) dirs.add(dirname(resolve(root, p)));
|
|
112
|
+
}
|
|
113
|
+
index.root = root;
|
|
114
|
+
const top = resolve(root);
|
|
115
|
+
const absent = (p) => Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
|
|
116
|
+
const bytes = (p) => {
|
|
117
|
+
const b = store.get(p);
|
|
118
|
+
if (!b) throw absent(p);
|
|
119
|
+
return b;
|
|
120
|
+
};
|
|
121
|
+
const kindOf = classify(bytes);
|
|
122
|
+
classifyUnder(top, kindOf);
|
|
123
|
+
const text = (p) => decodeSource(bytes(p));
|
|
124
|
+
const readText = (p) => {
|
|
125
|
+
if (store.has(p)) return text(p).text;
|
|
126
|
+
const r = resolve(root, p);
|
|
127
|
+
if (r === top || r.startsWith(top + sep)) throw absent(p);
|
|
128
|
+
return readSource(p).text;
|
|
129
|
+
};
|
|
104
130
|
return {
|
|
105
|
-
kind
|
|
131
|
+
kind,
|
|
106
132
|
root,
|
|
107
|
-
|
|
133
|
+
kindOf,
|
|
134
|
+
index: { index, copyDirs: [...dirs].sort(), unreadableDirs, symlinks },
|
|
108
135
|
list: () => [...store.keys()].sort(),
|
|
109
|
-
bytes
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
return b;
|
|
113
|
-
},
|
|
114
|
-
text: (p) => {
|
|
115
|
-
const b = store.get(p);
|
|
116
|
-
if (!b) throw Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
|
|
117
|
-
return { text: b.toString('latin1'), encoding: 'latin1' };
|
|
118
|
-
},
|
|
119
|
-
rel: (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), ''),
|
|
136
|
+
bytes,
|
|
137
|
+
text,
|
|
138
|
+
rel,
|
|
120
139
|
// Containment by construction: a path this tree does not hold is a path outside it.
|
|
121
140
|
contains: (p) => store.has(p),
|
|
122
|
-
parse: () => {
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
)
|
|
127
|
-
|
|
141
|
+
parse: (file, t) => parseSource(t ?? text(file).text, file, {
|
|
142
|
+
format: 'auto',
|
|
143
|
+
includeDirs: [...dirs].sort(),
|
|
144
|
+
fileIndex: index,
|
|
145
|
+
mainDir: dirname(resolve(root, file)),
|
|
146
|
+
copyFormat: 'auto',
|
|
147
|
+
systemDirs,
|
|
148
|
+
readText,
|
|
149
|
+
}),
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// A tree that was never on disk, for tests. `files` maps a path to its contents, as a string or a
|
|
154
|
+
// Buffer. Paths are used exactly as given, so a test reads the way it writes.
|
|
155
|
+
export function memoryTree(files, { root = '/memory', systemDirs = [] } = {}) {
|
|
156
|
+
const store = new Map(Object.entries(files).map(([p, v]) => [p, Buffer.isBuffer(v) ? v : Buffer.from(v, 'latin1')]));
|
|
157
|
+
const rel = (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), '');
|
|
158
|
+
return heldTree({ kind: 'memory', root, store, rel, systemDirs });
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// A git revision, read with plumbing into memory: nothing is written to disk. Its root is a path
|
|
162
|
+
// beside the repository that does not exist, `<repo>@<tree>`, so every path it answers for is
|
|
163
|
+
// absolute and none of them can be found on disk. The tree object is the containment: it cannot
|
|
164
|
+
// name a path outside itself, and a path that would climb, name a drive or stream, or enter .git
|
|
165
|
+
// is left out, as it is when a revision is written to disk. Links and submodules are not read.
|
|
166
|
+
export function gitTree(repo, ref, opts = {}) {
|
|
167
|
+
const oid = treeOf(repo, ref);
|
|
168
|
+
const { blobs, links } = revisionBlobs(repo, oid, ref);
|
|
169
|
+
const root = `${resolve(repo)}@${oid.slice(0, 12)}`;
|
|
170
|
+
const store = new Map();
|
|
171
|
+
for (const b of blobs) {
|
|
172
|
+
const parts = treePathParts(b.path);
|
|
173
|
+
if (parts) store.set(resolve(root, ...parts), b.bytes);
|
|
174
|
+
}
|
|
175
|
+
const tree = heldTree({ kind: 'git', root, store, rel: (p) => relPath(root, p), systemDirs: opts.systemDirs || [],
|
|
176
|
+
symlinks: { followed: 0, outside: 0, broken: 0, notRead: links } });
|
|
177
|
+
return Object.assign(tree, { ref, oid });
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// z/OSMF's record mode (zowe --record) writes each record after its length, four bytes big-endian,
|
|
181
|
+
// and nothing between records. Read as that only when the lengths chain exactly to the end of the
|
|
182
|
+
// file, and given back as lines; anything else is the file as it is.
|
|
183
|
+
const MAX_LRECL = 32760;
|
|
184
|
+
export function unframeRecords(buf) {
|
|
185
|
+
let records = 0;
|
|
186
|
+
for (let at = 0; at < buf.length; records++) {
|
|
187
|
+
if (buf.length - at < 4) return buf;
|
|
188
|
+
const n = buf.readUInt32BE(at);
|
|
189
|
+
if (n > MAX_LRECL || buf.length - at - 4 < n) return buf;
|
|
190
|
+
at += 4 + n;
|
|
191
|
+
}
|
|
192
|
+
if (!records || records * 4 === buf.length) return buf;
|
|
193
|
+
const lines = (end) => {
|
|
194
|
+
const out = Buffer.alloc(buf.length - records * 3);
|
|
195
|
+
let w = 0;
|
|
196
|
+
for (let at = 0; at < buf.length;) {
|
|
197
|
+
const n = buf.readUInt32BE(at);
|
|
198
|
+
w += buf.copy(out, w, at + 4, at + 4 + n);
|
|
199
|
+
out[w++] = end;
|
|
200
|
+
at += 4 + n;
|
|
201
|
+
}
|
|
202
|
+
return out;
|
|
128
203
|
};
|
|
204
|
+
const ebcdic = lines(0x15);
|
|
205
|
+
return looksEbcdic(ebcdic) ? ebcdic : lines(0x0A);
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
// A member is read from the file the walk found. A link put in its place since, or a file that is
|
|
209
|
+
// not a regular one, is refused rather than followed or waited on.
|
|
210
|
+
function readMember(file) {
|
|
211
|
+
const fd = openSync(file, constants.O_RDONLY | (constants.O_NOFOLLOW || 0) | (constants.O_NONBLOCK || 0));
|
|
212
|
+
try {
|
|
213
|
+
const st = fstatSync(fd);
|
|
214
|
+
if (!st.isFile()) throw Object.assign(new Error(`not a regular file: ${file}`), { code: 'ENOTFILE' });
|
|
215
|
+
const buf = Buffer.alloc(st.size);
|
|
216
|
+
let n = 0;
|
|
217
|
+
while (n < st.size) {
|
|
218
|
+
const r = readSync(fd, buf, n, st.size - n, n);
|
|
219
|
+
if (!r) break;
|
|
220
|
+
n += r;
|
|
221
|
+
}
|
|
222
|
+
return unframeRecords(buf.subarray(0, n));
|
|
223
|
+
} finally { closeSync(fd); }
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
const QUALIFIER = /^[A-Z@#$][A-Z0-9@#$-]{0,7}$/;
|
|
227
|
+
const MEMBER = /^([A-Z@#$][A-Z0-9@#$]{0,7})(?:\.[^.]*)?$/i;
|
|
228
|
+
const MAX_LISTED = 8;
|
|
229
|
+
|
|
230
|
+
// Partitioned data sets exported to a directory, a member to a file: what
|
|
231
|
+
// `zowe zos-files download all-members` writes (ibmuser/new/cntl/member.txt, lower-cased unless
|
|
232
|
+
// --preserve-original-letter-case, with whatever extension -e gave), or a directory named for the
|
|
233
|
+
// data set with its members copied in. Each member is held as `DATA.SET.NAME/MEMBER` under a root
|
|
234
|
+
// beside the export that does not exist, so a finding names the member as z/OS does however it was
|
|
235
|
+
// downloaded, and nothing under that root is ever read from disk.
|
|
236
|
+
//
|
|
237
|
+
// The extension is the download's choice, not the language's, so members are classified by their
|
|
238
|
+
// contents. A file whose directory is not a data set name or whose name is not a member name is not
|
|
239
|
+
// read and is counted, as is a second file for a member already held. Every data set that holds
|
|
240
|
+
// anything but JCL and maps is a copy library, searched in name order: the export does not say
|
|
241
|
+
// what SYSLIB concatenated.
|
|
242
|
+
export function pdsExportTree(dir, opts = {}) {
|
|
243
|
+
const idx = buildFileIndex(dir);
|
|
244
|
+
const top = realpathSync(dir);
|
|
245
|
+
const root = `${resolve(dir)}@pds`;
|
|
246
|
+
const origins = new Map();
|
|
247
|
+
const found = new Map();
|
|
248
|
+
const notMembers = [];
|
|
249
|
+
const duplicates = [];
|
|
250
|
+
const dataSets = new Set();
|
|
251
|
+
for (const file of [...idx.index.values()].sort()) {
|
|
252
|
+
const where = relative(dir, dirname(file)).split(sep).filter(Boolean);
|
|
253
|
+
const dsn = where.join('.').toUpperCase();
|
|
254
|
+
const member = MEMBER.exec(basename(file));
|
|
255
|
+
if (!member || dsn.length > 44 || (dsn && !dsn.split('.').every((q) => QUALIFIER.test(q)))) {
|
|
256
|
+
notMembers.push(relPath(dir, file));
|
|
257
|
+
continue;
|
|
258
|
+
}
|
|
259
|
+
const at = resolve(root, ...(dsn ? [dsn] : []), member[1].toUpperCase());
|
|
260
|
+
if (found.has(at)) { duplicates.push(`${relPath(dir, file)}: ${relPath(root, at)} is ${found.get(at)}`); continue; }
|
|
261
|
+
let real;
|
|
262
|
+
try { real = realpathSync(file); } catch { real = file; }
|
|
263
|
+
if (real !== top && !real.startsWith(top + sep)) { notMembers.push(relPath(dir, file)); continue; }
|
|
264
|
+
origins.set(at, real);
|
|
265
|
+
found.set(at, relPath(dir, file));
|
|
266
|
+
dataSets.add(dirname(at));
|
|
267
|
+
}
|
|
268
|
+
const kinds = new Map();
|
|
269
|
+
const byKind = {};
|
|
270
|
+
const copyDirs = new Set();
|
|
271
|
+
for (const [at, file] of origins) {
|
|
272
|
+
let kind = null;
|
|
273
|
+
try { kind = kindOfBytes(readMember(file)); } catch { kind = null; }
|
|
274
|
+
kinds.set(at, kind);
|
|
275
|
+
byKind[kind || 'other'] = (byKind[kind || 'other'] || 0) + 1;
|
|
276
|
+
if (kind !== 'jcl' && kind !== 'bms') copyDirs.add(dirname(at));
|
|
277
|
+
}
|
|
278
|
+
const store = {
|
|
279
|
+
keys: () => origins.keys(),
|
|
280
|
+
has: (p) => origins.has(p),
|
|
281
|
+
get: (p) => (origins.has(p) ? readMember(origins.get(p)) : undefined),
|
|
282
|
+
};
|
|
283
|
+
const tree = heldTree({
|
|
284
|
+
kind: 'pds-export', root, store, rel: (p) => relPath(root, p), systemDirs: opts.systemDirs || [],
|
|
285
|
+
copyDirs: [...copyDirs], classify: () => (p) => kinds.get(p) ?? null, symlinks: idx.symlinks,
|
|
286
|
+
unreadableDirs: idx.unreadableDirs.map((d) => resolve(root, relative(dir, d))),
|
|
287
|
+
});
|
|
288
|
+
return Object.assign(tree, {
|
|
289
|
+
dir,
|
|
290
|
+
origin: (p) => found.get(p) ?? null,
|
|
291
|
+
export: {
|
|
292
|
+
dataSets: dataSets.size,
|
|
293
|
+
members: origins.size,
|
|
294
|
+
byKind,
|
|
295
|
+
...(notMembers.length ? { notMembers: notMembers.length, notMembersListed: notMembers.slice(0, MAX_LISTED) } : {}),
|
|
296
|
+
...(duplicates.length ? { duplicates: duplicates.length, duplicatesListed: duplicates.slice(0, MAX_LISTED) } : {}),
|
|
297
|
+
},
|
|
298
|
+
});
|
|
129
299
|
}
|
|
130
300
|
|
|
131
301
|
// What a rule set calls. Given a tree it uses it; given none it builds the directory one, so every
|
package/lib/parser.mjs
CHANGED
|
@@ -446,11 +446,18 @@ export function tokenize(norm, file) {
|
|
|
446
446
|
return { tokens: out, diags };
|
|
447
447
|
}
|
|
448
448
|
|
|
449
|
-
//
|
|
450
|
-
//
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
449
|
+
// A drive that stalls or drops for a moment answers a listing with ENOENT or an I/O error. Such a
|
|
450
|
+
// listing is tried again after a pause; the pauses add up to about 1.3 seconds.
|
|
451
|
+
const LISTING_RETRIED = new Set(['ENOENT', 'EIO', 'ETIMEDOUT', 'ENXIO', 'EBUSY', 'EAGAIN', 'EINTR', 'ESTALE', 'ENOTCONN']);
|
|
452
|
+
const LISTING_PAUSES_MS = [50, 250, 1000];
|
|
453
|
+
const pause = (ms) => Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
|
|
454
|
+
|
|
455
|
+
// Every rule set walks the tree through here. A directory that cannot be listed is recorded, not
|
|
456
|
+
// treated as empty: a scan over it is incomplete, not a scan of a smaller tree. Only the root's own
|
|
457
|
+
// ENOENT means absent; a directory the walk found in its parent was there. Symlinks are followed
|
|
458
|
+
// while they stay inside the tree, so a symlinked copy library is read; one pointing outside is
|
|
459
|
+
// counted and never followed, because a scan reads the tree it was given and nothing else.
|
|
460
|
+
export function buildFileIndex(root, { readdir = readdirSync, wait = pause } = {}) {
|
|
454
461
|
const index = new Map();
|
|
455
462
|
const dirs = new Set();
|
|
456
463
|
const unreadableDirs = [];
|
|
@@ -460,16 +467,20 @@ export function buildFileIndex(root) {
|
|
|
460
467
|
const inside = (p) => p === top || p.startsWith(top + sep);
|
|
461
468
|
const visited = new Set();
|
|
462
469
|
const addFile = (p, name, d) => { index.set(p.toLowerCase(), p); if (/\.(cpy|copy|inc|cbl|cob)$/i.test(name)) dirs.add(d); };
|
|
470
|
+
const list = (d) => {
|
|
471
|
+
for (let i = 0; ; i++) {
|
|
472
|
+
try { return readdir(d, { withFileTypes: true }); } catch (e) {
|
|
473
|
+
if (i >= LISTING_PAUSES_MS.length || !LISTING_RETRIED.has(e.code)) throw e;
|
|
474
|
+
wait(LISTING_PAUSES_MS[i]);
|
|
475
|
+
}
|
|
476
|
+
}
|
|
477
|
+
};
|
|
463
478
|
// Each real directory is walked once, however many links lead to it, or its programs count twice.
|
|
464
479
|
const walk = (d, real) => {
|
|
465
480
|
if (visited.has(real)) return;
|
|
466
481
|
visited.add(real);
|
|
467
482
|
let es;
|
|
468
|
-
try { es =
|
|
469
|
-
if (e.code === 'ENOENT') return;
|
|
470
|
-
if (e.code === 'EACCES' || e.code === 'EPERM') { unreadableDirs.push(d); return; }
|
|
471
|
-
throw e;
|
|
472
|
-
}
|
|
483
|
+
try { es = list(d); } catch { unreadableDirs.push(d); return; }
|
|
473
484
|
for (const e of es.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))) {
|
|
474
485
|
if (e.name === '.git' || e.name.startsWith('._')) continue;
|
|
475
486
|
const p = join(d, e.name);
|
|
@@ -634,7 +645,7 @@ function isProgramFile(p, ctx) {
|
|
|
634
645
|
const seen = (ctx.programFiles ||= new Map());
|
|
635
646
|
if (!seen.has(p)) {
|
|
636
647
|
let text = '';
|
|
637
|
-
try { text =
|
|
648
|
+
try { text = ctx.readText(p); } catch { /* an unreadable candidate is not known to be a program */ }
|
|
638
649
|
seen.set(p, /^[^*\n]{0,6}\s*PROGRAM-ID\s*\./im.test(text));
|
|
639
650
|
}
|
|
640
651
|
return seen.get(p);
|
|
@@ -787,7 +798,7 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
|
|
|
787
798
|
if (!path) return [];
|
|
788
799
|
if (stack.includes(path) || stack.length > 40) { record.status = 'recursive'; return []; }
|
|
789
800
|
if (++ctx.inclusions > MAX_INCLUSIONS || ctx.copyTokens > MAX_COPY_TOKENS) { record.status = 'expansion-limit'; return []; }
|
|
790
|
-
const src =
|
|
801
|
+
const src = ctx.readText(path);
|
|
791
802
|
// A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
|
|
792
803
|
// directive earlier in the including file may have changed from the file's starting format.
|
|
793
804
|
const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
|
|
@@ -1513,6 +1524,9 @@ function indexTokens(seg) {
|
|
|
1513
1524
|
const lone = (from, to) => (to - from === 1 && (seg[from].t === 'word' || seg[from].t === 'num') && /^\d+$/.test(seg[from].v) ? Number(seg[from].v) : null);
|
|
1514
1525
|
const refLength = colonAt < 0 ? null : lone(colonAt + 1, close);
|
|
1515
1526
|
const refStart = colonAt < 0 ? null : lone(k + 1, colonAt);
|
|
1527
|
+
// A length is judged against where its reference starts: a name, a name moved by a constant, or neither.
|
|
1528
|
+
const named = (from, to) => (seg[from].t === 'word' && !/^\d+$/.test(seg[from].v) && (to - from === 1 || (to - from === 3 && offsetOf(seg, from, k, close, colonAt))) ? { tok: seg[from], offset: offsetOf(seg, from, k, close, colonAt) } : null);
|
|
1529
|
+
const refFrom = colonAt < 0 || refStart != null ? null : named(k + 1, colonAt) || { tok: null, offset: 0 };
|
|
1516
1530
|
for (let j = k + 1; j < close; j++) {
|
|
1517
1531
|
const t = seg[j];
|
|
1518
1532
|
// A number is tokenized as a word, and no name is all digits.
|
|
@@ -1521,7 +1535,7 @@ function indexTokens(seg) {
|
|
|
1521
1535
|
if (prev && prev.t === 'word' && (prev.u === 'OF' || prev.u === 'IN' || prev.u === 'FUNCTION')) continue;
|
|
1522
1536
|
const kind = colonAt < 0 ? 'subscript' : j < colonAt ? 'refmod-offset' : 'refmod-length';
|
|
1523
1537
|
const other = kind === 'refmod-offset' ? refLength : kind === 'refmod-length' ? refStart : null;
|
|
1524
|
-
out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}) });
|
|
1538
|
+
out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}), ...(kind === 'refmod-length' && refFrom ? { from: refFrom } : {}) });
|
|
1525
1539
|
}
|
|
1526
1540
|
}
|
|
1527
1541
|
lastHost = host;
|
|
@@ -1848,7 +1862,7 @@ function collectCopybookDefines(src, format, ctx, depth) {
|
|
|
1848
1862
|
const path = resolveCopy(name, null, ctx);
|
|
1849
1863
|
if (!path || seen.has(path)) continue;
|
|
1850
1864
|
seen.add(path);
|
|
1851
|
-
const text =
|
|
1865
|
+
const text = ctx.readText(path);
|
|
1852
1866
|
const fmt = detectFormat(text) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : current;
|
|
1853
1867
|
normalize(text, fmt, ctx.defines, ctx.std);
|
|
1854
1868
|
collectCopybookDefines(text, fmt, ctx, depth + 1);
|
|
@@ -1880,7 +1894,9 @@ export function parseSource(src, file, opts = {}) {
|
|
|
1880
1894
|
const scheme = BINARY_SIZE[opts.std || 'default'] || BINARY_SIZE.default;
|
|
1881
1895
|
const ctx = { mainDir: opts.mainDir || dirname(file), includeDirs: opts.includeDirs || [], systemDirs: opts.systemDirs || [],
|
|
1882
1896
|
fileIndex: opts.fileIndex || null, allowAbsoluteCopy: !!opts.allowAbsoluteCopy, cache: new Map(), copies: [], diags: [], copyFormat: opts.copyFormat || format, defines: new Map(), std: opts.std,
|
|
1883
|
-
inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0
|
|
1897
|
+
inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0,
|
|
1898
|
+
// Where a copybook's text comes from: the disk, or the source tree the program was read from.
|
|
1899
|
+
readText: opts.readText || ((p) => readSource(p).text) };
|
|
1884
1900
|
collectCopybookDefines(src, format, ctx, 0);
|
|
1885
1901
|
const norm = normalize(src, format, ctx.defines, opts.std);
|
|
1886
1902
|
ctx.mainFormat = norm.finalFormat;
|
package/lib/revision.json
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"commit":"
|
|
1
|
+
{"commit":"b242a380e1b8b7d8bb8ff2d6e03e2c8a30242967","tag":"v0.3.0"}
|