@portll/cobolwork 0.2.140 → 0.2.150

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,60 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // Reading a revision out of a repository with git plumbing only.
3
+ import { spawnSync } from 'node:child_process';
4
+
5
+ // A reviewed repository's .git/config can name commands; with fsmonitor off and plumbing only, git runs none.
6
+ const GIT_ENV = { ...process.env, GIT_OPTIONAL_LOCKS: '0', GIT_TERMINAL_PROMPT: '0' };
7
+ export const git = (repo, args, opts = {}) => spawnSync('git', ['-c', 'core.fsmonitor=false', '-C', repo, ...args], { env: GIT_ENV, maxBuffer: 256 * 1024 * 1024, ...opts });
8
+ export const refused = (what, ref, why) => Object.assign(new Error(`${what} ${ref} failed: ${why}`), { code: 'EDIFFREF' });
9
+ const BATCH_BYTES = 64 * 1024 * 1024;
10
+
11
+ export function treeOf(repo, ref) {
12
+ if (!ref || String(ref).startsWith('-')) throw refused('git rev-parse', ref, 'a revision cannot be empty or start with "-"');
13
+ const r = git(repo, ['rev-parse', '--verify', '--quiet', '--end-of-options', `${ref}^{tree}`], { encoding: 'utf8' });
14
+ const oid = String(r.stdout || '').trim();
15
+ if (r.status !== 0 || !/^[0-9a-f]{40}([0-9a-f]{24})?$/.test(oid)) throw refused('git rev-parse', ref, String(r.stderr || '').trim() || 'not a revision');
16
+ return oid;
17
+ }
18
+
19
+ // The parts of a tree path, or null if a part is empty, climbs, names a drive or stream, or is .git.
20
+ export function treePathParts(path) {
21
+ const parts = path.split('/');
22
+ return parts.some((p) => !p || p === '.' || p === '..' || /[\\:\0]/.test(p) || p.toLowerCase() === '.git') ? null : parts;
23
+ }
24
+
25
+ // Every file in the tree `oid` as committed, as [{ path, bytes }] in tree order: blobs as stored,
26
+ // with no filter or line-ending conversion. Links and submodules are left out and counted.
27
+ export function revisionBlobs(repo, oid, ref = oid) {
28
+ const listed = git(repo, ['ls-tree', '-r', '-z', '-l', '--full-tree', oid]);
29
+ if (listed.status !== 0) throw refused('git ls-tree', ref, String(listed.stderr || '').trim() || `exit ${listed.status}`);
30
+ const blobs = [];
31
+ let links = 0;
32
+ for (const entry of listed.stdout.toString('utf8').split('\0')) {
33
+ const m = /^(\d{6}) blob ([0-9a-f]+) +(\d+)\t(.+)$/s.exec(entry);
34
+ if (!m) continue;
35
+ if (m[1] === '120000') { links++; continue; }
36
+ blobs.push({ oid: m[2], size: Number(m[3]), path: m[4] });
37
+ }
38
+ // Batched by bytes as well as count, and each batch's buffer sized to hold it, so one large blob
39
+ // cannot overflow a buffer sized for the others.
40
+ const batches = [];
41
+ for (const b of blobs) {
42
+ const last = batches[batches.length - 1];
43
+ if (!last || last.length >= 500 || last.bytes + b.size > BATCH_BYTES) batches.push(Object.assign([b], { bytes: b.size }));
44
+ else { last.push(b); last.bytes += b.size; }
45
+ }
46
+ for (const batch of batches) {
47
+ const r = git(repo, ['cat-file', '--batch'], { input: batch.map((b) => b.oid).join('\n') + '\n', maxBuffer: batch.bytes + batch.length * 128 + 4096 });
48
+ if (r.status !== 0) throw refused('git cat-file', ref, String(r.stderr || '').trim() || (r.error ? r.error.code || r.error.message : `exit ${r.status}`));
49
+ let at = 0;
50
+ for (const b of batch) {
51
+ const nl = r.stdout.indexOf(0x0a, at);
52
+ const head = r.stdout.toString('utf8', at, nl).split(' ');
53
+ if (head[1] !== 'blob') throw refused('git cat-file', ref, `${b.oid} is ${head[1] || 'missing'}`);
54
+ const size = Number(head[2]);
55
+ b.bytes = r.stdout.subarray(nl + 1, nl + 1 + size);
56
+ at = nl + 1 + size + 1;
57
+ }
58
+ }
59
+ return { blobs, links };
60
+ }
@@ -131,14 +131,15 @@ function codeText(line, kind, format) {
131
131
  return maskSecrets(s).replace(/\s+/g, ' ').trim();
132
132
  }
133
133
 
134
- // Reads each file once, however many findings it holds.
134
+ // Reads each file once, however many findings it holds, from disk or through the reader the
135
+ // caller's source tree supplies.
135
136
  function fileFacts(root, path, cache) {
136
137
  let facts = cache.get(path);
137
138
  if (facts) return facts;
138
139
  facts = { kind: 'other', lines: [], scopes: null, format: null };
139
140
  cache.set(path, facts);
140
141
  let src;
141
- try { src = readSource(join(root, path)).text; } catch { return facts; }
142
+ try { src = cache.read ? cache.read(path) : readSource(join(root, path)).text; } catch { return facts; }
142
143
  facts.lines = src.split(/\r?\n/);
143
144
  if (isJcl(join(root, path))) {
144
145
  facts.kind = 'jcl';
@@ -171,8 +172,8 @@ const digest = (parts) => createHash('sha256').update(parts.join('\u0000')).dige
171
172
  // Stamps `fingerprint` on every finding and returns how many shared one with an earlier finding.
172
173
  // `repo` names the repository when one report holds several, where the same program id in two
173
174
  // repositories is two programs.
174
- export function stampFingerprints(findings, { root, repo = '' } = {}) {
175
- const cache = new Map();
175
+ export function stampFingerprints(findings, { root, repo = '', read = null } = {}) {
176
+ const cache = Object.assign(new Map(), { read });
176
177
  const seen = new Set();
177
178
  let shared = 0;
178
179
  for (const f of findings) {
@@ -188,8 +189,8 @@ export function stampFingerprints(findings, { root, repo = '' } = {}) {
188
189
  // Looser keys for one finding seen in two trees whose line was edited between them: `scope` is the
189
190
  // fingerprint without the line's text, and `route` names a data-flow finding by the statement its
190
191
  // trace starts at. Null where a finding has no route.
191
- export function pairingKeys(findings, { root, repo = '' } = {}) {
192
- const cache = new Map();
192
+ export function pairingKeys(findings, { root, repo = '', read = null } = {}) {
193
+ const cache = Object.assign(new Map(), { read });
193
194
  return findings.map((f) => {
194
195
  const { scope, subject } = partsOf(f, root, cache);
195
196
  const src = f.evidence === 'path' && f.related && f.related[0];
@@ -26,13 +26,16 @@
26
26
  //
27
27
  // An adapter that does not have a filesystem underneath it has to make that decision itself, and
28
28
  // `contains` is where it makes it. For a tree held in memory it is key membership. For a git
29
- // revision it is membership of the tree object, which cannot name a path outside itself. Anything
30
- // cleverer than those two deserves the six containment tests pointed at it before it ships:
31
- // test/sources.test.mjs:44, :63, :88, :101, :144 and test/review.test.mjs:114.
32
- import { readFileSync, realpathSync } from 'node:fs';
33
- import { dirname, resolve, sep } from 'node:path';
29
+ // revision it is membership of the tree object, which cannot name a path outside itself. For a PDS
30
+ // export it is membership of the members the walk found, each read with no link followed. Anything
31
+ // cleverer than those deserves the six containment tests pointed at it before it ships:
32
+ // test/sources.test.mjs:44, :63, :88, :101, :144 and test/review.test.mjs:114, as
33
+ // test/git-tree.test.mjs and test/pds-export.test.mjs do.
34
+ import { closeSync, constants, fstatSync, openSync, readFileSync, readSync, realpathSync } from 'node:fs';
35
+ import { basename, dirname, relative, resolve, sep } from 'node:path';
34
36
  import { buildFileIndex, parseSource } from '../parser.mjs';
35
- import { readSource, relPath } from '../sources.mjs';
37
+ import { classifyUnder, decodeSource, diskClassifier, heldClassifier, kindOfBytes, looksEbcdic, readSource, relPath } from '../sources.mjs';
38
+ import { revisionBlobs, treeOf, treePathParts } from './git.mjs';
36
39
 
37
40
  // The shape every adapter answers to. Checked rather than documented, because an adapter missing a
38
41
  // method fails at the first rule set that happens to call it rather than at the boundary.
@@ -64,10 +67,14 @@ export function directoryTree(root, opts = {}) {
64
67
  const idx = buildFileIndex(root);
65
68
  const systemDirs = opts.systemDirs || [];
66
69
  const top = resolve(root);
70
+ const kindOf = diskClassifier();
71
+ classifyUnder(root, kindOf);
72
+ if (top !== root) classifyUnder(top, kindOf);
67
73
 
68
74
  return {
69
75
  kind: 'directory',
70
76
  root,
77
+ kindOf,
71
78
  // The index itself, for the two callers that need more than a list: the copybook set reads
72
79
  // copyDirs, and the inventory reports on unreadable directories and symlinks.
73
80
  index: idx,
@@ -92,40 +99,203 @@ export function directoryTree(root, opts = {}) {
92
99
  };
93
100
  }
94
101
 
95
- // A tree that was never on disk, for tests. `files` maps a path to its contents, as a string or a
96
- // Buffer. Paths are used exactly as given, so a test reads the way it writes.
97
- //
98
- // It cannot parse. parseSource resolves COPY statements against real directories and reads them
99
- // with readSource, so a program with copybooks needs a filesystem underneath it. A rule set that
100
- // only reads text - hidden, recon, vendor, jcl, build - works against this; one that parses does
101
- // not, and says so rather than parsing something unexpected.
102
- export function memoryTree(files, { root = '/memory' } = {}) {
103
- const store = new Map(Object.entries(files).map(([p, v]) => [p, Buffer.isBuffer(v) ? v : Buffer.from(v, 'latin1')]));
102
+ // A tree whose files are held rather than on disk. The parser finds copybooks in the index built
103
+ // here and reads them through `readText`, so a COPY resolves to a held file or to a system copy
104
+ // library outside the tree, never to a file on disk that happens to sit under this tree's root.
105
+ function heldTree({ kind, root, store, rel, systemDirs = [], copyDirs = null, classify = heldClassifier,
106
+ symlinks = { followed: 0, outside: 0, broken: 0 }, unreadableDirs = [] }) {
107
+ const index = new Map();
108
+ const dirs = new Set(copyDirs || []);
109
+ for (const p of store.keys()) {
110
+ index.set(resolve(root, p).toLowerCase(), p);
111
+ if (!copyDirs && /\.(cpy|copy|inc|cbl|cob)$/i.test(p)) dirs.add(dirname(resolve(root, p)));
112
+ }
113
+ index.root = root;
114
+ const top = resolve(root);
115
+ const absent = (p) => Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
116
+ const bytes = (p) => {
117
+ const b = store.get(p);
118
+ if (!b) throw absent(p);
119
+ return b;
120
+ };
121
+ const kindOf = classify(bytes);
122
+ classifyUnder(top, kindOf);
123
+ const text = (p) => decodeSource(bytes(p));
124
+ const readText = (p) => {
125
+ if (store.has(p)) return text(p).text;
126
+ const r = resolve(root, p);
127
+ if (r === top || r.startsWith(top + sep)) throw absent(p);
128
+ return readSource(p).text;
129
+ };
104
130
  return {
105
- kind: 'memory',
131
+ kind,
106
132
  root,
107
- index: { index: store, copyDirs: [], unreadableDirs: [], symlinks: { followed: 0, outside: 0, broken: 0 } },
133
+ kindOf,
134
+ index: { index, copyDirs: [...dirs].sort(), unreadableDirs, symlinks },
108
135
  list: () => [...store.keys()].sort(),
109
- bytes: (p) => {
110
- const b = store.get(p);
111
- if (!b) throw Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
112
- return b;
113
- },
114
- text: (p) => {
115
- const b = store.get(p);
116
- if (!b) throw Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
117
- return { text: b.toString('latin1'), encoding: 'latin1' };
118
- },
119
- rel: (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), ''),
136
+ bytes,
137
+ text,
138
+ rel,
120
139
  // Containment by construction: a path this tree does not hold is a path outside it.
121
140
  contains: (p) => store.has(p),
122
- parse: () => {
123
- throw Object.assign(
124
- new Error('a memory tree cannot parse: COPY resolution reads directories from disk'),
125
- { code: 'ETREEPARSE' },
126
- );
127
- },
141
+ parse: (file, t) => parseSource(t ?? text(file).text, file, {
142
+ format: 'auto',
143
+ includeDirs: [...dirs].sort(),
144
+ fileIndex: index,
145
+ mainDir: dirname(resolve(root, file)),
146
+ copyFormat: 'auto',
147
+ systemDirs,
148
+ readText,
149
+ }),
150
+ };
151
+ }
152
+
153
+ // A tree that was never on disk, for tests. `files` maps a path to its contents, as a string or a
154
+ // Buffer. Paths are used exactly as given, so a test reads the way it writes.
155
+ export function memoryTree(files, { root = '/memory', systemDirs = [] } = {}) {
156
+ const store = new Map(Object.entries(files).map(([p, v]) => [p, Buffer.isBuffer(v) ? v : Buffer.from(v, 'latin1')]));
157
+ const rel = (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), '');
158
+ return heldTree({ kind: 'memory', root, store, rel, systemDirs });
159
+ }
160
+
161
+ // A git revision, read with plumbing into memory: nothing is written to disk. Its root is a path
162
+ // beside the repository that does not exist, `<repo>@<tree>`, so every path it answers for is
163
+ // absolute and none of them can be found on disk. The tree object is the containment: it cannot
164
+ // name a path outside itself, and a path that would climb, name a drive or stream, or enter .git
165
+ // is left out, as it is when a revision is written to disk. Links and submodules are not read.
166
+ export function gitTree(repo, ref, opts = {}) {
167
+ const oid = treeOf(repo, ref);
168
+ const { blobs, links } = revisionBlobs(repo, oid, ref);
169
+ const root = `${resolve(repo)}@${oid.slice(0, 12)}`;
170
+ const store = new Map();
171
+ for (const b of blobs) {
172
+ const parts = treePathParts(b.path);
173
+ if (parts) store.set(resolve(root, ...parts), b.bytes);
174
+ }
175
+ const tree = heldTree({ kind: 'git', root, store, rel: (p) => relPath(root, p), systemDirs: opts.systemDirs || [],
176
+ symlinks: { followed: 0, outside: 0, broken: 0, notRead: links } });
177
+ return Object.assign(tree, { ref, oid });
178
+ }
179
+
180
+ // z/OSMF's record mode (zowe --record) writes each record after its length, four bytes big-endian,
181
+ // and nothing between records. Read as that only when the lengths chain exactly to the end of the
182
+ // file, and given back as lines; anything else is the file as it is.
183
+ const MAX_LRECL = 32760;
184
+ export function unframeRecords(buf) {
185
+ let records = 0;
186
+ for (let at = 0; at < buf.length; records++) {
187
+ if (buf.length - at < 4) return buf;
188
+ const n = buf.readUInt32BE(at);
189
+ if (n > MAX_LRECL || buf.length - at - 4 < n) return buf;
190
+ at += 4 + n;
191
+ }
192
+ if (!records || records * 4 === buf.length) return buf;
193
+ const lines = (end) => {
194
+ const out = Buffer.alloc(buf.length - records * 3);
195
+ let w = 0;
196
+ for (let at = 0; at < buf.length;) {
197
+ const n = buf.readUInt32BE(at);
198
+ w += buf.copy(out, w, at + 4, at + 4 + n);
199
+ out[w++] = end;
200
+ at += 4 + n;
201
+ }
202
+ return out;
128
203
  };
204
+ const ebcdic = lines(0x15);
205
+ return looksEbcdic(ebcdic) ? ebcdic : lines(0x0A);
206
+ }
207
+
208
+ // A member is read from the file the walk found. A link put in its place since, or a file that is
209
+ // not a regular one, is refused rather than followed or waited on.
210
+ function readMember(file) {
211
+ const fd = openSync(file, constants.O_RDONLY | (constants.O_NOFOLLOW || 0) | (constants.O_NONBLOCK || 0));
212
+ try {
213
+ const st = fstatSync(fd);
214
+ if (!st.isFile()) throw Object.assign(new Error(`not a regular file: ${file}`), { code: 'ENOTFILE' });
215
+ const buf = Buffer.alloc(st.size);
216
+ let n = 0;
217
+ while (n < st.size) {
218
+ const r = readSync(fd, buf, n, st.size - n, n);
219
+ if (!r) break;
220
+ n += r;
221
+ }
222
+ return unframeRecords(buf.subarray(0, n));
223
+ } finally { closeSync(fd); }
224
+ }
225
+
226
+ const QUALIFIER = /^[A-Z@#$][A-Z0-9@#$-]{0,7}$/;
227
+ const MEMBER = /^([A-Z@#$][A-Z0-9@#$]{0,7})(?:\.[^.]*)?$/i;
228
+ const MAX_LISTED = 8;
229
+
230
+ // Partitioned data sets exported to a directory, a member to a file: what
231
+ // `zowe zos-files download all-members` writes (ibmuser/new/cntl/member.txt, lower-cased unless
232
+ // --preserve-original-letter-case, with whatever extension -e gave), or a directory named for the
233
+ // data set with its members copied in. Each member is held as `DATA.SET.NAME/MEMBER` under a root
234
+ // beside the export that does not exist, so a finding names the member as z/OS does however it was
235
+ // downloaded, and nothing under that root is ever read from disk.
236
+ //
237
+ // The extension is the download's choice, not the language's, so members are classified by their
238
+ // contents. A file whose directory is not a data set name or whose name is not a member name is not
239
+ // read and is counted, as is a second file for a member already held. Every data set that holds
240
+ // anything but JCL and maps is a copy library, searched in name order: the export does not say
241
+ // what SYSLIB concatenated.
242
+ export function pdsExportTree(dir, opts = {}) {
243
+ const idx = buildFileIndex(dir);
244
+ const top = realpathSync(dir);
245
+ const root = `${resolve(dir)}@pds`;
246
+ const origins = new Map();
247
+ const found = new Map();
248
+ const notMembers = [];
249
+ const duplicates = [];
250
+ const dataSets = new Set();
251
+ for (const file of [...idx.index.values()].sort()) {
252
+ const where = relative(dir, dirname(file)).split(sep).filter(Boolean);
253
+ const dsn = where.join('.').toUpperCase();
254
+ const member = MEMBER.exec(basename(file));
255
+ if (!member || dsn.length > 44 || (dsn && !dsn.split('.').every((q) => QUALIFIER.test(q)))) {
256
+ notMembers.push(relPath(dir, file));
257
+ continue;
258
+ }
259
+ const at = resolve(root, ...(dsn ? [dsn] : []), member[1].toUpperCase());
260
+ if (found.has(at)) { duplicates.push(`${relPath(dir, file)}: ${relPath(root, at)} is ${found.get(at)}`); continue; }
261
+ let real;
262
+ try { real = realpathSync(file); } catch { real = file; }
263
+ if (real !== top && !real.startsWith(top + sep)) { notMembers.push(relPath(dir, file)); continue; }
264
+ origins.set(at, real);
265
+ found.set(at, relPath(dir, file));
266
+ dataSets.add(dirname(at));
267
+ }
268
+ const kinds = new Map();
269
+ const byKind = {};
270
+ const copyDirs = new Set();
271
+ for (const [at, file] of origins) {
272
+ let kind = null;
273
+ try { kind = kindOfBytes(readMember(file)); } catch { kind = null; }
274
+ kinds.set(at, kind);
275
+ byKind[kind || 'other'] = (byKind[kind || 'other'] || 0) + 1;
276
+ if (kind !== 'jcl' && kind !== 'bms') copyDirs.add(dirname(at));
277
+ }
278
+ const store = {
279
+ keys: () => origins.keys(),
280
+ has: (p) => origins.has(p),
281
+ get: (p) => (origins.has(p) ? readMember(origins.get(p)) : undefined),
282
+ };
283
+ const tree = heldTree({
284
+ kind: 'pds-export', root, store, rel: (p) => relPath(root, p), systemDirs: opts.systemDirs || [],
285
+ copyDirs: [...copyDirs], classify: () => (p) => kinds.get(p) ?? null, symlinks: idx.symlinks,
286
+ unreadableDirs: idx.unreadableDirs.map((d) => resolve(root, relative(dir, d))),
287
+ });
288
+ return Object.assign(tree, {
289
+ dir,
290
+ origin: (p) => found.get(p) ?? null,
291
+ export: {
292
+ dataSets: dataSets.size,
293
+ members: origins.size,
294
+ byKind,
295
+ ...(notMembers.length ? { notMembers: notMembers.length, notMembersListed: notMembers.slice(0, MAX_LISTED) } : {}),
296
+ ...(duplicates.length ? { duplicates: duplicates.length, duplicatesListed: duplicates.slice(0, MAX_LISTED) } : {}),
297
+ },
298
+ });
129
299
  }
130
300
 
131
301
  // What a rule set calls. Given a tree it uses it; given none it builds the directory one, so every
package/lib/parser.mjs CHANGED
@@ -634,7 +634,7 @@ function isProgramFile(p, ctx) {
634
634
  const seen = (ctx.programFiles ||= new Map());
635
635
  if (!seen.has(p)) {
636
636
  let text = '';
637
- try { text = readSource(p).text; } catch { /* an unreadable candidate is not known to be a program */ }
637
+ try { text = ctx.readText(p); } catch { /* an unreadable candidate is not known to be a program */ }
638
638
  seen.set(p, /^[^*\n]{0,6}\s*PROGRAM-ID\s*\./im.test(text));
639
639
  }
640
640
  return seen.get(p);
@@ -787,7 +787,7 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
787
787
  if (!path) return [];
788
788
  if (stack.includes(path) || stack.length > 40) { record.status = 'recursive'; return []; }
789
789
  if (++ctx.inclusions > MAX_INCLUSIONS || ctx.copyTokens > MAX_COPY_TOKENS) { record.status = 'expansion-limit'; return []; }
790
- const src = readSource(path).text;
790
+ const src = ctx.readText(path);
791
791
  // A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
792
792
  // directive earlier in the including file may have changed from the file's starting format.
793
793
  const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
@@ -1848,7 +1848,7 @@ function collectCopybookDefines(src, format, ctx, depth) {
1848
1848
  const path = resolveCopy(name, null, ctx);
1849
1849
  if (!path || seen.has(path)) continue;
1850
1850
  seen.add(path);
1851
- const text = readSource(path).text;
1851
+ const text = ctx.readText(path);
1852
1852
  const fmt = detectFormat(text) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : current;
1853
1853
  normalize(text, fmt, ctx.defines, ctx.std);
1854
1854
  collectCopybookDefines(text, fmt, ctx, depth + 1);
@@ -1880,7 +1880,9 @@ export function parseSource(src, file, opts = {}) {
1880
1880
  const scheme = BINARY_SIZE[opts.std || 'default'] || BINARY_SIZE.default;
1881
1881
  const ctx = { mainDir: opts.mainDir || dirname(file), includeDirs: opts.includeDirs || [], systemDirs: opts.systemDirs || [],
1882
1882
  fileIndex: opts.fileIndex || null, allowAbsoluteCopy: !!opts.allowAbsoluteCopy, cache: new Map(), copies: [], diags: [], copyFormat: opts.copyFormat || format, defines: new Map(), std: opts.std,
1883
- inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0 };
1883
+ inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0,
1884
+ // Where a copybook's text comes from: the disk, or the source tree the program was read from.
1885
+ readText: opts.readText || ((p) => readSource(p).text) };
1884
1886
  collectCopybookDefines(src, format, ctx, 0);
1885
1887
  const norm = normalize(src, format, ctx.defines, opts.std);
1886
1888
  ctx.mainFormat = norm.finalFormat;
package/lib/revision.json CHANGED
@@ -1 +1 @@
1
- {"commit":"54c9654b8c13bc22c8d5683b02551362d95f0d0a","tag":"v0.2.140"}
1
+ {"commit":"f5995391b095f62da69d501ea7f6bd11a75bdc6e","tag":"v0.2.150"}
package/lib/sbom.mjs CHANGED
@@ -102,7 +102,20 @@ export function sbom(root, opts = {}) {
102
102
  const idx = tree.index;
103
103
  const top = resolve(root);
104
104
  const within = (p) => resolve(p).startsWith(top + sep);
105
- const refOf = (p) => (within(p) ? relPath(root, p) : `copylib:${basename(p)}`);
105
+ // A member outside the tree is named by its file name, and by its directory's digest as well where
106
+ // another library holds a member of the same name.
107
+ const copylibRefs = new Map();
108
+ const claimed = new Map();
109
+ const refOf = (p) => {
110
+ if (within(p)) return relPath(root, p);
111
+ if (!copylibRefs.has(p)) {
112
+ let ref = `copylib:${basename(p)}`;
113
+ if (claimed.has(ref) && claimed.get(ref) !== p) ref = `copylib:${basename(p)}#${sha256(dirname(resolve(p))).slice(0, 8)}`;
114
+ claimed.set(ref, p);
115
+ copylibRefs.set(p, ref);
116
+ }
117
+ return copylibRefs.get(p);
118
+ };
106
119
 
107
120
  const entries = [];
108
121
  const programIds = new Map();
@@ -110,17 +123,24 @@ export function sbom(root, opts = {}) {
110
123
  const procIds = new Map();
111
124
  for (const f of tree.list().filter(inScope(opts)).sort()) {
112
125
  let bytes;
113
- let src;
114
- try { bytes = readFileSync(f); src = readSource(f).text; } catch { continue; }
115
- const kind = kindOf(f, src);
126
+ let src = null;
127
+ try { bytes = readFileSync(f); } catch { continue; }
128
+ try { src = readSource(f).text; } catch { /* its bytes are listed; its text could not be decoded */ }
129
+ const kind = src === null ? (isProgram(f) ? 'program' : isCopybook(f) ? 'copybook' : isJcl(f) ? 'jcl' : isBms(f) ? 'bms' : null) : kindOf(f, src);
116
130
  if (!kind) continue;
117
- const entry = { file: f, ref: relPath(root, f), bytes, src, kind };
131
+ const entry = { file: f, ref: relPath(root, f), bytes, src: src ?? '', kind, ...(src === null ? { unparsed: true, undecodable: true } : {}) };
132
+ if (src === null) { entries.push(entry); continue; }
118
133
  if (kind === 'program') {
119
134
  try {
120
135
  const res = parseSource(src, f, { format: 'auto', includeDirs: idx.copyDirs, fileIndex: idx.index, mainDir: dirname(f), copyFormat: 'auto', systemDirs: opts.systemDirs || [] });
121
136
  entry.programs = res.programs;
122
137
  entry.copies = res.copies;
123
- for (const p of res.programs) if (p.id) programIds.set(String(p.id).toUpperCase(), entry.ref);
138
+ for (const p of res.programs) {
139
+ if (!p.id) continue;
140
+ const id = String(p.id).toUpperCase();
141
+ if (!programIds.has(id)) programIds.set(id, []);
142
+ if (!programIds.get(id).includes(entry.ref)) programIds.get(id).push(entry.ref);
143
+ }
124
144
  } catch { entry.unparsed = true; }
125
145
  } else if (kind === 'jcl' || kind === 'proc') {
126
146
  try { entry.jcl = parseJcl(src, f); } catch { entry.unparsed = true; }
@@ -139,9 +159,10 @@ export function sbom(root, opts = {}) {
139
159
  const dependencies = new Map();
140
160
  const depend = (from, to) => { if (!dependencies.has(from)) dependencies.set(from, new Set()); dependencies.get(from).add(to); };
141
161
  // A program the estate does not hold is a component of its own, so an edge to it is not lost.
162
+ // A PROGRAM-ID two files hold is an edge to each: which one a run loads is the load library's order.
142
163
  const runs = (from, name) => {
143
- const target = programIds.get(name);
144
- if (target) { if (target !== from) depend(from, target); return; }
164
+ const targets = programIds.get(name);
165
+ if (targets) { for (const t of targets) if (t !== from) depend(from, t); return; }
145
166
  if (platformRoutine(name)) return;
146
167
  const ref = `program:${name}`;
147
168
  depend(from, ref);
@@ -153,9 +174,12 @@ export function sbom(root, opts = {}) {
153
174
  for (const e of entries) {
154
175
  const properties = [{ name: 'cobolwork:kind', value: e.kind }];
155
176
  if (e.unparsed) properties.push({ name: 'cobolwork:unparsed', value: 'true' });
177
+ if (e.undecodable) properties.push({ name: 'cobolwork:undecodable', value: 'true' });
156
178
  if (e.kind === 'program' && e.programs && e.programs.length) {
157
179
  const ids = e.programs.map((p) => p.id).filter(Boolean);
158
180
  if (ids.length) properties.push({ name: 'cobolwork:program-id', value: ids.join(',') });
181
+ const twins = [...new Set(ids.flatMap((id) => (programIds.get(String(id).toUpperCase()) || []).filter((r) => r !== e.ref)))].sort(order);
182
+ if (twins.length) properties.push({ name: 'cobolwork:program-id-also-in', value: twins.join(',') });
159
183
  properties.push({ name: 'cobolwork:format', value: detectFormat(e.src) });
160
184
  const options = optionCards(e.src).flatMap((c) => c.options);
161
185
  if (options.length) properties.push({ name: 'cobolwork:options', value: options.join(' ') });
@@ -166,7 +190,7 @@ export function sbom(root, opts = {}) {
166
190
  if (!within(c.path) && !components.has(ref)) {
167
191
  let b = null;
168
192
  try { b = readFileSync(c.path); } catch { /* read at parse time, gone since */ }
169
- components.set(ref, { 'bom-ref': ref, type: 'file', name: ref, ...(b ? { hashes: [{ alg: 'SHA-256', content: sha256(b) }] } : {}), properties: [{ name: 'cobolwork:kind', value: 'copybook' }, { name: 'cobolwork:source', value: 'copy library' }] });
193
+ components.set(ref, { 'bom-ref': ref, type: 'file', name: ref, ...(b ? { hashes: [{ alg: 'SHA-256', content: sha256(b) }] } : {}), properties: [{ name: 'cobolwork:kind', value: 'copybook' }, { name: 'cobolwork:source', value: 'copy library' }, ...(b ? [] : [{ name: 'cobolwork:unread', value: 'true' }])] });
170
194
  }
171
195
  }
172
196
  const calls = e.programs.flatMap((p) => p.calls || []);
package/lib/scan.mjs CHANGED
@@ -1,5 +1,5 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
- import { join } from 'node:path';
2
+ import { join, resolve } from 'node:path';
3
3
  import { sortFindings, tally, evidenceMap, EVIDENCE, EXPLOITABILITY } from './kernel/findings.mjs';
4
4
  import { directoryTree } from './kernel/source-tree.mjs';
5
5
  import { REGISTRY, RULE_SETS, ALL_RULES, reportKey } from './kernel/registry.mjs';
@@ -13,6 +13,7 @@ import { rollup } from './sets/flow.mjs';
13
13
  import { loadSite } from './site.mjs';
14
14
  import { reachFeedPaths, loadReachFeed, reachResolver, stampReach, stampEffect } from './reach.mjs';
15
15
  import { stampExploitability, byVerdict, HANDLING, witnessFeedPaths, loadWitnessFeed, applyWitness } from './exploitability.mjs';
16
+ import { executionFeedPaths, loadExecutionFeed, applyExecution } from './execution.mjs';
16
17
 
17
18
  import { FLOW_MODEL, SCHEMA_VERSION, TOOL_VERSION } from './version.mjs';
18
19
 
@@ -133,7 +134,7 @@ function scanEach(root, opts) {
133
134
  // Reach and effect annotate rather than re-rank; where no fact is declared, a note says so rather
134
135
  // than the report reading as a clean bill.
135
136
  export function applyEstateFacts(findings, checked, root, opts = {}) {
136
- const site = loadSite(root, opts.site || null);
137
+ const site = loadSite(root, opts.site || null, opts.tree);
137
138
  const feedRoot = opts.feedRoot || root;
138
139
  const reachFeeds = reachFeedPaths(opts).map((p) => loadReachFeed(p, { root: feedRoot }));
139
140
  const reach = reachResolver({ feeds: reachFeeds, site });
@@ -151,6 +152,8 @@ export function applyEstateFacts(findings, checked, root, opts = {}) {
151
152
  const witnessed = applyWitness(findings, checked, witnessFeeds);
152
153
  if (witnessed.overruled.length) sortFindings(findings);
153
154
  const byExploitability = byVerdict(findings);
155
+ const executionFeeds = executionFeedPaths(opts).map((p) => loadExecutionFeed(p));
156
+ const { byExecution } = applyExecution(findings, root, executionFeeds);
154
157
  const driven = byExploitability ? byExploitability['attacker-driven'] : 0;
155
158
  const loaded = witnessFeeds.filter((f) => !f.problem);
156
159
  return {
@@ -164,6 +167,8 @@ export function applyEstateFacts(findings, checked, root, opts = {}) {
164
167
  ...(byExploitability?.exploitable || byExploitability?.confirmed ? { handling: HANDLING } : {}),
165
168
  ...(loaded.length ? { witnessFeeds: loaded.map((f) => ({ file: f.file, witness: f.witness, recorded: f.recorded })), byWitness: witnessed.byWitness } : {}),
166
169
  ...(witnessFeeds.some((f) => f.problem || f.refused.length) ? { witnessFeedProblems: witnessFeeds.flatMap((f) => (f.problem ? [`${f.file}: ${f.problem}`] : f.refused.map((r) => `${f.file}: ${r.why}`))) } : {}),
170
+ ...(byExecution ? { executionFeeds: executionFeeds.filter((f) => !f.problem).map((f) => ({ file: f.file, programs: f.programs.size })), byExecution } : {}),
171
+ ...(executionFeeds.some((f) => f.problem) ? { executionFeedProblems: executionFeeds.filter((f) => f.problem).map((f) => `${f.file}: ${f.problem}`) } : {}),
167
172
  // A result no path finding carries: the code changed since the test, or the route is gone.
168
173
  ...(witnessed.unmatched.length ? { witnessUnmatched: witnessed.unmatched } : {}),
169
174
  // Routes the reading refuted that the estate reproduced: the check model was wrong about them.
@@ -234,10 +239,12 @@ export function scanAll(root, opts = {}) {
234
239
  const startedBy = parts.find((p) => p.tool === 'cobolwork-flow')?.startedBy || {};
235
240
  for (const p of parts) for (const f of p.findings) findings.push(!f.startedBy && f.program && startedBy[f.program] ? { ...f, startedBy: startedBy[f.program] } : { ...f });
236
241
  sortFindings(findings);
237
- const identity = stampFingerprints(findings, { root, repo: opts.repoName || '' });
242
+ // A tree held in memory is read for the fingerprint's line text as it was for the finding.
243
+ const read = tree.kind === 'directory' ? null : (path) => tree.text(resolve(tree.root, path)).text;
244
+ const identity = stampFingerprints(findings, { root, repo: opts.repoName || '', read });
238
245
  // Routes a check stops keep their identity too, so a finding a fix clears can be found again here.
239
246
  const checked = parts.find((p) => p.tool === 'cobolwork-flow')?.checked || [];
240
- stampFingerprints(checked, { root, repo: opts.repoName || '' });
247
+ stampFingerprints(checked, { root, repo: opts.repoName || '', read });
241
248
 
242
249
  const estate = applyEstateFacts(findings, checked, root, opts);
243
250
 
@@ -132,7 +132,7 @@ function withFeeds(coverage, feeds) {
132
132
  }
133
133
 
134
134
  export function scanBuild(root, opts = {}) {
135
- const site = loadSite(root, opts.site || null);
135
+ const site = loadSite(root, opts.site || null, opts.tree);
136
136
  const tree = treeFor(root, opts);
137
137
  const files = tree.list().filter(inScope(opts)).filter(isBuildFile);
138
138
  const findings = [];
package/lib/sets/cics.mjs CHANGED
@@ -136,6 +136,31 @@ function execOptions(exec) {
136
136
 
137
137
  const subtreeRead = (x) => x.directRefs > 0 || x.children.some(subtreeRead);
138
138
 
139
+ // Whether EIBCALEN bounds what the program reads: a condition tests it, or a field it was moved or
140
+ // computed into; it is the start or length of a reference modification, or what an OCCURS DEPENDING
141
+ // ON counts by. A CALL handed it, or the whole EIB, may test it where this program cannot see.
142
+ // Named anywhere else - logged, passed on to a LINK - it checks nothing.
143
+ function testsLength(p) {
144
+ // A statement's own tokens, since its sources leave out what a condition or an expression wraps
145
+ // in parentheses: IF (EIBCALEN > 0).
146
+ const tokens = (p.proc && p.proc.tokens) || [];
147
+ const own = (st) => (st.at != null ? tokens.slice(st.at + 1, st.end) : st.sources || []);
148
+ const carriers = new Set(['EIBCALEN']);
149
+ for (let grew = true; grew;) {
150
+ grew = false;
151
+ for (const st of p.statements || []) {
152
+ if (!own(st).some((t) => carriers.has(t.u))) continue;
153
+ for (const t of st.targets || []) if (!carriers.has(t.u)) { carriers.add(t.u); grew = true; }
154
+ }
155
+ }
156
+ const carries = (toks) => (toks || []).some((t) => t && carriers.has(t.u));
157
+ return p.items.some((i) => i.dependingOn === 'EIBCALEN') || (p.statements || []).some((st) =>
158
+ ((st.verb === 'IF' || st.verb === 'EVALUATE' || st.verb === 'WHEN') && carries(own(st)))
159
+ || (st.loops || []).some((l) => carries(l.until))
160
+ || (st.indexes || []).some((x) => x.kind.startsWith('refmod-') && carries([x.tok]))
161
+ || (st.verb === 'CALL' && (st.sources || []).some((t) => t.u === 'EIBCALEN' || t.u === 'DFHEIBLK')));
162
+ }
163
+
139
164
  // Commands whose failure a program ordinarily lets pass: removing a temporary-storage queue that may
140
165
  // not exist, a line written to a transient-data log, the clock, a screen sent to a terminal that may
141
166
  // have gone, an ASSIGN, and the RETURN that ends the task anyway.
@@ -191,7 +216,6 @@ export function scanCics(root, opts = {}) {
191
216
  // The whole of what the second pass reads, computed once, while the tree is still here.
192
217
  function summarise(p, path) {
193
218
  const isSignOn = [...new Set([...p.refs.map(x => x.tok.u), ...p.execs.flatMap(e => e.toks.filter(t => t.t === 'word').map(t => t.u))])].some((w) => PASSWORD.test(w));
194
- const words = new Set([...p.refs.map(x => x.tok.u), ...p.execs.flatMap(e => e.toks.filter(t => t.t === 'word').map(t => t.u))]);
195
219
  const commarea = p.items.find(i => i.name === 'DFHCOMMAREA' && i.section === 'LINKAGE');
196
220
  // A variable naming the program usually holds one literal from its VALUE clause.
197
221
  const valueOf = (name) => {
@@ -236,7 +260,7 @@ export function scanCics(root, opts = {}) {
236
260
  ownCics: p.execs.some((e) => e.kind === 'CICS'),
237
261
  pointsItself,
238
262
  commarea: commarea ? { ...where(commarea, path), size: commarea.size, read: subtreeRead(commarea) } : null,
239
- checksLength: words.has('EIBCALEN'),
263
+ checksLength: testsLength(p),
240
264
  transfers,
241
265
  signOnBypasses: isSignOn ? signOnBypasses(p, path) : [],
242
266
  // Gated the same way: a program with no password-named field is not answering a sign-on,
@@ -393,7 +417,7 @@ export function scanCics(root, opts = {}) {
393
417
  findings.push({
394
418
  rule: 'cics-commarea-without-length-check', path: s.commarea.path, line: s.commarea.line, program: s.id,
395
419
  ...(covered ? { sev: 'low' } : {}),
396
- detail: `${s.id} reads DFHCOMMAREA (${s.commarea.size} bytes declared) and never references EIBCALEN, so a shorter or absent communication area is read as if it were whole${covered ? `; every caller in the tree passes at least ${s.commarea.size} bytes, and no transaction starts it` : ''}`,
420
+ detail: `${s.id} reads DFHCOMMAREA (${s.commarea.size} bytes declared) and never tests EIBCALEN, so a shorter or absent communication area is read as if it were whole${covered ? `; every caller in the tree passes at least ${s.commarea.size} bytes, and no transaction starts it` : ''}`,
397
421
  });
398
422
  }
399
423
  }
@@ -7,7 +7,7 @@ import { finish } from '../kernel/findings.mjs';
7
7
  import { report } from '../kernel/ruleset.mjs';
8
8
  import { treeFor, noteUnread, noteUnparsed } from '../kernel/source-tree.mjs';
9
9
  import { eachWithinMemory } from '../kernel/memory.mjs';
10
- import { cobolCard } from '../card.mjs';
10
+ import { cobolCard } from '../cards.mjs';
11
11
 
12
12
  // A copybook is resolved by name along a search path: the program's own directory first, then the
13
13
  // include directories, then the system library. Two files answering to one name mean the record