@portll/cobolwork 0.2.140 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/inventory.mjs CHANGED
@@ -1,7 +1,6 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
- import { dirname } from 'node:path';
3
- import { parseSource, buildFileIndex, detectFormat } from './parser.mjs';
4
- import { inScope, isCopybook, isJcl, isProgram, readSource, relPath } from './sources.mjs';
2
+ import { detectFormat } from './parser.mjs';
3
+ import { inScope, isCopybook, isJcl, isProgram, relPath } from './sources.mjs';
5
4
  import { SCHEMA_VERSION, TOOL_VERSION } from './version.mjs';
6
5
  import { treeFor } from './kernel/source-tree.mjs';
7
6
 
@@ -36,19 +35,19 @@ export function inventory(root, opts = {}) {
36
35
  out.summary.programFiles++;
37
36
  let src;
38
37
  try {
39
- const s = readSource(f);
38
+ const s = tree.text(f);
40
39
  src = s.text;
41
- if (s.encoding === 'ebcdic') { out.summary.filesEbcdic++; out.ebcdic.push(relPath(root, f)); }
42
- } catch (e) { out.summary.filesUnreadable++; out.unreadable.push(`${relPath(root, f)}: ${e.code || e.name}`); continue; }
40
+ if (s.encoding === 'ebcdic') { out.summary.filesEbcdic++; out.ebcdic.push(tree.rel(f)); }
41
+ } catch (e) { out.summary.filesUnreadable++; out.unreadable.push(`${tree.rel(f)}: ${e.code || e.name}`); continue; }
43
42
  // A file with a program extension and no program in it is usually a copybook named .cbl; it is
44
43
  // counted so the difference between program files and programs read is accounted for.
45
44
  if (!/PROCEDURE\s+DIVISION|PROGRAM-ID/i.test(src)) { out.summary.notPrograms++; continue; }
46
45
  let res;
47
46
  try {
48
- res = parseSource(src, f, { format: 'auto', includeDirs: idx.copyDirs, fileIndex: idx.index, mainDir: dirname(f), copyFormat: 'auto', systemDirs: opts.systemDirs || [] });
47
+ res = tree.parse(f, src);
49
48
  } catch (e) {
50
49
  out.summary.filesUnreadable++;
51
- out.unreadable.push(`${relPath(root, f)}: ${e.code || e.name}`);
50
+ out.unreadable.push(`${tree.rel(f)}: ${e.code || e.name}`);
52
51
  continue;
53
52
  }
54
53
  out.summary.filesScanned++;
@@ -60,7 +59,7 @@ export function inventory(root, opts = {}) {
60
59
  if (/EXEC\s+DLI/i.test(src)) out.summary.withDli++;
61
60
  for (const c of res.copies) {
62
61
  if (c.status === 'resolved') out.summary.copiesResolved++;
63
- else if (String(c.status).startsWith('refused')) { out.summary.copiesRefused++; out.refusedCopies.push({ name: c.name, from: relPath(root, c.file), line: c.line, why: c.status }); }
62
+ else if (String(c.status).startsWith('refused')) { out.summary.copiesRefused++; out.refusedCopies.push({ name: c.name, from: tree.rel(c.file), line: c.line, why: c.status }); }
64
63
  // Before the system names, which a copybook in the tree is free to take.
65
64
  else if (c.status === 'expansion-limit') out.summary.copiesOverLimit = (out.summary.copiesOverLimit || 0) + 1;
66
65
  else if (c.status === 'system' || SYSTEM_COPY.test(c.name)) out.summary.copiesSystem++;
package/lib/jcl.mjs CHANGED
@@ -13,7 +13,9 @@
13
13
  // there.
14
14
  import { readSource } from './sources.mjs';
15
15
  import { copiesOf, tsoCommands } from './utilities.mjs';
16
- import { statementCard } from './card.mjs';
16
+ import { statementCard, parseOperands } from './cards.mjs';
17
+
18
+ export { splitOperands, parseOperands } from './cards.mjs';
17
19
 
18
20
  // The operations a statement may carry. Anything else in the operation field is a statement this
19
21
  // reader does not know, which is a diagnostic rather than a silent skip.
@@ -38,53 +40,6 @@ const SYSTEM_SYMBOLS = new Set(`SYSUID SYSALVL SYSCLONE SYSNAME SYSOSLVL SYSPLEX
38
40
  YYMMDD LYYMMDD HHMMSS LHHMMSS DAY HR MIN SEC JDAY MON YR2 YR4 WDAY
39
41
  LDATE LDAY LHR LMIN LSEC LJDAY LMON LYR2 LYR4 LWDAY LTIME JOBNAME DS SEQ DATE TIME`.split(/\s+/));
40
42
 
41
- // Splits an operand field on commas that are not inside parentheses or quotes. JCL nests both, and
42
- // a naive split on comma turns DISP=(NEW,CATLG,DELETE) into three operands.
43
- export function splitOperands(s) {
44
- const out = [];
45
- let depth = 0, quoted = false, start = 0;
46
- for (let i = 0; i < s.length; i++) {
47
- const c = s[i];
48
- if (quoted) { if (c === "'") { if (s[i + 1] === "'") i++; else quoted = false; } continue; }
49
- if (c === "'") quoted = true;
50
- else if (c === '(') depth++;
51
- else if (c === ')') depth = Math.max(0, depth - 1);
52
- else if (c === ',' && depth === 0) { out.push(s.slice(start, i)); start = i + 1; }
53
- }
54
- out.push(s.slice(start));
55
- return out.filter((x, i, a) => x.length || i < a.length - 1);
56
- }
57
-
58
- // Operands are positional until the first KEY=VALUE, and keyword after it. Both forms matter:
59
- // EXEC takes its procedure name positionally, and everything interesting on DD is a keyword.
60
- export function parseOperands(field) {
61
- const positional = [];
62
- const keywords = new Map();
63
- for (const raw of splitOperands(field)) {
64
- const part = raw.trim();
65
- if (!part) continue;
66
- const eq = keywordSplit(part);
67
- if (eq < 0) positional.push(part);
68
- else keywords.set(part.slice(0, eq).toUpperCase(), part.slice(eq + 1));
69
- }
70
- return { positional, keywords };
71
- }
72
-
73
- // The first '=' that is not inside parentheses or quotes. DCB=(RECFM=FB,LRECL=80) is one keyword
74
- // whose value happens to contain more of them.
75
- function keywordSplit(part) {
76
- let depth = 0, quoted = false;
77
- for (let i = 0; i < part.length; i++) {
78
- const c = part[i];
79
- if (quoted) { if (c === "'") { if (part[i + 1] === "'") i++; else quoted = false; } continue; }
80
- if (c === "'") quoted = true;
81
- else if (c === '(') depth++;
82
- else if (c === ')') depth = Math.max(0, depth - 1);
83
- else if (c === '=' && depth === 0) return i;
84
- }
85
- return -1;
86
- }
87
-
88
43
  // Symbolic substitution. &NAME ends at a non-name character, and a trailing dot is a separator
89
44
  // that is consumed rather than kept: &PREFIX..DATA resolves to <prefix>.DATA.
90
45
  export function substitute(text, symbols) {
@@ -0,0 +1,60 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // Reading a revision out of a repository with git plumbing only.
3
+ import { spawnSync } from 'node:child_process';
4
+
5
+ // A reviewed repository's .git/config can name commands; with fsmonitor off and plumbing only, git runs none.
6
+ const GIT_ENV = { ...process.env, GIT_OPTIONAL_LOCKS: '0', GIT_TERMINAL_PROMPT: '0' };
7
+ export const git = (repo, args, opts = {}) => spawnSync('git', ['-c', 'core.fsmonitor=false', '-C', repo, ...args], { env: GIT_ENV, maxBuffer: 256 * 1024 * 1024, ...opts });
8
+ export const refused = (what, ref, why) => Object.assign(new Error(`${what} ${ref} failed: ${why}`), { code: 'EDIFFREF' });
9
+ const BATCH_BYTES = 64 * 1024 * 1024;
10
+
11
+ export function treeOf(repo, ref) {
12
+ if (!ref || String(ref).startsWith('-')) throw refused('git rev-parse', ref, 'a revision cannot be empty or start with "-"');
13
+ const r = git(repo, ['rev-parse', '--verify', '--quiet', '--end-of-options', `${ref}^{tree}`], { encoding: 'utf8' });
14
+ const oid = String(r.stdout || '').trim();
15
+ if (r.status !== 0 || !/^[0-9a-f]{40}([0-9a-f]{24})?$/.test(oid)) throw refused('git rev-parse', ref, String(r.stderr || '').trim() || 'not a revision');
16
+ return oid;
17
+ }
18
+
19
+ // The parts of a tree path, or null if a part is empty, climbs, names a drive or stream, or is .git.
20
+ export function treePathParts(path) {
21
+ const parts = path.split('/');
22
+ return parts.some((p) => !p || p === '.' || p === '..' || /[\\:\0]/.test(p) || p.toLowerCase() === '.git') ? null : parts;
23
+ }
24
+
25
+ // Every file in the tree `oid` as committed, as [{ path, bytes }] in tree order: blobs as stored,
26
+ // with no filter or line-ending conversion. Links and submodules are left out and counted.
27
+ export function revisionBlobs(repo, oid, ref = oid) {
28
+ const listed = git(repo, ['ls-tree', '-r', '-z', '-l', '--full-tree', oid]);
29
+ if (listed.status !== 0) throw refused('git ls-tree', ref, String(listed.stderr || '').trim() || `exit ${listed.status}`);
30
+ const blobs = [];
31
+ let links = 0;
32
+ for (const entry of listed.stdout.toString('utf8').split('\0')) {
33
+ const m = /^(\d{6}) blob ([0-9a-f]+) +(\d+)\t(.+)$/s.exec(entry);
34
+ if (!m) continue;
35
+ if (m[1] === '120000') { links++; continue; }
36
+ blobs.push({ oid: m[2], size: Number(m[3]), path: m[4] });
37
+ }
38
+ // Batched by bytes as well as count, and each batch's buffer sized to hold it, so one large blob
39
+ // cannot overflow a buffer sized for the others.
40
+ const batches = [];
41
+ for (const b of blobs) {
42
+ const last = batches[batches.length - 1];
43
+ if (!last || last.length >= 500 || last.bytes + b.size > BATCH_BYTES) batches.push(Object.assign([b], { bytes: b.size }));
44
+ else { last.push(b); last.bytes += b.size; }
45
+ }
46
+ for (const batch of batches) {
47
+ const r = git(repo, ['cat-file', '--batch'], { input: batch.map((b) => b.oid).join('\n') + '\n', maxBuffer: batch.bytes + batch.length * 128 + 4096 });
48
+ if (r.status !== 0) throw refused('git cat-file', ref, String(r.stderr || '').trim() || (r.error ? r.error.code || r.error.message : `exit ${r.status}`));
49
+ let at = 0;
50
+ for (const b of batch) {
51
+ const nl = r.stdout.indexOf(0x0a, at);
52
+ const head = r.stdout.toString('utf8', at, nl).split(' ');
53
+ if (head[1] !== 'blob') throw refused('git cat-file', ref, `${b.oid} is ${head[1] || 'missing'}`);
54
+ const size = Number(head[2]);
55
+ b.bytes = r.stdout.subarray(nl + 1, nl + 1 + size);
56
+ at = nl + 1 + size + 1;
57
+ }
58
+ }
59
+ return { blobs, links };
60
+ }
@@ -131,14 +131,15 @@ function codeText(line, kind, format) {
131
131
  return maskSecrets(s).replace(/\s+/g, ' ').trim();
132
132
  }
133
133
 
134
- // Reads each file once, however many findings it holds.
134
+ // Reads each file once, however many findings it holds, from disk or through the reader the
135
+ // caller's source tree supplies.
135
136
  function fileFacts(root, path, cache) {
136
137
  let facts = cache.get(path);
137
138
  if (facts) return facts;
138
139
  facts = { kind: 'other', lines: [], scopes: null, format: null };
139
140
  cache.set(path, facts);
140
141
  let src;
141
- try { src = readSource(join(root, path)).text; } catch { return facts; }
142
+ try { src = cache.read ? cache.read(path) : readSource(join(root, path)).text; } catch { return facts; }
142
143
  facts.lines = src.split(/\r?\n/);
143
144
  if (isJcl(join(root, path))) {
144
145
  facts.kind = 'jcl';
@@ -171,8 +172,8 @@ const digest = (parts) => createHash('sha256').update(parts.join('\u0000')).dige
171
172
  // Stamps `fingerprint` on every finding and returns how many shared one with an earlier finding.
172
173
  // `repo` names the repository when one report holds several, where the same program id in two
173
174
  // repositories is two programs.
174
- export function stampFingerprints(findings, { root, repo = '' } = {}) {
175
- const cache = new Map();
175
+ export function stampFingerprints(findings, { root, repo = '', read = null } = {}) {
176
+ const cache = Object.assign(new Map(), { read });
176
177
  const seen = new Set();
177
178
  let shared = 0;
178
179
  for (const f of findings) {
@@ -188,8 +189,8 @@ export function stampFingerprints(findings, { root, repo = '' } = {}) {
188
189
  // Looser keys for one finding seen in two trees whose line was edited between them: `scope` is the
189
190
  // fingerprint without the line's text, and `route` names a data-flow finding by the statement its
190
191
  // trace starts at. Null where a finding has no route.
191
- export function pairingKeys(findings, { root, repo = '' } = {}) {
192
- const cache = new Map();
192
+ export function pairingKeys(findings, { root, repo = '', read = null } = {}) {
193
+ const cache = Object.assign(new Map(), { read });
193
194
  return findings.map((f) => {
194
195
  const { scope, subject } = partsOf(f, root, cache);
195
196
  const src = f.evidence === 'path' && f.related && f.related[0];
@@ -26,13 +26,16 @@
26
26
  //
27
27
  // An adapter that does not have a filesystem underneath it has to make that decision itself, and
28
28
  // `contains` is where it makes it. For a tree held in memory it is key membership. For a git
29
- // revision it is membership of the tree object, which cannot name a path outside itself. Anything
30
- // cleverer than those two deserves the six containment tests pointed at it before it ships:
31
- // test/sources.test.mjs:44, :63, :88, :101, :144 and test/review.test.mjs:114.
32
- import { readFileSync, realpathSync } from 'node:fs';
33
- import { dirname, resolve, sep } from 'node:path';
29
+ // revision it is membership of the tree object, which cannot name a path outside itself. For a PDS
30
+ // export it is membership of the members the walk found, each read with no link followed. Anything
31
+ // cleverer than those deserves the six containment tests pointed at it before it ships:
32
+ // test/sources.test.mjs:44, :63, :88, :101, :144 and test/review.test.mjs:114, as
33
+ // test/git-tree.test.mjs and test/pds-export.test.mjs do.
34
+ import { closeSync, constants, fstatSync, openSync, readFileSync, readSync, realpathSync } from 'node:fs';
35
+ import { basename, dirname, relative, resolve, sep } from 'node:path';
34
36
  import { buildFileIndex, parseSource } from '../parser.mjs';
35
- import { readSource, relPath } from '../sources.mjs';
37
+ import { classifyUnder, decodeSource, diskClassifier, heldClassifier, kindOfBytes, looksEbcdic, readSource, relPath } from '../sources.mjs';
38
+ import { revisionBlobs, treeOf, treePathParts } from './git.mjs';
36
39
 
37
40
  // The shape every adapter answers to. Checked rather than documented, because an adapter missing a
38
41
  // method fails at the first rule set that happens to call it rather than at the boundary.
@@ -64,10 +67,14 @@ export function directoryTree(root, opts = {}) {
64
67
  const idx = buildFileIndex(root);
65
68
  const systemDirs = opts.systemDirs || [];
66
69
  const top = resolve(root);
70
+ const kindOf = diskClassifier();
71
+ classifyUnder(root, kindOf);
72
+ if (top !== root) classifyUnder(top, kindOf);
67
73
 
68
74
  return {
69
75
  kind: 'directory',
70
76
  root,
77
+ kindOf,
71
78
  // The index itself, for the two callers that need more than a list: the copybook set reads
72
79
  // copyDirs, and the inventory reports on unreadable directories and symlinks.
73
80
  index: idx,
@@ -92,40 +99,203 @@ export function directoryTree(root, opts = {}) {
92
99
  };
93
100
  }
94
101
 
95
- // A tree that was never on disk, for tests. `files` maps a path to its contents, as a string or a
96
- // Buffer. Paths are used exactly as given, so a test reads the way it writes.
97
- //
98
- // It cannot parse. parseSource resolves COPY statements against real directories and reads them
99
- // with readSource, so a program with copybooks needs a filesystem underneath it. A rule set that
100
- // only reads text - hidden, recon, vendor, jcl, build - works against this; one that parses does
101
- // not, and says so rather than parsing something unexpected.
102
- export function memoryTree(files, { root = '/memory' } = {}) {
103
- const store = new Map(Object.entries(files).map(([p, v]) => [p, Buffer.isBuffer(v) ? v : Buffer.from(v, 'latin1')]));
102
+ // A tree whose files are held rather than on disk. The parser finds copybooks in the index built
103
+ // here and reads them through `readText`, so a COPY resolves to a held file or to a system copy
104
+ // library outside the tree, never to a file on disk that happens to sit under this tree's root.
105
+ function heldTree({ kind, root, store, rel, systemDirs = [], copyDirs = null, classify = heldClassifier,
106
+ symlinks = { followed: 0, outside: 0, broken: 0 }, unreadableDirs = [] }) {
107
+ const index = new Map();
108
+ const dirs = new Set(copyDirs || []);
109
+ for (const p of store.keys()) {
110
+ index.set(resolve(root, p).toLowerCase(), p);
111
+ if (!copyDirs && /\.(cpy|copy|inc|cbl|cob)$/i.test(p)) dirs.add(dirname(resolve(root, p)));
112
+ }
113
+ index.root = root;
114
+ const top = resolve(root);
115
+ const absent = (p) => Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
116
+ const bytes = (p) => {
117
+ const b = store.get(p);
118
+ if (!b) throw absent(p);
119
+ return b;
120
+ };
121
+ const kindOf = classify(bytes);
122
+ classifyUnder(top, kindOf);
123
+ const text = (p) => decodeSource(bytes(p));
124
+ const readText = (p) => {
125
+ if (store.has(p)) return text(p).text;
126
+ const r = resolve(root, p);
127
+ if (r === top || r.startsWith(top + sep)) throw absent(p);
128
+ return readSource(p).text;
129
+ };
104
130
  return {
105
- kind: 'memory',
131
+ kind,
106
132
  root,
107
- index: { index: store, copyDirs: [], unreadableDirs: [], symlinks: { followed: 0, outside: 0, broken: 0 } },
133
+ kindOf,
134
+ index: { index, copyDirs: [...dirs].sort(), unreadableDirs, symlinks },
108
135
  list: () => [...store.keys()].sort(),
109
- bytes: (p) => {
110
- const b = store.get(p);
111
- if (!b) throw Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
112
- return b;
113
- },
114
- text: (p) => {
115
- const b = store.get(p);
116
- if (!b) throw Object.assign(new Error(`no such file in this tree: ${p}`), { code: 'ENOENT' });
117
- return { text: b.toString('latin1'), encoding: 'latin1' };
118
- },
119
- rel: (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), ''),
136
+ bytes,
137
+ text,
138
+ rel,
120
139
  // Containment by construction: a path this tree does not hold is a path outside it.
121
140
  contains: (p) => store.has(p),
122
- parse: () => {
123
- throw Object.assign(
124
- new Error('a memory tree cannot parse: COPY resolution reads directories from disk'),
125
- { code: 'ETREEPARSE' },
126
- );
127
- },
141
+ parse: (file, t) => parseSource(t ?? text(file).text, file, {
142
+ format: 'auto',
143
+ includeDirs: [...dirs].sort(),
144
+ fileIndex: index,
145
+ mainDir: dirname(resolve(root, file)),
146
+ copyFormat: 'auto',
147
+ systemDirs,
148
+ readText,
149
+ }),
150
+ };
151
+ }
152
+
153
+ // A tree that was never on disk, for tests. `files` maps a path to its contents, as a string or a
154
+ // Buffer. Paths are used exactly as given, so a test reads the way it writes.
155
+ export function memoryTree(files, { root = '/memory', systemDirs = [] } = {}) {
156
+ const store = new Map(Object.entries(files).map(([p, v]) => [p, Buffer.isBuffer(v) ? v : Buffer.from(v, 'latin1')]));
157
+ const rel = (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), '');
158
+ return heldTree({ kind: 'memory', root, store, rel, systemDirs });
159
+ }
160
+
161
+ // A git revision, read with plumbing into memory: nothing is written to disk. Its root is a path
162
+ // beside the repository that does not exist, `<repo>@<tree>`, so every path it answers for is
163
+ // absolute and none of them can be found on disk. The tree object is the containment: it cannot
164
+ // name a path outside itself, and a path that would climb, name a drive or stream, or enter .git
165
+ // is left out, as it is when a revision is written to disk. Links and submodules are not read.
166
+ export function gitTree(repo, ref, opts = {}) {
167
+ const oid = treeOf(repo, ref);
168
+ const { blobs, links } = revisionBlobs(repo, oid, ref);
169
+ const root = `${resolve(repo)}@${oid.slice(0, 12)}`;
170
+ const store = new Map();
171
+ for (const b of blobs) {
172
+ const parts = treePathParts(b.path);
173
+ if (parts) store.set(resolve(root, ...parts), b.bytes);
174
+ }
175
+ const tree = heldTree({ kind: 'git', root, store, rel: (p) => relPath(root, p), systemDirs: opts.systemDirs || [],
176
+ symlinks: { followed: 0, outside: 0, broken: 0, notRead: links } });
177
+ return Object.assign(tree, { ref, oid });
178
+ }
179
+
180
+ // z/OSMF's record mode (zowe --record) writes each record after its length, four bytes big-endian,
181
+ // and nothing between records. Read as that only when the lengths chain exactly to the end of the
182
+ // file, and given back as lines; anything else is the file as it is.
183
+ const MAX_LRECL = 32760;
184
+ export function unframeRecords(buf) {
185
+ let records = 0;
186
+ for (let at = 0; at < buf.length; records++) {
187
+ if (buf.length - at < 4) return buf;
188
+ const n = buf.readUInt32BE(at);
189
+ if (n > MAX_LRECL || buf.length - at - 4 < n) return buf;
190
+ at += 4 + n;
191
+ }
192
+ if (!records || records * 4 === buf.length) return buf;
193
+ const lines = (end) => {
194
+ const out = Buffer.alloc(buf.length - records * 3);
195
+ let w = 0;
196
+ for (let at = 0; at < buf.length;) {
197
+ const n = buf.readUInt32BE(at);
198
+ w += buf.copy(out, w, at + 4, at + 4 + n);
199
+ out[w++] = end;
200
+ at += 4 + n;
201
+ }
202
+ return out;
128
203
  };
204
+ const ebcdic = lines(0x15);
205
+ return looksEbcdic(ebcdic) ? ebcdic : lines(0x0A);
206
+ }
207
+
208
+ // A member is read from the file the walk found. A link put in its place since, or a file that is
209
+ // not a regular one, is refused rather than followed or waited on.
210
+ function readMember(file) {
211
+ const fd = openSync(file, constants.O_RDONLY | (constants.O_NOFOLLOW || 0) | (constants.O_NONBLOCK || 0));
212
+ try {
213
+ const st = fstatSync(fd);
214
+ if (!st.isFile()) throw Object.assign(new Error(`not a regular file: ${file}`), { code: 'ENOTFILE' });
215
+ const buf = Buffer.alloc(st.size);
216
+ let n = 0;
217
+ while (n < st.size) {
218
+ const r = readSync(fd, buf, n, st.size - n, n);
219
+ if (!r) break;
220
+ n += r;
221
+ }
222
+ return unframeRecords(buf.subarray(0, n));
223
+ } finally { closeSync(fd); }
224
+ }
225
+
226
+ const QUALIFIER = /^[A-Z@#$][A-Z0-9@#$-]{0,7}$/;
227
+ const MEMBER = /^([A-Z@#$][A-Z0-9@#$]{0,7})(?:\.[^.]*)?$/i;
228
+ const MAX_LISTED = 8;
229
+
230
+ // Partitioned data sets exported to a directory, a member to a file: what
231
+ // `zowe zos-files download all-members` writes (ibmuser/new/cntl/member.txt, lower-cased unless
232
+ // --preserve-original-letter-case, with whatever extension -e gave), or a directory named for the
233
+ // data set with its members copied in. Each member is held as `DATA.SET.NAME/MEMBER` under a root
234
+ // beside the export that does not exist, so a finding names the member as z/OS does however it was
235
+ // downloaded, and nothing under that root is ever read from disk.
236
+ //
237
+ // The extension is the download's choice, not the language's, so members are classified by their
238
+ // contents. A file whose directory is not a data set name or whose name is not a member name is not
239
+ // read and is counted, as is a second file for a member already held. Every data set that holds
240
+ // anything but JCL and maps is a copy library, searched in name order: the export does not say
241
+ // what SYSLIB concatenated.
242
+ export function pdsExportTree(dir, opts = {}) {
243
+ const idx = buildFileIndex(dir);
244
+ const top = realpathSync(dir);
245
+ const root = `${resolve(dir)}@pds`;
246
+ const origins = new Map();
247
+ const found = new Map();
248
+ const notMembers = [];
249
+ const duplicates = [];
250
+ const dataSets = new Set();
251
+ for (const file of [...idx.index.values()].sort()) {
252
+ const where = relative(dir, dirname(file)).split(sep).filter(Boolean);
253
+ const dsn = where.join('.').toUpperCase();
254
+ const member = MEMBER.exec(basename(file));
255
+ if (!member || dsn.length > 44 || (dsn && !dsn.split('.').every((q) => QUALIFIER.test(q)))) {
256
+ notMembers.push(relPath(dir, file));
257
+ continue;
258
+ }
259
+ const at = resolve(root, ...(dsn ? [dsn] : []), member[1].toUpperCase());
260
+ if (found.has(at)) { duplicates.push(`${relPath(dir, file)}: ${relPath(root, at)} is ${found.get(at)}`); continue; }
261
+ let real;
262
+ try { real = realpathSync(file); } catch { real = file; }
263
+ if (real !== top && !real.startsWith(top + sep)) { notMembers.push(relPath(dir, file)); continue; }
264
+ origins.set(at, real);
265
+ found.set(at, relPath(dir, file));
266
+ dataSets.add(dirname(at));
267
+ }
268
+ const kinds = new Map();
269
+ const byKind = {};
270
+ const copyDirs = new Set();
271
+ for (const [at, file] of origins) {
272
+ let kind = null;
273
+ try { kind = kindOfBytes(readMember(file)); } catch { kind = null; }
274
+ kinds.set(at, kind);
275
+ byKind[kind || 'other'] = (byKind[kind || 'other'] || 0) + 1;
276
+ if (kind !== 'jcl' && kind !== 'bms') copyDirs.add(dirname(at));
277
+ }
278
+ const store = {
279
+ keys: () => origins.keys(),
280
+ has: (p) => origins.has(p),
281
+ get: (p) => (origins.has(p) ? readMember(origins.get(p)) : undefined),
282
+ };
283
+ const tree = heldTree({
284
+ kind: 'pds-export', root, store, rel: (p) => relPath(root, p), systemDirs: opts.systemDirs || [],
285
+ copyDirs: [...copyDirs], classify: () => (p) => kinds.get(p) ?? null, symlinks: idx.symlinks,
286
+ unreadableDirs: idx.unreadableDirs.map((d) => resolve(root, relative(dir, d))),
287
+ });
288
+ return Object.assign(tree, {
289
+ dir,
290
+ origin: (p) => found.get(p) ?? null,
291
+ export: {
292
+ dataSets: dataSets.size,
293
+ members: origins.size,
294
+ byKind,
295
+ ...(notMembers.length ? { notMembers: notMembers.length, notMembersListed: notMembers.slice(0, MAX_LISTED) } : {}),
296
+ ...(duplicates.length ? { duplicates: duplicates.length, duplicatesListed: duplicates.slice(0, MAX_LISTED) } : {}),
297
+ },
298
+ });
129
299
  }
130
300
 
131
301
  // What a rule set calls. Given a tree it uses it; given none it builds the directory one, so every
package/lib/parser.mjs CHANGED
@@ -446,11 +446,18 @@ export function tokenize(norm, file) {
446
446
  return { tokens: out, diags };
447
447
  }
448
448
 
449
- // Every rule set walks the tree through here. A directory it may not read is recorded, not treated
450
- // as empty: only ENOENT means absent. Symlinks are followed while they stay inside the tree, so a
451
- // symlinked copy library is read; one pointing outside is counted and never followed, because a
452
- // scan reads the tree it was given and nothing else.
453
- export function buildFileIndex(root) {
449
+ // A drive that stalls or drops for a moment answers a listing with ENOENT or an I/O error. Such a
450
+ // listing is tried again after a pause; the pauses add up to about 1.3 seconds.
451
+ const LISTING_RETRIED = new Set(['ENOENT', 'EIO', 'ETIMEDOUT', 'ENXIO', 'EBUSY', 'EAGAIN', 'EINTR', 'ESTALE', 'ENOTCONN']);
452
+ const LISTING_PAUSES_MS = [50, 250, 1000];
453
+ const pause = (ms) => Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
454
+
455
+ // Every rule set walks the tree through here. A directory that cannot be listed is recorded, not
456
+ // treated as empty: a scan over it is incomplete, not a scan of a smaller tree. Only the root's own
457
+ // ENOENT means absent; a directory the walk found in its parent was there. Symlinks are followed
458
+ // while they stay inside the tree, so a symlinked copy library is read; one pointing outside is
459
+ // counted and never followed, because a scan reads the tree it was given and nothing else.
460
+ export function buildFileIndex(root, { readdir = readdirSync, wait = pause } = {}) {
454
461
  const index = new Map();
455
462
  const dirs = new Set();
456
463
  const unreadableDirs = [];
@@ -460,16 +467,20 @@ export function buildFileIndex(root) {
460
467
  const inside = (p) => p === top || p.startsWith(top + sep);
461
468
  const visited = new Set();
462
469
  const addFile = (p, name, d) => { index.set(p.toLowerCase(), p); if (/\.(cpy|copy|inc|cbl|cob)$/i.test(name)) dirs.add(d); };
470
+ const list = (d) => {
471
+ for (let i = 0; ; i++) {
472
+ try { return readdir(d, { withFileTypes: true }); } catch (e) {
473
+ if (i >= LISTING_PAUSES_MS.length || !LISTING_RETRIED.has(e.code)) throw e;
474
+ wait(LISTING_PAUSES_MS[i]);
475
+ }
476
+ }
477
+ };
463
478
  // Each real directory is walked once, however many links lead to it, or its programs count twice.
464
479
  const walk = (d, real) => {
465
480
  if (visited.has(real)) return;
466
481
  visited.add(real);
467
482
  let es;
468
- try { es = readdirSync(d, { withFileTypes: true }); } catch (e) {
469
- if (e.code === 'ENOENT') return;
470
- if (e.code === 'EACCES' || e.code === 'EPERM') { unreadableDirs.push(d); return; }
471
- throw e;
472
- }
483
+ try { es = list(d); } catch { unreadableDirs.push(d); return; }
473
484
  for (const e of es.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))) {
474
485
  if (e.name === '.git' || e.name.startsWith('._')) continue;
475
486
  const p = join(d, e.name);
@@ -634,7 +645,7 @@ function isProgramFile(p, ctx) {
634
645
  const seen = (ctx.programFiles ||= new Map());
635
646
  if (!seen.has(p)) {
636
647
  let text = '';
637
- try { text = readSource(p).text; } catch { /* an unreadable candidate is not known to be a program */ }
648
+ try { text = ctx.readText(p); } catch { /* an unreadable candidate is not known to be a program */ }
638
649
  seen.set(p, /^[^*\n]{0,6}\s*PROGRAM-ID\s*\./im.test(text));
639
650
  }
640
651
  return seen.get(p);
@@ -787,7 +798,7 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
787
798
  if (!path) return [];
788
799
  if (stack.includes(path) || stack.length > 40) { record.status = 'recursive'; return []; }
789
800
  if (++ctx.inclusions > MAX_INCLUSIONS || ctx.copyTokens > MAX_COPY_TOKENS) { record.status = 'expansion-limit'; return []; }
790
- const src = readSource(path).text;
801
+ const src = ctx.readText(path);
791
802
  // A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
792
803
  // directive earlier in the including file may have changed from the file's starting format.
793
804
  const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
@@ -1513,6 +1524,9 @@ function indexTokens(seg) {
1513
1524
  const lone = (from, to) => (to - from === 1 && (seg[from].t === 'word' || seg[from].t === 'num') && /^\d+$/.test(seg[from].v) ? Number(seg[from].v) : null);
1514
1525
  const refLength = colonAt < 0 ? null : lone(colonAt + 1, close);
1515
1526
  const refStart = colonAt < 0 ? null : lone(k + 1, colonAt);
1527
+ // A length is judged against where its reference starts: a name, a name moved by a constant, or neither.
1528
+ const named = (from, to) => (seg[from].t === 'word' && !/^\d+$/.test(seg[from].v) && (to - from === 1 || (to - from === 3 && offsetOf(seg, from, k, close, colonAt))) ? { tok: seg[from], offset: offsetOf(seg, from, k, close, colonAt) } : null);
1529
+ const refFrom = colonAt < 0 || refStart != null ? null : named(k + 1, colonAt) || { tok: null, offset: 0 };
1516
1530
  for (let j = k + 1; j < close; j++) {
1517
1531
  const t = seg[j];
1518
1532
  // A number is tokenized as a word, and no name is all digits.
@@ -1521,7 +1535,7 @@ function indexTokens(seg) {
1521
1535
  if (prev && prev.t === 'word' && (prev.u === 'OF' || prev.u === 'IN' || prev.u === 'FUNCTION')) continue;
1522
1536
  const kind = colonAt < 0 ? 'subscript' : j < colonAt ? 'refmod-offset' : 'refmod-length';
1523
1537
  const other = kind === 'refmod-offset' ? refLength : kind === 'refmod-length' ? refStart : null;
1524
- out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}) });
1538
+ out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}), ...(kind === 'refmod-length' && refFrom ? { from: refFrom } : {}) });
1525
1539
  }
1526
1540
  }
1527
1541
  lastHost = host;
@@ -1848,7 +1862,7 @@ function collectCopybookDefines(src, format, ctx, depth) {
1848
1862
  const path = resolveCopy(name, null, ctx);
1849
1863
  if (!path || seen.has(path)) continue;
1850
1864
  seen.add(path);
1851
- const text = readSource(path).text;
1865
+ const text = ctx.readText(path);
1852
1866
  const fmt = detectFormat(text) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : current;
1853
1867
  normalize(text, fmt, ctx.defines, ctx.std);
1854
1868
  collectCopybookDefines(text, fmt, ctx, depth + 1);
@@ -1880,7 +1894,9 @@ export function parseSource(src, file, opts = {}) {
1880
1894
  const scheme = BINARY_SIZE[opts.std || 'default'] || BINARY_SIZE.default;
1881
1895
  const ctx = { mainDir: opts.mainDir || dirname(file), includeDirs: opts.includeDirs || [], systemDirs: opts.systemDirs || [],
1882
1896
  fileIndex: opts.fileIndex || null, allowAbsoluteCopy: !!opts.allowAbsoluteCopy, cache: new Map(), copies: [], diags: [], copyFormat: opts.copyFormat || format, defines: new Map(), std: opts.std,
1883
- inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0 };
1897
+ inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0,
1898
+ // Where a copybook's text comes from: the disk, or the source tree the program was read from.
1899
+ readText: opts.readText || ((p) => readSource(p).text) };
1884
1900
  collectCopybookDefines(src, format, ctx, 0);
1885
1901
  const norm = normalize(src, format, ctx.defines, opts.std);
1886
1902
  ctx.mainFormat = norm.finalFormat;
package/lib/revision.json CHANGED
@@ -1 +1 @@
1
- {"commit":"54c9654b8c13bc22c8d5683b02551362d95f0d0a","tag":"v0.2.140"}
1
+ {"commit":"b242a380e1b8b7d8bb8ff2d6e03e2c8a30242967","tag":"v0.3.0"}