@portll/cobolwork 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +49 -18
- package/STABILITY.md +76 -0
- package/bin/cobolwork.mjs +30 -7
- package/lib/bms.mjs +21 -7
- package/lib/build.mjs +108 -53
- package/lib/capabilities.mjs +9 -10
- package/lib/cics-commands.mjs +501 -9
- package/lib/compliance.mjs +11 -0
- package/lib/consequence.mjs +13 -0
- package/lib/control-workers.mjs +161 -0
- package/lib/control.mjs +52 -3
- package/lib/dataflow.mjs +176 -125
- package/lib/db2/cursor.mjs +15 -0
- package/lib/db2/read.mjs +170 -0
- package/lib/db2/rules.mjs +77 -0
- package/lib/db2/stmt/alter.mjs +473 -0
- package/lib/db2/stmt/grant.mjs +125 -0
- package/lib/db2/stmt/index.mjs +141 -0
- package/lib/db2/stmt/misc.mjs +325 -0
- package/lib/db2/stmt/routine.mjs +564 -0
- package/lib/db2/stmt/storage.mjs +146 -0
- package/lib/db2/stmt/table.mjs +540 -0
- package/lib/db2/stmt/view.mjs +146 -0
- package/lib/diff.mjs +17 -5
- package/lib/evidence/cli.mjs +14 -3
- package/lib/evidence/record.mjs +1 -1
- package/lib/evidence/store.mjs +44 -31
- package/lib/exec-reading.mjs +51 -0
- package/lib/execution.mjs +3 -2
- package/lib/explain.mjs +2 -0
- package/lib/exploitability.mjs +11 -2
- package/lib/hlasm/asm/data.mjs +13 -1
- package/lib/hlasm/asm/sections.mjs +3 -1
- package/lib/hlasm/instr.mjs +25 -0
- package/lib/hlasm/macro/authorization.mjs +24 -4
- package/lib/hlasm/macro/datasets.mjs +30 -17
- package/lib/hlasm/macro/io.mjs +68 -45
- package/lib/hlasm/macro/le.mjs +2 -2
- package/lib/hlasm/macro/operator.mjs +32 -21
- package/lib/hlasm/macro/program.mjs +126 -84
- package/lib/hlasm/macro/recovery.mjs +10 -5
- package/lib/hlasm/macro/storage.mjs +64 -14
- package/lib/hlasm/macro/structured.mjs +1 -1
- package/lib/hlasm/model.mjs +35 -10
- package/lib/hlasm/mvs38.mjs +47 -0
- package/lib/hlasm/operands.mjs +10 -0
- package/lib/hlasm/optable.mjs +61 -0
- package/lib/hlasm/read.mjs +22 -8
- package/lib/hlasm.mjs +44 -7
- package/lib/ims/dli.mjs +37 -0
- package/lib/ims/macro/dbd.mjs +299 -0
- package/lib/ims/macro/psb.mjs +286 -0
- package/lib/ims/model.mjs +149 -0
- package/lib/ims/operands.mjs +23 -0
- package/lib/ims/read.mjs +37 -0
- package/lib/ims/rules.mjs +135 -0
- package/lib/ironwork.mjs +17 -13
- package/lib/kernel/pds-archive.mjs +256 -0
- package/lib/kernel/registry.mjs +14 -8
- package/lib/kernel/shared-pass.mjs +34 -9
- package/lib/kernel/source-tree.mjs +104 -36
- package/lib/kernel/version-key.mjs +17 -0
- package/lib/layout.mjs +26 -31
- package/lib/parser.mjs +136 -17
- package/lib/pli/cursor.mjs +15 -0
- package/lib/pli/expr.mjs +101 -0
- package/lib/pli/include.mjs +82 -0
- package/lib/pli/layout.mjs +125 -0
- package/lib/pli/lex.mjs +198 -0
- package/lib/pli/program.mjs +280 -0
- package/lib/pli/rules/based.mjs +95 -0
- package/lib/pli/rules/conditions.mjs +68 -0
- package/lib/pli/rules/entry.mjs +130 -0
- package/lib/pli/rules/index.mjs +24 -0
- package/lib/pli/rules/preprocessor.mjs +55 -0
- package/lib/pli/statements.mjs +130 -0
- package/lib/pli/stmt/alloc.mjs +45 -0
- package/lib/pli/stmt/assignment.mjs +56 -0
- package/lib/pli/stmt/call.mjs +104 -0
- package/lib/pli/stmt/conditions.mjs +94 -0
- package/lib/pli/stmt/control.mjs +219 -0
- package/lib/pli/stmt/declare.mjs +149 -0
- package/lib/pli/stmt/exec.mjs +55 -0
- package/lib/pli/stmt/io.mjs +239 -0
- package/lib/pli/stmt/misc.mjs +4 -0
- package/lib/pli/stmt/preprocessor.mjs +242 -0
- package/lib/pli/stmt/procedure.mjs +258 -0
- package/lib/pli/stmt/stream.mjs +283 -0
- package/lib/pli/storage.mjs +129 -0
- package/lib/precompile-check.mjs +124 -0
- package/lib/precompile-cics.mjs +7 -3
- package/lib/reach.mjs +11 -2
- package/lib/revision.json +1 -1
- package/lib/sarif.mjs +41 -3
- package/lib/scan.mjs +7 -1
- package/lib/sets/abend.mjs +16 -6
- package/lib/sets/cics.mjs +15 -35
- package/lib/sets/compile.mjs +41 -15
- package/lib/sets/crypto.mjs +5 -3
- package/lib/sets/ddl.mjs +36 -0
- package/lib/sets/flow.mjs +18 -1
- package/lib/sets/hidden.mjs +5 -3
- package/lib/sets/hlasm.mjs +72 -12
- package/lib/sets/ims.mjs +139 -0
- package/lib/sets/log.mjs +6 -6
- package/lib/sets/opaque.mjs +27 -7
- package/lib/sets/pli.mjs +40 -0
- package/lib/sets/recon.mjs +5 -3
- package/lib/sets/secrets.mjs +5 -3
- package/lib/sets/semantics.mjs +3 -0
- package/lib/sets/web.mjs +36 -22
- package/lib/site.mjs +10 -0
- package/lib/sources.mjs +80 -20
- package/lib/statement-cursor.mjs +67 -0
- package/lib/verify.mjs +3 -2
- package/lib/version.mjs +6 -0
- package/package.json +3 -2
- package/rules/compliance-cobit2019.json +432 -1
- package/rules/compliance-dora.json +414 -1
- package/rules/compliance-ffiec.json +414 -1
- package/rules/compliance-nist80053.json +466 -1
- package/rules/hlasm-optables.json +8024 -0
- package/schema/cobolwork-baseline.schema.json +36 -0
- package/schema/cobolwork-build-provenance.schema.json +187 -0
- package/schema/cobolwork-build.schema.json +382 -0
- package/schema/cobolwork-capabilities.schema.json +239 -0
- package/schema/cobolwork-diff.schema.json +217 -0
- package/schema/cobolwork-evidence.schema.json +161 -0
- package/schema/cobolwork-execution.schema.json +53 -0
- package/schema/cobolwork-explain.schema.json +360 -0
- package/schema/cobolwork-finding.schema.json +465 -0
- package/schema/cobolwork-flow.schema.json +465 -0
- package/schema/cobolwork-gate.schema.json +211 -0
- package/schema/cobolwork-inventory.schema.json +206 -0
- package/schema/cobolwork-parse.schema.json +105 -0
- package/schema/cobolwork-reach.schema.json +74 -0
- package/schema/cobolwork-report.schema.json +559 -0
- package/schema/cobolwork-witness.schema.json +107 -0
- package/schema/cobolwork.baseline.schema.json +101 -0
- package/schema/cobolwork.site.schema.json +116 -0
package/lib/pli/expr.mjs
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// PL/I expressions and references, as the statement parsers consume them: an operator tree by
|
|
3
|
+
// PL/I precedence, every token consumed, and the references the expression reads.
|
|
4
|
+
import { cursor } from './cursor.mjs';
|
|
5
|
+
|
|
6
|
+
// Higher binds tighter. Infix ¬ is exclusive-or, at the level of |.
|
|
7
|
+
const PREC = {
|
|
8
|
+
'**': 7, '*': 6, '/': 6, '+': 5, '-': 5, '||': 4,
|
|
9
|
+
'=': 3, '¬=': 3, '<': 3, '>': 3, '<=': 3, '>=': 3, '¬<': 3, '¬>': 3, '<>': 3,
|
|
10
|
+
'&': 2, '|': 1, '¬': 1,
|
|
11
|
+
};
|
|
12
|
+
const PREFIX = new Set(['+', '-', '¬']);
|
|
13
|
+
|
|
14
|
+
// A reference: a name, then any mix of (arguments), '.' name, and '->' or '=>' locator
|
|
15
|
+
// qualification. args holds one list per parenthesised group, each argument an expression or
|
|
16
|
+
// { t: 'star' } for '*'.
|
|
17
|
+
export function parseReference(c) {
|
|
18
|
+
const start = c.pos;
|
|
19
|
+
const first = c.word();
|
|
20
|
+
if (!first) c.fail('a reference');
|
|
21
|
+
const path = [first.u];
|
|
22
|
+
const locators = [];
|
|
23
|
+
const args = [];
|
|
24
|
+
for (;;) {
|
|
25
|
+
if (c.isOp('(')) { args.push(c.items().map(argument)); continue; }
|
|
26
|
+
if (c.isOp('.') && c.isWord(undefined, 1)) { c.next(); path.push(c.next().u); continue; }
|
|
27
|
+
if (c.isOp(['->', '=>']) && c.isWord(undefined, 1)) { c.next(); locators.push(path.length); path.push(c.next().u); continue; }
|
|
28
|
+
break;
|
|
29
|
+
}
|
|
30
|
+
return { t: 'ref', name: path[path.length - 1], path, locators, args, toks: c.slice(start) };
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function argument(toks) {
|
|
34
|
+
if (toks.length === 1 && toks[0].t === 'op' && toks[0].v === '*') return { t: 'star', toks };
|
|
35
|
+
const s = cursor(toks);
|
|
36
|
+
const e = parseExpression(s);
|
|
37
|
+
if (!s.done()) s.fail('the end of the argument');
|
|
38
|
+
return e;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function primary(c) {
|
|
42
|
+
const t = c.peek();
|
|
43
|
+
if (!t) c.fail('an operand');
|
|
44
|
+
if (t.t === 'num') return { t: 'num', tok: c.next() };
|
|
45
|
+
if (t.t === 'lit') return { t: 'lit', tok: c.next() };
|
|
46
|
+
if (t.t === 'word') return parseReference(c);
|
|
47
|
+
if (t.t === 'op' && t.v === '(') {
|
|
48
|
+
c.next();
|
|
49
|
+
const inner = parseExpression(c);
|
|
50
|
+
c.expectOp(')');
|
|
51
|
+
// (3)'AB' and (N)'0'B repeat the literal: a parenthesis is never otherwise followed by one.
|
|
52
|
+
if (c.peek() && c.peek().t === 'lit') return { t: 'lit', tok: c.next(), factor: inner };
|
|
53
|
+
return { t: 'paren', expr: inner };
|
|
54
|
+
}
|
|
55
|
+
return c.fail('an operand');
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// A prefix operator applies to its operand's ** chain: -A**2 is -(A**2).
|
|
59
|
+
function unary(c) {
|
|
60
|
+
const t = c.peek();
|
|
61
|
+
if (t && t.t === 'op' && PREFIX.has(t.v)) {
|
|
62
|
+
c.next();
|
|
63
|
+
return { op: `prefix${t.v}`, args: [binary(c, 7)] };
|
|
64
|
+
}
|
|
65
|
+
return primary(c);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function binary(c, min, stopOps = []) {
|
|
69
|
+
let left = unary(c);
|
|
70
|
+
for (;;) {
|
|
71
|
+
const t = c.peek();
|
|
72
|
+
if (!t || t.t !== 'op' || !(t.v in PREC) || stopOps.includes(t.v)) return left;
|
|
73
|
+
const p = PREC[t.v];
|
|
74
|
+
if (p < min) return left;
|
|
75
|
+
c.next();
|
|
76
|
+
const right = binary(c, t.v === '**' ? p : p + 1, stopOps);
|
|
77
|
+
left = { op: t.v, args: [left, right] };
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function collect(node, out) {
|
|
82
|
+
if (!node) return;
|
|
83
|
+
if (node.t === 'ref') {
|
|
84
|
+
for (const at of node.locators) out.push(node.path.slice(0, at).join('.'));
|
|
85
|
+
out.push(node.path.join('.'));
|
|
86
|
+
for (const list of node.args) for (const a of list) collect(a.tree ?? a, out);
|
|
87
|
+
} else if (node.t === 'paren') collect(node.expr.tree, out);
|
|
88
|
+
else if (node.t === 'lit' && node.factor) collect(node.factor.tree, out);
|
|
89
|
+
else if (node.args) for (const a of node.args) collect(a, out);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// An expression, stopping before the first token that cannot continue it: ',' or ')', a word after a
|
|
93
|
+
// complete operand (so a caller's stopWords need no test), an assignment operator, or an op in
|
|
94
|
+
// stopOps. { t: 'expr', tree, toks, refs }.
|
|
95
|
+
export function parseExpression(c, { stopOps = [] } = {}) {
|
|
96
|
+
const start = c.pos;
|
|
97
|
+
const tree = binary(c, 1, stopOps);
|
|
98
|
+
const refs = [];
|
|
99
|
+
collect(tree, refs);
|
|
100
|
+
return { t: 'expr', tree, toks: c.slice(start), refs: [...new Set(refs)] };
|
|
101
|
+
}
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// %INCLUDE expansion at the token level, as the preprocessor does it before statements exist: a
|
|
3
|
+
// member can hold the rest of a DECLARE (`DCL 1 REC, %INCLUDE RECFLDS;`), so a member's tokens
|
|
4
|
+
// replace the directive in the token stream and statements are split afterwards. Every token keeps
|
|
5
|
+
// the file and line it was read from.
|
|
6
|
+
import { basename, dirname, extname } from 'node:path';
|
|
7
|
+
import { tokenize, statements } from './lex.mjs';
|
|
8
|
+
|
|
9
|
+
// Upper-case member name -> paths of the files that could hold it, in path order.
|
|
10
|
+
export function membersOf(paths) {
|
|
11
|
+
const out = new Map();
|
|
12
|
+
for (const p of [...paths].sort()) {
|
|
13
|
+
const name = basename(p, extname(p)).toUpperCase();
|
|
14
|
+
if (!out.has(name)) out.set(name, []);
|
|
15
|
+
out.get(name).push(p);
|
|
16
|
+
}
|
|
17
|
+
return out;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
// The member a directive in `from` means: one in the same directory if there is one, else the first.
|
|
21
|
+
export function chooseMember(candidates, from) {
|
|
22
|
+
if (!candidates || !candidates.length) return null;
|
|
23
|
+
return candidates.find((p) => dirname(p) === dirname(from)) || candidates[0];
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// The members a %INCLUDE or %XINCLUDE directive names: name, 'name' or ddname(name), comma-separated.
|
|
27
|
+
function directiveMembers(toks) {
|
|
28
|
+
const names = [];
|
|
29
|
+
for (let k = 2; k < toks.length; k++) {
|
|
30
|
+
const t = toks[k];
|
|
31
|
+
if (t.t === 'op' && t.v === ',') continue;
|
|
32
|
+
if (t.t === 'word' && toks[k + 1]?.t === 'op' && toks[k + 1].v === '(' && toks[k + 3]?.t === 'op' && toks[k + 3].v === ')') {
|
|
33
|
+
const m = toks[k + 2];
|
|
34
|
+
names.push(String(m.u ?? m.v).toUpperCase());
|
|
35
|
+
k += 3;
|
|
36
|
+
} else if (t.t === 'word') names.push(t.u);
|
|
37
|
+
else if (t.t === 'lit') names.push(t.v.toUpperCase());
|
|
38
|
+
}
|
|
39
|
+
return names;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// readMember(name, fromFile) -> { path, text } or null.
|
|
43
|
+
export function expandIncludes(tokens, { file, readMember, maxDepth = 16 }) {
|
|
44
|
+
const included = [];
|
|
45
|
+
const unresolved = [];
|
|
46
|
+
const cycles = [];
|
|
47
|
+
const seen = new Set();
|
|
48
|
+
const walk = (toks, from, chain) => {
|
|
49
|
+
const out = [];
|
|
50
|
+
for (let i = 0; i < toks.length; i++) {
|
|
51
|
+
const t = toks[i];
|
|
52
|
+
const d = toks[i + 1];
|
|
53
|
+
if (!(t.t === 'op' && t.v === '%' && d && d.t === 'word' && (d.u === 'INCLUDE' || d.u === 'XINCLUDE'))) { out.push(t); continue; }
|
|
54
|
+
let end = i;
|
|
55
|
+
while (end < toks.length && toks[end].t !== 'semi') end++;
|
|
56
|
+
const directive = toks.slice(i, end + 1);
|
|
57
|
+
const keep = [];
|
|
58
|
+
for (const name of directiveMembers(directive)) {
|
|
59
|
+
if (d.u === 'XINCLUDE' && seen.has(name)) continue;
|
|
60
|
+
if (chain.includes(name)) { cycles.push({ chain: [...chain, name] }); continue; }
|
|
61
|
+
if (chain.length >= maxDepth) { unresolved.push({ name, file: t.file ?? from, line: t.line, why: 'too deep' }); continue; }
|
|
62
|
+
const m = readMember(name, from);
|
|
63
|
+
if (!m) { unresolved.push({ name, file: t.file ?? from, line: t.line }); keep.push(name); continue; }
|
|
64
|
+
seen.add(name);
|
|
65
|
+
included.push({ name, path: m.path, from: { file: t.file ?? from, line: t.line } });
|
|
66
|
+
out.push(...walk(tokenize(m.text, { file: m.path }).tokens, m.path, [...chain, name]));
|
|
67
|
+
}
|
|
68
|
+
// An unresolved member keeps its directive, so the statement is still there to be counted.
|
|
69
|
+
if (keep.length) out.push(...directive);
|
|
70
|
+
i = end;
|
|
71
|
+
}
|
|
72
|
+
return out;
|
|
73
|
+
};
|
|
74
|
+
return { tokens: walk(tokens, file, []), included, unresolved, cycles };
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// readPli with %INCLUDE members spliced in.
|
|
78
|
+
export function readPliExpanded(text, { file, readMember }) {
|
|
79
|
+
const { tokens, diags, process, margins } = tokenize(text, { file });
|
|
80
|
+
const ex = expandIncludes(tokens, { file, readMember });
|
|
81
|
+
return { statements: statements(ex.tokens), diags, process, margins, included: ex.included, unresolved: ex.unresolved, cycles: ex.cycles };
|
|
82
|
+
}
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// PL/I structure mapping: DECLARE items into structure trees, and each member's offset and size by the
|
|
3
|
+
// Enterprise PL/I Language Reference's rules, innermost minor structures first, every pair of units
|
|
4
|
+
// placed with the first shifted toward the second as far as its alignment allows.
|
|
5
|
+
import { storageOf } from './storage.mjs';
|
|
6
|
+
|
|
7
|
+
const DW = 64;
|
|
8
|
+
const mod = (a, m) => ((a % m) + m) % m;
|
|
9
|
+
|
|
10
|
+
// The trees a DECLARE's items describe. A member's logical level is one deeper than the structure
|
|
11
|
+
// that holds it, whatever level numbers the source wrote.
|
|
12
|
+
export function structuresOf(items) {
|
|
13
|
+
const roots = [];
|
|
14
|
+
const stack = [];
|
|
15
|
+
for (const it of items) {
|
|
16
|
+
const node = { name: it.name, level: it.level, dims: it.dims || [], attributes: it.attributes || [], line: it.line, ...(it.file ? { file: it.file } : {}), children: [] };
|
|
17
|
+
if (it.level == null || it.level <= 1) { roots.push(node); stack.length = 0; if (it.level != null) stack.push(node); node.logical = 1; continue; }
|
|
18
|
+
while (stack.length && stack[stack.length - 1].level >= it.level) stack.pop();
|
|
19
|
+
const parent = stack[stack.length - 1];
|
|
20
|
+
if (!parent) { roots.push(node); node.logical = 1; node.orphan = true; stack.push(node); continue; }
|
|
21
|
+
node.logical = parent.logical + 1;
|
|
22
|
+
parent.children.push(node);
|
|
23
|
+
stack.push(node);
|
|
24
|
+
}
|
|
25
|
+
return roots;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// The number of elements a dimension list gives, or null when a bound is not a constant.
|
|
29
|
+
function extent(dims) {
|
|
30
|
+
let n = 1;
|
|
31
|
+
for (const d of dims) {
|
|
32
|
+
const parts = [[]];
|
|
33
|
+
for (const t of d) { if (t.t === 'op' && t.v === ':') parts.push([]); else parts[parts.length - 1].push(t); }
|
|
34
|
+
const num = (p) => {
|
|
35
|
+
if (p.length === 1 && p[0].t === 'num' && /^\d+$/.test(p[0].v)) return Number(p[0].v);
|
|
36
|
+
if (p.length === 2 && p[0].t === 'op' && (p[0].v === '-' || p[0].v === '+') && p[1].t === 'num' && /^\d+$/.test(p[1].v)) return (p[0].v === '-' ? -1 : 1) * Number(p[1].v);
|
|
37
|
+
return null;
|
|
38
|
+
};
|
|
39
|
+
const [lo, hi] = parts.length === 2 ? [num(parts[0]), num(parts[1])] : [1, num(parts[0])];
|
|
40
|
+
if (lo == null || hi == null) return null;
|
|
41
|
+
n *= Math.max(0, hi - lo + 1);
|
|
42
|
+
}
|
|
43
|
+
return n;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
const explicitAlignment = (node) => (node.attributes.some((a) => a.name === 'ALIGNED') ? true : node.attributes.some((a) => a.name === 'UNALIGNED') ? false : null);
|
|
47
|
+
|
|
48
|
+
// Pairs two units: the first begins at its offset from a doubleword boundary, the second at the
|
|
49
|
+
// first position after it that its alignment allows, then the first moves toward the second by
|
|
50
|
+
// whole multiples of its own alignment. Positions are in bits.
|
|
51
|
+
function pair(a, b) {
|
|
52
|
+
const end = a.offset + a.bits;
|
|
53
|
+
const start = b.structure ? end + mod(b.offset - end, b.align) : Math.ceil(end / b.align) * b.align;
|
|
54
|
+
const shift = Math.floor((start - end) / a.align) * a.align;
|
|
55
|
+
const first = a.offset + shift;
|
|
56
|
+
return { first, second: start, unit: { offset: mod(first, DW), bits: start + b.bits - first, align: Math.max(a.align, b.align), structure: true } };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// Maps one node: an element takes its storage; a structure maps its members into one unit, a union
|
|
60
|
+
// overlays them. Every node gets `at`, its start in bits within the unit of its parent.
|
|
61
|
+
function mapNode(node, inherited, problems) {
|
|
62
|
+
const own = explicitAlignment(node);
|
|
63
|
+
const inherit = own ?? inherited;
|
|
64
|
+
const count = node.dims.length ? extent(node.dims) : 1;
|
|
65
|
+
let unit;
|
|
66
|
+
if (!node.children.length) {
|
|
67
|
+
const s = storageOf(node.attributes, { inherited, name: node.name });
|
|
68
|
+
node.storage = s;
|
|
69
|
+
if (!s.known) problems.push({ name: node.name, line: node.line, why: s.type === 'TYPE' ? `type ${s.typeName} is defined by a DEFINE this reading does not resolve` : s.type ? `the extent of ${s.type} is not a constant` : 'no data attributes' });
|
|
70
|
+
unit = { offset: 0, bits: s.bits ?? 0, align: s.align, structure: false };
|
|
71
|
+
} else if (node.attributes.some((a) => a.name === 'UNION')) {
|
|
72
|
+
const members = node.children.map((ch) => mapNode(ch, inherit, problems));
|
|
73
|
+
let len = 0;
|
|
74
|
+
node.children.forEach((ch, k) => { ch.at = mod(members[k].offset, members[k].align); len = Math.max(len, ch.at + members[k].bits); });
|
|
75
|
+
unit = { offset: 0, bits: len, align: Math.max(...members.map((m) => m.align)), structure: true };
|
|
76
|
+
} else {
|
|
77
|
+
const members = node.children.map((ch) => mapNode(ch, inherit, problems));
|
|
78
|
+
let acc = { ...members[0] };
|
|
79
|
+
const starts = [members[0].offset];
|
|
80
|
+
for (let k = 1; k < members.length; k++) {
|
|
81
|
+
const { first, second, unit: u } = pair(acc, members[k]);
|
|
82
|
+
const moved = first - acc.offset;
|
|
83
|
+
for (let j = 0; j < k; j++) starts[j] += moved;
|
|
84
|
+
starts.push(starts[0] + (second - first));
|
|
85
|
+
acc = u;
|
|
86
|
+
}
|
|
87
|
+
const base = starts[0];
|
|
88
|
+
node.children.forEach((ch, k) => { ch.at = starts[k] - base; });
|
|
89
|
+
unit = { offset: acc.offset, bits: acc.bits, align: acc.align, structure: true };
|
|
90
|
+
}
|
|
91
|
+
if (count == null) problems.push({ name: node.name, line: node.line, why: 'a dimension bound is not a constant' });
|
|
92
|
+
node.count = count;
|
|
93
|
+
// Each element of an array starts on the same boundary, so an element is padded to its alignment.
|
|
94
|
+
const stride = count > 1 ? Math.ceil(unit.bits / unit.align) * unit.align : unit.bits;
|
|
95
|
+
node.strideBits = stride;
|
|
96
|
+
return count > 1 ? { ...unit, bits: stride * count } : unit;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// Byte and bit offsets from the start of the major structure, and sizes, set on every node.
|
|
100
|
+
function place(node, atBits) {
|
|
101
|
+
node.offsetBits = atBits;
|
|
102
|
+
node.offset = Math.floor(atBits / 8);
|
|
103
|
+
if (atBits % 8) node.bitOffset = atBits % 8;
|
|
104
|
+
node.size = Math.ceil(node.strideBits / 8);
|
|
105
|
+
node.occurs = node.count ?? null;
|
|
106
|
+
for (const ch of node.children) place(ch, atBits + ch.at);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
// Lays out every structure a DECLARE describes. Returns the trees, each node carrying offset (bytes
|
|
110
|
+
// from its major structure), bitOffset when not on a byte, size (bytes of one element), occurs, and
|
|
111
|
+
// storage for an element; problems lists what could not be sized.
|
|
112
|
+
export function layout(items) {
|
|
113
|
+
const roots = structuresOf(items);
|
|
114
|
+
const problems = [];
|
|
115
|
+
for (const r of roots) {
|
|
116
|
+
const u = mapNode(r, null, problems);
|
|
117
|
+
// Storage begins at the byte holding the first bit; unaligned bits shifted toward their
|
|
118
|
+
// successor leave their padding there.
|
|
119
|
+
const lead = u.offset % 8;
|
|
120
|
+
r.totalBits = u.bits;
|
|
121
|
+
place(r, lead);
|
|
122
|
+
r.total = Math.ceil((lead + u.bits) / 8);
|
|
123
|
+
}
|
|
124
|
+
return { roots, problems };
|
|
125
|
+
}
|
package/lib/pli/lex.mjs
ADDED
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
// PL/I source as tokens and statements: the margins a file is read with, comments and literals
|
|
3
|
+
// removed or kept as tokens, and the semicolon-terminated statements with their label and condition
|
|
4
|
+
// prefixes. Classification by statement kind is in lib/pli/statements.mjs.
|
|
5
|
+
|
|
6
|
+
const PROCESS_LINE = /^[*%]PROCESS\b(.*)$/i;
|
|
7
|
+
const CARRIAGE = new Set([' ', '0', '1', '-', '+']);
|
|
8
|
+
const SEQUENCE = /^[ A-Za-z0-9]{0,8}$/;
|
|
9
|
+
|
|
10
|
+
// Enterprise PL/I reads columns 2 to 72 unless *PROCESS MARGINS says otherwise. Public source is
|
|
11
|
+
// often written from column 1 to any width, and read with MARGINS(2,72) it loses its first column
|
|
12
|
+
// and its tail, so the margins are taken from the file itself and the reason recorded.
|
|
13
|
+
export function marginsOf(text) {
|
|
14
|
+
const lines = text.split(/\r?\n/);
|
|
15
|
+
for (const l of lines) {
|
|
16
|
+
const m = PROCESS_LINE.exec(l);
|
|
17
|
+
const opt = m && /\bMAR(?:GINS)?\s*\(\s*(\d+)\s*,\s*(\d+)\s*(?:,\s*(\d+)\s*)?\)/i.exec(m[1]);
|
|
18
|
+
if (opt) return { left: Number(opt[1]), right: Number(opt[2]), carriage: opt[3] ? Number(opt[3]) : 0, reason: 'process-option' };
|
|
19
|
+
}
|
|
20
|
+
const body = lines.filter((l) => !PROCESS_LINE.test(l));
|
|
21
|
+
const long = body.filter((l) => l.replace(/\s+$/, '').length > 72);
|
|
22
|
+
const sequenceLike = long.filter((l) => l.length <= 80 && SEQUENCE.test(l.slice(72).replace(/\s+$/, '')) && /\d/.test(l.slice(72))).length;
|
|
23
|
+
const sequenced = long.length > 0 && sequenceLike >= 0.95 * long.length;
|
|
24
|
+
const firstColumnFree = body.every((l) => !l.length || CARRIAGE.has(l[0]));
|
|
25
|
+
const left = firstColumnFree ? 2 : 1;
|
|
26
|
+
const right = long.length === 0 || sequenced ? 72 : Infinity;
|
|
27
|
+
const carriage = left === 2 && body.some((l) => l.length && l[0] !== ' ') ? 1 : 0;
|
|
28
|
+
const reason = right === Infinity ? 'text-beyond-72' : sequenced ? 'sequence-area' : 'default';
|
|
29
|
+
return { left, right, carriage, reason };
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
const NOT = new Set(['¬', '^', '~']);
|
|
33
|
+
const OPS3 = ['||=', '**=', '¬=>', '^=>'];
|
|
34
|
+
const OPS2 = ['->', '=>', '¬=', '^=', '~=', '<>', '<=', '>=', '¬<', '¬>', '^<', '^>', '~<', '~>', '||', '!!', '**', '+=', '-=', '*=', '/=', '|=', '&='];
|
|
35
|
+
const OPS1 = '+-*/=<>&|!:.,()%?';
|
|
36
|
+
|
|
37
|
+
const LIT_SUFFIX = /^(?:B[1-4]?X?|BX|XN|XU|GX|WX|UX|X|G|M|W|A|E|U)(?![A-Za-z0-9_$#@])/i;
|
|
38
|
+
const NUMBER = /^(?:\d[\d_]*(?:\.\d*)?|\.\d+)(?:[EeSsDdQq][+-]?\d+)?(?:[Bb](?![A-Za-z0-9_$#@]))?(?:[Ii](?![A-Za-z0-9_$#@]))?/;
|
|
39
|
+
const WORD_START = /[A-Za-z$#@_À-ɏ]/;
|
|
40
|
+
const WORD_REST = /[A-Za-z0-9$#@_À-ɏ]/;
|
|
41
|
+
|
|
42
|
+
// Tokens over the whole text, so a comment or literal that runs across lines is one token.
|
|
43
|
+
// { t: 'word', v, u } | { t: 'lit', v, suffix } | { t: 'num', v } | { t: 'op', v } | { t: 'semi' },
|
|
44
|
+
// each with line and col as the source has them.
|
|
45
|
+
export function tokenize(text, { file = null, margins = marginsOf(text) } = {}) {
|
|
46
|
+
const diags = [];
|
|
47
|
+
const process = [];
|
|
48
|
+
const physical = text.split(/\r?\n/);
|
|
49
|
+
const rows = physical.map((l, k) => {
|
|
50
|
+
if (PROCESS_LINE.test(l)) { process.push({ line: k + 1, options: PROCESS_LINE.exec(l)[1].replace(/;\s*$/, '').trim() }); return ''; }
|
|
51
|
+
return l.slice(margins.left - 1, margins.right === Infinity ? undefined : margins.right).replace(/\t/g, ' ');
|
|
52
|
+
});
|
|
53
|
+
const tokens = [];
|
|
54
|
+
const at = (line, col) => ({ line, col: col + margins.left, ...(file ? { file } : {}) });
|
|
55
|
+
let line = 0;
|
|
56
|
+
let i = 0;
|
|
57
|
+
const source = rows;
|
|
58
|
+
while (line < source.length) {
|
|
59
|
+
const s = source[line];
|
|
60
|
+
if (i >= s.length) { line++; i = 0; continue; }
|
|
61
|
+
const c = s[i];
|
|
62
|
+
if (c === ' ' || c === '\f' || c === '\r' || c === '\v') { i++; continue; }
|
|
63
|
+
if (c === '/' && s[i + 1] === '*') {
|
|
64
|
+
const start = at(line + 1, i);
|
|
65
|
+
let l = line, k = i + 2, closed = false;
|
|
66
|
+
while (l < source.length) {
|
|
67
|
+
const e = source[l].indexOf('*/', k);
|
|
68
|
+
if (e >= 0) { line = l; i = e + 2; closed = true; break; }
|
|
69
|
+
l++; k = 0;
|
|
70
|
+
}
|
|
71
|
+
if (!closed) { diags.push({ sev: 'error', kind: 'unterminated-comment', ...start }); line = source.length; }
|
|
72
|
+
continue;
|
|
73
|
+
}
|
|
74
|
+
if (c === "'" || c === '"') {
|
|
75
|
+
const start = at(line + 1, i);
|
|
76
|
+
let v = '', l = line, k = i + 1, closed = false;
|
|
77
|
+
while (l < source.length) {
|
|
78
|
+
const row = source[l];
|
|
79
|
+
if (k >= row.length) { l++; k = 0; continue; }
|
|
80
|
+
if (row[k] === c) {
|
|
81
|
+
if (row[k + 1] === c) { v += c; k += 2; continue; }
|
|
82
|
+
closed = true; k++; break;
|
|
83
|
+
}
|
|
84
|
+
v += row[k++];
|
|
85
|
+
}
|
|
86
|
+
if (!closed) { diags.push({ sev: 'error', kind: 'unterminated-literal', ...start }); line = source.length; i = 0; tokens.push({ t: 'lit', v, suffix: '', ...start }); continue; }
|
|
87
|
+
line = l; i = k;
|
|
88
|
+
const suf = LIT_SUFFIX.exec(source[line].slice(i));
|
|
89
|
+
const suffix = suf ? suf[0].toUpperCase() : '';
|
|
90
|
+
if (suf) i += suf[0].length;
|
|
91
|
+
tokens.push({ t: 'lit', v, suffix, quote: c, ...start });
|
|
92
|
+
continue;
|
|
93
|
+
}
|
|
94
|
+
const num = /[0-9.]/.test(c) ? NUMBER.exec(s.slice(i)) : null;
|
|
95
|
+
if (num && !(c === '.' && !/[0-9]/.test(s[i + 1] || ''))) {
|
|
96
|
+
tokens.push({ t: 'num', v: num[0], ...at(line + 1, i) });
|
|
97
|
+
i += num[0].length;
|
|
98
|
+
continue;
|
|
99
|
+
}
|
|
100
|
+
if (WORD_START.test(c)) {
|
|
101
|
+
let j = i + 1;
|
|
102
|
+
while (j < s.length && WORD_REST.test(s[j])) j++;
|
|
103
|
+
const v = s.slice(i, j);
|
|
104
|
+
tokens.push({ t: 'word', v, u: v.toUpperCase(), ...at(line + 1, i) });
|
|
105
|
+
i = j;
|
|
106
|
+
continue;
|
|
107
|
+
}
|
|
108
|
+
if (c === ';') { tokens.push({ t: 'semi', ...at(line + 1, i) }); i++; continue; }
|
|
109
|
+
const three = s.slice(i, i + 3);
|
|
110
|
+
const two = s.slice(i, i + 2);
|
|
111
|
+
const op = OPS3.find((o) => o === three) || OPS2.find((o) => o === two) || (OPS1.includes(c) || NOT.has(c) ? c : null);
|
|
112
|
+
if (op) {
|
|
113
|
+
tokens.push({ t: 'op', v: normalOp(op), ...at(line + 1, i) });
|
|
114
|
+
i += op.length;
|
|
115
|
+
continue;
|
|
116
|
+
}
|
|
117
|
+
diags.push({ sev: 'warn', kind: 'unexpected-char', char: c, ...at(line + 1, i) });
|
|
118
|
+
i++;
|
|
119
|
+
}
|
|
120
|
+
return { tokens, diags, process, margins };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const normalOp = (op) => op.replace(/^[\^~]/, '¬').replace(/^!!$/, '||').replace(/^!$/, '|');
|
|
124
|
+
|
|
125
|
+
// A leading `name:` is a label, `name(3):` a subscripted label, and `(SIZE, NOFOFL):` a condition
|
|
126
|
+
// prefix. They may repeat and mix, and are taken off before the statement is classified.
|
|
127
|
+
function prefixes(toks) {
|
|
128
|
+
const labels = [];
|
|
129
|
+
const conditions = [];
|
|
130
|
+
let i = 0;
|
|
131
|
+
for (;;) {
|
|
132
|
+
const a = toks[i], b = toks[i + 1];
|
|
133
|
+
if (a && a.t === 'word' && b && b.t === 'op' && b.v === ':') { labels.push({ name: a.u, line: a.line }); i += 2; continue; }
|
|
134
|
+
if (a && a.t === 'word' && b && b.t === 'op' && b.v === '(') {
|
|
135
|
+
const close = closing(toks, i + 1);
|
|
136
|
+
if (close > 0 && toks[close + 1] && toks[close + 1].t === 'op' && toks[close + 1].v === ':' && toks.slice(i + 2, close).every((t) => t.t === 'num' || (t.t === 'op' && (t.v === ',' || t.v === '-' || t.v === '+')))) {
|
|
137
|
+
labels.push({ name: a.u, line: a.line, subscript: toks.slice(i + 2, close).map((t) => t.v).join('') });
|
|
138
|
+
i = close + 2;
|
|
139
|
+
continue;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
if (a && a.t === 'op' && a.v === '(') {
|
|
143
|
+
const close = closing(toks, i);
|
|
144
|
+
if (close > 0 && toks[close + 1] && toks[close + 1].t === 'op' && toks[close + 1].v === ':' && toks.slice(i + 1, close).every((t) => t.t === 'word' || (t.t === 'op' && t.v === ','))) {
|
|
145
|
+
for (const t of toks.slice(i + 1, close)) if (t.t === 'word') conditions.push(t.u);
|
|
146
|
+
i = close + 2;
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return { labels, conditions, at: i };
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// The index of the parenthesis closing the one at `open`, or -1 when it is not closed.
|
|
155
|
+
export function closing(toks, open) {
|
|
156
|
+
let depth = 0;
|
|
157
|
+
for (let k = open; k < toks.length; k++) {
|
|
158
|
+
const t = toks[k];
|
|
159
|
+
if (t.t !== 'op') continue;
|
|
160
|
+
if (t.v === '(') depth++;
|
|
161
|
+
else if (t.v === ')' && --depth === 0) return k;
|
|
162
|
+
}
|
|
163
|
+
return -1;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
// Statements split at semicolons. Each carries its prefixes and the tokens after them; a trailing
|
|
167
|
+
// run with no semicolon is kept as a statement and flagged, never dropped.
|
|
168
|
+
export function statements(tokens) {
|
|
169
|
+
const out = [];
|
|
170
|
+
let cur = [];
|
|
171
|
+
const close = (semi) => {
|
|
172
|
+
if (!cur.length && !semi) return;
|
|
173
|
+
const { labels, conditions, at } = prefixes(cur);
|
|
174
|
+
const toks = cur.slice(at);
|
|
175
|
+
const first = cur[0] || semi;
|
|
176
|
+
const last = cur[cur.length - 1] || semi;
|
|
177
|
+
out.push({ labels, conditions, toks, line: first.line, endLine: (semi || last).line, ...(first.file ? { file: first.file } : {}), ...(semi ? {} : { unterminated: true }) });
|
|
178
|
+
cur = [];
|
|
179
|
+
};
|
|
180
|
+
for (const t of tokens) {
|
|
181
|
+
if (t.t === 'semi') close(t);
|
|
182
|
+
else cur.push(t);
|
|
183
|
+
}
|
|
184
|
+
close(null);
|
|
185
|
+
return out;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// UTF-8 where the bytes are valid UTF-8, else Latin-1: a ¬ read as Latin-1 from UTF-8 is two
|
|
189
|
+
// characters and moves every column after it.
|
|
190
|
+
export function sourceText(buf) {
|
|
191
|
+
const utf8 = buf.toString('utf8');
|
|
192
|
+
return utf8.includes('�') ? buf.toString('latin1') : utf8;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
export function readPli(text, opts = {}) {
|
|
196
|
+
const { tokens, diags, process, margins } = tokenize(text, opts);
|
|
197
|
+
return { statements: statements(tokens), diags, process, margins };
|
|
198
|
+
}
|