@portll/cobolwork 0.2.76 → 0.2.140

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +66 -14
  2. package/bin/cobolwork.mjs +126 -9
  3. package/lib/arith.mjs +235 -0
  4. package/lib/baseline.mjs +5 -0
  5. package/lib/build.mjs +112 -30
  6. package/lib/capabilities.mjs +12 -7
  7. package/lib/card.mjs +18 -0
  8. package/lib/cics-commands.mjs +136 -125
  9. package/lib/consequence.mjs +29 -0
  10. package/lib/control.mjs +1169 -91
  11. package/lib/csd.mjs +17 -2
  12. package/lib/dataflow.mjs +276 -126
  13. package/lib/embedded-sql.mjs +110 -0
  14. package/lib/enterprise-options.mjs +255 -0
  15. package/lib/equivalence.mjs +124 -0
  16. package/lib/evidence/cli.mjs +65 -0
  17. package/lib/evidence/journal.mjs +94 -0
  18. package/lib/evidence/record.mjs +110 -0
  19. package/lib/evidence/run.mjs +111 -0
  20. package/lib/evidence/seal.mjs +177 -0
  21. package/lib/evidence/slsa.mjs +64 -0
  22. package/lib/evidence/sshsig.mjs +172 -0
  23. package/lib/evidence/store.mjs +170 -0
  24. package/lib/evidence/verify.mjs +167 -0
  25. package/lib/explain.mjs +9 -5
  26. package/lib/exploitability.mjs +263 -0
  27. package/lib/ftp.mjs +136 -0
  28. package/lib/ironwork.mjs +109 -0
  29. package/lib/jcl.mjs +27 -7
  30. package/lib/kernel/findings.mjs +11 -0
  31. package/lib/kernel/registry.mjs +4 -0
  32. package/lib/layout.mjs +202 -0
  33. package/lib/option-diff.mjs +49 -0
  34. package/lib/options.mjs +109 -46
  35. package/lib/parser.mjs +80 -209
  36. package/lib/policy.mjs +8 -3
  37. package/lib/precompile.mjs +5 -84
  38. package/lib/reach.mjs +29 -12
  39. package/lib/revision.json +1 -1
  40. package/lib/sarif.mjs +1 -0
  41. package/lib/sbom.mjs +235 -0
  42. package/lib/scan.mjs +93 -27
  43. package/lib/sets/cics.mjs +119 -2
  44. package/lib/sets/copybook.mjs +2 -1
  45. package/lib/sets/flow.mjs +82 -2
  46. package/lib/sets/hidden.mjs +4 -8
  47. package/lib/sets/jcl.mjs +70 -106
  48. package/lib/sets/recon.mjs +2 -1
  49. package/lib/sets/semantics.mjs +377 -0
  50. package/lib/sets/web.mjs +34 -10
  51. package/lib/sets/zowe.mjs +246 -0
  52. package/lib/sources.mjs +10 -0
  53. package/lib/tui/app.mjs +31 -13
  54. package/lib/tui/model.mjs +17 -2
  55. package/lib/utilities.mjs +229 -15
  56. package/lib/verify.mjs +128 -0
  57. package/package.json +1 -1
  58. package/rules/compliance-dora.json +507 -0
  59. package/rules/compliance-ffiec.json +507 -0
  60. package/rules/compliance-nist80053.json +507 -0
  61. package/schema/cobolwork.policy.schema.json +94 -16
@@ -0,0 +1,377 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // What Enterprise COBOL's generated code does that the source does not say, read from ironwork's
3
+ // model of the compiler: a binary store under TRUNC(OPT) that can exceed its PICTURE, an intermediate
4
+ // result wider than ARITH lets the compiler carry, and a range of characters whose order EBCDIC and
5
+ // ASCII disagree on. The first two rest on entries in ironwork's assumption register
6
+ // (crates/numeric/src/assumptions.rs), C2 and C1, which no Enterprise COBOL compile has settled yet,
7
+ // and each finding says so. cobolwork links nothing of ironwork; the rules are its model written down.
8
+ import { basename, extname } from 'node:path';
9
+ import { inScope, isProgram, isJcl, relPath, ebcdicByte } from '../sources.mjs';
10
+ import { report } from '../kernel/ruleset.mjs';
11
+ import { treeFor, noteUnread, noteUnparsed } from '../kernel/source-tree.mjs';
12
+ import { eachWithinMemory } from '../kernel/memory.mjs';
13
+ import { loadSite } from '../site.mjs';
14
+ import { parseJcl } from '../jcl.mjs';
15
+ import { optionCards, optionTokens, compileStepOptions } from '../options.mjs';
16
+
17
+ export const SEMANTICS_RULES = {
18
+ 'binary-store-exceeds-picture-under-trunc-opt': {
19
+ sev: 'low', evidence: 'construct', cwe: 'CWE-758',
20
+ text: 'Under TRUNC(OPT) a binary field receives a value that can have more digits than its PICTURE',
21
+ impact: 'TRUNC(OPT) lets the compiler assume every value fits the PICTURE, so when one does not, whether the field keeps the excess digits or loses them depends on the code generated for that statement and can change with the compiler level or its optimisation',
22
+ remedy: 'Make the receiver COMP-5 or wide enough for the value, add ON SIZE ERROR, or compile the program with TRUNC(STD) or TRUNC(BIN)',
23
+ },
24
+ 'intermediate-result-loses-high-order-digits': {
25
+ sev: 'low', evidence: 'construct', cwe: 'CWE-197',
26
+ text: 'A COMPUTE has an intermediate result wider than the compiler carries, so high-order digits can be lost',
27
+ impact: 'In ironwork\'s model of Enterprise COBOL a fixed-point intermediate carries at most 30 digits (31 under ARITH(EXTEND)) and, beyond that, keeps the decimal places the statement needs and drops high-order integer digits without raising SIZE ERROR, so a large enough value computes a wrong result silently',
28
+ remedy: 'Split the expression so each step fits, carry fewer decimal places in the operands and receivers, or compile with ARITH(EXTEND) where one more digit is enough',
29
+ },
30
+ 'character-range-reverses-in-ascii': {
31
+ sev: 'low', evidence: 'construct', cwe: 'CWE-474',
32
+ text: 'A range of characters is in order in EBCDIC and reversed in ASCII, or the other way round, so it holds values in one and none in the other',
33
+ impact: 'The same source tests a different set of characters when it is compiled with an ASCII collating sequence - GnuCOBOL, Micro Focus, a migration target - so a test on one platform passes or fails for a reason the source does not show',
34
+ remedy: 'Test the class the range means (IS NUMERIC, IS ALPHABETIC, a CLASS condition) or list the values, rather than relying on the order of the code page',
35
+ },
36
+ };
37
+
38
+ const MAX_SHOWN = 3;
39
+ const memberName = (file) => basename(file, extname(file)).toUpperCase();
40
+
41
+ // The last setting of an option family across the estate's defaults, the JCL step that compiles the
42
+ // member and its own CBL and PROCESS cards, with where it was set. null where no level sets it.
43
+ function lastSetting(names, { site, steps, cards }) {
44
+ const settings = [
45
+ ...site.flatMap((s) => optionTokens(s).map((token) => ({ token, where: 'compilerOptions in cobolwork.site.json' }))),
46
+ ...steps.flatMap((s) => s.options.map((token) => ({ token, where: `the compile step at ${s.file}:${s.line}` }))),
47
+ ...cards.flatMap((c) => c.options.map((token) => ({ token, where: `the ${c.level} statement at line ${c.line}` }))),
48
+ ];
49
+ for (let k = settings.length - 1; k >= 0; k--) {
50
+ const m = /^([A-Z0-9-]+)(?:\((.*)\))?$/.exec(settings[k].token);
51
+ if (m && names.includes(m[1])) return { sub: (m[2] || '').trim(), token: settings[k].token, where: settings[k].where };
52
+ }
53
+ return null;
54
+ }
55
+
56
+ // Integer and decimal places of a numeric PICTURE, with P scaling: P before the 9s adds decimal
57
+ // places, after them integer places. null for anything that is not a fixed-point number.
58
+ export function placesOf(picture) {
59
+ if (!picture) return null;
60
+ const pic = String(picture).toUpperCase().replace(/(.)\((\d+)\)/g, (_, c, n) => c.repeat(Number(n)));
61
+ if (!/^S?[9PV]+$/.test(pic) || !pic.includes('9')) return null;
62
+ const body = pic.replace(/^S/, '');
63
+ const v = body.indexOf('V');
64
+ const [left, right] = v < 0 ? [body, ''] : [body.slice(0, v), body.slice(v + 1)];
65
+ const leadingP = /^P+/.test(left) ? left.match(/^P+/)[0].length : 0;
66
+ const nines = (s) => (s.match(/9/g) || []).length;
67
+ const ps = (s) => (s.match(/P/g) || []).length;
68
+ if (leadingP || (v >= 0 && /^P/.test(right))) return { int: 0, dec: nines(left) + nines(right) + ps(left) + ps(right) };
69
+ return { int: nines(left) + ps(left), dec: nines(right) };
70
+ }
71
+
72
+ const BINARY_USAGE = new Set(['COMP', 'COMP-4', 'BINARY', 'COMPUTATIONAL', 'COMPUTATIONAL-4']);
73
+ const FLOAT_USAGE = new Set(['COMP-1', 'COMP-2', 'COMPUTATIONAL-1', 'COMPUTATIONAL-2', 'FLOAT-SHORT', 'FLOAT-LONG']);
74
+ const numberPlaces = (text) => {
75
+ const m = /^[+-]?(\d*)(?:[.,](\d*))?$/.exec(String(text));
76
+ if (!m || !(m[1] || m[2])) return null;
77
+ return { int: m[1].replace(/^0+(?=\d)/, '').length, dec: (m[2] || '').length };
78
+ };
79
+
80
+ // IBM's places for an intermediate, as ironwork's numeric/src/precision.rs computes them.
81
+ const sum = (a, b) => ({ int: Math.max(a.int, b.int) + 1, dec: Math.max(a.dec, b.dec) });
82
+ const product = (a, b) => ({ int: a.int + b.int, dec: a.dec + b.dec });
83
+ const quotient = (a, b, dmax) => ({ int: a.int + b.dec, dec: Math.max(a.dec, dmax) });
84
+ function carried(ir, dmax, n) {
85
+ if (ir.int + ir.dec <= n) return ir;
86
+ if (ir.dec <= dmax) return { int: Math.max(0, n - ir.dec), dec: ir.dec };
87
+ if (ir.int + dmax <= n) return { int: ir.int, dec: n - ir.int };
88
+ return { int: Math.max(0, n - dmax), dec: dmax };
89
+ }
90
+
91
+ // A COMPUTE's expression as operands and operators: an identifier (with its subscript skipped), a
92
+ // numeric literal, parentheses, unary and binary + - * /. Exponentiation and functions are left
93
+ // unread: IBM computes them in floating point or by rules of their own.
94
+ function expression(toks) {
95
+ const out = [];
96
+ for (let k = 0; k < toks.length; k++) {
97
+ const t = toks[k];
98
+ if (t.t === 'op' && ['+', '-', '*', '/'].includes(t.v)) { out.push({ op: t.v }); continue; }
99
+ if (t.t === 'op' && t.v === '**') return null;
100
+ if (t.t === 'sep' && (t.v === '(' || t.v === ')')) { out.push({ paren: t.v }); continue; }
101
+ if (t.t === 'num' || (t.t === 'word' && /^[+-]?\d+([.,]\d+)?$/.test(t.v))) { out.push({ literal: String(t.v) }); continue; }
102
+ if (t.t === 'word' && t.u === 'FUNCTION') return null;
103
+ if (t.t === 'word') {
104
+ const name = [t.u];
105
+ while (toks[k + 1] && toks[k + 1].t === 'word' && (toks[k + 1].u === 'OF' || toks[k + 1].u === 'IN') && toks[k + 2]) { name.push(toks[k + 2].u); k += 2; }
106
+ out.push({ name: name[0], qualified: name.length > 1 });
107
+ if (toks[k + 1] && toks[k + 1].t === 'sep' && toks[k + 1].v === '(') {
108
+ let depth = 0;
109
+ for (k++; k < toks.length; k++) {
110
+ if (toks[k].t === 'sep' && toks[k].v === '(') depth++;
111
+ if (toks[k].t === 'sep' && toks[k].v === ')' && --depth === 0) break;
112
+ }
113
+ }
114
+ continue;
115
+ }
116
+ return null;
117
+ }
118
+ return out;
119
+ }
120
+
121
+ // Walks the expression by precedence, carrying each intermediate as IBM would, and returns the first
122
+ // intermediate whose integer places the carry cut, or null. `lookup` gives an operand's places.
123
+ function firstLoss(items, { lookup, dmax, n }) {
124
+ let at = 0;
125
+ let loss = null;
126
+ const combine = (a, op, b) => {
127
+ const ir = op === '+' || op === '-' ? sum(a, b) : op === '*' ? product(a, b) : quotient(a, b, dmax);
128
+ const kept = carried(ir, dmax, n);
129
+ if (!loss && kept.int < ir.int) loss = { ir, kept, op };
130
+ return kept;
131
+ };
132
+ const primary = () => {
133
+ const x = items[at++];
134
+ if (!x) throw new Error('end');
135
+ if (x.op === '+' || x.op === '-') return primary();
136
+ if (x.paren === '(') { const v = additive(); if (!items[at] || items[at].paren !== ')') throw new Error('paren'); at++; return v; }
137
+ const p = x.literal !== undefined ? numberPlaces(x.literal) : lookup(x);
138
+ if (!p) throw new Error('operand');
139
+ return p;
140
+ };
141
+ const multiplicative = () => {
142
+ let v = primary();
143
+ while (items[at] && (items[at].op === '*' || items[at].op === '/')) { const op = items[at++].op; v = combine(v, op, primary()); }
144
+ return v;
145
+ };
146
+ const additive = () => {
147
+ let v = multiplicative();
148
+ while (items[at] && (items[at].op === '+' || items[at].op === '-')) { const op = items[at++].op; v = combine(v, op, multiplicative()); }
149
+ return v;
150
+ };
151
+ try { additive(); } catch { return { unread: true }; }
152
+ return at === items.length ? loss : { unread: true };
153
+ }
154
+
155
+ // The integer places a value from `items` can have: no carry for + and -, as a counter's increment
156
+ // is not a value the program lets grow past its field; a product or quotient as IBM places it.
157
+ function naturalInt(items, lookup) {
158
+ let at = 0;
159
+ const primary = () => {
160
+ const x = items[at++];
161
+ if (!x) throw new Error('end');
162
+ if (x.op === '+' || x.op === '-') return primary();
163
+ if (x.paren === '(') { const v = additive(); at++; return v; }
164
+ const p = x.literal !== undefined ? numberPlaces(x.literal) : lookup(x);
165
+ if (!p) throw new Error('operand');
166
+ return p;
167
+ };
168
+ const multiplicative = () => {
169
+ let v = primary();
170
+ while (items[at] && (items[at].op === '*' || items[at].op === '/')) {
171
+ const op = items[at++].op;
172
+ const b = primary();
173
+ v = op === '*' ? product(v, b) : { int: v.int + b.dec, dec: v.dec };
174
+ }
175
+ return v;
176
+ };
177
+ const additive = () => {
178
+ let v = multiplicative();
179
+ while (items[at] && (items[at].op === '+' || items[at].op === '-')) { at++; const b = multiplicative(); v = { int: Math.max(v.int, b.int), dec: Math.max(v.dec, b.dec) }; }
180
+ return v;
181
+ };
182
+ try { return additive().int; } catch { return null; }
183
+ }
184
+
185
+ // Two literals compared as the collating sequence would: shorter padded with spaces, by code.
186
+ function order(a, b, code) {
187
+ const len = Math.max(a.length, b.length);
188
+ for (let i = 0; i < len; i++) {
189
+ const x = code(a[i] ?? ' ');
190
+ const y = code(b[i] ?? ' ');
191
+ if (x === null || y === null) return null;
192
+ if (x !== y) return x < y ? -1 : 1;
193
+ }
194
+ return 0;
195
+ }
196
+ const asciiCode = (ch) => { const c = ch.codePointAt(0); return c >= 0x20 && c <= 0x7E ? c : null; };
197
+
198
+ // The literal pairs `lo THRU hi` in a run of VALUE or WHEN tokens.
199
+ function thruPairs(toks) {
200
+ const out = [];
201
+ for (let k = 0; k + 2 < toks.length; k++) {
202
+ const [lo, thru, hi] = [toks[k], toks[k + 1], toks[k + 2]];
203
+ if (lo.t === 'lit' && thru.t === 'word' && (thru.u === 'THRU' || thru.u === 'THROUGH') && hi.t === 'lit') out.push({ lo: String(lo.v), hi: String(hi.v), line: lo.line });
204
+ }
205
+ return out;
206
+ }
207
+ const WHEN_WORDS = new Set(['THRU', 'THROUGH', 'ALSO', 'OR', 'NOT', 'ANY', 'OTHER', 'TRUE', 'FALSE']);
208
+
209
+ export function scanSemantics(root, opts = {}) {
210
+ const tree = treeFor(root, opts);
211
+ const all = tree.list().filter(inScope(opts));
212
+ const files = all.filter(isProgram);
213
+ const findings = [];
214
+ const stats = {
215
+ filesScanned: 0, filesUnreadable: 0, filesUnparsed: 0,
216
+ truncOptPrograms: 0, truncUnknownPrograms: 0, arithExtendPrograms: 0, beyondEnterprisePrograms: 0, computesRead: 0, computesUnread: 0,
217
+ rangesRead: 0, collatingDeclaredPrograms: 0,
218
+ };
219
+ const site = loadSite(root, opts.site || null).compilerOptions || [];
220
+ const parsedJcl = [];
221
+ for (const f of all.filter(isJcl)) {
222
+ try { parsedJcl.push({ ...parseJcl(tree.text(f).text, f), file: relPath(root, f) }); } catch { /* the jcl set reports it */ }
223
+ }
224
+ const stepsOf = new Map();
225
+ for (const s of compileStepOptions(parsedJcl)) {
226
+ if (!stepsOf.has(s.member)) stepsOf.set(s.member, []);
227
+ stepsOf.get(s.member).push(s);
228
+ }
229
+
230
+ function judge(p, path, levels, collatingDeclared) {
231
+ const byName = new Map();
232
+ for (const it of p.items) {
233
+ const k = String(it.name).toUpperCase();
234
+ byName.set(k, byName.has(k) ? null : it);
235
+ }
236
+ const numeric = (x) => {
237
+ const it = byName.get(x.name);
238
+ if (!it || FLOAT_USAGE.has(String(it.effectiveUsage || '').toUpperCase())) return null;
239
+ return placesOf(it.picture);
240
+ };
241
+ const trunc = lastSetting(['TRUNC'], levels);
242
+ const arith = lastSetting(['ARITH', 'AR'], levels);
243
+ // ARITH(COMPAT) allows 18 digits in an item and ARITH(EXTEND) 31 (ironwork's
244
+ // Arith::max_picture_digits): a wider item says which one the program needs, or that it is not
245
+ // Enterprise COBOL at all.
246
+ const widest = Math.max(0, ...p.items.map((it) => { const x = placesOf(it.picture); return x ? x.int + x.dec : 0; }));
247
+ const extendByItem = !arith && widest > 18;
248
+ const n = (arith && /^(EXTEND|E)$/.test(arith.sub)) || extendByItem ? 31 : 30;
249
+ const beyondIbm = widest > 31;
250
+ if (beyondIbm) stats.beyondEnterprisePrograms++;
251
+ const truncOpt = trunc && trunc.sub === 'OPT';
252
+ if (truncOpt) stats.truncOptPrograms++;
253
+ if (!trunc) stats.truncUnknownPrograms++;
254
+ if (n === 31) stats.arithExtendPrograms++;
255
+ const toks = p.proc ? p.proc.tokens : [];
256
+ const narrowed = [];
257
+
258
+ for (const st of p.statements) {
259
+ const seg = toks.slice(st.at + 1, st.end ?? toks.length);
260
+ const sizeError = seg.some((t, k) => t.t === 'word' && t.u === 'SIZE' && seg[k + 1] && seg[k + 1].u === 'ERROR');
261
+
262
+ if (st.verb === 'COMPUTE') {
263
+ const eq = seg.findIndex((t) => (t.t === 'op' && t.v === '=') || (t.t === 'word' && (t.u === 'EQUAL' || t.u === 'EQUALS')));
264
+ if (eq < 0) continue;
265
+ let stop = seg.findIndex((t, k) => k > eq && t.t === 'word' && (t.u === 'ON' || t.u === 'NOT' || t.u === 'SIZE' || t.u === 'END-COMPUTE'));
266
+ if (stop < 0) stop = seg.length;
267
+ const items = expression(seg.slice(eq + 1, stop));
268
+ const receivers = seg.slice(0, eq).filter((t) => t.t === 'word' && t.u !== 'ROUNDED').map((t) => byName.get(t.u)).filter(Boolean);
269
+ if (!items || beyondIbm) { stats.computesUnread++; continue; }
270
+ const dmax = Math.max(0, ...receivers.map((r) => (placesOf(r.picture) || { dec: 0 }).dec),
271
+ ...items.filter((x, k) => !(items[k - 1] && items[k - 1].op === '/')).map((x) => (x.name ? (numeric(x) || { dec: 0 }).dec : x.literal ? (numberPlaces(x.literal) || { dec: 0 }).dec : 0)));
272
+ const loss = firstLoss(items, { lookup: numeric, dmax, n });
273
+ if (loss && loss.unread) { stats.computesUnread++; continue; }
274
+ stats.computesRead++;
275
+ if (loss) {
276
+ findings.push({
277
+ rule: 'intermediate-result-loses-high-order-digits', path, line: st.line, program: p.id,
278
+ detail: `${p.id}: a ${loss.op === '*' ? 'product' : loss.op === '/' ? 'quotient' : 'sum'} in this COMPUTE has ${loss.ir.int} integer and ${loss.ir.dec} decimal places; under ARITH(${n === 31 ? 'EXTEND' : 'COMPAT'}) (${arith ? `set by ${arith.where}` : extendByItem ? `which its ${widest}-digit item needs, since no level this set reads sets ARITH` : 'IBM\'s default: no level this set reads sets ARITH'}) the compiler carries ${n} digits and keeps ${loss.kept.int} integer places. This follows ironwork's intermediate table, assumption C1, which no Enterprise COBOL compile has settled yet`,
279
+ });
280
+ }
281
+ if (truncOpt && !sizeError) {
282
+ const int = naturalInt(items, numeric);
283
+ for (const r of receivers) {
284
+ const rp = placesOf(r.picture);
285
+ if (rp && BINARY_USAGE.has(String(r.effectiveUsage || '').toUpperCase()) && int !== null && int > rp.int) narrowed.push({ st, r, rp, int, via: 'COMPUTE' });
286
+ }
287
+ }
288
+ continue;
289
+ }
290
+
291
+ if (!truncOpt || sizeError) continue;
292
+ if (st.verb === 'MOVE' && !st.corresponding) {
293
+ const to = seg.findIndex((t) => t.t === 'word' && t.u === 'TO');
294
+ if (to !== 1) continue;
295
+ const src = seg[0];
296
+ const sp = src.t === 'num' || (src.t === 'word' && /^[+-]?\d/.test(src.v)) ? numberPlaces(src.v) : src.t === 'word' ? numeric({ name: src.u }) : null;
297
+ if (!sp) continue;
298
+ for (const t of seg.slice(to + 1)) {
299
+ if (t.t !== 'word') continue;
300
+ const r = byName.get(t.u);
301
+ const rp = r && placesOf(r.picture);
302
+ if (rp && BINARY_USAGE.has(String(r.effectiveUsage || '').toUpperCase()) && sp.int > rp.int) narrowed.push({ st, r, rp, int: sp.int, via: 'MOVE' });
303
+ }
304
+ continue;
305
+ }
306
+ if (['ADD', 'SUBTRACT', 'MULTIPLY', 'DIVIDE'].includes(st.verb)) {
307
+ const operands = st.sources.map((t) => numeric({ name: t.u })).filter(Boolean);
308
+ const literal = st.literals.map((l) => numberPlaces(l.v)).filter(Boolean);
309
+ const all = [...operands, ...literal];
310
+ if (!all.length) continue;
311
+ for (const t of st.targets) {
312
+ const r = byName.get(t.u);
313
+ const rp = r && placesOf(r.picture);
314
+ if (!rp || !BINARY_USAGE.has(String(r.effectiveUsage || '').toUpperCase())) continue;
315
+ const giving = seg.some((x) => x.t === 'word' && x.u === 'GIVING');
316
+ const int = st.verb === 'MULTIPLY' ? all.reduce((a, b) => a + b.int, giving ? 0 : rp.int)
317
+ : st.verb === 'DIVIDE' ? Math.max(...all.map((x) => x.int))
318
+ : Math.max(...all.map((x) => x.int));
319
+ if (int > rp.int) narrowed.push({ st, r, rp, int, via: st.verb });
320
+ }
321
+ }
322
+ }
323
+
324
+ // One finding per statement: the receivers it narrows, and where TRUNC(OPT) was set.
325
+ const byStatement = new Map();
326
+ for (const x of narrowed) {
327
+ if (!byStatement.has(x.st)) byStatement.set(x.st, []);
328
+ byStatement.get(x.st).push(x);
329
+ }
330
+ for (const [st, xs] of byStatement) {
331
+ const shown = xs.slice(0, MAX_SHOWN).map((x) => `${x.r.name} (${x.r.picture} ${String(x.r.effectiveUsage).toUpperCase()}, ${x.rp.int} integer digits) a value with up to ${x.int} integer digits`).join('; ');
332
+ findings.push({
333
+ rule: 'binary-store-exceeds-picture-under-trunc-opt', path, line: st.line, program: p.id,
334
+ detail: `${p.id}: this ${xs[0].via} gives ${shown}, under ${trunc.token} set by ${trunc.where}; what the field then holds depends on the generated code (ironwork's model keeps the binary width, assumption C2, which no Enterprise COBOL compile has settled yet)`,
335
+ });
336
+ }
337
+
338
+ if (collatingDeclared) { stats.collatingDeclaredPrograms++; return; }
339
+ const ranges = [];
340
+ for (const it of p.items) if (it.level === 88) for (const r of thruPairs(it.values || [])) ranges.push({ ...r, where: `88 ${it.name}` });
341
+ for (const st of p.statements) {
342
+ if (st.verb !== 'WHEN') continue;
343
+ const run = [];
344
+ for (let k = st.at + 1; k < toks.length; k++) {
345
+ const t = toks[k];
346
+ if (t.t === 'lit' || (t.t === 'word' && WHEN_WORDS.has(t.u))) run.push(t); else break;
347
+ }
348
+ for (const r of thruPairs(run)) ranges.push({ ...r, line: st.line, where: 'WHEN' });
349
+ }
350
+ for (const r of ranges) {
351
+ stats.rangesRead++;
352
+ const e = order(r.lo, r.hi, ebcdicByte);
353
+ const a = order(r.lo, r.hi, asciiCode);
354
+ if (e === null || a === null || e === 0 || a === 0 || e === a) continue;
355
+ findings.push({
356
+ rule: 'character-range-reverses-in-ascii', path, line: r.line, program: p.id,
357
+ detail: `${p.id}: ${r.where} '${r.lo}' THRU '${r.hi}' is ${e < 0 ? 'in order' : 'reversed'} in EBCDIC and ${a < 0 ? 'in order' : 'reversed'} in ASCII, so it is empty ${e < 0 ? 'under an ASCII collating sequence' : 'in EBCDIC, on the mainframe'}`,
358
+ });
359
+ }
360
+ }
361
+
362
+ const run = eachWithinMemory(files, (f) => {
363
+ let src;
364
+ try { src = tree.text(f).text; } catch (e) { noteUnread(stats, tree, f, e); return 0; }
365
+ let r;
366
+ try { r = tree.parse(f, src); } catch (e) { noteUnparsed(stats, tree, f, e); return src.length; }
367
+ stats.filesScanned++;
368
+ const path = relPath(root, f);
369
+ const levels = { site, steps: stepsOf.get(memberName(f)) || [], cards: optionCards(src) };
370
+ const collatingDeclared = /\bPROGRAM\s+COLLATING\s+SEQUENCE\b/i.test(src);
371
+ for (const p of r.programs) judge(p, path, levels, collatingDeclared);
372
+ r = null;
373
+ return src.length;
374
+ }, { label: 'semantics', maxBytes: opts.maxSourceBytes ?? Infinity });
375
+
376
+ return report('semantics', { rules: SEMANTICS_RULES, findings, stats, run });
377
+ }
package/lib/sets/web.mjs CHANGED
@@ -121,6 +121,25 @@ const TELLS = /^(server|x-powered-by|x-aspnet-version|x-generator|via)$/i;
121
121
  // STRING from a field whose name says what it holds - lib/sets/log.mjs already judges the name.
122
122
  const CREDENTIAL_PARAM = /[?&/][a-z0-9_-]*(password|passwd|pwd|token|apikey|api_key|secret|credential)[a-z0-9_-]*=/i;
123
123
 
124
+ // What a request can make a program change: a file record, a row, or a transaction started. A
125
+ // temporary-storage queue is where a web program keeps its own conversation, so it is not counted.
126
+ function stateChange(e) {
127
+ const w = e.toks.filter((t) => t.t === 'word').map((t) => t.u);
128
+ if (e.kind === 'SQL') return ['UPDATE', 'INSERT', 'DELETE', 'MERGE'].includes(w[0]) ? `EXEC SQL ${w[0]}` : null;
129
+ if (e.kind !== 'CICS') return null;
130
+ if (['WRITE', 'REWRITE', 'DELETE'].includes(w[0]) && ['FILE', 'DATASET'].includes(w[1])) return `EXEC CICS ${w[0]} ${w[1]}`;
131
+ if (w[0] === 'START' && w.includes('TRANSID')) return 'EXEC CICS START TRANSID';
132
+ return null;
133
+ }
134
+ const CONDITION_VERBS = new Set(['IF', 'EVALUATE', 'WHEN']);
135
+ // A WHEN's statement runs to the next verb; its condition is the few words after it.
136
+ const WHEN_WINDOW = 12;
137
+ const CONDITION_WORDS = new Set(['IS', 'NOT', 'EQUAL', 'EQUALS', 'TO', 'THAN', 'GREATER', 'LESS', 'AND', 'OR', 'THEN', 'OF', 'IN', 'TRUE', 'FALSE', 'ALSO', 'OTHER', 'ANY']);
138
+ const FIGURATIVE = /^(SPACES?|ZEROS?|ZEROES|LOW-VALUES?|HIGH-VALUES?|NULLS?|QUOTES?)$/;
139
+ const TOKEN_WORDS = new Set(['TOKEN', 'NONCE', 'CSRF', 'ANTIFORGERY', 'XSRF']);
140
+ const namesToken = (t) => String(t.u).split('-').some((w) => TOKEN_WORDS.has(w));
141
+ const HTTP_METHOD = /^(GET|POST|PUT|PATCH|DELETE)$/i;
142
+
124
143
  // A definition that says nothing about SSL accepts cleartext, because CICS defaults it to NO. So
125
144
  // the absence and the explicit NO are one finding, and PROTOCOL only changes what is in the clear.
126
145
  const encrypted = (svc) => /^(YES|CLIENTAUTH|CLIENTCERT)$/i.test(String(svc.ssl || ''));
@@ -198,16 +217,18 @@ export function scanWeb(root, opts = {}) {
198
217
  const w = e.toks.filter((t) => t.t === 'word').map((t) => t.u);
199
218
  return w[0] === 'WEB' && (w[1] === 'RECEIVE' || w[1] === 'READ');
200
219
  });
201
- const changes = p.execs.filter((e) => {
202
- if (e.kind !== 'CICS') return false;
203
- const w = e.toks.filter((t) => t.t === 'word').map((t) => t.u);
204
- return ['WRITE', 'REWRITE', 'DELETE'].includes(w[0]) && ['FILE', 'DATASET'].includes(w[1]);
220
+ const changes = p.execs.map((e) => ({ e, what: stateChange(e) })).filter((c) => c.what);
221
+ // A token counts where a condition compares it with another field: moved, displayed or tested
222
+ // against SPACES, it verifies nothing.
223
+ const conditions = (p.statements || []).filter((st) => CONDITION_VERBS.has(st.verb))
224
+ .map((st) => (p.proc ? p.proc.tokens.slice(st.at + 1, st.end ?? st.at + 1 + WHEN_WINDOW) : []));
225
+ const comparesToken = conditions.some((c) => {
226
+ const words = c.filter((t) => t.t === 'word' && !CONDITION_WORDS.has(t.u) && !FIGURATIVE.test(t.u));
227
+ return words.some(namesToken) && words.length >= 2;
205
228
  });
206
- const tokenWords = new Set(['TOKEN', 'NONCE', 'CSRF', 'ANTIFORGERY', 'XSRF']);
207
- const comparesToken = (p.statements || []).some((st) => (st.sources || [])
208
- .some((t) => t.t === 'word' && String(t.u).split('-').some((w) => tokenWords.has(w))));
229
+ const testsMethod = conditions.some((c) => c.some((t) => t.t === 'lit' && HTTP_METHOD.test(String(t.v).trim())));
209
230
 
210
- return { headers, sends, cookies, computedHeaders, docText, looksHtml, id: p.id, uris, built, webRequest, changes, comparesToken };
231
+ return { headers, sends, cookies, computedHeaders, docText, looksHtml, id: p.id, uris, built, webRequest, changes, comparesToken, testsMethod };
211
232
  }
212
233
 
213
234
  // The listener that accepts the request these responses answer. It is defined in a CSD rather
@@ -287,9 +308,12 @@ export function scanWeb(root, opts = {}) {
287
308
  });
288
309
  }
289
310
  if (v.webRequest && v.changes.length && !v.comparesToken) {
311
+ const what = [...new Set(v.changes.map((c) => c.what))];
290
312
  findings.push({
291
- rule: 'web-request-changes-state-without-a-token', ...where(v.changes[0], path), program: v.id,
292
- detail: `${v.id} changes state on a request it received from the web and compares no token; nothing here follows a token checked in another program reached by LINK, so that is what this could not rule out`,
313
+ rule: 'web-request-changes-state-without-a-token', ...where(v.changes[0].e, path), program: v.id,
314
+ detail: `${v.id} changes state (${what.slice(0, 3).join(', ')}${what.length > 3 ? ', …' : ''}) on a request it received from the web and compares no token with another field`
315
+ + (v.testsMethod ? '; it tests the HTTP method, which a forged request sets too' : '')
316
+ + '; nothing here follows a token checked in another program reached by LINK, so that is what this could not rule out',
293
317
  });
294
318
  }
295
319
  const html = v.sends.filter((s) => s.html === true).length > 0 || (v.looksHtml && v.sends.length > 0);