@portll/cobolwork 0.0.1 → 0.2.76
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/LICENSING.md +93 -0
- package/NOTICE +9 -0
- package/README.md +325 -3
- package/THIRD-PARTY-NOTICES.md +118 -0
- package/bin/cobolwork.mjs +354 -0
- package/lib/advisories.mjs +133 -0
- package/lib/baseline.mjs +154 -0
- package/lib/bms.mjs +453 -0
- package/lib/build.mjs +402 -0
- package/lib/capabilities.mjs +79 -0
- package/lib/cics-commands.mjs +281 -0
- package/lib/compliance.mjs +81 -0
- package/lib/consequence.mjs +139 -0
- package/lib/control.mjs +1515 -0
- package/lib/csd.mjs +77 -0
- package/lib/dataflow.mjs +1506 -0
- package/lib/diff.mjs +344 -0
- package/lib/explain.mjs +145 -0
- package/lib/gate.mjs +383 -0
- package/lib/index.mjs +6 -0
- package/lib/inventory.mjs +79 -0
- package/lib/jcl.mjs +478 -0
- package/lib/kernel/findings.mjs +94 -0
- package/lib/kernel/identity.mjs +216 -0
- package/lib/kernel/memory.mjs +217 -0
- package/lib/kernel/printable.mjs +6 -0
- package/lib/kernel/registry.mjs +79 -0
- package/lib/kernel/ruleset.mjs +72 -0
- package/lib/kernel/source-tree.mjs +159 -0
- package/lib/kev.mjs +27 -0
- package/lib/options.mjs +512 -0
- package/lib/packs.mjs +148 -0
- package/lib/parser.mjs +2055 -0
- package/lib/policy.mjs +163 -0
- package/lib/precompile-cics.mjs +169 -0
- package/lib/precompile.mjs +544 -0
- package/lib/reach.mjs +122 -0
- package/lib/revision.json +1 -0
- package/lib/revision.mjs +89 -0
- package/lib/sarif.mjs +222 -0
- package/lib/scan.mjs +272 -0
- package/lib/sets/build.mjs +234 -0
- package/lib/sets/cics.mjs +306 -0
- package/lib/sets/compile.mjs +187 -0
- package/lib/sets/copybook.mjs +174 -0
- package/lib/sets/flow.mjs +487 -0
- package/lib/sets/hidden.mjs +216 -0
- package/lib/sets/jcl.mjs +440 -0
- package/lib/sets/log.mjs +406 -0
- package/lib/sets/opaque.mjs +102 -0
- package/lib/sets/priv.mjs +322 -0
- package/lib/sets/recon.mjs +267 -0
- package/lib/sets/vendor.mjs +117 -0
- package/lib/sets/web.mjs +327 -0
- package/lib/site.mjs +164 -0
- package/lib/sources.mjs +156 -0
- package/lib/tui/app.mjs +325 -0
- package/lib/tui/keys.mjs +39 -0
- package/lib/tui/model.mjs +96 -0
- package/lib/tui/run.mjs +38 -0
- package/lib/tui/screen.mjs +59 -0
- package/lib/tui/terminal.mjs +46 -0
- package/lib/utilities.mjs +296 -0
- package/lib/version.mjs +15 -0
- package/lib/words.mjs +318 -0
- package/package.json +45 -6
- package/rules/advisories.json +264 -0
- package/rules/compliance-dora.json +2151 -0
- package/rules/compliance-ffiec.json +2134 -0
- package/rules/compliance-nist80053.json +2134 -0
- package/rules/gitleaks-mainframe.toml +57 -0
- package/rules/kev-ids.json +1729 -0
- package/rules/packs/broadcom.json +124 -0
- package/rules/packs/connectdirect.json +116 -0
- package/rules/packs/controlm.json +114 -0
- package/rules/system-layouts.json +28 -0
- package/schema/cobolwork-coverage.schema.json +65 -0
- package/schema/cobolwork.policy.schema.json +54 -0
package/lib/dataflow.mjs
ADDED
|
@@ -0,0 +1,1506 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
import { inScope, isProgram, isBms, readSource, relPath } from './sources.mjs';
|
|
3
|
+
import { parseBms, symbolicNames } from './bms.mjs';
|
|
4
|
+
import { dirname, join } from 'node:path';
|
|
5
|
+
import { parseSource, buildFileIndex } from './parser.mjs';
|
|
6
|
+
import { parseJcl } from './jcl.mjs';
|
|
7
|
+
import { parseCsd, ddOfQueue } from './csd.mjs';
|
|
8
|
+
import { loadSite } from './site.mjs';
|
|
9
|
+
import { ssrangeAbends } from './options.mjs';
|
|
10
|
+
import { isJcl } from './sources.mjs';
|
|
11
|
+
import { eachWithinMemory, watchMemoryBuffer } from './kernel/memory.mjs';
|
|
12
|
+
// A flow finding is located by source and sink rather than by one path and line, so it keeps its
|
|
13
|
+
// own comparator. The text comparison underneath it is the shared one.
|
|
14
|
+
import { byText } from './kernel/findings.mjs';
|
|
15
|
+
import { buildControl, creditOf, hasFact } from './control.mjs';
|
|
16
|
+
import { directoryTree } from './kernel/source-tree.mjs';
|
|
17
|
+
|
|
18
|
+
const OS_COMMAND_ROUTINE = /^(SYSTEM|C\$SYSTEM|CBL_EXEC_RUN_UNIT|CBL_GC_HOSTED|BXPSYSTM)$/i;
|
|
19
|
+
|
|
20
|
+
export const SOURCE_KINDS = {
|
|
21
|
+
'argv-or-env': 'ACCEPT FROM COMMAND-LINE, ARGUMENT-VALUE or ENVIRONMENT',
|
|
22
|
+
'cics-terminal': 'EXEC CICS RECEIVE',
|
|
23
|
+
'cics-web': 'EXEC CICS WEB RECEIVE or EXTRACT',
|
|
24
|
+
'file-record': 'record read from a file',
|
|
25
|
+
'database': 'host variable filled by EXEC SQL',
|
|
26
|
+
'jcl-parm': 'PARM= on a JCL EXEC statement',
|
|
27
|
+
'jcl-instream': 'in-stream data on a JCL DD statement',
|
|
28
|
+
'cics-protected-field': 'a field the BMS map protects, read back after EXEC CICS RECEIVE MAP',
|
|
29
|
+
'system-response': 'a response code or SQL error the system sets: a command\'s RESP or RESP2, EIBRESP, SQLCODE, SQLSTATE, SQLERRMC',
|
|
30
|
+
};
|
|
31
|
+
export const SINK_KINDS = {
|
|
32
|
+
'os-command': 'CALL of an operating-system command routine',
|
|
33
|
+
'dynamic-program-load': 'CALL whose target is a variable',
|
|
34
|
+
'cics-dynamic-transfer': 'EXEC CICS LINK, XCTL or START with a variable program or transaction',
|
|
35
|
+
'dynamic-sql': 'EXEC SQL EXECUTE IMMEDIATE or PREPARE from a host variable',
|
|
36
|
+
'dynamic-file-path': 'SELECT ... ASSIGN TO a variable',
|
|
37
|
+
'outbound-http': 'EXEC CICS WEB CONVERSE, WEB SEND on a client session, or the containers INVOKE SERVICE sends',
|
|
38
|
+
'socket-send': 'CALL EZASOKET or EZACICAL with SEND, SENDTO or WRITE',
|
|
39
|
+
'message-queue': 'CALL MQPUT or MQPUT1',
|
|
40
|
+
'internal-reader': 'a record written where the internal reader submits it as a job: EXEC CICS WRITEQ TD to a queue the estate declares, or a DD the job sends to SYSOUT=(class,INTRDR)',
|
|
41
|
+
'extrapartition-queue': 'EXEC CICS WRITEQ TD to a queue the CSD defines as extrapartition, which leaves the region through a DD',
|
|
42
|
+
'web-response': 'the HTTP response this program returns: EXEC CICS WEB SEND, or a document built for it',
|
|
43
|
+
'http-header': 'a header on that response: EXEC CICS WEB WRITE HTTPHEADER',
|
|
44
|
+
'outbound-host': 'the host or path of a request this program makes: EXEC CICS WEB OPEN or CONVERSE',
|
|
45
|
+
'queue-name': 'the name of the queue an EXEC CICS command acts on, rather than what it writes',
|
|
46
|
+
'subscript': 'a name used as a subscript of a table',
|
|
47
|
+
'reference-modification': 'a name used as the start or length of a reference modification',
|
|
48
|
+
'occurs-depending-count': 'the object of OCCURS DEPENDING ON, which sets how many entries a table holds',
|
|
49
|
+
'loop-bound': 'what a PERFORM VARYING counter is compared with to stop, where the counter subscripts a table',
|
|
50
|
+
'arithmetic': 'a zoned or packed decimal operand of COMPUTE, ADD, SUBTRACT, MULTIPLY or DIVIDE',
|
|
51
|
+
'record-key': 'the RIDFLD of EXEC CICS READ, STARTBR or RESETBR, which decides which record is read',
|
|
52
|
+
'record-update': 'the RIDFLD of a READ UPDATE whose record the program then rewrites or deletes, or of a DELETE',
|
|
53
|
+
'log': 'what a program in a CICS region writes to a log: DISPLAY, EXEC CICS WRITEQ TD to a queue that neither starts a transaction nor feeds the internal reader, WRITE JOURNALNAME or WRITE OPERATOR',
|
|
54
|
+
};
|
|
55
|
+
|
|
56
|
+
const ARITHMETIC = new Set(['COMPUTE', 'ADD', 'SUBTRACT', 'MULTIPLY', 'DIVIDE']);
|
|
57
|
+
// Intrinsic functions whose result is a number the program computed, never the argument's bytes.
|
|
58
|
+
const NUMERIC_FUNCTIONS = new Set(`NUMVAL NUMVAL-C NUMVAL-F TEST-NUMVAL TEST-NUMVAL-C TEST-NUMVAL-F INTEGER INTEGER-PART
|
|
59
|
+
INTEGER-OF-DATE INTEGER-OF-DAY INTEGER-OF-FORMATTED-DATE DATE-OF-INTEGER DAY-OF-INTEGER DATE-TO-YYYYMMDD DAY-TO-YYYYDDD
|
|
60
|
+
YEAR-TO-YYYY LENGTH BYTE-LENGTH ORD ORD-MAX ORD-MIN MOD REM ABS SIGN SUM MEAN MEDIAN MIDRANGE RANGE VARIANCE
|
|
61
|
+
STANDARD-DEVIATION RANDOM SQRT FACTORIAL LOG LOG10 EXP EXP10 PI E ACOS ASIN ATAN COS SIN TAN ANNUITY PRESENT-VALUE
|
|
62
|
+
SECONDS-PAST-MIDNIGHT`.split(/\s+/));
|
|
63
|
+
const computes = (st) => ARITHMETIC.has(st.verb) || st.counts === true || (st.fns || []).some((f) => NUMERIC_FUNCTIONS.has(f));
|
|
64
|
+
const computedOnRoute = (state) => { for (let s = state; s; s = s.prev) if (s.why && s.why.computes) return true; return false; };
|
|
65
|
+
|
|
66
|
+
// Every literal a program holds that could be a program name: its VALUE clauses and the literals its
|
|
67
|
+
// statements and commands use. A menu that XCTLs through a table of names starts every one of them.
|
|
68
|
+
function literalNames(prog) {
|
|
69
|
+
const out = new Set();
|
|
70
|
+
const take = (v) => { const s = String(v).trim().toUpperCase(); if (/^[A-Z$#@][A-Z0-9$#@-]{0,7}$/.test(s)) out.add(s); };
|
|
71
|
+
for (const it of prog.items) for (const v of it.values || []) if (v.t === 'lit') take(v.v);
|
|
72
|
+
for (const st of prog.statements) for (const l of st.literals || []) if (l.t === 'lit') take(l.v);
|
|
73
|
+
for (const e of prog.execs) for (const t of e.toks) if (t.t === 'lit') take(t.v);
|
|
74
|
+
return out;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// The value of the last NAME(value) option in a list: the last one wins, at either level.
|
|
78
|
+
function lastOption(name, list) {
|
|
79
|
+
for (let k = (list || []).length - 1; k >= 0; k--) {
|
|
80
|
+
const m = new RegExp(`^${name}\\(([A-Z]+)\\)$`).exec(list[k]);
|
|
81
|
+
if (m) return m[1];
|
|
82
|
+
}
|
|
83
|
+
return null;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
// Whether an item's bytes are read as decimal digits: zoned (DISPLAY with a numeric picture) or
|
|
87
|
+
// packed. A group, an edited picture, binary and floating point are not.
|
|
88
|
+
function holdsDecimal(it) {
|
|
89
|
+
if ((it.children || []).some((c) => c.level !== 88)) return false;
|
|
90
|
+
const usage = String(it.effectiveUsage || 'DISPLAY').replace('COMPUTATIONAL', 'COMP');
|
|
91
|
+
if (usage === 'COMP-3' || usage === 'PACKED-DECIMAL' || usage === 'COMP-6') return true;
|
|
92
|
+
return usage === 'DISPLAY' && !!it.picture && /9/.test(it.picture) && /^[S9VP()0-9]+$/i.test(it.picture);
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
const RELATION = new Set(['>', '<', '>=', '<=', '=', '<>']);
|
|
96
|
+
const RELATION_WORDS = new Set(['GREATER', 'LESS', 'EQUAL']);
|
|
97
|
+
const FILLER_WORDS = new Set(['IS', 'NOT', 'THAN', 'TO', 'OR', 'GREATER', 'LESS', 'EQUAL', 'OF', 'IN']);
|
|
98
|
+
// The names a loop counter is compared with in its UNTIL condition: I > N, N < I, I NOT LESS THAN N.
|
|
99
|
+
// GREATER THAN OR EQUAL TO is one relation, not two conditions joined by OR.
|
|
100
|
+
function boundsOf(counter, until) {
|
|
101
|
+
const parts = [[]];
|
|
102
|
+
let depth = 0;
|
|
103
|
+
until.forEach((x, k) => {
|
|
104
|
+
if (x.t === 'sep') depth += x.v === '(' ? 1 : -1;
|
|
105
|
+
else if (depth === 0 && x.t === 'word' && (x.u === 'AND' || x.u === 'OR') && !(until[k + 1] && until[k + 1].u === 'EQUAL')) { parts.push([]); return; }
|
|
106
|
+
parts[parts.length - 1].push(x);
|
|
107
|
+
});
|
|
108
|
+
const out = [];
|
|
109
|
+
for (const p of parts) {
|
|
110
|
+
const at = p.findIndex((x) => (x.t === 'op' && RELATION.has(x.v)) || (x.t === 'word' && RELATION_WORDS.has(x.u)));
|
|
111
|
+
if (at < 0) continue;
|
|
112
|
+
const names = (side) => side.filter((x, k) => x.t === 'word' && !FILLER_WORDS.has(x.u) && !/^\d+$/.test(x.v) && !(side[k - 1] && (side[k - 1].u === 'OF' || side[k - 1].u === 'IN')));
|
|
113
|
+
const left = names(p.slice(0, at));
|
|
114
|
+
const right = names(p.slice(at + 1));
|
|
115
|
+
if (left.length === 1 && left[0].u === counter) out.push(...right);
|
|
116
|
+
else if (right.length === 1 && right[0].u === counter) out.push(...left);
|
|
117
|
+
}
|
|
118
|
+
return out;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
// Input from outside the program's own data. The bounds sinks follow only these: in batch COBOL
|
|
122
|
+
// nearly every subscript descends from a file record, and reporting those would be reporting batch.
|
|
123
|
+
const FROM_OUTSIDE = ['argv-or-env', 'cics-terminal', 'cics-web', 'jcl-parm', 'jcl-instream'];
|
|
124
|
+
// What can reach a CICS command: a batch job never issues one.
|
|
125
|
+
const IN_CICS = ['cics-web', 'cics-terminal', 'file-record', 'database'];
|
|
126
|
+
const REFLECTED = ['cics-web', 'cics-terminal'];
|
|
127
|
+
// What the system says about a failure: a command's RESP, the EIB's copy of it, the SQLCA. Sent to a
|
|
128
|
+
// web client it describes the system behind the program (CWE-209); written anywhere else it is the
|
|
129
|
+
// program's own business, so these reach only a response.
|
|
130
|
+
const SYSTEM_FIELDS = new Set(['EIBRESP', 'EIBRESP2', 'SQLCODE', 'SQLSTATE', 'SQLERRMC']);
|
|
131
|
+
const TO_CLIENT = ['web-response', 'http-header'];
|
|
132
|
+
|
|
133
|
+
// Exfiltration is data at rest leaving the program. Terminal or web input reaching the same channel
|
|
134
|
+
// is the program relaying its own caller's data, and pairing it would fire on every gateway.
|
|
135
|
+
|
|
136
|
+
// A trace longer than this keeps its two ends and says how many hops it dropped.
|
|
137
|
+
export const TRACE_MAX = 64;
|
|
138
|
+
export const TRACE_KEEP = 24;
|
|
139
|
+
// Edges one source's walk may examine, and all walks together: about fifteen times what the largest
|
|
140
|
+
// repository of a 3,184-repository corpus needs, 294,370 for one walk and 33.5 million in all.
|
|
141
|
+
export const WALK_EDGES = 5_000_000;
|
|
142
|
+
export const TOTAL_EDGES = 500_000_000;
|
|
143
|
+
|
|
144
|
+
const DATA_AT_REST = ['database', 'file-record'];
|
|
145
|
+
// The class tests a condition can make, which restrict what a field holds rather than compare it.
|
|
146
|
+
const CLASS_TEST = new Set(['NUMERIC', 'ALPHABETIC', 'ALPHABETIC-LOWER', 'ALPHABETIC-UPPER', 'POSITIVE', 'NEGATIVE']);
|
|
147
|
+
const SOCKET_ROUTINE = /^(EZASOKET|EZACICAL)$/i;
|
|
148
|
+
const SOCKET_SEND = /^(SEND|SENDTO|WRITE|SENDMSG|WRITEV)$/i;
|
|
149
|
+
// MQPUT's buffer is its sixth argument, MQPUT1's its seventh.
|
|
150
|
+
// MQPUT(Hconn, Hobj, MsgDesc, PutMsgOpts, BufferLength, Buffer, …) and MQPUT1(Hconn, ObjDesc,
|
|
151
|
+
// MsgDesc, PutMsgOpts, BufferLength, Buffer, …): the buffer is the sixth argument of both.
|
|
152
|
+
const MQ_BUFFER_ARG = { MQPUT: 5, MQPUT1: 5 };
|
|
153
|
+
|
|
154
|
+
function execOptions(exec) {
|
|
155
|
+
const opts = new Map();
|
|
156
|
+
const toks = exec.toks;
|
|
157
|
+
const words = [];
|
|
158
|
+
for (let i = 0; i < toks.length; i++) {
|
|
159
|
+
const t = toks[i];
|
|
160
|
+
if (t.t !== 'word') continue;
|
|
161
|
+
words.push(t.u);
|
|
162
|
+
if (toks[i + 1] && toks[i + 1].t === 'sep' && toks[i + 1].v === '(') {
|
|
163
|
+
const inner = [];
|
|
164
|
+
let depth = 1;
|
|
165
|
+
for (let k = i + 2; k < toks.length && depth > 0; k++) {
|
|
166
|
+
if (toks[k].t === 'sep') { depth += toks[k].v === '(' ? 1 : -1; if (depth === 0) break; continue; }
|
|
167
|
+
inner.push(toks[k]);
|
|
168
|
+
}
|
|
169
|
+
if (!opts.has(t.u)) opts.set(t.u, inner);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
return { opts, words };
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
function sqlHostVars(exec) {
|
|
176
|
+
const out = [];
|
|
177
|
+
for (let i = 0; i < exec.toks.length; i++) {
|
|
178
|
+
const t = exec.toks[i];
|
|
179
|
+
if (t.t === 'op' && t.v === ':' && exec.toks[i + 1] && exec.toks[i + 1].t === 'word') out.push({ tok: exec.toks[i + 1], at: i + 1 });
|
|
180
|
+
}
|
|
181
|
+
return out;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
// The host variables a SELECT or FETCH fills are the ones in its INTO list, which ends at FROM.
|
|
185
|
+
// Taking every variable at or after the INTO line made the input parameters in a WHERE clause read
|
|
186
|
+
// as database values — and they are whatever the caller sent, under a rule that says data at rest.
|
|
187
|
+
function sqlIntoRange(exec) {
|
|
188
|
+
const words = exec.toks.map(t => (t.t === 'word' ? t.u : ''));
|
|
189
|
+
const into = words.indexOf('INTO');
|
|
190
|
+
if (into < 0) return null;
|
|
191
|
+
let end = exec.toks.length;
|
|
192
|
+
for (let i = into + 1; i < words.length; i++) if (words[i] === 'FROM') { end = i; break; }
|
|
193
|
+
return [into, end];
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
export function analyze(root, opts = {}) {
|
|
197
|
+
const repos = opts.repos || [''];
|
|
198
|
+
const stats = { files: 0, programs: 0, edges: 0, nodes: 0, threw: 0, overBudget: 0, unreadable: 0, ebcdic: 0,
|
|
199
|
+
jclFiles: 0, jclSteps: 0, jclStepsResolved: 0, jclCrossings: 0,
|
|
200
|
+
filesNotReached: 0, stoppedBy: null, peakHeapBytes: 0, programsUnordered: 0,
|
|
201
|
+
bmsFiles: 0, mapsReceived: 0, mapsNoSource: 0, mapsNoSymbolic: 0, protectedFields: 0 };
|
|
202
|
+
// One watcher across every repository in the run, so a --repos scan cannot spend the whole heap
|
|
203
|
+
// on the first repository and then report the rest as clean.
|
|
204
|
+
const memoryWatcher = watchMemoryBuffer();
|
|
205
|
+
// Reading everything is the default. A caller with less memory than the repository needs can set
|
|
206
|
+
// a source budget; files past it are counted as unread, never silently dropped.
|
|
207
|
+
const budget = opts.maxSourceBytes ?? (Number(process.env.COBOLWORK_MAX_SOURCE_BYTES) || Infinity);
|
|
208
|
+
// Every hop of every trace, for the reader who is following one chain rather than counting
|
|
209
|
+
// findings. Off by default: a chain-shaped program turns this quadratic in the chain length.
|
|
210
|
+
const fullTrace = opts.fullTrace === true;
|
|
211
|
+
let held = 0;
|
|
212
|
+
|
|
213
|
+
// Which files, not just how many. A count names nothing to go and look at, and it cannot tell
|
|
214
|
+
// a permission denied from this code being wrong.
|
|
215
|
+
const unread = [];
|
|
216
|
+
const unparsed = [];
|
|
217
|
+
const nodes = [];
|
|
218
|
+
const programs = [];
|
|
219
|
+
const jclPending = [];
|
|
220
|
+
// CICS definitions, from CSD extracts and from DFHCSDUP job input, and the estate's own statement
|
|
221
|
+
// of which DDs and queues reach the internal reader.
|
|
222
|
+
const csd = { tdqueues: new Map(), transactions: new Map() };
|
|
223
|
+
const addCsd = (text, file, lineBase = 0) => {
|
|
224
|
+
const c = parseCsd(text);
|
|
225
|
+
for (const [k, v] of c.tdqueues) if (!csd.tdqueues.has(k)) csd.tdqueues.set(k, v);
|
|
226
|
+
for (const [k, v] of c.transactions) if (!csd.transactions.has(k)) csd.transactions.set(k, { ...v, file, line: v.line + lineBase });
|
|
227
|
+
};
|
|
228
|
+
const jobSteps = [];
|
|
229
|
+
// The maps of the repository being read, by MAPSET/MAP and by MAP alone.
|
|
230
|
+
const bmsMaps = new Map();
|
|
231
|
+
const site = loadSite(root, opts.site || null);
|
|
232
|
+
const relOf = new Map();
|
|
233
|
+
const rel = (p) => { let r = relOf.get(p); if (r === undefined) { r = relPath(root, p); relOf.set(p, r); } return r; };
|
|
234
|
+
// `at` is the statement an edge that is not a statement's own is read at: the CALL or LINK that
|
|
235
|
+
// passes an argument, the PUT or GET of a container.
|
|
236
|
+
const edge = (a, b, why, dir, at) => {
|
|
237
|
+
if (!a || !b || a === b) return;
|
|
238
|
+
a.edgesOut.push(at != null ? { to: b, why, dir, at } : { to: b, why, dir });
|
|
239
|
+
stats.edges++;
|
|
240
|
+
};
|
|
241
|
+
|
|
242
|
+
// Everything the graph needs from one program is taken while its parse tree is in hand, and the
|
|
243
|
+
// tree is then released. Holding every tree until the end is what exhausted a 4 GB heap on an
|
|
244
|
+
// 8,024-file repository; nodes, edges and a few cross-program sites are all that survive.
|
|
245
|
+
function summarise(prog, file, pk, options = []) {
|
|
246
|
+
const byItem = new Map();
|
|
247
|
+
const byName = new Map();
|
|
248
|
+
for (const it of prog.items) if (!byName.has(it.name)) byName.set(it.name, it);
|
|
249
|
+
const nodeOf = (item) => {
|
|
250
|
+
let n = byItem.get(item);
|
|
251
|
+
if (!n) {
|
|
252
|
+
// Byte position within the record, so partial taint can be followed to the bytes it covers.
|
|
253
|
+
// Under an OCCURS the position repeats, and a range there would claim more than is known.
|
|
254
|
+
let repeats = false;
|
|
255
|
+
let table = null;
|
|
256
|
+
for (let a = item; a; a = a.parent) if ((a.occurs || 1) > 1) { repeats = true; table = a; }
|
|
257
|
+
n = { id: nodes.length, pk, program: prog.id, file, name: item.name, off: item.offset ?? null, size: item.size ?? null, repeats, edgesOut: [], sources: [], sinks: [] };
|
|
258
|
+
// Where the outermost table holding this item sits in the record: every occurrence of the
|
|
259
|
+
// item is somewhere inside it. A group's size already counts its occurrences; an
|
|
260
|
+
// elementary item's is one element.
|
|
261
|
+
if (table && table.offset != null && table.size != null) {
|
|
262
|
+
n.tableOff = table.offset;
|
|
263
|
+
n.tableEnd = table.offset + (table.children && table.children.length ? table.size : table.size * table.occurs);
|
|
264
|
+
}
|
|
265
|
+
nodes.push(n);
|
|
266
|
+
byItem.set(item, n);
|
|
267
|
+
stats.nodes++;
|
|
268
|
+
}
|
|
269
|
+
return n;
|
|
270
|
+
};
|
|
271
|
+
// Names inside EXEC blocks never pass through the parser's reference resolution, so fall back to the name.
|
|
272
|
+
const itemOfToken = (tok) => (tok ? prog.resolved.get(tok) || (tok.t === 'word' ? byName.get(tok.u) || null : null) : null);
|
|
273
|
+
// An INDEXED BY name is not a data item, so it needs a node of its own to carry taint.
|
|
274
|
+
const indexNames = new Set(prog.items.flatMap((it) => it.indexNames || []));
|
|
275
|
+
const indexNodes = new Map();
|
|
276
|
+
const indexNode = (name) => {
|
|
277
|
+
let n = indexNodes.get(name);
|
|
278
|
+
if (!n) {
|
|
279
|
+
n = { id: nodes.length, pk, program: prog.id, file, name, off: null, size: null, repeats: false, edgesOut: [], sources: [], sinks: [] };
|
|
280
|
+
nodes.push(n);
|
|
281
|
+
indexNodes.set(name, n);
|
|
282
|
+
stats.nodes++;
|
|
283
|
+
}
|
|
284
|
+
return n;
|
|
285
|
+
};
|
|
286
|
+
// A system field the program never declares - the EIB's, or an SQLCA the precompiler supplies -
|
|
287
|
+
// still holds a value, so it needs a node of its own, as an index name does.
|
|
288
|
+
const systemNodes = new Map();
|
|
289
|
+
const systemSource = (n, detail, at) => {
|
|
290
|
+
if (!n.sources.some((x) => x.kind === 'system-response')) n.sources.push({ kind: 'system-response', onlyTo: TO_CLIENT, noCredit: true, ...at, detail });
|
|
291
|
+
};
|
|
292
|
+
const systemNode = (tok) => {
|
|
293
|
+
let n = systemNodes.get(tok.u);
|
|
294
|
+
if (!n) {
|
|
295
|
+
n = { id: nodes.length, pk, program: prog.id, file, name: tok.u, off: null, size: null, repeats: false, edgesOut: [], sources: [], sinks: [] };
|
|
296
|
+
nodes.push(n);
|
|
297
|
+
systemNodes.set(tok.u, n);
|
|
298
|
+
stats.nodes++;
|
|
299
|
+
systemSource(n, `${tok.u}, which the system sets`, { file: tok.file || file, line: tok.line });
|
|
300
|
+
}
|
|
301
|
+
return n;
|
|
302
|
+
};
|
|
303
|
+
const nodeOfToken = (tok) => {
|
|
304
|
+
const it = itemOfToken(tok);
|
|
305
|
+
if (it) return nodeOf(it);
|
|
306
|
+
if (!tok || tok.t !== 'word') return null;
|
|
307
|
+
if (indexNames.has(tok.u)) return indexNode(tok.u);
|
|
308
|
+
return SYSTEM_FIELDS.has(tok.u) ? systemNode(tok) : null;
|
|
309
|
+
};
|
|
310
|
+
// The program's statement order, where its structure could be read. Without it a check is
|
|
311
|
+
// credited as it always was, wherever its field is used, and never clears anything.
|
|
312
|
+
let ctl = null;
|
|
313
|
+
try {
|
|
314
|
+
ctl = buildControl(prog, (tok) => itemOfToken(tok) || (tok && tok.t === 'word' && indexNames.has(tok.u) ? { index: tok.u } : null));
|
|
315
|
+
} catch { stats.programsUnordered++; }
|
|
316
|
+
const pointOf = (x) => (ctl ? ctl.nodeOf.get(x) : undefined);
|
|
317
|
+
// Where a name inside a condition is read, which can know more than the statement does.
|
|
318
|
+
const placeOf = (tok, st) => (ctl && ctl.posOf.get(tok)) ?? pointOf(st);
|
|
319
|
+
|
|
320
|
+
for (const it of prog.items) if (SYSTEM_FIELDS.has(it.name)) systemSource(nodeOf(it), `${it.name}, which the system sets`, { file: it.file || file, line: it.line });
|
|
321
|
+
// What a program in a CICS region DISPLAYs goes to its CESE and CESO queues, a log. Terminal or web
|
|
322
|
+
// input written there unchecked can carry the control characters that start a line the program
|
|
323
|
+
// did not write (CWE-117). A batch program's DISPLAY is its job's own log, which no terminal
|
|
324
|
+
// reaches, so it is no sink.
|
|
325
|
+
const inRegion = prog.execs.some((e) => e.kind === 'CICS');
|
|
326
|
+
for (const st of inRegion ? prog.statements : []) {
|
|
327
|
+
if (st.verb !== 'DISPLAY') continue;
|
|
328
|
+
const upon = st.sources.findIndex((t) => t.t === 'word' && t.u === 'UPON');
|
|
329
|
+
const device = upon >= 0 && st.sources[upon + 1] ? ` UPON ${st.sources[upon + 1].u}` : '';
|
|
330
|
+
for (const t of upon >= 0 ? st.sources.slice(0, upon) : st.sources) {
|
|
331
|
+
const n = nodeOfToken(t);
|
|
332
|
+
if (n) n.sinks.push({ kind: 'log', onlyFrom: REFLECTED, file: st.file, line: st.line, point: pointOf(st), detail: `DISPLAY ${n.name}${device}` });
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
for (const it of prog.items) {
|
|
336
|
+
const n = nodeOf(it);
|
|
337
|
+
// A tainted child makes its group partly tainted; a tainted group taints every child. The
|
|
338
|
+
// walk must not go up and back down: that path taints a sibling no statement ever wrote.
|
|
339
|
+
if (it.parent) { const p = nodeOf(it.parent); edge(n, p, 'group', 'up'); edge(p, n, 'group', 'down'); }
|
|
340
|
+
// The sibling this item redefines, resolved by the parser; the same bytes read two ways.
|
|
341
|
+
if (it.redefinesItem) { const t = nodeOf(it.redefinesItem); edge(n, t, 'redefines'); edge(t, n, 'redefines'); }
|
|
342
|
+
// A RENAMES alias covers a run of items. It is not their parent, so the partial rule would
|
|
343
|
+
// block a value that reached one of them from reaching the alias that spans it.
|
|
344
|
+
for (const covered of it.renamesSpan || []) {
|
|
345
|
+
const c = nodeOf(covered);
|
|
346
|
+
edge(c, n, 'renames', 'up');
|
|
347
|
+
edge(n, c, 'renames', 'down');
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
for (const st of prog.statements) {
|
|
351
|
+
const targets = st.targets.map(nodeOfToken).filter(Boolean);
|
|
352
|
+
if (!targets.length) continue;
|
|
353
|
+
const sources = st.sources.map(nodeOfToken).filter(Boolean);
|
|
354
|
+
if (!sources.length) continue;
|
|
355
|
+
// One reason object per statement, shared by every edge it makes and formatted only if a
|
|
356
|
+
// finding's path passes through it.
|
|
357
|
+
const why = { verb: st.verb, file: rel(st.file), line: st.line, ...(computes(st) ? { computes: true } : {}), ...(ctl ? { point: pointOf(st) } : {}) };
|
|
358
|
+
for (const tgt of targets) for (const src of sources) edge(src, tgt, why);
|
|
359
|
+
}
|
|
360
|
+
// Where the order is known, a check belongs to every node holding the bytes it tests: the field,
|
|
361
|
+
// what lies inside it, and what redefines those bytes. Whether it counts is decided per route,
|
|
362
|
+
// at the statement the value leaves the node by (lib/control.mjs).
|
|
363
|
+
if (ctl) {
|
|
364
|
+
const byRecord = new Map();
|
|
365
|
+
const recordOf = (it) => { let top = it; while (top.parent) top = top.parent; return top; };
|
|
366
|
+
for (const it of prog.items) { const top = recordOf(it); if (!byRecord.has(top)) byRecord.set(top, []); byRecord.get(top).push(it); }
|
|
367
|
+
const extent = (it) => (it.offset != null && it.size != null ? [it.offset, it.offset + (it.contributes || it.size)] : null);
|
|
368
|
+
for (const c of ctl.checks) {
|
|
369
|
+
const entry = { x: c.x, t: c.t, f: c.f, tCons: c.tCons, fCons: c.fCons, file: c.file, line: c.line, item: c.field.index || c.field.name, ...(c.bounds ? { bounds: c.bounds } : {}) };
|
|
370
|
+
if (c.field.index) { (indexNode(c.field.index).checks ||= []).push(entry); continue; }
|
|
371
|
+
const span = extent(c.field);
|
|
372
|
+
for (const it of byRecord.get(recordOf(c.field)) || [c.field]) {
|
|
373
|
+
if (it.level === 88) continue;
|
|
374
|
+
const e = extent(it);
|
|
375
|
+
if (it === c.field || (span && e && e[0] >= span[0] && e[1] <= span[1])) (nodeOf(it).checks ||= []).push(entry);
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
}
|
|
379
|
+
// Without the order, a condition that restricts what an item may hold is credited wherever the
|
|
380
|
+
// item is used: the finding says where the check is and is lowered one step, never cleared.
|
|
381
|
+
// Restricting means a class test, an ordering comparison or a condition-name. Equality is not:
|
|
382
|
+
// on the corpus IF X = SPACES tests that a field was filled, and IF X = "@STDIN" picks a branch.
|
|
383
|
+
for (const st of ctl ? [] : prog.statements) {
|
|
384
|
+
if (st.verb !== 'IF' && st.verb !== 'EVALUATE' && st.verb !== 'WHEN') continue;
|
|
385
|
+
const operands = st.sources.map((tok) => ({ tok, item: itemOfToken(tok) }));
|
|
386
|
+
const items = operands.filter((o) => o.item).map((o) => o.item);
|
|
387
|
+
const words = operands.filter((o) => !o.item).map((o) => o.tok.u);
|
|
388
|
+
const restricts = words.some((w) => CLASS_TEST.has(w) || w === 'GREATER' || w === 'LESS')
|
|
389
|
+
|| (st.ops || []).some((op) => op === '>' || op === '<' || op === '>=' || op === '<=')
|
|
390
|
+
|| items.some((it) => it.level === 88);
|
|
391
|
+
if (!restricts) continue;
|
|
392
|
+
const checked = items.map((it) => nodeOf(it.level === 88 && it.parent ? it.parent : it));
|
|
393
|
+
for (const o of operands) if (!o.item && indexNames.has(o.tok.u)) checked.push(indexNode(o.tok.u));
|
|
394
|
+
for (const n of checked) if (!n.guard) n.guard = { file: st.file, line: st.line };
|
|
395
|
+
}
|
|
396
|
+
// Where a statement reads or writes, taken from what it indexes with: a subscript of a table,
|
|
397
|
+
// the start or length of a reference modification, and the count an OCCURS DEPENDING ON table
|
|
398
|
+
// is sized by. Every use is a sink, judged where it reads, and a route reports one use per index
|
|
399
|
+
// and the name it indexes, the least checked, so a loop that indexes one table forty times is one
|
|
400
|
+
// thing to look at. SSRANGE, set on a CBL or PROCESS card, turns a bad index into an abend; the
|
|
401
|
+
// finding carries it. The last option wins across both levels, and SSRANGE(MSG) reports and
|
|
402
|
+
// carries on, so it does not count.
|
|
403
|
+
const ssrange = ssrangeAbends([...(site.compilerOptions || []), ...options]) ?? false;
|
|
404
|
+
const tableOf = (it) => { for (let a = it; a; a = a.parent) if ((a.occurs || 1) > 1) return a; return null; };
|
|
405
|
+
// How many entries or bytes an index may name, where one number says it: a table of one
|
|
406
|
+
// dimension and a fixed size, or an item of a fixed length.
|
|
407
|
+
const variable = (it) => !!it.dependingOn || (it.children || []).some(variable);
|
|
408
|
+
const dimensions = (it) => { let n = 0; for (let a = it; a; a = a.parent) if ((a.occurs || 1) > 1) n++; return n; };
|
|
409
|
+
const limitOf = (host, table) => {
|
|
410
|
+
if (table) return dimensions(host) === 1 && !variable(table) && !table.dependingOn ? table.occurs : null;
|
|
411
|
+
if (host.size == null || variable(host)) return null;
|
|
412
|
+
return (host.children || []).some((c) => c.level !== 88) ? host.size / Math.max(1, host.occurs || 1) : host.size;
|
|
413
|
+
};
|
|
414
|
+
const indexed = new Set();
|
|
415
|
+
const subscripting = new Map();
|
|
416
|
+
for (const st of prog.statements) {
|
|
417
|
+
for (const x of st.indexes || []) {
|
|
418
|
+
const host = itemOfToken(x.host);
|
|
419
|
+
const idx = itemOfToken(x.tok);
|
|
420
|
+
if (!host || (idx && idx.level === 88)) continue;
|
|
421
|
+
const at = idx ? nodeOf(idx) : nodeOfToken(x.tok);
|
|
422
|
+
if (!at) continue;
|
|
423
|
+
const kind = x.kind === 'subscript' ? 'subscript' : 'reference-modification';
|
|
424
|
+
const table = kind === 'subscript' ? tableOf(host) : null;
|
|
425
|
+
if (kind === 'subscript' && !table) continue;
|
|
426
|
+
const key = `${kind}|${host.name}|${at.name}`;
|
|
427
|
+
if (table && !subscripting.has(at.name)) subscripting.set(at.name, { host, table });
|
|
428
|
+
const point = placeOf(x.tok, st);
|
|
429
|
+
const use = `${key}|${point ?? `s${st.at}`}`;
|
|
430
|
+
if (indexed.has(use)) continue;
|
|
431
|
+
indexed.add(use);
|
|
432
|
+
at.sinks.push({
|
|
433
|
+
kind, onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point, group: `${pk}|${key}`, limit: limitOf(host, table), ...(ssrange ? { ssrange } : {}),
|
|
434
|
+
detail: kind === 'subscript'
|
|
435
|
+
? `${at.name} subscripts ${host.name}, a table of ${table.occurs}`
|
|
436
|
+
: `${at.name} sets the ${x.kind === 'refmod-offset' ? 'start' : 'length'} of a reference to ${host.name}, which is ${host.size} bytes`,
|
|
437
|
+
});
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
// A counter that subscripts a table walks as far as its loop's condition lets it. PERFORM
|
|
441
|
+
// VARYING I ... UNTIL I > N with N from outside puts the table's bound in the caller's hands, and
|
|
442
|
+
// no subscript sink sees it: I itself is only ever a literal plus one. The subscript may be
|
|
443
|
+
// anywhere in the program; which statements the loop body holds is not known here.
|
|
444
|
+
for (const st of prog.statements) {
|
|
445
|
+
for (const loop of st.loops || []) {
|
|
446
|
+
const counter = loop.counter && (itemOfToken(loop.counter)?.name || (indexNames.has(loop.counter.u) ? loop.counter.u : null));
|
|
447
|
+
const walks = counter && subscripting.get(counter);
|
|
448
|
+
if (!walks) continue;
|
|
449
|
+
for (const tok of boundsOf(counter, loop.until)) {
|
|
450
|
+
const it = itemOfToken(tok);
|
|
451
|
+
if (!it || it.level === 88 || it.name === counter) continue;
|
|
452
|
+
const key = `loop|${counter}|${it.name}`;
|
|
453
|
+
const use = `${key}|${pointOf(st) ?? `s${st.at}`}`;
|
|
454
|
+
if (indexed.has(use)) continue;
|
|
455
|
+
indexed.add(use);
|
|
456
|
+
nodeOf(it).sinks.push({
|
|
457
|
+
kind: 'loop-bound', onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point: pointOf(st), group: `${pk}|${key}`, limit: limitOf(walks.host, walks.table), ...(ssrange ? { ssrange } : {}),
|
|
458
|
+
detail: `${it.name} decides where PERFORM VARYING ${counter} stops, and ${counter} subscripts ${walks.host.name}, a table of ${walks.table.occurs}`,
|
|
459
|
+
});
|
|
460
|
+
}
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
// Arithmetic reads a zoned or packed operand as decimal digits, and a byte that is not one stops
|
|
464
|
+
// the program: S0C7 in batch, ASRA under CICS. Binary and floating-point fields cannot hold an
|
|
465
|
+
// invalid value, and an alphanumeric one converted by NUMVAL is not an operand here, so neither
|
|
466
|
+
// is a sink. Every use is a sink, and a route reports the least checked, as for an index.
|
|
467
|
+
const numproc = lastOption('NUMPROC', options) ?? lastOption('NUMPROC', site.compilerOptions);
|
|
468
|
+
for (const st of prog.statements) {
|
|
469
|
+
if (!ARITHMETIC.has(st.verb)) continue;
|
|
470
|
+
for (const tok of st.sources) {
|
|
471
|
+
const it = itemOfToken(tok);
|
|
472
|
+
if (!it || !holdsDecimal(it)) continue;
|
|
473
|
+
const n = nodeOf(it);
|
|
474
|
+
const use = `arithmetic|${n.id}|${pointOf(st) ?? `s${st.at}`}`;
|
|
475
|
+
if (indexed.has(use)) continue;
|
|
476
|
+
indexed.add(use);
|
|
477
|
+
n.sinks.push({
|
|
478
|
+
kind: 'arithmetic', onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point: pointOf(st), group: `${pk}|arithmetic|${n.id}`,
|
|
479
|
+
detail: `${it.name} (PIC ${it.picture}${it.effectiveUsage && it.effectiveUsage !== 'DISPLAY' ? ` ${it.effectiveUsage}` : ''}) is an operand of ${st.verb}${numproc === 'PFD' ? ', compiled NUMPROC(PFD), which repairs no sign' : ''}`,
|
|
480
|
+
});
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
for (const it of prog.items) {
|
|
484
|
+
if (!it.dependingOn) continue;
|
|
485
|
+
const count = byName.get(it.dependingOn);
|
|
486
|
+
if (!count) continue;
|
|
487
|
+
nodeOf(count).sinks.push({
|
|
488
|
+
kind: 'occurs-depending-count', onlyFrom: FROM_OUTSIDE, file: it.file, line: it.line, ...(ssrange ? { ssrange } : {}),
|
|
489
|
+
detail: `${count.name} sets how many ${it.name} entries there are, up to ${it.occurs}`,
|
|
490
|
+
});
|
|
491
|
+
}
|
|
492
|
+
for (const st of prog.statements) {
|
|
493
|
+
if (st.verb !== 'READ' && st.verb !== 'RETURN') continue;
|
|
494
|
+
for (const t of st.targets) {
|
|
495
|
+
const n = nodeOfToken(t);
|
|
496
|
+
if (n) n.sources.push({ kind: 'file-record', file: st.file, line: st.line, detail: `${st.verb} ... INTO` });
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
for (const fd of prog.fds || []) for (const rec of fd.records || []) {
|
|
500
|
+
const readsFile = prog.statements.some(st => (st.verb === 'READ') && st.sources.concat(st.targets).some(t => t.u === fd.name));
|
|
501
|
+
if (readsFile) nodeOf(rec).sources.push({ kind: 'file-record', file: fd.file, line: fd.line, detail: `record of file ${fd.name}` });
|
|
502
|
+
}
|
|
503
|
+
for (const a of prog.accepts) {
|
|
504
|
+
const item = (a.targetTok ? itemOfToken(a.targetTok) : null) || byName.get(a.target);
|
|
505
|
+
if (!item) continue;
|
|
506
|
+
const kind = /^(COMMAND-LINE|ARGUMENT-VALUE|ENVIRONMENT|ENVIRONMENT-VALUE)$/.test(a.from || '') ? 'argv-or-env' : null;
|
|
507
|
+
if (kind) nodeOf(item).sources.push({ kind, file: a.file, line: a.line, detail: `ACCEPT ... FROM ${a.from}` });
|
|
508
|
+
}
|
|
509
|
+
// A SELECT assigns either to a data item, which makes the path dynamic, or to an external
|
|
510
|
+
// name. On z/OS that external name is the DD name a JCL step fills, and it is how a job's
|
|
511
|
+
// in-stream data reaches a program's record without any COBOL statement naming the job.
|
|
512
|
+
const ddFiles = [];
|
|
513
|
+
for (const f of prog.files) {
|
|
514
|
+
if (!f.assign) continue;
|
|
515
|
+
if (f.assign.t === 'word') {
|
|
516
|
+
const item = byName.get(f.assign.v);
|
|
517
|
+
if (item) {
|
|
518
|
+
nodeOf(item).sinks.push({ kind: 'dynamic-file-path', file: f.file, line: f.line, detail: `SELECT ${f.name} ASSIGN TO ${f.assign.v}` });
|
|
519
|
+
continue; // a variable path is not a DD name
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
// "DD:SYSIN", 'S-SYSIN' and SYSIN all name the same DD once the dialect prefix is off.
|
|
523
|
+
const external = String(f.assign.v || '').replace(/^["']|["']$/g, '').replace(/^(DD:|S-|-)/i, '').toUpperCase();
|
|
524
|
+
if (!external) continue;
|
|
525
|
+
const fd = (prog.fds || []).find((x) => x.name === f.name);
|
|
526
|
+
const recs = (fd?.records || []).map(nodeOf);
|
|
527
|
+
if (recs.length) ddFiles.push({ dd: external, nodes: recs, file: f.file, line: f.line, select: f.name });
|
|
528
|
+
}
|
|
529
|
+
const topOf = (it) => { let t = it; while (t && t.parent) t = t.parent; return t; };
|
|
530
|
+
const ownParams = new Set((prog.paramTokens || []).map((pt) => topOf(itemOfToken(pt.tok))).filter(Boolean));
|
|
531
|
+
// A LINKAGE record that is not one of this program's own parameters is storage it addressed
|
|
532
|
+
// itself, through a pointer: its declared size says nothing about what the pointer points at.
|
|
533
|
+
const argExtent = (it) => (it && topOf(it).section === 'LINKAGE' && !ownParams.has(topOf(it)) ? null : extentOf(it));
|
|
534
|
+
// How far into each of its own parameters this program writes: a statement's targets, and
|
|
535
|
+
// anything it passes on by reference, which the program it calls may write.
|
|
536
|
+
const writtenEnd = new Map();
|
|
537
|
+
const noteWrite = (it) => {
|
|
538
|
+
const top = topOf(it);
|
|
539
|
+
if (!it || !ownParams.has(top) || it.offset == null || it.size == null) return;
|
|
540
|
+
const end = it.offset - (top.offset ?? 0) + it.size * Math.max(1, it.occurs || 1);
|
|
541
|
+
if (end > (writtenEnd.get(top) || 0)) writtenEnd.set(top, end);
|
|
542
|
+
};
|
|
543
|
+
for (const st of prog.statements) for (const t of st.targets || []) noteWrite(itemOfToken(t));
|
|
544
|
+
for (const c of prog.calls) for (const a of c.using) if (a.tok && a.mode === 'REFERENCE') noteWrite(itemOfToken(a.tok));
|
|
545
|
+
const callSites = [];
|
|
546
|
+
for (const c of prog.calls) {
|
|
547
|
+
const point = pointOf(prog.statements[c.stmtIndex]);
|
|
548
|
+
if (c.kind === 'I') {
|
|
549
|
+
const n = nodeOfToken(c.targetTok);
|
|
550
|
+
if (n) n.sinks.push({ kind: 'dynamic-program-load', file: c.file, line: c.line, point, detail: `CALL ${c.name}` });
|
|
551
|
+
continue;
|
|
552
|
+
}
|
|
553
|
+
callSites.push({ name: String(c.name).toUpperCase(), display: c.name, file: c.file, line: c.line, point, args: c.using.map(a => ({ node: a.tok ? nodeOfToken(a.tok) : null, mode: a.mode, extent: a.tok ? argExtent(itemOfToken(a.tok)) : null })) });
|
|
554
|
+
if (OS_COMMAND_ROUTINE.test(c.name)) {
|
|
555
|
+
for (const a of c.using) {
|
|
556
|
+
const n = a.tok ? nodeOfToken(a.tok) : null;
|
|
557
|
+
if (n) n.sinks.push({ kind: 'os-command', file: c.file, line: c.line, point, detail: `CALL '${c.name}' USING ${a.word}` });
|
|
558
|
+
}
|
|
559
|
+
} else if (SOCKET_ROUTINE.test(c.name)) {
|
|
560
|
+
// IBM's own examples hold the function code in a field: 01 SOC-FUNCTION PIC X(16) VALUE 'SEND'.
|
|
561
|
+
const first = c.using[0];
|
|
562
|
+
const fnItem = first && first.tok ? itemOfToken(first.tok) : null;
|
|
563
|
+
const fnValue = fnItem && (fnItem.values || []).find(v => v.t === 'lit');
|
|
564
|
+
const fn = (first && first.lit) || (fnValue && fnValue.v);
|
|
565
|
+
if (fn && SOCKET_SEND.test(String(fn).trim())) {
|
|
566
|
+
for (const a of c.using.slice(1)) {
|
|
567
|
+
const n = a.tok ? nodeOfToken(a.tok) : null;
|
|
568
|
+
if (n) n.sinks.push({ kind: 'socket-send', onlyFrom: DATA_AT_REST, file: c.file, line: c.line, point, detail: `CALL '${c.name}' ${String(fn).trim()} USING ${a.word}` });
|
|
569
|
+
}
|
|
570
|
+
}
|
|
571
|
+
} else if (Object.hasOwn(MQ_BUFFER_ARG, String(c.name).toUpperCase())) {
|
|
572
|
+
const a = c.using[MQ_BUFFER_ARG[String(c.name).toUpperCase()]];
|
|
573
|
+
const n = a && a.tok ? nodeOfToken(a.tok) : null;
|
|
574
|
+
if (n) n.sinks.push({ kind: 'message-queue', onlyFrom: DATA_AT_REST, file: c.file, line: c.line, point, detail: `CALL '${c.name}' with buffer ${a.word}` });
|
|
575
|
+
}
|
|
576
|
+
}
|
|
577
|
+
const transfers = [];
|
|
578
|
+
const channelOf = new Map();
|
|
579
|
+
const invokes = [];
|
|
580
|
+
const transfersTo = new Set();
|
|
581
|
+
const startsTransactions = new Set();
|
|
582
|
+
const tdWrites = [];
|
|
583
|
+
// A container is storage with a name rather than a field, so it gets a node of its own and the
|
|
584
|
+
// PUT and GET statements are edges into and out of it. Without one, data put in a container and
|
|
585
|
+
// sent by WEB CONVERSE CONTAINER(...) left no trace at all.
|
|
586
|
+
const containers = new Map();
|
|
587
|
+
const containerNode = (name) => {
|
|
588
|
+
let n = containers.get(name);
|
|
589
|
+
if (!n) {
|
|
590
|
+
n = { id: nodes.length, pk, program: prog.id, file, name: `CONTAINER(${name})`, edgesOut: [], sources: [], sinks: [] };
|
|
591
|
+
nodes.push(n);
|
|
592
|
+
containers.set(name, n);
|
|
593
|
+
stats.nodes++;
|
|
594
|
+
}
|
|
595
|
+
return n;
|
|
596
|
+
};
|
|
597
|
+
const optionName = (o, key) => {
|
|
598
|
+
const t = (o.get(key) || [])[0];
|
|
599
|
+
return t ? String(t.t === 'lit' ? t.v : t.u).trim().toUpperCase() : null;
|
|
600
|
+
};
|
|
601
|
+
// A map named through a data name is named by its VALUE, where nothing writes the item: CardDemo
|
|
602
|
+
// writes MAP(LIT-THISMAP) for a map it never renames.
|
|
603
|
+
let written = null;
|
|
604
|
+
const constOption = (o, key) => {
|
|
605
|
+
const t = (o.get(key) || [])[0];
|
|
606
|
+
if (!t) return null;
|
|
607
|
+
if (t.t === 'lit') return String(t.v).trim().toUpperCase();
|
|
608
|
+
const it = itemOfToken(t);
|
|
609
|
+
if (!it) return null;
|
|
610
|
+
written ||= new Set(prog.statements.flatMap((st) => st.targets.map(itemOfToken).filter(Boolean)));
|
|
611
|
+
const v = !written.has(it) && (it.values || [])[0];
|
|
612
|
+
return v && v.t === 'lit' ? String(v.v).trim().toUpperCase() : null;
|
|
613
|
+
};
|
|
614
|
+
// The fields of a received map that the map protects, hides or restricts to digits. The terminal
|
|
615
|
+
// enforces those attributes and CICS does not, so a modified 3270 client writes a protected field
|
|
616
|
+
// and reads a dark one. Each such field's NAMEI carries what the map says; a protected one is also
|
|
617
|
+
// a source of its own, for the one use a typed field makes ordinary and a protected one does not:
|
|
618
|
+
// choosing which record to read.
|
|
619
|
+
// The programs a data name can name at a transfer: the constant it holds there on every route,
|
|
620
|
+
// from the program's own order, or else every literal the program puts in it. CardDemo names
|
|
621
|
+
// each XCTL this way - MOVE 'COTRN01C' TO CDEMO-TO-PROGRAM - so without it no COMMAREA crossed
|
|
622
|
+
// one of its transfers.
|
|
623
|
+
let movedIn = null;
|
|
624
|
+
const literalsInto = (it) => {
|
|
625
|
+
if (!movedIn) {
|
|
626
|
+
movedIn = new Map();
|
|
627
|
+
const add = (item, v) => {
|
|
628
|
+
const name = String(v).trim().toUpperCase();
|
|
629
|
+
if (!/^[A-Z$#@][A-Z0-9$#@-]{0,7}$/.test(name)) return;
|
|
630
|
+
if (!movedIn.has(item)) movedIn.set(item, new Set());
|
|
631
|
+
movedIn.get(item).add(name);
|
|
632
|
+
};
|
|
633
|
+
for (const i of prog.items) for (const v of i.values || []) if (v.t === 'lit') add(i, v.v);
|
|
634
|
+
for (const st of prog.statements) {
|
|
635
|
+
if (st.verb !== 'MOVE' || st.sources.length) continue;
|
|
636
|
+
const lit = (st.literals || []).find((l) => l.t === 'lit');
|
|
637
|
+
if (lit) for (const t of st.targets) { const ti = itemOfToken(t); if (ti) add(ti, lit.v); }
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
return [...(movedIn.get(it) || [])];
|
|
641
|
+
};
|
|
642
|
+
const programsNamed = (tok, e) => {
|
|
643
|
+
const it = itemOfToken(tok);
|
|
644
|
+
if (!it) return [];
|
|
645
|
+
const bits = ctl && ctl.facts.get(ctl.nodeOf.get(e));
|
|
646
|
+
if (bits) {
|
|
647
|
+
const must = ctl.checks.filter((c) => c.assigned && c.field === it && hasFact(bits, c.t)).map((c) => [...c.tCons.set][0]);
|
|
648
|
+
if (must.length) return [...new Set(must.map((v) => String(v).trim().toUpperCase()))];
|
|
649
|
+
}
|
|
650
|
+
return literalsInto(it);
|
|
651
|
+
};
|
|
652
|
+
const receiveMap = (e, o, into, at) => {
|
|
653
|
+
stats.mapsReceived++;
|
|
654
|
+
const mapName = constOption(o, 'MAP');
|
|
655
|
+
const setName = constOption(o, 'MAPSET');
|
|
656
|
+
const hit = mapName && ((setName && bmsMaps.get(`${setName}/${mapName}`)) || bmsMaps.get(mapName));
|
|
657
|
+
if (!hit) { stats.mapsNoSource++; return; }
|
|
658
|
+
const { mapset, map } = hit[0];
|
|
659
|
+
const record = into || byName.get(map.symbolic.input);
|
|
660
|
+
if (!record) { stats.mapsNoSymbolic++; return; }
|
|
661
|
+
const within = new Map();
|
|
662
|
+
const walk = (it) => { if (!within.has(it.name)) within.set(it.name, it); for (const c of it.children || []) walk(c); };
|
|
663
|
+
walk(record);
|
|
664
|
+
for (const field of map.fields) {
|
|
665
|
+
if (!field.name) continue;
|
|
666
|
+
const marks = ['PROT', 'ASKIP', 'DRK', 'NUM'].filter((a) => field.effective.has(a));
|
|
667
|
+
if (!marks.length) continue;
|
|
668
|
+
const iname = symbolicNames(map, field).find((n) => n.suffix === 'I');
|
|
669
|
+
const item = iname && within.get(iname.name);
|
|
670
|
+
if (!item) continue;
|
|
671
|
+
const n = nodeOf(item);
|
|
672
|
+
n.screen = { item: item.name, field: field.name, map: map.name, mapset: mapset.name, marks, declared: !!field.attrb };
|
|
673
|
+
if (!marks.includes('PROT') && !marks.includes('ASKIP')) continue;
|
|
674
|
+
stats.protectedFields++;
|
|
675
|
+
const back = field.effective.has('FSET') ? ' with FSET' : '';
|
|
676
|
+
n.sources.push({ kind: 'cics-protected-field', onlyTo: ['record-key', 'record-update'], ...at,
|
|
677
|
+
detail: `EXEC CICS RECEIVE MAP(${map.name}) returns ${item.name}, field ${field.name} of map ${map.name} in mapset ${mapset.name}, which the map marks ${marks.join(' and ')}${field.attrb ? '' : ' by default'}${back}` });
|
|
678
|
+
}
|
|
679
|
+
};
|
|
680
|
+
// The files whose held record this program rewrites or deletes, by the operand naming them. The
|
|
681
|
+
// REWRITE or DELETE carries no key: the READ UPDATE before it chose the record.
|
|
682
|
+
const changes = new Map();
|
|
683
|
+
for (const e of prog.execs) {
|
|
684
|
+
if (e.kind !== 'CICS') continue;
|
|
685
|
+
const { opts: o, words } = execOptions(e);
|
|
686
|
+
if (words[0] !== 'REWRITE' && !(words[0] === 'DELETE' && !o.has('RIDFLD'))) continue;
|
|
687
|
+
const file = optionName(o, 'DATASET') || optionName(o, 'FILE');
|
|
688
|
+
if (file && !changes.has(file)) changes.set(file, words[0] === 'REWRITE' ? 'rewrites' : 'deletes');
|
|
689
|
+
}
|
|
690
|
+
for (const e of prog.execs) {
|
|
691
|
+
const { opts: o, words } = execOptions(e);
|
|
692
|
+
const at = { file: e.file, line: e.line, point: pointOf(e) };
|
|
693
|
+
if (e.kind === 'CICS') {
|
|
694
|
+
const verb = words[0];
|
|
695
|
+
const intoTok = (o.get('INTO') || o.get('SET') || [])[0];
|
|
696
|
+
for (const opt of ['RESP', 'RESP2']) {
|
|
697
|
+
const r = nodeOfToken((o.get(opt) || [])[0]);
|
|
698
|
+
if (r) systemSource(r, `the ${opt} of EXEC CICS ${verb}`, at);
|
|
699
|
+
}
|
|
700
|
+
if (verb === 'WRITE' && ['OPERATOR', 'JOURNALNAME', 'JOURNALNUM'].includes(words[1])) {
|
|
701
|
+
const opt = words[1] === 'OPERATOR' ? 'TEXT' : 'FROM';
|
|
702
|
+
const n = nodeOfToken((o.get(opt) || [])[0]);
|
|
703
|
+
if (n) n.sinks.push({ kind: 'log', onlyFrom: REFLECTED, ...at, detail: `EXEC CICS WRITE ${words[1]} ${opt}(${n.name}), ${words[1] === 'OPERATOR' ? 'to the operator console' : 'to a journal'}` });
|
|
704
|
+
}
|
|
705
|
+
if (verb === 'RECEIVE') {
|
|
706
|
+
const isWeb = words.includes('WEB');
|
|
707
|
+
const n = nodeOfToken(intoTok);
|
|
708
|
+
if (n) n.sources.push({ kind: isWeb ? 'cics-web' : 'cics-terminal', ...at, detail: `EXEC CICS ${isWeb ? 'WEB ' : ''}RECEIVE` });
|
|
709
|
+
// RECEIVE MAP with no INTO fills the symbolic map, whose input record is <map>I.
|
|
710
|
+
const map = optionName(o, 'MAP');
|
|
711
|
+
if (!n && map) {
|
|
712
|
+
const sym = byName.get(`${map}I`);
|
|
713
|
+
if (sym) nodeOf(sym).sources.push({ kind: 'cics-terminal', ...at, detail: `EXEC CICS RECEIVE MAP(${map}) into the symbolic map` });
|
|
714
|
+
}
|
|
715
|
+
if (!isWeb && o.has('MAP')) receiveMap(e, o, itemOfToken(intoTok), at);
|
|
716
|
+
}
|
|
717
|
+
// RIDFLD decides which record a READ or DELETE reaches, or where a browse starts. A key the
|
|
718
|
+
// user types is how a lookup works; a key the program put in a protected field and read back
|
|
719
|
+
// is its own state, held where a modified client can change it.
|
|
720
|
+
if (['READ', 'DELETE', 'STARTBR', 'RESETBR'].includes(verb) && o.has('RIDFLD')) {
|
|
721
|
+
const k = nodeOfToken((o.get('RIDFLD') || [])[0]);
|
|
722
|
+
const file = optionName(o, 'DATASET') || optionName(o, 'FILE');
|
|
723
|
+
const then = verb === 'READ' && words.includes('UPDATE') ? changes.get(file) : null;
|
|
724
|
+
const detail = verb === 'DELETE' ? `EXEC CICS DELETE RIDFLD(${k && k.name}), which decides which record it deletes`
|
|
725
|
+
: then ? `EXEC CICS READ UPDATE RIDFLD(${k && k.name}), which decides which record the program then ${then}`
|
|
726
|
+
: `EXEC CICS ${verb} RIDFLD(${k && k.name}), which decides which record it reaches`;
|
|
727
|
+
if (k) k.sinks.push({ kind: verb === 'DELETE' || then ? 'record-update' : 'record-key', onlyFrom: ['cics-protected-field'], noCredit: true, ...at, detail });
|
|
728
|
+
}
|
|
729
|
+
if (verb === 'PUT' && o.has('CONTAINER')) {
|
|
730
|
+
const name = optionName(o, 'CONTAINER');
|
|
731
|
+
const from = nodeOfToken((o.get('FROM') || [])[0]);
|
|
732
|
+
if (name && from) edge(from, containerNode(name), `EXEC CICS PUT CONTAINER(${name})`, undefined, at.point);
|
|
733
|
+
if (name) channelOf.set(name, optionName(o, 'CHANNEL'));
|
|
734
|
+
}
|
|
735
|
+
// A service call sends every container on its channel, and its URI decides where they go.
|
|
736
|
+
if (verb === 'INVOKE' && (words[1] === 'SERVICE' || words[1] === 'WEBSERVICE')) {
|
|
737
|
+
invokes.push({ at, what: words[1], channel: optionName(o, 'CHANNEL') });
|
|
738
|
+
for (const opt of ['URI', 'URIMAP']) {
|
|
739
|
+
const n = nodeOfToken((o.get(opt) || [])[0]);
|
|
740
|
+
if (n) n.sinks.push({ kind: 'outbound-host', onlyFrom: IN_CICS, ...at, detail: `EXEC CICS INVOKE ${words[1]} ${opt}(...), which decides where the request goes` });
|
|
741
|
+
}
|
|
742
|
+
}
|
|
743
|
+
if (verb === 'GET' && o.has('CONTAINER')) {
|
|
744
|
+
const name = optionName(o, 'CONTAINER');
|
|
745
|
+
const into = nodeOfToken((o.get('INTO') || o.get('SET') || [])[0]);
|
|
746
|
+
if (name && into) edge(containerNode(name), into, `EXEC CICS GET CONTAINER(${name})`, undefined, at.point);
|
|
747
|
+
}
|
|
748
|
+
// CONVERSE is always a client call; SEND is one only on a session this program opened. A
|
|
749
|
+
// plain WEB SEND answers the request that started the program, which is its job.
|
|
750
|
+
if (verb === 'WEB' && (words[1] === 'CONVERSE' || (words[1] === 'SEND' && o.has('SESSTOKEN')))) {
|
|
751
|
+
const n = nodeOfToken((o.get('FROM') || [])[0]);
|
|
752
|
+
if (n) n.sinks.push({ kind: 'outbound-http', onlyFrom: DATA_AT_REST, ...at, detail: `EXEC CICS WEB ${words[1]} to a remote server` });
|
|
753
|
+
// The request may be a container rather than a field.
|
|
754
|
+
const sent = o.has('CONTAINER') ? optionName(o, 'CONTAINER') : null;
|
|
755
|
+
if (sent) containerNode(sent).sinks.push({ kind: 'outbound-http', onlyFrom: DATA_AT_REST, ...at, detail: `EXEC CICS WEB ${words[1]} sending CONTAINER(${sent})` });
|
|
756
|
+
}
|
|
757
|
+
// The response this program returns to its own caller, and the document built for it. CICS
|
|
758
|
+
// escapes nothing on the way out - DOCUMENT SET's UNESCAPED option is about URL decoding,
|
|
759
|
+
// not HTML - so whatever reaches here is what the browser is handed.
|
|
760
|
+
if (verb === 'WEB' && words[1] === 'SEND' && !o.has('SESSTOKEN')) {
|
|
761
|
+
const n = nodeOfToken((o.get('FROM') || [])[0]);
|
|
762
|
+
// A stored value becomes script only where the reply is markup, and a web program
|
|
763
|
+
// answering with a database row is the ordinary shape of one. Input echoed back is a
|
|
764
|
+
// defect whatever the reply holds.
|
|
765
|
+
const media = (o.get('MEDIATYPE') || [])[0];
|
|
766
|
+
const markup = media && media.t === 'lit' && /html|xml|svg/i.test(String(media.v));
|
|
767
|
+
if (n) n.sinks.push({ kind: 'web-response', onlyFrom: [...(markup ? IN_CICS : REFLECTED), 'system-response'], ...at, detail: `EXEC CICS WEB SEND answering this program's caller${markup ? ` as ${String(media.v).trim()}` : ''}` });
|
|
768
|
+
}
|
|
769
|
+
if (verb === 'DOCUMENT') {
|
|
770
|
+
const what = words[1] === 'SET' ? ['VALUE', 'SYMBOLLIST'] : ['TEXT', 'FROM', 'BINARY'];
|
|
771
|
+
for (const opt of what) {
|
|
772
|
+
const n = nodeOfToken((o.get(opt) || [])[0]);
|
|
773
|
+
if (n) n.sinks.push({ kind: 'web-response', onlyFrom: [...IN_CICS, 'system-response'], ...at, detail: `EXEC CICS DOCUMENT ${words[1]} ${opt}(...), which is sent as the response` });
|
|
774
|
+
}
|
|
775
|
+
}
|
|
776
|
+
if (verb === 'WEB' && words[1] === 'WRITE' && o.has('HTTPHEADER')) {
|
|
777
|
+
for (const opt of ['HTTPHEADER', 'VALUE']) {
|
|
778
|
+
const n = nodeOfToken((o.get(opt) || [])[0]);
|
|
779
|
+
if (n) n.sinks.push({ kind: 'http-header', onlyFrom: [...IN_CICS, 'system-response'], ...at, detail: `EXEC CICS WEB WRITE HTTPHEADER ${opt}(...)` });
|
|
780
|
+
}
|
|
781
|
+
}
|
|
782
|
+
if (verb === 'WEB' && (words[1] === 'OPEN' || words[1] === 'CONVERSE')) {
|
|
783
|
+
for (const opt of ['HOST', 'URIMAP', 'PATH']) {
|
|
784
|
+
const n = nodeOfToken((o.get(opt) || [])[0]);
|
|
785
|
+
if (n) n.sinks.push({ kind: 'outbound-host', onlyFrom: IN_CICS, ...at, detail: `EXEC CICS WEB ${words[1]} ${opt}(...), which decides where the request goes` });
|
|
786
|
+
}
|
|
787
|
+
}
|
|
788
|
+
// The queue a command acts on, as distinct from what it writes there: a name from input
|
|
789
|
+
// reads, overwrites or deletes whatever queue the input asks for.
|
|
790
|
+
if (['WRITEQ', 'READQ', 'DELETEQ'].includes(verb)) {
|
|
791
|
+
const ts = words[1] === 'TS' || !words.includes('TD');
|
|
792
|
+
for (const opt of ['QUEUE', 'QNAME']) {
|
|
793
|
+
const n = nodeOfToken((o.get(opt) || [])[0]);
|
|
794
|
+
if (n) n.sinks.push({ kind: 'queue-name', onlyFrom: IN_CICS, ...at, detail: `EXEC CICS ${verb} ${ts ? 'TS' : 'TD'} ${opt}(...) names the queue` });
|
|
795
|
+
}
|
|
796
|
+
}
|
|
797
|
+
if (verb === 'READ' || verb === 'READNEXT' || verb === 'READPREV') {
|
|
798
|
+
const n = nodeOfToken(intoTok);
|
|
799
|
+
if (n && (o.has('FILE') || o.has('DATASET'))) n.sources.push({ kind: 'file-record', ...at, detail: `EXEC CICS ${verb} FILE ... INTO` });
|
|
800
|
+
}
|
|
801
|
+
// A CONVERSE response is the remote server's data, which is untrusted input like any other.
|
|
802
|
+
if ((verb === 'WEB' && words[1] !== 'SEND') || verb === 'EXTRACT') {
|
|
803
|
+
const n = nodeOfToken(intoTok);
|
|
804
|
+
if (n) n.sources.push({ kind: 'cics-web', ...at, detail: `EXEC CICS ${words.slice(0, 2).join(' ')}` });
|
|
805
|
+
}
|
|
806
|
+
// Where a transient-data write lands is not in the program: the CSD maps the queue to a DD
|
|
807
|
+
// and the region's JCL maps the DD to a destination. Collected here, resolved once every
|
|
808
|
+
// definition has been read.
|
|
809
|
+
if (verb === 'WRITEQ' && words[1] === 'TD') {
|
|
810
|
+
const from = nodeOfToken((o.get('FROM') || [])[0]);
|
|
811
|
+
const qTok = (o.get('QUEUE') || o.get('TDQUEUE') || [])[0];
|
|
812
|
+
const qItem = qTok && qTok.t === 'word' ? itemOfToken(qTok) : null;
|
|
813
|
+
const qValue = qItem && (qItem.values || []).find((v) => v.t === 'lit');
|
|
814
|
+
const queue = qTok && qTok.t === 'lit' ? qTok.v : qValue ? qValue.v : null;
|
|
815
|
+
if (from) tdWrites.push({ node: from, queue: queue == null ? null : String(queue).trim().toUpperCase(), ...at });
|
|
816
|
+
}
|
|
817
|
+
if (['LINK', 'XCTL', 'START'].includes(verb)) {
|
|
818
|
+
const pgm = (o.get('PROGRAM') || o.get('TRANSID') || [])[0];
|
|
819
|
+
if (pgm && pgm.t === 'word') {
|
|
820
|
+
const n = nodeOfToken(pgm);
|
|
821
|
+
if (n) n.sinks.push({ kind: 'cics-dynamic-transfer', ...at, detail: `EXEC CICS ${verb} with a variable program name` });
|
|
822
|
+
}
|
|
823
|
+
const commarea = (o.get('COMMAREA') || o.get('FROM') || [])[0];
|
|
824
|
+
const argNode = commarea ? nodeOfToken(commarea) : null;
|
|
825
|
+
if (pgm && pgm.t === 'lit' && argNode) transfers.push({ verb, callee: pgm.v.toUpperCase(), argNode, point: at.point });
|
|
826
|
+
else if (pgm && pgm.t === 'word' && argNode && o.has('PROGRAM')) {
|
|
827
|
+
for (const callee of programsNamed(pgm, e)) { transfers.push({ verb, callee, argNode, point: at.point, through: pgm.u }); transfersTo.add(callee); }
|
|
828
|
+
}
|
|
829
|
+
if (pgm && pgm.t === 'lit') (o.has('TRANSID') && !o.has('PROGRAM') ? startsTransactions : transfersTo).add(String(pgm.v).trim().toUpperCase());
|
|
830
|
+
}
|
|
831
|
+
// RETURN TRANSID names the transaction the terminal's next input starts.
|
|
832
|
+
if (verb === 'RETURN' && o.has('TRANSID')) {
|
|
833
|
+
const t = (o.get('TRANSID') || [])[0];
|
|
834
|
+
if (t && t.t === 'lit') startsTransactions.add(String(t.v).trim().toUpperCase());
|
|
835
|
+
}
|
|
836
|
+
}
|
|
837
|
+
if (e.kind === 'SQL') {
|
|
838
|
+
// The parser's host variables name the item a qualified :G.F resolves to, where the first
|
|
839
|
+
// word after the colon would name the group G and so every field in it.
|
|
840
|
+
const hv = e.hostVariables || sqlHostVars(e);
|
|
841
|
+
const verb = words[0];
|
|
842
|
+
const range = e.hostVariables ? null : ['SELECT', 'FETCH'].includes(verb) && sqlIntoRange(e);
|
|
843
|
+
for (const h of hv) {
|
|
844
|
+
if (e.hostVariables ? !h.written : !range || h.at < range[0] || h.at > range[1]) continue;
|
|
845
|
+
const n = nodeOfToken(h.tok);
|
|
846
|
+
if (n) n.sources.push({ kind: 'database', ...at, detail: `EXEC SQL ${verb} INTO host variable` });
|
|
847
|
+
}
|
|
848
|
+
const dynamic = words.some((w, i) => w === 'PREPARE' || (w === 'EXECUTE' && words[i + 1] === 'IMMEDIATE'));
|
|
849
|
+
if (dynamic) for (const h of hv) {
|
|
850
|
+
const n = nodeOfToken(h.tok);
|
|
851
|
+
if (n) n.sinks.push({ kind: 'dynamic-sql', ...at, detail: `EXEC SQL ${words.slice(0, 2).join(' ')} from a host variable` });
|
|
852
|
+
}
|
|
853
|
+
}
|
|
854
|
+
}
|
|
855
|
+
for (const inv of invokes) {
|
|
856
|
+
for (const [name, channel] of channelOf) {
|
|
857
|
+
if (inv.channel && channel && channel !== inv.channel) continue;
|
|
858
|
+
containerNode(name).sinks.push({ kind: 'outbound-http', onlyFrom: DATA_AT_REST, ...inv.at, detail: `EXEC CICS INVOKE ${inv.what} sending CONTAINER(${name})${inv.channel ? ` on CHANNEL(${inv.channel})` : ''}` });
|
|
859
|
+
}
|
|
860
|
+
}
|
|
861
|
+
const commareaItem = byName.get('DFHCOMMAREA');
|
|
862
|
+
return {
|
|
863
|
+
id: prog.id, pk, callSites, transfers, ddFiles, tdWrites,
|
|
864
|
+
params: prog.paramTokens ? prog.paramTokens.map(p => {
|
|
865
|
+
const it = itemOfToken(p.tok);
|
|
866
|
+
const extent = extentOf(it);
|
|
867
|
+
return { node: nodeOfToken(p.tok), mode: p.mode, extent: extent && { ...extent, written: writtenEnd.get(topOf(it)) || 0 } };
|
|
868
|
+
}) : null,
|
|
869
|
+
commarea: commareaItem ? nodeOf(commareaItem) : null,
|
|
870
|
+
// Only the facts survive the parse tree: which checks hold before each statement, and so which
|
|
871
|
+
// statements a route reaches at all.
|
|
872
|
+
ordered: !!ctl,
|
|
873
|
+
reachKnown: !!ctl && !ctl.partial,
|
|
874
|
+
reached: ctl ? ctl.reached : null,
|
|
875
|
+
facts: ctl ? ctl.facts : null,
|
|
876
|
+
file, line: prog.line,
|
|
877
|
+
calls: new Set(prog.calls.filter((c) => c.kind === 'L').map((c) => String(c.name).toUpperCase())),
|
|
878
|
+
transfersTo, startsTransactions,
|
|
879
|
+
names: literalNames(prog),
|
|
880
|
+
};
|
|
881
|
+
}
|
|
882
|
+
|
|
883
|
+
for (const repo of repos) {
|
|
884
|
+
const base = repo ? join(root, repo) : root;
|
|
885
|
+
// The scan's tree covers the root it was built for. A --repos run scans a different directory
|
|
886
|
+
// per repository, so each gets its own; a single-root run reuses the one already walked.
|
|
887
|
+
const tree = (!repo && opts.tree) ? opts.tree : directoryTree(base, opts);
|
|
888
|
+
const idx = tree.index;
|
|
889
|
+
const files = tree.list().filter(isProgram).filter(inScope(opts));
|
|
890
|
+
for (const f of tree.list().filter(isJcl).filter(inScope(opts))) jclPending.push(f);
|
|
891
|
+
for (const f of tree.list().filter((p) => /\.csd$/i.test(p)).filter(inScope(opts))) {
|
|
892
|
+
try { addCsd(tree.text(f).text, f); } catch (e) { stats.unreadable++; unread.push(`${tree.rel(f)}: ${e.code || e.name}`); }
|
|
893
|
+
}
|
|
894
|
+
bmsMaps.clear();
|
|
895
|
+
for (const f of tree.list().filter(isBms).filter(inScope(opts))) {
|
|
896
|
+
let bms;
|
|
897
|
+
try { bms = parseBms(tree.text(f).text); } catch (e) { stats.unreadable++; unread.push(`${tree.rel(f)}: ${e.code || e.name}`); continue; }
|
|
898
|
+
stats.bmsFiles++;
|
|
899
|
+
for (const ms of bms.mapsets) for (const map of ms.maps) for (const k of [`${ms.name}/${map.name}`, map.name]) {
|
|
900
|
+
if (!bmsMaps.has(k)) bmsMaps.set(k, []);
|
|
901
|
+
bmsMaps.get(k).push({ mapset: ms, map });
|
|
902
|
+
}
|
|
903
|
+
}
|
|
904
|
+
// The byte budget bounds the source read. It does not bound the graph built from it: nodes and
|
|
905
|
+
// edges are proportional to data items and statements, not to bytes, so a repository of small
|
|
906
|
+
// programs with large record layouts can exhaust a heap the budget believes is untouched. The
|
|
907
|
+
// shared loop watches the heap itself, and the budget stays as the cheaper first line.
|
|
908
|
+
const run = eachWithinMemory(files, (f) => {
|
|
909
|
+
let src;
|
|
910
|
+
try { const s = tree.text(f); src = s.text; if (s.encoding === 'ebcdic') stats.ebcdic++; } catch (e) { stats.unreadable++; unread.push(`${tree.rel(f)}: ${e.code || e.name}`); return 0; }
|
|
911
|
+
if (!/PROCEDURE\s+DIVISION|PROGRAM-ID/i.test(src)) return 0;
|
|
912
|
+
if (held + src.length > budget) { stats.overBudget++; return 0; }
|
|
913
|
+
held += src.length;
|
|
914
|
+
let r;
|
|
915
|
+
try {
|
|
916
|
+
r = tree.parse(f, src);
|
|
917
|
+
} catch (e) { stats.threw++; unparsed.push(`${tree.rel(f)}: ${e.code || e.name}`); return src.length; }
|
|
918
|
+
stats.files++;
|
|
919
|
+
for (const p of r.programs) { programs.push(summarise(p, f, programs.length, r.options)); stats.programs++; }
|
|
920
|
+
return src.length;
|
|
921
|
+
}, { label: 'flow', watcher: memoryWatcher });
|
|
922
|
+
|
|
923
|
+
// A file the loop never reached is not the same as one past the byte budget, and they are
|
|
924
|
+
// counted separately so a report can say which limit it hit.
|
|
925
|
+
stats.filesNotReached += run.skipped.length;
|
|
926
|
+
if (run.stoppedBy) stats.stoppedBy = run.stoppedBy;
|
|
927
|
+
stats.peakHeapBytes = Math.max(stats.peakHeapBytes, run.peakHeapBytes);
|
|
928
|
+
}
|
|
929
|
+
|
|
930
|
+
const byId = new Map();
|
|
931
|
+
const idCount = new Map();
|
|
932
|
+
for (const p of programs) {
|
|
933
|
+
if (p.id && !byId.has(p.id)) byId.set(p.id, p);
|
|
934
|
+
if (p.id) idCount.set(p.id, (idCount.get(p.id) || 0) + 1);
|
|
935
|
+
}
|
|
936
|
+
const constructs = [];
|
|
937
|
+
for (const p of programs) {
|
|
938
|
+
for (const t of p.transfers) {
|
|
939
|
+
const callee = byId.get(t.callee);
|
|
940
|
+
if (!callee || !callee.commarea) continue;
|
|
941
|
+
edge(t.argNode, callee.commarea, `EXEC CICS ${t.verb} COMMAREA to ${t.callee}${t.through ? `, the program ${t.through} names` : ''}`, undefined, t.point);
|
|
942
|
+
if (t.verb === 'LINK') edge(callee.commarea, t.argNode, `COMMAREA returned from ${t.callee}`);
|
|
943
|
+
}
|
|
944
|
+
for (const c of p.callSites) {
|
|
945
|
+
const callee = byId.get(c.name);
|
|
946
|
+
if (!callee || !callee.params) continue;
|
|
947
|
+
c.args.forEach((arg, i) => {
|
|
948
|
+
const param = callee.params[i];
|
|
949
|
+
if (!param || !arg.node || !param.node) return;
|
|
950
|
+
edge(arg.node, param.node, `CALL '${c.display}' argument ${i + 1} at ${rel(c.file)}:${c.line}`, undefined, c.point);
|
|
951
|
+
if (arg.mode === 'REFERENCE' && param.mode === 'REFERENCE') edge(param.node, arg.node, `CALL '${c.display}' argument ${i + 1} written back`);
|
|
952
|
+
});
|
|
953
|
+
// A callee that declares more bytes than its caller passes reads and writes whatever follows
|
|
954
|
+
// the argument in the caller's storage. Measured over 8,640 bound pairs in 125 repositories:
|
|
955
|
+
// where the extra bytes stay inside the caller's own record they are its neighbouring fields,
|
|
956
|
+
// sometimes on purpose (ACAS passes the first field of a group to reach the whole group);
|
|
957
|
+
// where they run past the record they are someone else's storage. Those are different
|
|
958
|
+
// claims, so they are different rules.
|
|
959
|
+
//
|
|
960
|
+
// Not compared: a program id defined twice in the repository, where which callee runs is not
|
|
961
|
+
// known here; BY VALUE, which passes no storage; and a parameter holding an OCCURS DEPENDING
|
|
962
|
+
// ON table, whose declared size is its largest rather than what it holds.
|
|
963
|
+
if ((idCount.get(c.name) || 0) > 1) continue;
|
|
964
|
+
const over = [];
|
|
965
|
+
c.args.forEach((arg, i) => {
|
|
966
|
+
const param = callee.params[i];
|
|
967
|
+
const a = arg.extent;
|
|
968
|
+
const q = param && param.extent;
|
|
969
|
+
if (!a || !q || q.variable || arg.mode === 'VALUE' || param.mode === 'VALUE' || q.bytes <= a.bytes) return;
|
|
970
|
+
over.push({ position: i + 1, argument: a.name, argumentBytes: a.bytes, bytesToEndOfRecord: a.room,
|
|
971
|
+
parameter: q.name, parameterBytes: q.bytes, writesPast: q.written > a.room, declared: { file: rel(q.file), line: q.line } });
|
|
972
|
+
});
|
|
973
|
+
for (const [rule, past] of [['call-parameter-exceeds-caller-record', true], ['call-parameter-exceeds-argument', false]]) {
|
|
974
|
+
const these = over.filter(o => (o.parameterBytes > o.bytesToEndOfRecord) === past);
|
|
975
|
+
// Past the caller's record, a callee that only reads takes a wrong value; one that writes
|
|
976
|
+
// overwrites storage that is not the caller's.
|
|
977
|
+
const readsOnly = past && these.every((o) => !o.writesPast);
|
|
978
|
+
if (these.length) constructs.push({ rule, file: rel(c.file), line: c.line, program: p.id, callee: callee.id, args: these, pk: p.pk, ...(readsOnly ? { sev: 'low' } : {}) });
|
|
979
|
+
}
|
|
980
|
+
}
|
|
981
|
+
}
|
|
982
|
+
constructs.sort((a, b) => byText(a.file, b.file) || a.line - b.line || byText(a.rule, b.rule));
|
|
983
|
+
|
|
984
|
+
// The entry point above the program boundary. Until this ran, the flow engine began where a
|
|
985
|
+
// program began, and the real beginning is a JCL step: it chooses the program, hands it a
|
|
986
|
+
// parameter, and fills the DD names the program reads. Both of those are written in files that
|
|
987
|
+
// anyone who can commit to the repository can edit, which makes them untrusted in the same sense
|
|
988
|
+
// ACCEPT FROM COMMAND-LINE is untrusted.
|
|
989
|
+
for (const f of jclPending) {
|
|
990
|
+
let src;
|
|
991
|
+
try { src = readSource(f).text; } catch (e) { stats.unreadable++; unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
|
|
992
|
+
let job;
|
|
993
|
+
try { job = parseJcl(src, f); } catch (e) { stats.threw++; unparsed.push(`${rel(f)}: ${e.code || e.name}`); continue; }
|
|
994
|
+
stats.jclFiles++;
|
|
995
|
+
|
|
996
|
+
for (const step of job.steps) {
|
|
997
|
+
if (!step.pgm) continue;
|
|
998
|
+
stats.jclSteps++;
|
|
999
|
+
// A DFHCSDUP step's input is CSD definitions, kept in a job rather than an extract.
|
|
1000
|
+
if (step.pgm.toUpperCase() === 'DFHCSDUP') {
|
|
1001
|
+
for (const dd of step.dds) if (dd.name && dd.name.toUpperCase() === 'SYSIN' && dd.inStream) addCsd(dd.inStream.map((l) => l.text).join('\n'), f, dd.inStream[0].line - 1);
|
|
1002
|
+
}
|
|
1003
|
+
jobSteps.push({ file: f, job: job.jobs[0]?.name || null, step: step.name, line: step.line, pgm: step.pgm.toUpperCase() });
|
|
1004
|
+
// A TSO batch step starts a program by name in its commands: DSN ... RUN PROGRAM(X).
|
|
1005
|
+
if (/^IKJEFT(01|1A|1B)$/i.test(step.pgm)) {
|
|
1006
|
+
for (const dd of step.dds) for (const l of dd.inStream || []) {
|
|
1007
|
+
const run = /\bRUN\s+PROGRAM\s*\(\s*([A-Z0-9$#@]{1,8})\s*\)/i.exec(l.text);
|
|
1008
|
+
if (run) jobSteps.push({ file: f, job: job.jobs[0]?.name || null, step: step.name, line: l.line, pgm: run[1].toUpperCase() });
|
|
1009
|
+
}
|
|
1010
|
+
}
|
|
1011
|
+
const callee = byId.get(step.pgm.toUpperCase());
|
|
1012
|
+
if (!callee) continue; // a system utility, or a program not in this tree
|
|
1013
|
+
stats.jclStepsResolved++;
|
|
1014
|
+
|
|
1015
|
+
// PARM arrives in the first PROCEDURE DIVISION USING item: on z/OS a halfword length
|
|
1016
|
+
// followed by the text. A program with no USING cannot receive one, and saying it does
|
|
1017
|
+
// would be a path nobody could follow.
|
|
1018
|
+
if (step.parm !== null && callee.params && callee.params[0] && callee.params[0].node) {
|
|
1019
|
+
callee.params[0].node.sources.push({
|
|
1020
|
+
kind: 'jcl-parm', file: f, line: step.line,
|
|
1021
|
+
detail: `PARM= on step ${step.name || '(unnamed)'} of ${rel(f)}, which runs ${step.pgm}`,
|
|
1022
|
+
});
|
|
1023
|
+
stats.jclCrossings++;
|
|
1024
|
+
}
|
|
1025
|
+
|
|
1026
|
+
// In-stream data reaches whatever record the program reads from that DD. The COBOL says
|
|
1027
|
+
// ASSIGN TO SYSIN and the job says //SYSIN DD *; neither half names the other, and the
|
|
1028
|
+
// join is the whole point of reading both.
|
|
1029
|
+
for (const dd of step.dds) {
|
|
1030
|
+
// The same join in the other direction: what the program writes through SELECT ... ASSIGN
|
|
1031
|
+
// TO a DD the job sends to the internal reader is submitted as a job.
|
|
1032
|
+
if (dd.name && dd.sysout && /\bINTRDR\b/i.test(dd.sysout)) {
|
|
1033
|
+
for (const m of callee.ddFiles || []) {
|
|
1034
|
+
if (m.dd !== dd.name.toUpperCase()) continue;
|
|
1035
|
+
for (const n of m.nodes) {
|
|
1036
|
+
n.sinks.push({ kind: 'internal-reader', file: m.file, line: m.line,
|
|
1037
|
+
detail: `records written through SELECT ${m.select} to //${dd.name}, which step ${step.name || '(unnamed)'} of ${rel(f)} sends to the internal reader` });
|
|
1038
|
+
}
|
|
1039
|
+
}
|
|
1040
|
+
}
|
|
1041
|
+
if (!dd.inStream || !dd.inStream.length || !dd.name) continue;
|
|
1042
|
+
for (const m of callee.ddFiles || []) {
|
|
1043
|
+
if (m.dd !== dd.name.toUpperCase()) continue;
|
|
1044
|
+
for (const n of m.nodes) {
|
|
1045
|
+
n.sources.push({
|
|
1046
|
+
kind: 'jcl-instream', file: f, line: dd.line,
|
|
1047
|
+
detail: `${dd.inStream.length} lines of in-stream data on //${dd.name} in step ${step.name || '(unnamed)'}, read through SELECT ${m.select}`,
|
|
1048
|
+
});
|
|
1049
|
+
stats.jclCrossings++;
|
|
1050
|
+
}
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1053
|
+
}
|
|
1054
|
+
}
|
|
1055
|
+
|
|
1056
|
+
// Where each program is started from: the transactions the CSD defines for it, the job steps that
|
|
1057
|
+
// run it, and whatever those programs call, link or transfer to, or start by transaction on the
|
|
1058
|
+
// way. A mainframe team triages by transaction, and a program nothing here starts is one no route
|
|
1059
|
+
// in this tree reaches.
|
|
1060
|
+
const roots = [];
|
|
1061
|
+
for (const [name, t] of csd.transactions) if (t.program && byId.has(t.program)) roots.push({ transaction: name, file: rel(t.file), line: t.line, program: t.program });
|
|
1062
|
+
for (const j of jobSteps) if (byId.has(j.pgm)) roots.push({ job: j.job, step: j.step, file: rel(j.file), line: j.line, program: j.pgm });
|
|
1063
|
+
const STARTED_BY_LISTED = 8;
|
|
1064
|
+
for (const r of roots) {
|
|
1065
|
+
const { program, ...entry } = r;
|
|
1066
|
+
const seenIds = new Set([program]);
|
|
1067
|
+
const queue = [program];
|
|
1068
|
+
for (let i = 0; i < queue.length; i++) {
|
|
1069
|
+
const p = byId.get(queue[i]);
|
|
1070
|
+
if (!p) continue;
|
|
1071
|
+
p.startedCount = (p.startedCount || 0) + 1;
|
|
1072
|
+
if (!p.startedBy) p.startedBy = [];
|
|
1073
|
+
if (p.startedBy.length < STARTED_BY_LISTED) p.startedBy.push(entry);
|
|
1074
|
+
const next = [...p.calls, ...p.transfersTo];
|
|
1075
|
+
for (const t of p.startsTransactions) { const d = csd.transactions.get(t); if (d && d.program) next.push(d.program); }
|
|
1076
|
+
for (const n of next) if (!seenIds.has(n) && byId.has(n)) { seenIds.add(n); queue.push(n); }
|
|
1077
|
+
}
|
|
1078
|
+
}
|
|
1079
|
+
// A program nothing starts, where the tree holds its entries at all. One whose name another
|
|
1080
|
+
// program spells in a literal may be started through a variable, and is left alone. Nothing is
|
|
1081
|
+
// said when any program went unread: its caller may be among them.
|
|
1082
|
+
const spelled = new Set();
|
|
1083
|
+
for (const p of programs) for (const n of p.names) if (n !== p.id) spelled.add(n);
|
|
1084
|
+
const readAll = !stats.stoppedBy && !stats.filesNotReached && !stats.overBudget && !stats.threw && !stats.unreadable;
|
|
1085
|
+
const unstarted = roots.length && readAll ? programs.filter((p) => p.id && !p.startedBy && !spelled.has(p.id) && byId.get(p.id) === p)
|
|
1086
|
+
.map((p) => ({ program: p.id, file: rel(p.file), line: p.line })) : [];
|
|
1087
|
+
const startedByOf = (pk) => {
|
|
1088
|
+
const p = programs[pk];
|
|
1089
|
+
return p && p.startedBy ? { startedBy: p.startedBy, ...(p.startedCount > p.startedBy.length ? { startedByMore: p.startedCount - p.startedBy.length } : {}) } : {};
|
|
1090
|
+
};
|
|
1091
|
+
|
|
1092
|
+
// A transient-data write reaches the internal reader when the estate says its queue does, or says
|
|
1093
|
+
// the DD the CSD maps it to does. Anything else extrapartition is undecided rather than clean:
|
|
1094
|
+
// the region's JCL is what knows, and it is rarely in the repository. An intrapartition queue has
|
|
1095
|
+
// no DD and is decided; a queue named by a variable is undecided whatever the estate declares.
|
|
1096
|
+
const readerDds = new Set(site.internalReaderDds);
|
|
1097
|
+
const readerQueues = new Set(site.internalReaderQueues);
|
|
1098
|
+
const declared = readerDds.size > 0 || readerQueues.size > 0;
|
|
1099
|
+
stats.tdWritesUndecided = 0;
|
|
1100
|
+
for (const p of programs) {
|
|
1101
|
+
for (const w of p.tdWrites) {
|
|
1102
|
+
const def = w.queue ? csd.tdqueues.get(w.queue) : null;
|
|
1103
|
+
const dd = w.queue ? ddOfQueue(csd, w.queue) : null;
|
|
1104
|
+
if (w.queue && (readerQueues.has(w.queue) || (dd && readerDds.has(dd)))) {
|
|
1105
|
+
w.node.sinks.push({ kind: 'internal-reader', file: w.file, line: w.line, point: w.point,
|
|
1106
|
+
detail: `EXEC CICS WRITEQ TD QUEUE('${w.queue}')${dd ? `, which the CSD sends to DD ${dd}` : ''}, declared as reaching the internal reader` });
|
|
1107
|
+
continue;
|
|
1108
|
+
}
|
|
1109
|
+
// Any other queue is where a region writes what it logs or prints, unless it starts a
|
|
1110
|
+
// transaction, when it is a channel to that transaction instead.
|
|
1111
|
+
if (!(def && def.transid)) {
|
|
1112
|
+
w.node.sinks.push({ kind: 'log', onlyFrom: REFLECTED, file: w.file, line: w.line, point: w.point,
|
|
1113
|
+
detail: `EXEC CICS WRITEQ TD QUEUE(${w.queue ? `'${w.queue}'` : 'a name the program computes'})` });
|
|
1114
|
+
}
|
|
1115
|
+
if (!w.queue) { stats.tdWritesUndecided++; continue; }
|
|
1116
|
+
if (def && def.type === 'INTRA') continue;
|
|
1117
|
+
// A queue the CSD sends to a DD leaves the region: to a dataset, a printer or another job.
|
|
1118
|
+
if (dd) {
|
|
1119
|
+
w.node.sinks.push({ kind: 'extrapartition-queue', onlyFrom: DATA_AT_REST, file: w.file, line: w.line, point: w.point,
|
|
1120
|
+
detail: `EXEC CICS WRITEQ TD QUEUE('${w.queue}'), which the CSD sends out of the region to DD ${dd}` });
|
|
1121
|
+
}
|
|
1122
|
+
if (!declared) stats.tdWritesUndecided++;
|
|
1123
|
+
else if (!def && !readerQueues.size) stats.tdWritesUndecided++;
|
|
1124
|
+
}
|
|
1125
|
+
}
|
|
1126
|
+
|
|
1127
|
+
// Only nodes that can reach some sink are worth walking into. One backward pass over a transient
|
|
1128
|
+
// reverse index answers that for every source at once; it is a superset, so nothing is lost.
|
|
1129
|
+
// Two answers, one bit each: whether a node reaches a sink a source at rest could be reported at,
|
|
1130
|
+
// and whether it reaches any sink. The bounds sinks take only input from outside, and a file
|
|
1131
|
+
// record's walk is not widened by sinks that could never report it - batch programs index
|
|
1132
|
+
// everything, and those are the walks that would pay.
|
|
1133
|
+
const AT_REST = 1, ANY = 2;
|
|
1134
|
+
const canReach = new Uint8Array(nodes.length);
|
|
1135
|
+
{
|
|
1136
|
+
const revStart = new Int32Array(nodes.length + 1);
|
|
1137
|
+
for (const n of nodes) for (const e of n.edgesOut) revStart[e.to.id + 1]++;
|
|
1138
|
+
for (let i = 0; i < nodes.length; i++) revStart[i + 1] += revStart[i];
|
|
1139
|
+
const fill = revStart.slice(0, nodes.length);
|
|
1140
|
+
const rev = new Int32Array(revStart[nodes.length]);
|
|
1141
|
+
for (const n of nodes) for (const e of n.edgesOut) rev[fill[e.to.id]++] = n.id;
|
|
1142
|
+
const mark = (bit, seeds) => {
|
|
1143
|
+
const queue = [];
|
|
1144
|
+
for (const n of nodes) if (seeds(n)) { canReach[n.id] |= bit; queue.push(n.id); }
|
|
1145
|
+
for (let qi = 0; qi < queue.length; qi++) {
|
|
1146
|
+
const id = queue[qi];
|
|
1147
|
+
for (let k = revStart[id]; k < revStart[id + 1]; k++) if (!(canReach[rev[k]] & bit)) { canReach[rev[k]] |= bit; queue.push(rev[k]); }
|
|
1148
|
+
}
|
|
1149
|
+
};
|
|
1150
|
+
mark(ANY, (n) => n.sinks.length > 0);
|
|
1151
|
+
mark(AT_REST, (n) => n.sinks.some((s) => !s.onlyFrom || s.onlyFrom.some((k) => DATA_AT_REST.includes(k))));
|
|
1152
|
+
}
|
|
1153
|
+
|
|
1154
|
+
const fmtWhy = (w) => (typeof w === 'string' ? w : `${w.verb} at ${w.file}:${w.line}`);
|
|
1155
|
+
|
|
1156
|
+
// Whether a check carries a value at the point it leaves a node, and how far: 2 when what the node
|
|
1157
|
+
// holds there is safe for the sink, 1 when a check has run on every route to that point, 0 when
|
|
1158
|
+
// neither. A program whose order could not be read credits its checks the old way, wherever the
|
|
1159
|
+
// field is used and never past 1.
|
|
1160
|
+
const STRUCTURAL = new Set(['group', 'redefines', 'renames']);
|
|
1161
|
+
const pointOfEdge = (e) => (e.why && typeof e.why === 'object' ? e.why.point : e.at);
|
|
1162
|
+
const moves = (e) => !(typeof e.why === 'string' && STRUCTURAL.has(e.why));
|
|
1163
|
+
const NONE = { level: 0, check: null };
|
|
1164
|
+
// `limit` is how many entries or bytes the sink's index may name, where one number says it.
|
|
1165
|
+
function creditAt(n, point, kind, limit = null) {
|
|
1166
|
+
const p = programs[n.pk];
|
|
1167
|
+
if (!p || !p.ordered) return n.guard ? { level: 1, check: n.guard } : NONE;
|
|
1168
|
+
if (!n.checks || point == null) return NONE;
|
|
1169
|
+
const bits = p.facts.get(point);
|
|
1170
|
+
return bits ? creditOf(n.checks, bits, kind, limit) : NONE;
|
|
1171
|
+
}
|
|
1172
|
+
// The best credit on a route. Each node is judged at the statement its value leaves by, which for
|
|
1173
|
+
// the last is the sink's own. A structural hop - into a group, across a REDEFINES - moves no value
|
|
1174
|
+
// in time, so the node before it is judged where the value next moves.
|
|
1175
|
+
function routeCredit(hops, sink, src) {
|
|
1176
|
+
if (sink.noCredit || (src && src.noCredit)) return { best: NONE, by: null, missed: null };
|
|
1177
|
+
let best = NONE;
|
|
1178
|
+
let by = null;
|
|
1179
|
+
let missed = null;
|
|
1180
|
+
let point = sink.point;
|
|
1181
|
+
for (let h = hops.length - 1; h >= 0; h--) {
|
|
1182
|
+
const n = hops[h].node;
|
|
1183
|
+
const c = creditAt(n, point, sink.kind, sink.limit);
|
|
1184
|
+
if (c.level > best.level) { best = c; by = n; }
|
|
1185
|
+
const any = (n.checks && n.checks.find((c) => c.x != null)) || n.guard;
|
|
1186
|
+
if (!c.level && any && !missed) missed = { node: n, check: any };
|
|
1187
|
+
if (h > 0 && hops[h].moves) point = hops[h].point;
|
|
1188
|
+
}
|
|
1189
|
+
return { best, by, missed };
|
|
1190
|
+
}
|
|
1191
|
+
// The credited findings of one source whose sink can be reached without the credit: walking from
|
|
1192
|
+
// the source, a value may not leave a node - or any node it reached by a structural hop since it
|
|
1193
|
+
// last moved - at a point where the credit reaches `level`.
|
|
1194
|
+
function refuse(start, reach, credited, level, kind) {
|
|
1195
|
+
walk++;
|
|
1196
|
+
const want = credited.filter((c) => c.level >= level && c.sink.kind === kind);
|
|
1197
|
+
// Past the budget no route can be ruled out, so every credit in question is lost: a finding
|
|
1198
|
+
// may be reported unchecked, never dropped.
|
|
1199
|
+
if (edgesWalked >= totalBudget) { stats.creditUndecided++; return new Set(want); }
|
|
1200
|
+
const at = new Map();
|
|
1201
|
+
for (const c of want) { if (!at.has(c.node.id)) at.set(c.node.id, []); at.get(c.node.id).push(c); }
|
|
1202
|
+
// Along the way no sink's limit applies, so a bound that holds only against one blocks nothing.
|
|
1203
|
+
const holds = (since, point, k, limit = null) => since.some((n) => creditAt(n, point, k, limit).level >= level);
|
|
1204
|
+
const lost = new Set();
|
|
1205
|
+
const dirty = new Map();
|
|
1206
|
+
const parts = new Map();
|
|
1207
|
+
const open = [{ node: start, partial: false, lo: 0, hi: 0, vague: false, since: [start] }];
|
|
1208
|
+
stampWhole[start.id] = walk;
|
|
1209
|
+
const began = edgesWalked;
|
|
1210
|
+
for (let i = 0; i < open.length; i++) {
|
|
1211
|
+
const st = open[i];
|
|
1212
|
+
for (const c of at.get(st.node.id) || []) if (!holds(st.since, c.sink.point, c.sink.kind, c.sink.limit)) lost.add(c);
|
|
1213
|
+
for (const e of st.node.edgesOut) {
|
|
1214
|
+
edgesWalked++;
|
|
1215
|
+
const to = e.to;
|
|
1216
|
+
if (!(canReach[to.id] & reach)) continue;
|
|
1217
|
+
const mv = moves(e);
|
|
1218
|
+
if (mv && holds(st.since, pointOfEdge(e), kind)) continue;
|
|
1219
|
+
const t = cross(st, e);
|
|
1220
|
+
if (!t) continue;
|
|
1221
|
+
const since = mv ? [to] : [...st.since, to];
|
|
1222
|
+
// An arrival carrying a checked node is refused more than one that does not, so it must not
|
|
1223
|
+
// stand in for it.
|
|
1224
|
+
const guarded = since.some((n) => n.checks || n.guard);
|
|
1225
|
+
if (t.partial) {
|
|
1226
|
+
const key = `${to.id}|${guarded}`;
|
|
1227
|
+
const held = parts.get(key);
|
|
1228
|
+
if (held && held.some((r) => (r[2] ? t.vague : !t.vague && r[0] <= t.lo && t.hi <= r[1]))) continue;
|
|
1229
|
+
if (held) held.push([t.lo, t.hi, t.vague]); else parts.set(key, [[t.lo, t.hi, t.vague]]);
|
|
1230
|
+
} else if (guarded) {
|
|
1231
|
+
if (stampWhole[to.id] === walk || dirty.get(to.id) === walk) continue;
|
|
1232
|
+
dirty.set(to.id, walk);
|
|
1233
|
+
} else {
|
|
1234
|
+
if (stampWhole[to.id] === walk) continue;
|
|
1235
|
+
stampWhole[to.id] = walk;
|
|
1236
|
+
}
|
|
1237
|
+
open.push({ node: to, ...t, since });
|
|
1238
|
+
}
|
|
1239
|
+
if (edgesWalked - began > walkBudget) { stats.creditUndecided++; return new Set(want); }
|
|
1240
|
+
}
|
|
1241
|
+
return lost;
|
|
1242
|
+
}
|
|
1243
|
+
const findings = [];
|
|
1244
|
+
const seen = new Set();
|
|
1245
|
+
// A sink at a statement no route from any entry reaches does not run as the program stands.
|
|
1246
|
+
const unreached = (n, sink) => {
|
|
1247
|
+
const p = programs[n.pk];
|
|
1248
|
+
return !!p && p.reachKnown && sink.point != null && p.reached[sink.point] !== 1;
|
|
1249
|
+
};
|
|
1250
|
+
// The findings of one source, one table and one index name, from which the least checked is kept.
|
|
1251
|
+
const useGroup = new Map();
|
|
1252
|
+
// Visit stamps instead of a per-walk Set: one walk per source, and a Set of string keys per walk
|
|
1253
|
+
// was most of the time spent on large repositories.
|
|
1254
|
+
const stampWhole = new Int32Array(nodes.length);
|
|
1255
|
+
let walk = 0;
|
|
1256
|
+
// Every walk is bounded, and so is the analysis. One source of an estate whose programs share
|
|
1257
|
+
// their records through thousands of copybooks reached millions of nodes, and 22,905 such walks
|
|
1258
|
+
// would have run for hours. Edges examined are counted rather than seconds, so the same tree always
|
|
1259
|
+
// stops at the same place.
|
|
1260
|
+
const walkBudget = opts.walkEdges ?? WALK_EDGES;
|
|
1261
|
+
const totalBudget = opts.totalEdges ?? TOTAL_EDGES;
|
|
1262
|
+
let edgesWalked = 0;
|
|
1263
|
+
stats.walksCut = 0;
|
|
1264
|
+
stats.sourcesNotWalked = 0;
|
|
1265
|
+
stats.creditUndecided = 0;
|
|
1266
|
+
|
|
1267
|
+
// How taint crosses one edge: whether it reaches the far item, and which of its bytes. Null when
|
|
1268
|
+
// it does not. Both walks below cross edges through this, so neither can take a route the other
|
|
1269
|
+
// would refuse.
|
|
1270
|
+
const known = (n) => n.off != null && n.size != null && !n.repeats;
|
|
1271
|
+
function cross(state, e) {
|
|
1272
|
+
const cur = state.node;
|
|
1273
|
+
const to = e.to;
|
|
1274
|
+
let partial = false, lo = 0, hi = 0, vague = false;
|
|
1275
|
+
if (opts.legacyGroupEdges) { /* every edge carries the whole item */ }
|
|
1276
|
+
else if (e.dir === 'up') {
|
|
1277
|
+
partial = true;
|
|
1278
|
+
if (known(cur) && known(to) && !state.vague) {
|
|
1279
|
+
const shift = cur.off - to.off;
|
|
1280
|
+
lo = shift + (state.partial ? state.lo : 0);
|
|
1281
|
+
hi = shift + (state.partial ? state.hi : cur.size);
|
|
1282
|
+
} else vague = true;
|
|
1283
|
+
} else if (e.dir === 'down') {
|
|
1284
|
+
if (state.partial && to.repeats) {
|
|
1285
|
+
// A table inside the group. Which element the tainted bytes fall in is not known, but
|
|
1286
|
+
// whether they fall inside the table is; if they do, some element holds them, and the walk
|
|
1287
|
+
// goes on without claiming which bytes. A table they miss stays clean.
|
|
1288
|
+
if (state.vague || !known(cur) || to.tableOff == null) return null;
|
|
1289
|
+
const tlo = cur.off + state.lo;
|
|
1290
|
+
const thi = cur.off + state.hi;
|
|
1291
|
+
if (Math.max(tlo, to.tableOff) >= Math.min(thi, to.tableEnd)) return null;
|
|
1292
|
+
partial = true; vague = true;
|
|
1293
|
+
} else if (state.partial) {
|
|
1294
|
+
if (state.vague || !known(cur) || !known(to)) return null;
|
|
1295
|
+
const start = to.off - cur.off;
|
|
1296
|
+
lo = Math.max(state.lo, start) - start;
|
|
1297
|
+
hi = Math.min(state.hi, start + to.size) - start;
|
|
1298
|
+
if (hi <= lo) return null;
|
|
1299
|
+
partial = !(lo === 0 && hi === to.size);
|
|
1300
|
+
}
|
|
1301
|
+
} else if (state.partial) {
|
|
1302
|
+
// A statement other than MOVE builds its result from the operand, so the tainted bytes are in
|
|
1303
|
+
// the result at a position nobody can name: the whole result is tainted.
|
|
1304
|
+
const copies = typeof e.why === 'string' || (e.why && e.why.verb === 'MOVE');
|
|
1305
|
+
if (copies) {
|
|
1306
|
+
partial = true; vague = state.vague; lo = state.lo; hi = state.hi;
|
|
1307
|
+
if (!vague && to.size != null) { hi = Math.min(hi, to.size); if (hi <= lo) return null; }
|
|
1308
|
+
}
|
|
1309
|
+
}
|
|
1310
|
+
return { partial, lo, hi, vague };
|
|
1311
|
+
}
|
|
1312
|
+
for (const start of nodes) {
|
|
1313
|
+
if (!start.sources.length || !canReach[start.id]) continue;
|
|
1314
|
+
for (const src of start.sources) {
|
|
1315
|
+
const reach = DATA_AT_REST.includes(src.kind) ? AT_REST : ANY;
|
|
1316
|
+
if (!(canReach[start.id] & reach)) continue;
|
|
1317
|
+
if (edgesWalked >= totalBudget) { stats.sourcesNotWalked++; continue; }
|
|
1318
|
+
walk++;
|
|
1319
|
+
const credited = [];
|
|
1320
|
+
// A state is a node plus which of its bytes are tainted. Going up from a child taints only
|
|
1321
|
+
// that child's bytes of the group; going down reaches a child only if those bytes overlap it,
|
|
1322
|
+
// which is what keeps a sibling nobody wrote out of the path. The bytes travel unchanged
|
|
1323
|
+
// through anything that copies or overlays storage: a group MOVE, a REDEFINES, an argument
|
|
1324
|
+
// passed by CALL or a communication area. Where a position is unknown (an OCCURS, a record
|
|
1325
|
+
// without a layout) partial taint does not descend at all.
|
|
1326
|
+
const queue = [{ node: start, partial: false, lo: 0, hi: 0, vague: false, prev: null, why: 'source' }];
|
|
1327
|
+
const partSeen = new Map();
|
|
1328
|
+
stampWhole[start.id] = walk;
|
|
1329
|
+
let cut = false;
|
|
1330
|
+
const began = edgesWalked;
|
|
1331
|
+
for (let qi = 0; qi < queue.length && !cut; qi++) {
|
|
1332
|
+
const state = queue[qi];
|
|
1333
|
+
const cur = state.node;
|
|
1334
|
+
for (const sink of cur.sinks) {
|
|
1335
|
+
if (sink.onlyFrom && !sink.onlyFrom.includes(src.kind)) continue;
|
|
1336
|
+
if (src.onlyTo && !src.onlyTo.includes(sink.kind)) continue;
|
|
1337
|
+
// Uses of an index are told apart by name and place, so one line's two uses are both judged.
|
|
1338
|
+
const key = `${src.kind}|${src.file}:${src.line}|${sink.kind}|${sink.file}:${sink.line}${sink.group ? `|${sink.group}|${sink.point}` : ''}`;
|
|
1339
|
+
if (seen.has(key)) continue;
|
|
1340
|
+
// A value the program computed is valid decimal whatever the input was, so a route through
|
|
1341
|
+
// arithmetic or a numeric function carries no bad bytes to the next one. Only this route is
|
|
1342
|
+
// dropped: a later one that copies the bytes still reports.
|
|
1343
|
+
if (sink.kind === 'arithmetic' && computedOnRoute(state)) continue;
|
|
1344
|
+
seen.add(key);
|
|
1345
|
+
// The ends of a long path are what a reader uses: where the value came from, and what it
|
|
1346
|
+
// reached. Keeping every hop of a thousand-hop chain, for a thousand findings, is the
|
|
1347
|
+
// quadratic cost in a chain-shaped program, so the middle is dropped and said to be.
|
|
1348
|
+
const hops = [];
|
|
1349
|
+
for (let s = state; s; s = s.prev) hops.push(s);
|
|
1350
|
+
hops.reverse();
|
|
1351
|
+
const hop = (s) => ({ program: s.node.program, item: s.node.name, file: rel(s.node.file), via: fmtWhy(s.why), ...(s.dir ? { dir: s.dir } : {}) });
|
|
1352
|
+
const path = fullTrace || hops.length <= TRACE_MAX
|
|
1353
|
+
? hops.map(hop)
|
|
1354
|
+
: [...hops.slice(0, TRACE_KEEP).map(hop),
|
|
1355
|
+
{ program: null, item: null, file: null, via: `… ${hops.length - TRACE_KEEP * 2} hops not listed`, elided: hops.length - TRACE_KEEP * 2 },
|
|
1356
|
+
...hops.slice(-TRACE_KEEP).map(hop)];
|
|
1357
|
+
const { best, by, missed } = routeCredit(hops, sink, src);
|
|
1358
|
+
const screen = hops.find((h) => h.node.screen)?.node.screen;
|
|
1359
|
+
findings.push({
|
|
1360
|
+
rule: `${src.kind}-to-${sink.kind}`,
|
|
1361
|
+
source: { kind: src.kind, detail: src.detail, file: rel(src.file), line: src.line, program: start.program },
|
|
1362
|
+
sink: { kind: sink.kind, detail: sink.detail, file: rel(sink.file), line: sink.line, program: cur.program },
|
|
1363
|
+
crossProgram: start.pk !== cur.pk,
|
|
1364
|
+
hops: hops.length,
|
|
1365
|
+
path,
|
|
1366
|
+
...(sink.ssrange ? { ssrange: true } : {}),
|
|
1367
|
+
...(unreached(cur, sink) ? { unreached: true } : {}),
|
|
1368
|
+
...(screen ? { screen } : {}),
|
|
1369
|
+
...startedByOf(cur.pk),
|
|
1370
|
+
...(best.level ? { guard: { program: by.program, item: by.name, file: rel(best.check.file), line: best.check.line, ...(best.level === 2 ? { stops: true } : {}) } } : {}),
|
|
1371
|
+
...(!best.level && missed ? { checkElsewhere: { program: missed.node.program, item: missed.node.name, file: rel(missed.check.file), line: missed.check.line } } : {}),
|
|
1372
|
+
});
|
|
1373
|
+
if (best.level) credited.push({ finding: findings[findings.length - 1], node: cur, sink, level: best.level });
|
|
1374
|
+
if (sink.group) useGroup.set(findings[findings.length - 1], `${src.kind}|${src.file}:${src.line}|${sink.kind}|${sink.group}`);
|
|
1375
|
+
}
|
|
1376
|
+
for (const e of cur.edgesOut) {
|
|
1377
|
+
if (++edgesWalked - began > walkBudget) { cut = true; break; }
|
|
1378
|
+
if (!(canReach[e.to.id] & reach)) continue;
|
|
1379
|
+
const to = e.to;
|
|
1380
|
+
const crossed = cross(state, e);
|
|
1381
|
+
if (!crossed) continue;
|
|
1382
|
+
const { partial, lo, hi, vague } = crossed;
|
|
1383
|
+
if (partial) {
|
|
1384
|
+
const held = partSeen.get(to.id);
|
|
1385
|
+
if (held && held.some(r => (r[2] ? vague : !vague && r[0] <= lo && hi <= r[1]))) continue;
|
|
1386
|
+
if (held) held.push([lo, hi, vague]); else partSeen.set(to.id, [[lo, hi, vague]]);
|
|
1387
|
+
} else {
|
|
1388
|
+
if (stampWhole[to.id] === walk) continue;
|
|
1389
|
+
stampWhole[to.id] = walk;
|
|
1390
|
+
}
|
|
1391
|
+
queue.push({ node: to, partial, lo, hi, vague, prev: state, why: e.why, dir: e.dir, point: pointOfEdge(e), moves: moves(e) });
|
|
1392
|
+
}
|
|
1393
|
+
}
|
|
1394
|
+
if (cut) stats.walksCut++;
|
|
1395
|
+
// The walk keeps the shortest route to each sink, and credit on that route says nothing if
|
|
1396
|
+
// another reaches the sink without it. So the walk runs again, refusing to let the value leave
|
|
1397
|
+
// a node where the credit holds, and a sink it still reaches loses the credit: first for any
|
|
1398
|
+
// check that ran, then, per kind of sink, for one that stops the value. It crosses edges through
|
|
1399
|
+
// cross() like the first walk, so it cannot take a route the first would refuse.
|
|
1400
|
+
if (credited.length) {
|
|
1401
|
+
// Per kind of sink, because what a check leaves may stop one kind and say nothing to another.
|
|
1402
|
+
for (const kind of new Set(credited.map((c) => c.sink.kind))) {
|
|
1403
|
+
for (const c of refuse(start, reach, credited, 1, kind)) {
|
|
1404
|
+
const g = c.finding.guard;
|
|
1405
|
+
delete c.finding.guard;
|
|
1406
|
+
c.finding.checkElsewhere = { program: g.program, item: g.item, file: g.file, line: g.line };
|
|
1407
|
+
c.level = 0;
|
|
1408
|
+
}
|
|
1409
|
+
}
|
|
1410
|
+
for (const kind of new Set(credited.filter((c) => c.level === 2).map((c) => c.sink.kind))) {
|
|
1411
|
+
for (const c of refuse(start, reach, credited, 2, kind)) if (c.level === 2) { delete c.finding.guard.stops; c.level = 1; }
|
|
1412
|
+
}
|
|
1413
|
+
}
|
|
1414
|
+
}
|
|
1415
|
+
}
|
|
1416
|
+
|
|
1417
|
+
// One use per source, table and index: the least checked a route reaches, then the first.
|
|
1418
|
+
if (useGroup.size) {
|
|
1419
|
+
const rank = (f) => (f.unreached ? 3 : f.guard ? (f.guard.stops ? 2 : 1) : 0);
|
|
1420
|
+
const kept = new Map();
|
|
1421
|
+
for (const [f, key] of useGroup) {
|
|
1422
|
+
const k = kept.get(key);
|
|
1423
|
+
if (!k || rank(f) < rank(k) || (rank(f) === rank(k) && (byText(f.sink.file, k.sink.file) || f.sink.line - k.sink.line) < 0)) kept.set(key, f);
|
|
1424
|
+
}
|
|
1425
|
+
const keep = new Set(kept.values());
|
|
1426
|
+
let w = 0;
|
|
1427
|
+
for (const f of findings) if (!useGroup.has(f) || keep.has(f)) findings[w++] = f;
|
|
1428
|
+
findings.length = w;
|
|
1429
|
+
}
|
|
1430
|
+
findings.sort((a, b) => byText(a.rule, b.rule) || byText(a.source.file, b.source.file) || a.source.line - b.source.line || a.sink.line - b.sink.line);
|
|
1431
|
+
// The counts say how much was missed; these say what. A flow result over a tree where eleven
|
|
1432
|
+
// programs would not parse is a different claim from one over a tree where all of them did, and
|
|
1433
|
+
// the reader cannot tell which without the names.
|
|
1434
|
+
if (unread.length) stats.unreadableFiles = unread;
|
|
1435
|
+
if (unparsed.length) stats.unparsedFiles = unparsed;
|
|
1436
|
+
// Every sink the graph holds, reached or not, once per site and operand. A labelling worksheet
|
|
1437
|
+
// built from findings alone could measure precision and never recall.
|
|
1438
|
+
const sinks = opts.listSinks ? [] : null;
|
|
1439
|
+
if (sinks) {
|
|
1440
|
+
const listed = new Set();
|
|
1441
|
+
for (const n of nodes) for (const s of n.sinks) {
|
|
1442
|
+
const at = { program: n.program, programFile: rel(n.file), file: rel(s.file), line: s.line, kind: s.kind, item: n.name, detail: s.detail, ...(s.onlyFrom ? { onlyFrom: s.onlyFrom } : {}) };
|
|
1443
|
+
const key = `${at.programFile}|${at.program}|${at.file}|${at.line}|${at.kind}|${at.item}`;
|
|
1444
|
+
if (!listed.has(key)) { listed.add(key); sinks.push(at); }
|
|
1445
|
+
}
|
|
1446
|
+
}
|
|
1447
|
+
const sources = opts.listSources ? [] : null;
|
|
1448
|
+
if (sources) {
|
|
1449
|
+
const listed = new Set();
|
|
1450
|
+
for (const n of nodes) for (const s of n.sources) {
|
|
1451
|
+
if (!s.file) continue;
|
|
1452
|
+
const at = { program: n.program, programFile: rel(n.file), file: rel(s.file), line: s.line, kind: s.kind, item: n.name };
|
|
1453
|
+
const key = `${at.programFile}|${at.program}|${at.file}|${at.line}|${at.kind}|${at.item}`;
|
|
1454
|
+
if (!listed.has(key)) { listed.add(key); sources.push(at); }
|
|
1455
|
+
}
|
|
1456
|
+
}
|
|
1457
|
+
for (const c of constructs) { Object.assign(c, startedByOf(c.pk)); delete c.pk; }
|
|
1458
|
+
stats.entryPoints = { transactions: csd.transactions.size, jobSteps: jobSteps.length, roots: roots.length, programsStarted: programs.filter((p) => p.startedBy).length, programsWithoutEntry: unstarted.length };
|
|
1459
|
+
const startedBy = {};
|
|
1460
|
+
for (const p of programs) if (p.id && p.startedBy && !startedBy[p.id]) startedBy[p.id] = p.startedBy;
|
|
1461
|
+
return { findings, constructs, unstarted, startedBy, stats, sourceKinds: SOURCE_KINDS, sinkKinds: SINK_KINDS, ...(sinks ? { sinks } : {}), ...(sources ? { sources } : {}) };
|
|
1462
|
+
}
|
|
1463
|
+
|
|
1464
|
+
// What the size check across a CALL needs from an item: its bytes, the bytes from it to the end of
|
|
1465
|
+
// the record it sits in, whether a table inside it is variable, and where it was declared.
|
|
1466
|
+
function extentOf(item) {
|
|
1467
|
+
if (!item || item.size == null) return null;
|
|
1468
|
+
let top = item;
|
|
1469
|
+
while (top.parent) top = top.parent;
|
|
1470
|
+
const room = top.size != null && item.offset != null ? top.size - (item.offset - (top.offset ?? 0)) : item.size;
|
|
1471
|
+
return { name: item.name, bytes: item.size, room, variable: hasVariableTable(item), file: item.file, line: item.line };
|
|
1472
|
+
}
|
|
1473
|
+
|
|
1474
|
+
function hasVariableTable(item) {
|
|
1475
|
+
if (item.dependingOn) return true;
|
|
1476
|
+
for (const c of item.children || []) if (hasVariableTable(c)) return true;
|
|
1477
|
+
return false;
|
|
1478
|
+
}
|
|
1479
|
+
|
|
1480
|
+
if (process.argv[1] && process.argv[1].endsWith('dataflow.mjs')) {
|
|
1481
|
+
const [root, out] = process.argv.slice(2);
|
|
1482
|
+
const repos = process.env.DF_REPOS ? process.env.DF_REPOS.split(',') : (process.env.DF_FLAT ? [''] : null);
|
|
1483
|
+
const { readdirSync, writeFileSync } = await import('node:fs');
|
|
1484
|
+
const list = repos || readdirSync(root, { withFileTypes: true }).filter(d => d.isDirectory() && !d.name.startsWith('.')).map(d => d.name);
|
|
1485
|
+
const t0 = Date.now();
|
|
1486
|
+
// One graph per repository: program names are only unique within a repository, so a single
|
|
1487
|
+
// corpus-wide graph would link a CALL in one repository to a same-named program in another.
|
|
1488
|
+
const res = { findings: [], stats: { files: 0, programs: 0, edges: 0, nodes: 0, threw: 0 }, perRepo: {} };
|
|
1489
|
+
for (const repo of list) {
|
|
1490
|
+
const one = analyze(root, { repos: [repo] });
|
|
1491
|
+
for (const k of Object.keys(res.stats)) res.stats[k] += one.stats[k];
|
|
1492
|
+
res.findings.push(...one.findings);
|
|
1493
|
+
res.perRepo[repo] = { files: one.stats.files, programs: one.stats.programs, findings: one.findings.length };
|
|
1494
|
+
if (process.env.DF_PROGRESS) console.error(`${repo} files=${one.stats.files} findings=${one.findings.length}`);
|
|
1495
|
+
}
|
|
1496
|
+
res.sourceKinds = SOURCE_KINDS;
|
|
1497
|
+
res.sinkKinds = SINK_KINDS;
|
|
1498
|
+
res.secs = Math.round((Date.now() - t0) / 1000);
|
|
1499
|
+
if (out) writeFileSync(out, JSON.stringify(res, null, 1));
|
|
1500
|
+
const byRule = {};
|
|
1501
|
+
for (const f of res.findings) byRule[f.rule] = (byRule[f.rule] || 0) + 1;
|
|
1502
|
+
const cross = res.findings.filter(f => f.crossProgram).length;
|
|
1503
|
+
console.log(`files=${res.stats.files} programs=${res.stats.programs} nodes=${res.stats.nodes} edges=${res.stats.edges} secs=${res.secs}`);
|
|
1504
|
+
console.log(`findings=${res.findings.length} crossProgram=${cross}`);
|
|
1505
|
+
console.log(JSON.stringify(byRule, null, 1));
|
|
1506
|
+
}
|