@portll/cobolwork 0.2.117 → 0.2.150

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +13 -1
  2. package/THIRD-PARTY-NOTICES.md +3 -2
  3. package/bin/cobolwork.mjs +16 -1
  4. package/lib/baseline.mjs +8 -6
  5. package/lib/bms.mjs +1 -1
  6. package/lib/capabilities.mjs +2 -1
  7. package/lib/cards.mjs +66 -0
  8. package/lib/cics-commands.mjs +136 -125
  9. package/lib/compliance.mjs +6 -5
  10. package/lib/control.mjs +295 -6
  11. package/lib/dataflow.mjs +72 -88
  12. package/lib/diff.mjs +65 -89
  13. package/lib/embedded-sql.mjs +110 -0
  14. package/lib/equivalence.mjs +30 -12
  15. package/lib/evidence/journal.mjs +11 -2
  16. package/lib/evidence/record.mjs +3 -1
  17. package/lib/evidence/slsa.mjs +17 -2
  18. package/lib/execution.mjs +77 -0
  19. package/lib/explain.mjs +5 -4
  20. package/lib/ftp.mjs +136 -0
  21. package/lib/inventory.mjs +8 -9
  22. package/lib/ironwork.mjs +4 -6
  23. package/lib/jcl.mjs +29 -54
  24. package/lib/kernel/git.mjs +60 -0
  25. package/lib/kernel/identity.mjs +7 -6
  26. package/lib/kernel/source-tree.mjs +204 -34
  27. package/lib/layout.mjs +202 -0
  28. package/lib/parser.mjs +58 -207
  29. package/lib/precompile.mjs +5 -88
  30. package/lib/revision.json +1 -1
  31. package/lib/sbom.mjs +125 -13
  32. package/lib/scan.mjs +11 -4
  33. package/lib/sets/build.mjs +1 -1
  34. package/lib/sets/cics.mjs +32 -5
  35. package/lib/sets/copybook.mjs +2 -1
  36. package/lib/sets/hidden.mjs +4 -8
  37. package/lib/sets/jcl.mjs +17 -120
  38. package/lib/sets/priv.mjs +1 -1
  39. package/lib/sets/recon.mjs +3 -2
  40. package/lib/sets/semantics.mjs +1 -1
  41. package/lib/sets/vendor.mjs +1 -1
  42. package/lib/sets/zowe.mjs +34 -8
  43. package/lib/site.mjs +17 -7
  44. package/lib/sources.mjs +82 -38
  45. package/lib/utilities.mjs +146 -4
  46. package/package.json +1 -1
  47. package/rules/compliance-cobit2019.json +2438 -0
  48. package/rules/compliance-dora.json +1 -1
  49. package/rules/compliance-ffiec.json +1 -1
  50. package/rules/compliance-nist80053.json +1 -1
  51. package/schema/cobolwork.policy.schema.json +94 -17
package/lib/dataflow.mjs CHANGED
@@ -1,9 +1,12 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
- import { inScope, isProgram, isBms, readSource, relPath } from './sources.mjs';
2
+ import { inScope, isProgram, isBms, relPath } from './sources.mjs';
3
3
  import { parseBms, symbolicNames } from './bms.mjs';
4
4
  import { dirname, join } from 'node:path';
5
5
  import { parseSource, buildFileIndex } from './parser.mjs';
6
6
  import { parseJcl } from './jcl.mjs';
7
+ import { holdsDecimal } from './layout.mjs';
8
+ import { hostVariablesIn } from './embedded-sql.mjs';
9
+ import { tsoCommands } from './utilities.mjs';
7
10
  import { parseCsd, ddOfQueue } from './csd.mjs';
8
11
  import { loadSite } from './site.mjs';
9
12
  import { ssrangeAbends } from './options.mjs';
@@ -68,6 +71,7 @@ STANDARD-DEVIATION RANDOM SQRT FACTORIAL LOG LOG10 EXP EXP10 PI E ACOS ASIN ATAN
68
71
  SECONDS-PAST-MIDNIGHT`.split(/\s+/));
69
72
  const computes = (st) => ARITHMETIC.has(st.verb) || st.counts === true || (st.fns || []).some((f) => NUMERIC_FUNCTIONS.has(f));
70
73
  const computedOnRoute = (state) => { for (let s = state; s; s = s.prev) if (s.why && s.why.computes) return true; return false; };
74
+ const typedOnRoute = (state) => { for (let s = state; s; s = s.prev) if (s.node.typed) return true; return false; };
71
75
 
72
76
  // Every literal a program holds that could be a program name: its VALUE clauses and the literals its
73
77
  // statements and commands use. A menu that XCTLs through a table of names starts every one of them.
@@ -89,15 +93,6 @@ function lastOption(name, list) {
89
93
  return null;
90
94
  }
91
95
 
92
- // Whether an item's bytes are read as decimal digits: zoned (DISPLAY with a numeric picture) or
93
- // packed. A group, an edited picture, binary and floating point are not.
94
- function holdsDecimal(it) {
95
- if ((it.children || []).some((c) => c.level !== 88)) return false;
96
- const usage = String(it.effectiveUsage || 'DISPLAY').replace('COMPUTATIONAL', 'COMP');
97
- if (usage === 'COMP-3' || usage === 'PACKED-DECIMAL' || usage === 'COMP-6') return true;
98
- return usage === 'DISPLAY' && !!it.picture && /9/.test(it.picture) && /^[S9VP()0-9]+$/i.test(it.picture);
99
- }
100
-
101
96
  const RELATION = new Set(['>', '<', '>=', '<=', '=', '<>']);
102
97
  const RELATION_WORDS = new Set(['GREATER', 'LESS', 'EQUAL']);
103
98
  const FILLER_WORDS = new Set(['IS', 'NOT', 'THAN', 'TO', 'OR', 'GREATER', 'LESS', 'EQUAL', 'OF', 'IN']);
@@ -178,27 +173,6 @@ function execOptions(exec) {
178
173
  return { opts, words };
179
174
  }
180
175
 
181
- function sqlHostVars(exec) {
182
- const out = [];
183
- for (let i = 0; i < exec.toks.length; i++) {
184
- const t = exec.toks[i];
185
- if (t.t === 'op' && t.v === ':' && exec.toks[i + 1] && exec.toks[i + 1].t === 'word') out.push({ tok: exec.toks[i + 1], at: i + 1 });
186
- }
187
- return out;
188
- }
189
-
190
- // The host variables a SELECT or FETCH fills are the ones in its INTO list, which ends at FROM.
191
- // Taking every variable at or after the INTO line made the input parameters in a WHERE clause read
192
- // as database values — and they are whatever the caller sent, under a rule that says data at rest.
193
- function sqlIntoRange(exec) {
194
- const words = exec.toks.map(t => (t.t === 'word' ? t.u : ''));
195
- const into = words.indexOf('INTO');
196
- if (into < 0) return null;
197
- let end = exec.toks.length;
198
- for (let i = into + 1; i < words.length; i++) if (words[i] === 'FROM') { end = i; break; }
199
- return [into, end];
200
- }
201
-
202
176
  export function analyze(root, opts = {}) {
203
177
  const repos = opts.repos || [''];
204
178
  const stats = { files: 0, programs: 0, edges: 0, nodes: 0, threw: 0, overBudget: 0, unreadable: 0, ebcdic: 0,
@@ -223,6 +197,7 @@ export function analyze(root, opts = {}) {
223
197
  const nodes = [];
224
198
  const programs = [];
225
199
  const jclPending = [];
200
+ const jclTrees = new Map();
226
201
  // CICS definitions, from CSD extracts and from DFHCSDUP job input, and the estate's own statement
227
202
  // of which DDs and queues reach the internal reader.
228
203
  const csd = { tdqueues: new Map(), transactions: new Map(), urimaps: new Map() };
@@ -235,7 +210,7 @@ export function analyze(root, opts = {}) {
235
210
  const jobSteps = [];
236
211
  // The maps of the repository being read, by MAPSET/MAP and by MAP alone.
237
212
  const bmsMaps = new Map();
238
- const site = loadSite(root, opts.site || null);
213
+ const site = loadSite(root, opts.site || null, opts.tree);
239
214
  const relOf = new Map();
240
215
  const rel = (p) => { let r = relOf.get(p); if (r === undefined) { r = relPath(root, p); relOf.set(p, r); } return r; };
241
216
  // `at` is the statement an edge that is not a statement's own is read at: the CALL or LINK that
@@ -457,11 +432,14 @@ export function analyze(root, opts = {}) {
457
432
  usesOf.get(at.name).push(rec);
458
433
  }
459
434
  const offset = x.offset || 0;
435
+ // A constant length past the start, or a constant start before the length, takes that many bytes off the top.
436
+ const base = limitOf(host, table);
437
+ const limit = kind === 'reference-modification' && base != null && x.span ? base - x.span + 1 : base;
460
438
  const use = `${key}|${point ?? `s${st.at}`}|${offset}`;
461
439
  if (indexed.has(use)) continue;
462
440
  indexed.add(use);
463
441
  at.sinks.push({
464
- kind, onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point, group: `${pk}|${key}`, limit: limitOf(host, table), ...(offset ? { offset } : {}), ...(offset > 0 && idx && unsignedWhole(idx) ? { nonNeg: true } : {}), ...(ssrange ? { ssrange } : {}),
442
+ kind, onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point, group: `${pk}|${key}`, limit, ...(offset ? { offset } : {}), ...(offset > 0 && idx && unsignedWhole(idx) ? { nonNeg: true } : {}), ...(ssrange ? { ssrange } : {}),
465
443
  detail: kind === 'subscript'
466
444
  ? `${at.name} subscripts ${host.name}, a table of ${table.occurs}`
467
445
  : `${at.name} sets the ${x.kind === 'refmod-offset' ? 'start' : 'length'} of a reference to ${host.name}, which is ${host.size} bytes`,
@@ -724,14 +702,16 @@ export function analyze(root, opts = {}) {
724
702
  walk(record);
725
703
  for (const field of map.fields) {
726
704
  if (!field.name) continue;
727
- const marks = ['PROT', 'ASKIP', 'DRK', 'NUM'].filter((a) => field.effective.has(a));
728
- if (!marks.length) continue;
729
705
  const iname = symbolicNames(map, field).find((n) => n.suffix === 'I');
730
706
  const item = iname && within.get(iname.name);
731
707
  if (!item) continue;
708
+ const marks = ['PROT', 'ASKIP', 'DRK', 'NUM'].filter((a) => field.effective.has(a));
709
+ const typed = !marks.includes('PROT') && !marks.includes('ASKIP');
710
+ if (typed) nodeOf(item).typed = true;
711
+ if (!marks.length) continue;
732
712
  const n = nodeOf(item);
733
713
  n.screen = { item: item.name, field: field.name, map: map.name, mapset: mapset.name, marks, declared: !!field.attrb };
734
- if (!marks.includes('PROT') && !marks.includes('ASKIP')) continue;
714
+ if (typed) continue;
735
715
  stats.protectedFields++;
736
716
  const back = field.effective.has('FSET') ? ' with FSET' : '';
737
717
  n.sources.push({ kind: 'cics-protected-field', onlyTo: ['record-key', 'record-update'], ...at,
@@ -921,15 +901,14 @@ export function analyze(root, opts = {}) {
921
901
  }
922
902
  }
923
903
  if (e.kind === 'SQL') {
924
- // The parser's host variables name the item a qualified :G.F resolves to, where the first
925
- // word after the colon would name the group G and so every field in it.
926
- const hv = e.hostVariables || sqlHostVars(e);
904
+ // A host variable names the item a qualified :G.F resolves to; the statement fills the ones
905
+ // Db2 writes, and those hold what the database held.
906
+ const hv = e.hostVariables || hostVariablesIn(e.toks);
927
907
  const verb = words[0];
928
- const range = e.hostVariables ? null : ['SELECT', 'FETCH'].includes(verb) && sqlIntoRange(e);
929
908
  for (const h of hv) {
930
- if (e.hostVariables ? !h.written : !range || h.at < range[0] || h.at > range[1]) continue;
909
+ if (!h.written) continue;
931
910
  const n = nodeOfToken(h.tok);
932
- if (n) n.sources.push({ kind: 'database', ...at, detail: `EXEC SQL ${verb} INTO host variable` });
911
+ if (n) n.sources.push({ kind: 'database', ...at, detail: `EXEC SQL ${verb} ${verb === 'SELECT' || verb === 'FETCH' ? 'INTO' : 'writes'} host variable` });
933
912
  }
934
913
  // The location CONNECT TO or SET CONNECTION names, where a host variable holds it; what
935
914
  // follows USER and USING is the credential, not the target.
@@ -981,7 +960,7 @@ export function analyze(root, opts = {}) {
981
960
  const tree = (!repo && opts.tree) ? opts.tree : directoryTree(base, opts);
982
961
  const idx = tree.index;
983
962
  const files = tree.list().filter(isProgram).filter(inScope(opts));
984
- for (const f of tree.list().filter(isJcl).filter(inScope(opts))) jclPending.push(f);
963
+ for (const f of tree.list().filter(isJcl).filter(inScope(opts))) { jclPending.push(f); jclTrees.set(f, tree); }
985
964
  for (const f of tree.list().filter((p) => /\.csd$/i.test(p)).filter(inScope(opts))) {
986
965
  try { addCsd(tree.text(f).text, f); } catch (e) { stats.unreadable++; unread.push(`${tree.rel(f)}: ${e.code || e.name}`); }
987
966
  }
@@ -1110,7 +1089,7 @@ export function analyze(root, opts = {}) {
1110
1089
  // ACCEPT FROM COMMAND-LINE is untrusted.
1111
1090
  for (const f of jclPending) {
1112
1091
  let src;
1113
- try { src = readSource(f).text; } catch (e) { stats.unreadable++; unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
1092
+ try { src = jclTrees.get(f).text(f).text; } catch (e) { stats.unreadable++; unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
1114
1093
  let job;
1115
1094
  try { job = parseJcl(src, f); } catch (e) { stats.threw++; unparsed.push(`${rel(f)}: ${e.code || e.name}`); continue; }
1116
1095
  stats.jclFiles++;
@@ -1123,59 +1102,61 @@ export function analyze(root, opts = {}) {
1123
1102
  for (const dd of step.dds) if (dd.name && dd.name.toUpperCase() === 'SYSIN' && dd.inStream) addCsd(dd.inStream.map((l) => l.text).join('\n'), f, dd.inStream[0].line - 1);
1124
1103
  }
1125
1104
  jobSteps.push({ file: f, job: job.jobs[0]?.name || null, step: step.name, line: step.line, pgm: step.pgm.toUpperCase() });
1126
- // A TSO batch step starts a program by name in its commands: DSN ... RUN PROGRAM(X).
1127
- if (/^IKJEFT(01|1A|1B)$/i.test(step.pgm)) {
1128
- for (const dd of step.dds) for (const l of dd.inStream || []) {
1129
- const run = /\bRUN\s+PROGRAM\s*\(\s*([A-Z0-9$#@]{1,8})\s*\)/i.exec(l.text);
1130
- if (run) jobSteps.push({ file: f, job: job.jobs[0]?.name || null, step: step.name, line: l.line, pgm: run[1].toUpperCase() });
1131
- }
1105
+ // A TSO batch step starts programs by name in its commands, TSO CALL and DSN RUN PROGRAM, and
1106
+ // each is handed the step's DDs and its own parameter as EXEC PGM= would be.
1107
+ const runs = [{ pgm: step.pgm.toUpperCase(), parm: step.parm, line: step.line, how: `PARM= on step ${step.name || '(unnamed)'} of ${rel(f)}` }];
1108
+ for (const r of tsoCommands(step).runs) {
1109
+ jobSteps.push({ file: f, job: job.jobs[0]?.name || null, step: step.name, line: r.line, pgm: r.program });
1110
+ runs.push({ pgm: r.program, parm: r.parm, line: r.line, how: `the parameter ${r.via} passes in step ${step.name || '(unnamed)'} of ${rel(f)}` });
1132
1111
  }
1133
- const holders = allById.get(step.pgm.toUpperCase()) || [];
1134
- if (!holders.length) continue; // a system utility, or a program not in this tree
1135
- stats.jclStepsResolved++;
1136
- // A program id held several times has no caller file to choose by, so each holder is a candidate.
1137
- const amb = holders.length > 1 ? `, any of ${holders.length} programs named ${step.pgm}` : '';
1138
- for (const callee of holders) {
1112
+ for (const run of runs) {
1113
+ const holders = allById.get(run.pgm) || [];
1114
+ if (!holders.length) continue; // a system utility, or a program not in this tree
1115
+ stats.jclStepsResolved++;
1116
+ // A program id held several times has no caller file to choose by, so each holder is a candidate.
1117
+ const amb = holders.length > 1 ? `, any of ${holders.length} programs named ${run.pgm}` : '';
1118
+ for (const callee of holders) {
1139
1119
 
1140
- // PARM arrives in the first PROCEDURE DIVISION USING item: on z/OS a halfword length
1141
- // followed by the text. A program with no USING cannot receive one, and saying it does
1142
- // would be a path nobody could follow.
1143
- if (step.parm !== null && callee.params && callee.params[0] && callee.params[0].node) {
1144
- callee.params[0].node.sources.push({
1145
- kind: 'jcl-parm', file: f, line: step.line,
1146
- detail: `PARM= on step ${step.name || '(unnamed)'} of ${rel(f)}, which runs ${step.pgm}${amb}`,
1147
- });
1148
- stats.jclCrossings++;
1149
- }
1120
+ // PARM arrives in the first PROCEDURE DIVISION USING item: on z/OS a halfword length
1121
+ // followed by the text. A program with no USING cannot receive one, and saying it does
1122
+ // would be a path nobody could follow.
1123
+ if (run.parm !== null && callee.params && callee.params[0] && callee.params[0].node) {
1124
+ callee.params[0].node.sources.push({
1125
+ kind: 'jcl-parm', file: f, line: run.line,
1126
+ detail: `${run.how}, which runs ${run.pgm}${amb}`,
1127
+ });
1128
+ stats.jclCrossings++;
1129
+ }
1150
1130
 
1151
- // In-stream data reaches whatever record the program reads from that DD. The COBOL says
1152
- // ASSIGN TO SYSIN and the job says //SYSIN DD *; neither half names the other, and the
1153
- // join is the whole point of reading both.
1154
- for (const dd of step.dds) {
1155
- // The same join in the other direction: what the program writes through SELECT ... ASSIGN
1156
- // TO a DD the job sends to the internal reader is submitted as a job.
1157
- if (dd.name && dd.sysout && /\bINTRDR\b/i.test(dd.sysout)) {
1131
+ // In-stream data reaches whatever record the program reads from that DD. The COBOL says
1132
+ // ASSIGN TO SYSIN and the job says //SYSIN DD *; neither half names the other, and the
1133
+ // join is the whole point of reading both.
1134
+ for (const dd of step.dds) {
1135
+ // The same join in the other direction: what the program writes through SELECT ... ASSIGN
1136
+ // TO a DD the job sends to the internal reader is submitted as a job.
1137
+ if (dd.name && dd.sysout && /\bINTRDR\b/i.test(dd.sysout)) {
1138
+ for (const m of callee.ddFiles || []) {
1139
+ if (m.dd !== dd.name.toUpperCase()) continue;
1140
+ for (const n of m.nodes) {
1141
+ n.sinks.push({ kind: 'internal-reader', file: m.file, line: m.line,
1142
+ detail: `records written through SELECT ${m.select} to //${dd.name}, which step ${step.name || '(unnamed)'} of ${rel(f)} sends to the internal reader${amb}` });
1143
+ }
1144
+ }
1145
+ }
1146
+ if (!dd.inStream || !dd.inStream.length || !dd.name) continue;
1158
1147
  for (const m of callee.ddFiles || []) {
1159
1148
  if (m.dd !== dd.name.toUpperCase()) continue;
1160
1149
  for (const n of m.nodes) {
1161
- n.sinks.push({ kind: 'internal-reader', file: m.file, line: m.line,
1162
- detail: `records written through SELECT ${m.select} to //${dd.name}, which step ${step.name || '(unnamed)'} of ${rel(f)} sends to the internal reader${amb}` });
1150
+ n.sources.push({
1151
+ kind: 'jcl-instream', file: f, line: dd.line,
1152
+ detail: `${dd.inStream.length} lines of in-stream data on //${dd.name} in step ${step.name || '(unnamed)'}, read through SELECT ${m.select}${amb}`,
1153
+ });
1154
+ stats.jclCrossings++;
1163
1155
  }
1164
1156
  }
1165
- }
1166
- if (!dd.inStream || !dd.inStream.length || !dd.name) continue;
1167
- for (const m of callee.ddFiles || []) {
1168
- if (m.dd !== dd.name.toUpperCase()) continue;
1169
- for (const n of m.nodes) {
1170
- n.sources.push({
1171
- kind: 'jcl-instream', file: f, line: dd.line,
1172
- detail: `${dd.inStream.length} lines of in-stream data on //${dd.name} in step ${step.name || '(unnamed)'}, read through SELECT ${m.select}${amb}`,
1173
- });
1174
- stats.jclCrossings++;
1175
- }
1176
- }
1177
1157
 
1178
- }
1158
+ }
1159
+ }
1179
1160
  }
1180
1161
  }
1181
1162
  }
@@ -1494,6 +1475,9 @@ export function analyze(root, opts = {}) {
1494
1475
  // arithmetic or a numeric function carries no bad bytes to the next one. Only this route is
1495
1476
  // dropped: a later one that copies the bytes still reports.
1496
1477
  if (sink.kind === 'arithmetic' && computedOnRoute(state)) continue;
1478
+ // A protected field's value carried through a field the terminal may type into reaches the
1479
+ // key as typed input, which the user could have entered anyway: the protection was no control.
1480
+ if (src.kind === 'cics-protected-field' && typedOnRoute(state)) continue;
1497
1481
  seen.add(key);
1498
1482
  // The ends of a long path are what a reader uses: where the value came from, and what it
1499
1483
  // reached. Keeping every hop of a thousand-hop chain, for a thousand findings, is the
package/lib/diff.mjs CHANGED
@@ -1,16 +1,15 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
2
  import { existsSync, mkdirSync, mkdtempSync, realpathSync, rmSync, writeFileSync } from 'node:fs';
3
- import { spawnSync } from 'node:child_process';
4
3
  import { tmpdir } from 'node:os';
5
- import { dirname, join, resolve, sep } from 'node:path';
6
- import { parseSource } from './parser.mjs';
7
- import { inScope, isProgram, readSource, relPath } from './sources.mjs';
4
+ import { dirname, join, resolve } from 'node:path';
5
+ import { git, revisionBlobs, treeOf, treePathParts } from './kernel/git.mjs';
6
+ import { inScope, isProgram } from './sources.mjs';
8
7
  import { scanAll } from './scan.mjs';
9
8
  import { FLOW_MODEL, SCHEMA_VERSION, TOOL_VERSION } from './version.mjs';
10
9
  // A diff finding is located by program rather than by line, so it orders on its own key. The text
11
10
  // comparison underneath it is the shared one.
12
11
  import { byText, evidenceMap } from './kernel/findings.mjs';
13
- import { directoryTree } from './kernel/source-tree.mjs';
12
+ import { directoryTree, gitTree } from './kernel/source-tree.mjs';
14
13
  import { printable } from './kernel/printable.mjs';
15
14
  import { loadSite, SITE_FILE } from './site.mjs';
16
15
  import { loadBaseline, BASELINE_FILE, SUPPRESSING } from './baseline.mjs';
@@ -30,56 +29,18 @@ export const DIFF_RULES = {
30
29
 
31
30
  const MAX_LISTED = 8;
32
31
 
33
- // A reviewed repository's .git/config can name commands; with fsmonitor off and plumbing only, git runs none.
34
- const GIT_ENV = { ...process.env, GIT_OPTIONAL_LOCKS: '0', GIT_TERMINAL_PROMPT: '0' };
35
- const git = (repo, args, opts = {}) => spawnSync('git', ['-c', 'core.fsmonitor=false', '-C', repo, ...args], { env: GIT_ENV, maxBuffer: 256 * 1024 * 1024, ...opts });
36
- const refused = (what, ref, why) => Object.assign(new Error(`${what} ${ref} failed: ${why}`), { code: 'EDIFFREF' });
37
-
38
- function treeOf(repo, ref) {
39
- if (!ref || String(ref).startsWith('-')) throw refused('git rev-parse', ref, 'a revision cannot be empty or start with "-"');
40
- const r = git(repo, ['rev-parse', '--verify', '--quiet', '--end-of-options', `${ref}^{tree}`], { encoding: 'utf8' });
41
- const oid = String(r.stdout || '').trim();
42
- if (r.status !== 0 || !/^[0-9a-f]{40}([0-9a-f]{24})?$/.test(oid)) throw refused('git rev-parse', ref, String(r.stderr || '').trim() || 'not a revision');
43
- return oid;
44
- }
45
-
46
- // A tree path as a file under `dir`, or null if a part is empty, climbs, names a drive or stream, or is .git.
47
- function placeIn(dir, path) {
48
- const parts = path.split('/');
49
- if (parts.some((p) => !p || p === '.' || p === '..' || /[\\:\0]/.test(p) || p.toLowerCase() === '.git')) return null;
50
- const to = resolve(dir, ...parts);
51
- return to.startsWith(dir + sep) ? to : null;
52
- }
53
-
54
- // The revision as committed: blobs as stored, with no filter or line-ending conversion; links and submodules left out.
32
+ // The revision as committed, written into a temporary directory, for a caller that hands the tree
33
+ // to a program outside this process: build's compiler and ironwork read files, not blobs.
55
34
  function materialise(repo, ref) {
56
- const tree = treeOf(repo, ref);
57
- const listed = git(repo, ['ls-tree', '-r', '-z', '--full-tree', tree]);
58
- if (listed.status !== 0) throw refused('git ls-tree', ref, String(listed.stderr || '').trim() || `exit ${listed.status}`);
59
- const blobs = [];
60
- for (const entry of listed.stdout.toString('utf8').split('\0')) {
61
- const m = /^(\d{6}) blob ([0-9a-f]+)\t(.+)$/s.exec(entry);
62
- if (m && m[1] !== '120000') blobs.push({ oid: m[2], path: m[3] });
63
- }
35
+ const { blobs } = revisionBlobs(repo, treeOf(repo, ref), ref);
64
36
  const dir = realpathSync(mkdtempSync(join(tmpdir(), 'cobolwork-diff-')));
65
37
  try {
66
- for (let i = 0; i < blobs.length; i += 500) {
67
- const batch = blobs.slice(i, i + 500);
68
- const r = git(repo, ['cat-file', '--batch'], { input: batch.map((b) => b.oid).join('\n') + '\n' });
69
- if (r.status !== 0) throw refused('git cat-file', ref, String(r.stderr || '').trim() || `exit ${r.status}`);
70
- let at = 0;
71
- for (const b of batch) {
72
- const nl = r.stdout.indexOf(0x0a, at);
73
- const head = r.stdout.toString('utf8', at, nl).split(' ');
74
- if (head[1] !== 'blob') throw refused('git cat-file', ref, `${b.oid} is ${head[1] || 'missing'}`);
75
- const size = Number(head[2]);
76
- const to = placeIn(dir, b.path);
77
- if (to) {
78
- mkdirSync(dirname(to), { recursive: true });
79
- writeFileSync(to, r.stdout.subarray(nl + 1, nl + 1 + size));
80
- }
81
- at = nl + 1 + size + 1;
82
- }
38
+ for (const b of blobs) {
39
+ const parts = treePathParts(b.path);
40
+ if (!parts) continue;
41
+ const to = resolve(dir, ...parts);
42
+ mkdirSync(dirname(to), { recursive: true });
43
+ writeFileSync(to, b.bytes);
83
44
  }
84
45
  } catch (e) {
85
46
  rmSync(dir, { recursive: true, force: true });
@@ -102,9 +63,9 @@ const qualified = (it) => { const names = []; for (let x = it; x; x = x.parent)
102
63
 
103
64
  // A program that cannot be read or parsed is named in `unread`, never dropped: dropped, it would
104
65
  // read as removed from the tree, and every change inside it would go unreported.
105
- function facts(root, opts = {}) {
106
- // Each side of a diff is its own tree: the base is a revision materialised into a temporary
107
- // directory, the head is the working tree, and they are walked separately.
66
+ function facts(side, opts = {}) {
67
+ // Each side of a diff is its own tree: a revision read out of git, or the working tree, and they
68
+ // are walked separately.
108
69
  //
109
70
  // This used to build its own parse options with systemDirs: [], where every other caller passed
110
71
  // the caller's. Nobody decided that - it is what five copies of one object literal do. The
@@ -112,24 +73,24 @@ function facts(root, opts = {}) {
112
73
  // change inside one was invisible to the command whose whole job is what a change reaches. Both
113
74
  // sides now take the same configuration, which for a caller that passes no systemDirs is
114
75
  // exactly the old behaviour.
115
- const tree = directoryTree(root, opts);
116
- const idx = tree.index;
76
+ const tree = side.tree || directoryTree(side.root, opts);
77
+ const rel = (p) => tree.rel(p);
117
78
  const out = new Map();
118
79
  out.unread = [];
119
80
  for (const f of tree.list().filter(isProgram).filter(inScope(opts))) {
120
81
  let src;
121
- try { src = readSource(f).text; } catch (e) { out.unread.push(`${relPath(root, f)}: ${e.code || e.name}`); continue; }
82
+ try { src = tree.text(f).text; } catch (e) { out.unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
122
83
  if (!/PROCEDURE\s+DIVISION|PROGRAM-ID/i.test(src)) continue;
123
84
  let r;
124
85
  try {
125
86
  r = tree.parse(f, src);
126
- } catch (e) { out.unread.push(`${relPath(root, f)}: ${e.code || e.name}`); continue; }
127
- const file = relPath(root, f);
87
+ } catch (e) { out.unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
88
+ const file = rel(f);
128
89
  for (const p of r.programs) {
129
90
  const items = new Map();
130
91
  for (const it of p.items) {
131
92
  if (!it.name || it.name === 'FILLER' || it.level === 88 || it.level === 66 || it.level === 78) continue;
132
- items.set(qualified(it), { size: it.size, offset: it.offset, from: it.file ? relPath(root, it.file) : file, section: it.section, level: it.level });
93
+ items.set(qualified(it), { size: it.size, offset: it.offset, from: it.file ? rel(it.file) : file, section: it.section, level: it.level });
133
94
  }
134
95
  const calls = new Set();
135
96
  const dynamic = new Set();
@@ -149,7 +110,7 @@ function facts(root, opts = {}) {
149
110
  if (arg) commareas.add(arg.u);
150
111
  }
151
112
  const using = new Set(p.calls.flatMap(c => c.using.filter(a => a.word).map(a => a.word)));
152
- const copied = new Set(r.copies.filter((c) => c.status === 'resolved' && c.path).map((c) => relPath(root, c.path)));
113
+ const copied = new Set(r.copies.filter((c) => c.status === 'resolved' && c.path).map((c) => rel(c.path)));
153
114
  out.set(`${file}#${p.id}`, { id: p.id, file, text: src, items, calls, dynamic, commareas, using, copied });
154
115
  }
155
116
  }
@@ -168,12 +129,17 @@ function countByKey(findings) {
168
129
  return m;
169
130
  }
170
131
 
171
- export function diffTrees(baseRoot, headRoot, opts = {}) {
172
- // The allow list describes the working tree, so it never applies to a revision checked out into
173
- // a temporary directory: applied there it would match nothing and the base would read as empty.
132
+ // Each side is a directory, or a source tree: a git revision read without writing it out.
133
+ const sideOf = (x) => (typeof x === 'string' ? { root: x, tree: null } : { root: x.root, tree: x });
134
+
135
+ export function diffTrees(baseSide, headSide, opts = {}) {
136
+ const baseAt = sideOf(baseSide);
137
+ const headAt = sideOf(headSide);
138
+ // The allow list describes the working tree, so it never applies to a revision: applied there it
139
+ // would match nothing and the base would read as empty.
174
140
  const baseOpts = { ...opts, allow: null };
175
- const base = facts(baseRoot, baseOpts);
176
- const head = facts(headRoot, opts);
141
+ const base = facts(baseAt, baseOpts);
142
+ const head = facts(headAt, opts);
177
143
  const findings = [];
178
144
  const reach = { programsCompared: 0, programsAdded: 0, programsRemoved: 0, programsWithLayoutChange: 0, reachedOnlyThroughCopybooks: 0, copybooksImplicated: new Set() };
179
145
 
@@ -214,8 +180,9 @@ export function diffTrees(baseRoot, headRoot, opts = {}) {
214
180
 
215
181
  const listing = { listSinks: opts.listSinks === true, listSources: opts.listSources === true };
216
182
  const shared = { only: opts.only, fullTrace: opts.fullTrace, systemDirs: opts.systemDirs || [], ...listing };
217
- const before = scanAll(baseRoot, shared);
218
- const after = scanAll(headRoot, { ...shared, allow: opts.allow });
183
+ const withTree = (at) => (at.tree ? { tree: at.tree } : {});
184
+ const before = scanAll(baseAt.root, { ...shared, ...withTree(baseAt) });
185
+ const after = scanAll(headAt.root, { ...shared, ...withTree(headAt), allow: opts.allow });
219
186
  // A file neither tree could read cannot be compared. Its findings are neither introduced nor
220
187
  // resolved, and calling them resolved would report a scan failure as a fix.
221
188
  const unreadable = new Set([...base.unread, ...head.unread].map(u => u.split(': ')[0]));
@@ -237,7 +204,7 @@ export function diffTrees(baseRoot, headRoot, opts = {}) {
237
204
  return n > (afterCounts.get(k) || 0);
238
205
  });
239
206
  const uncomparable = [...new Set([...before.findings, ...after.findings].filter(f => !comparable(f)).map(f => f.path))];
240
- const configurationChanged = configurationChanges(baseRoot, headRoot);
207
+ const configurationChanged = configurationChanges(baseSide, headSide);
241
208
 
242
209
  for (const f of findings) { const m = DIFF_RULES[f.rule]; f.sev = m.sev; f.cwe = m.cwe; f.evidence = m.evidence; }
243
210
  findings.sort((a, b) => byText(a.rule, b.rule) || byText(a.path, b.path) || byText(String(a.program), String(b.program)));
@@ -272,13 +239,16 @@ export function diffTrees(baseRoot, headRoot, opts = {}) {
272
239
  const SITE_LISTS = ['productionQualifiers', 'productionJobPaths', 'nonProductionJobPaths', 'systemNames', 'vendorPacks',
273
240
  'internalReaderDds', 'internalReaderQueues', 'compilerOptions', 'apfLibraries', 'restrictedDatasets', 'surrogateUsers'];
274
241
 
275
- function siteChange(baseRoot, headRoot) {
276
- const was = existsSync(join(baseRoot, SITE_FILE));
277
- const now = existsSync(join(headRoot, SITE_FILE));
242
+ // Whether a side holds a file at its top: in the source tree it was read from, or on disk.
243
+ const holds = (at, name) => (at.tree && at.tree.kind !== 'directory' ? at.tree.contains(resolve(at.tree.root, name)) : existsSync(join(at.root, name)));
244
+
245
+ function siteChange(baseAt, headAt) {
246
+ const was = holds(baseAt, SITE_FILE);
247
+ const now = holds(headAt, SITE_FILE);
278
248
  if (!was && !now) return null;
279
249
  if (!now) return { file: SITE_FILE, change: 'removed, so the rules that need it do not run' };
280
- const b = loadSite(baseRoot);
281
- const h = loadSite(headRoot);
250
+ const b = loadSite(baseAt.root, null, baseAt.tree);
251
+ const h = loadSite(headAt.root, null, headAt.tree);
282
252
  const parts = was ? [] : ['added'];
283
253
  for (const k of SITE_LISTS) {
284
254
  const gone = b[k].filter((x) => !h[k].includes(x)).map((x) => `-${x}`);
@@ -291,16 +261,19 @@ function siteChange(baseRoot, headRoot) {
291
261
  return parts.length ? { file: SITE_FILE, change: printable(parts.join('; '), 400) } : null;
292
262
  }
293
263
 
294
- export const configurationChanges = (baseRoot, headRoot) =>
295
- [siteChange(baseRoot, headRoot), baselineChange(baseRoot, headRoot)].filter(Boolean);
264
+ export const configurationChanges = (baseSide, headSide) => {
265
+ const baseAt = sideOf(baseSide);
266
+ const headAt = sideOf(headSide);
267
+ return [siteChange(baseAt, headAt), baselineChange(baseAt, headAt)].filter(Boolean);
268
+ };
296
269
 
297
- function baselineChange(baseRoot, headRoot) {
298
- const was = existsSync(join(baseRoot, BASELINE_FILE));
299
- const now = existsSync(join(headRoot, BASELINE_FILE));
270
+ function baselineChange(baseAt, headAt) {
271
+ const was = holds(baseAt, BASELINE_FILE);
272
+ const now = holds(headAt, BASELINE_FILE);
300
273
  if (!was && !now) return null;
301
- const held = (root, there) => new Map((there ? loadBaseline(root)?.entries || [] : []).map((e) => [`${e.fingerprint}|${e.rule}`, e]));
302
- const b = held(baseRoot, was);
303
- const h = held(headRoot, now);
274
+ const held = (at, there) => new Map((there ? loadBaseline(at.root, { tree: at.tree })?.entries || [] : []).map((e) => [`${e.fingerprint}|${e.rule}`, e]));
275
+ const b = held(baseAt, was);
276
+ const h = held(headAt, now);
304
277
  const added = [...h.keys()].filter((k) => !b.has(k));
305
278
  const suppressing = added.filter((k) => SUPPRESSING.includes(h.get(k).action)).length;
306
279
  const removed = [...b.keys()].filter((k) => !h.has(k)).length;
@@ -334,11 +307,14 @@ export function withRefs(repo, baseRef, headRef, opts, fn) {
334
307
  }
335
308
  }
336
309
 
310
+ // Reads each revision out of git without writing it anywhere; the head is the working tree when
311
+ // headRef is null.
337
312
  export function diffRefs(repo, baseRef, headRef = null, opts = {}) {
338
- return withRefs(repo, baseRef, headRef, opts, (baseDir, headDir, scoped) => {
339
- const res = diffTrees(baseDir, headDir, scoped);
340
- res.summary.base = baseRef;
341
- res.summary.head = headRef || WORKING_TREE;
342
- return res;
343
- });
313
+ const trees = { systemDirs: opts.systemDirs || [] };
314
+ const base = gitTree(repo, baseRef, trees);
315
+ const head = headRef ? gitTree(repo, headRef, trees) : repo;
316
+ const res = diffTrees(base, head, headRef ? opts : { ...opts, allow: gitScope(repo) });
317
+ res.summary.base = baseRef;
318
+ res.summary.head = headRef || WORKING_TREE;
319
+ return res;
344
320
  }
@@ -0,0 +1,110 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // EXEC SQL as Db2 reads it, whichever language embeds it: a statement's words, host variables and
3
+ // literals, and which host variables it reads and which it writes.
4
+ // The words, host variables and literals of a statement. A host variable is :NAME, qualified as
5
+ // :GROUP.NAME, and may carry an indicator variable, :NAME:IND or :NAME INDICATOR :IND.
6
+ export function sqlTokens(sql) {
7
+ const toks = [];
8
+ const re = /'(?:[^']|'')*'|"(?:[^"]|"")*"|:\s*[A-Z0-9_$#@-]+(?:\.[A-Z0-9_$#@-]+)*|[A-Z0-9_$#@-]+|[(),=<>+*/;.]/gi;
9
+ for (const m of sql.matchAll(re)) {
10
+ const t = m[0];
11
+ if (t.startsWith(':')) toks.push({ host: t.slice(1).trim().toUpperCase() });
12
+ else if (/^['"]/.test(t)) toks.push({ lit: t });
13
+ else toks.push({ word: t.toUpperCase() });
14
+ }
15
+ return toks;
16
+ }
17
+
18
+ // Clauses that end an INTO list.
19
+ const AFTER_INTO = new Set(['FROM', 'WHERE', 'GROUP', 'HAVING', 'ORDER', 'FETCH', 'FOR', 'WITH', 'OPTIMIZE', 'QUERYNO', 'SKIP', 'UNION', 'USING', 'VALUES']);
20
+
21
+ // Which host variables a statement reads and which it writes, as Db2's SQL reference gives them. A
22
+ // variable both read and written is in both lists. INTO a host variable writes it (INSERT INTO and
23
+ // MERGE INTO name a table); so do SET's targets, GET DIAGNOSTICS' and ASSOCIATE LOCATORS' list. A
24
+ // procedure's argument that is a lone variable may be IN, OUT or INOUT, which only the server knows.
25
+ export function sqlRoles(toks) {
26
+ const sending = [];
27
+ const receiving = [];
28
+ const read = (h) => { if (!sending.includes(h)) sending.push(h); };
29
+ const write = (h) => { if (!receiving.includes(h)) receiving.push(h); };
30
+ const verb = toks.find((t) => t.word)?.word;
31
+ if (verb === 'CALL') return callRoles(toks, read, write), { sending, receiving };
32
+ // DESCRIBE and PREPARE read the SQLDA's SQLN and write the rest of it.
33
+ const intoBoth = verb === 'DESCRIBE' || verb === 'PREPARE';
34
+ let into = false;
35
+ let assigning = verb === 'SET';
36
+ let depth = 0;
37
+ for (let i = 0; i < toks.length; i++) {
38
+ const t = toks[i];
39
+ const next = toks[i + 1];
40
+ if (t.word === '(') depth++;
41
+ else if (t.word === ')') depth--;
42
+ else if (verb === 'SET' && t.word === '=' && depth === 0) assigning = false;
43
+ else if (verb === 'SET' && t.word === ',' && depth === 0) assigning = true;
44
+ if (t.word === 'INTO' && next && (next.host || next.word === 'DESCRIPTOR')) { into = true; continue; }
45
+ if (into && t.word && AFTER_INTO.has(t.word)) into = false;
46
+ if (t.word === 'DESCRIPTOR' && next?.host) {
47
+ // FETCH ... INTO DESCRIPTOR and USING DESCRIPTOR are synonyms: the program fills the SQLDA and Db2 writes where it points.
48
+ if (verb === 'FETCH' || intoBoth) { read(next.host); write(next.host); } else read(next.host);
49
+ i++;
50
+ continue;
51
+ }
52
+ if (!t.host) continue;
53
+ const written = into || assigning
54
+ || (verb === 'GET' && next?.word === '=')
55
+ || (verb === 'ASSOCIATE' && !toks.slice(0, i).some((x) => x.word === 'WITH'));
56
+ if (!written || (into && intoBoth)) read(t.host);
57
+ if (written) write(t.host);
58
+ }
59
+ return { sending, receiving };
60
+ }
61
+
62
+ // CALL :name reads the name. Each argument that is a lone variable, with or without its indicator,
63
+ // is read and may be written; any other argument is an expression, IN only.
64
+ function callRoles(toks, read, write) {
65
+ let depth = 0;
66
+ let arg = [];
67
+ const close = () => {
68
+ const hosts = arg.filter((t) => t.host);
69
+ const lone = hosts.length && arg.every((t) => t.host || t.word === 'INDICATOR');
70
+ for (const h of hosts) { read(h.host); if (lone) write(h.host); }
71
+ arg = [];
72
+ };
73
+ for (let i = 1; i < toks.length; i++) {
74
+ const t = toks[i];
75
+ if (t.word === 'DESCRIPTOR' && toks[i + 1]?.host) { read(toks[i + 1].host); write(toks[i + 1].host); i++; continue; }
76
+ if (t.word === '(') { if (depth++ > 0) arg.push(t); continue; }
77
+ if (t.word === ')') { if (--depth > 0) arg.push(t); else close(); continue; }
78
+ if (depth === 0) { if (t.host) read(t.host); continue; }
79
+ if (depth === 1 && t.word === ',') { close(); continue; }
80
+ arg.push(t);
81
+ }
82
+ }
83
+
84
+ export function hostVariableRoles(sql) {
85
+ return sqlRoles(sqlTokens(sql));
86
+ }
87
+
88
+ // The host variables in a parsed EXEC SQL block: a colon, then a name qualified by joined periods as
89
+ // :GROUP.FIELD. Each carries its path, the token naming the item, where it stands, and whether the
90
+ // statement writes it by sqlRoles.
91
+ export function hostVariablesIn(toks) {
92
+ const found = [];
93
+ const sql = [];
94
+ for (let k = 0; k < toks.length; k++) {
95
+ const t = toks[k];
96
+ if (t.t === 'op' && t.v === ':' && toks[k + 1] && toks[k + 1].t === 'word') {
97
+ const path = [toks[k + 1]];
98
+ let j = k + 2;
99
+ for (; j + 1 < toks.length && toks[j].t === 'period' && toks[j].joined && toks[j + 1].t === 'word'; j += 2) path.push(toks[j + 1]);
100
+ const host = path.map((p) => p.u).join('.');
101
+ found.push({ path, tok: path[path.length - 1], at: k + 1, host });
102
+ sql.push({ host });
103
+ k = j - 1;
104
+ continue;
105
+ }
106
+ sql.push(t.t === 'lit' ? { lit: t.v } : { word: String(t.u ?? t.v).toUpperCase() });
107
+ }
108
+ const { receiving } = sqlRoles(sql);
109
+ return found.map((h) => ({ ...h, written: receiving.includes(h.host) }));
110
+ }