@portll/cobolwork 0.2.117 → 0.2.150

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +13 -1
  2. package/THIRD-PARTY-NOTICES.md +3 -2
  3. package/bin/cobolwork.mjs +16 -1
  4. package/lib/baseline.mjs +8 -6
  5. package/lib/bms.mjs +1 -1
  6. package/lib/capabilities.mjs +2 -1
  7. package/lib/cards.mjs +66 -0
  8. package/lib/cics-commands.mjs +136 -125
  9. package/lib/compliance.mjs +6 -5
  10. package/lib/control.mjs +295 -6
  11. package/lib/dataflow.mjs +72 -88
  12. package/lib/diff.mjs +65 -89
  13. package/lib/embedded-sql.mjs +110 -0
  14. package/lib/equivalence.mjs +30 -12
  15. package/lib/evidence/journal.mjs +11 -2
  16. package/lib/evidence/record.mjs +3 -1
  17. package/lib/evidence/slsa.mjs +17 -2
  18. package/lib/execution.mjs +77 -0
  19. package/lib/explain.mjs +5 -4
  20. package/lib/ftp.mjs +136 -0
  21. package/lib/inventory.mjs +8 -9
  22. package/lib/ironwork.mjs +4 -6
  23. package/lib/jcl.mjs +29 -54
  24. package/lib/kernel/git.mjs +60 -0
  25. package/lib/kernel/identity.mjs +7 -6
  26. package/lib/kernel/source-tree.mjs +204 -34
  27. package/lib/layout.mjs +202 -0
  28. package/lib/parser.mjs +58 -207
  29. package/lib/precompile.mjs +5 -88
  30. package/lib/revision.json +1 -1
  31. package/lib/sbom.mjs +125 -13
  32. package/lib/scan.mjs +11 -4
  33. package/lib/sets/build.mjs +1 -1
  34. package/lib/sets/cics.mjs +32 -5
  35. package/lib/sets/copybook.mjs +2 -1
  36. package/lib/sets/hidden.mjs +4 -8
  37. package/lib/sets/jcl.mjs +17 -120
  38. package/lib/sets/priv.mjs +1 -1
  39. package/lib/sets/recon.mjs +3 -2
  40. package/lib/sets/semantics.mjs +1 -1
  41. package/lib/sets/vendor.mjs +1 -1
  42. package/lib/sets/zowe.mjs +34 -8
  43. package/lib/site.mjs +17 -7
  44. package/lib/sources.mjs +82 -38
  45. package/lib/utilities.mjs +146 -4
  46. package/package.json +1 -1
  47. package/rules/compliance-cobit2019.json +2438 -0
  48. package/rules/compliance-dora.json +1 -1
  49. package/rules/compliance-ffiec.json +1 -1
  50. package/rules/compliance-nist80053.json +1 -1
  51. package/schema/cobolwork.policy.schema.json +94 -17
package/lib/parser.mjs CHANGED
@@ -9,6 +9,8 @@ import {
9
9
  } from './words.mjs';
10
10
  import { CICS_COMMANDS, CICS_EVERY_COMMAND } from './cics-commands.mjs';
11
11
  import { cicsCommand } from './precompile-cics.mjs';
12
+ import { BINARY_SIZE, literalBytes, computeSizes, layoutReport } from './layout.mjs';
13
+ import { hostVariablesIn } from './embedded-sql.mjs';
12
14
  import { join, dirname, basename, resolve, isAbsolute, sep, delimiter } from 'node:path';
13
15
 
14
16
  const FIXED_INDICATORS = new Set([' ', '*', '/', '-', 'D', 'd', '$']);
@@ -24,7 +26,6 @@ const MAX_REPLACED_CHARS = 16000000;
24
26
  // an absent ATTRIBUTES.cpy as a system copybook, which made an incomplete tree read as covered.
25
27
  const SYSTEM_COPY = /^(?:(?:DFH|DSN|CEE|IGZ|EZA|BPX|CSQ|CMQ|DLI)[A-Z0-9$#@]{0,5}|SQLCA|SQLDA|ATTRIB)$/i;
26
28
  const EXEC_KINDS = new Set(['SQL', 'CICS', 'DLI', 'SQLIMS', 'ADO', 'HTML', 'ORACLE', 'TP', 'IDMS', 'XML']);
27
- const BINARY_SIZE = { default: '1-2-4-8', ibm: '2-4-8', mf: '1--8' };
28
29
 
29
30
  export const VERBS = new Set(`ACCEPT ADD ALLOCATE ALTER CALL CANCEL CLOSE COMMIT COMPUTE CONTINUE DELETE DISABLE DISPLAY DIVIDE
30
31
  ENABLE ENTRY EVALUATE EXHIBIT EXIT FREE GENERATE GO GOBACK IF INITIALIZE INITIALISE INITIATE INSPECT INVOKE JSON MERGE MOVE MULTIPLY
@@ -58,7 +59,7 @@ export function expandTabs(line, width = 8) {
58
59
  }
59
60
 
60
61
  export function detectFormat(src) {
61
- let nonblank = 0, badIndicator = 0, fixedMarkers = 0, starCol1Bad = 0, codeCol2to7 = 0, seqDigits = 0;
62
+ let nonblank = 0, badIndicator = 0, fixedMarkers = 0, starCol1Bad = 0, codeCol2to7 = 0, seqDigits = 0, columnSevenComments = 0;
62
63
  for (const raw of src.split(/\r?\n/, 600)) {
63
64
  const l = expandTabs(raw);
64
65
  if (!l.trim()) continue;
@@ -69,17 +70,20 @@ export function detectFormat(src) {
69
70
  if (/^(IDENTIFICATION|ID|PROCEDURE|DATA|ENVIRONMENT)\s+DIVISION/i.test(l) || /^(WORKING-STORAGE|LINKAGE|FILE|LOCAL-STORAGE)\s+SECTION/i.test(l)) return 'free';
70
71
  if (l.length >= 7 && !FIXED_INDICATORS.has(l[6])) badIndicator++;
71
72
  if (/^\d{6}/.test(l) || /^.{6}[*\/]/.test(l)) fixedMarkers++;
73
+ if (/^[ \d]{6}[*\/](?!>)/.test(l)) columnSevenComments++;
72
74
  }
73
75
  if (!nonblank) return 'fixed';
74
76
  if (starCol1Bad > 0 && codeCol2to7 > 0 && seqDigits === 0) return 'terminal';
75
77
  if (badIndicator / nonblank > 0.1 && badIndicator > fixedMarkers) return 'free';
76
- return seqDigits === 0 && codePastColumn72(src) ? 'free' : 'fixed';
78
+ // Code past 72 in a program that keeps column 7 for its comments is variable format: fixed columns, long lines.
79
+ if (seqDigits === 0 && codePastColumn72(src)) return columnSevenComments > 0 ? 'variable' : 'free';
80
+ return 'fixed';
77
81
  }
78
82
 
79
83
  // Code a fixed-form reading would cut off at column 72, in a program written for -free: a literal
80
84
  // still open there that the next line does not continue, a token running across the boundary, or a
81
85
  // line past 80 whose code does not end its sentence by 72. An inline comment, a comment entry and a
82
- // lone tag of up to eight characters are what fixed-form programs keep there. Of the 54,487 programs
86
+ // lone tag of up to eight characters whose parentheses balance are what fixed-form programs keep there. Of the 54,487 programs
83
87
  // cobc compiled in a 3,184-repository corpus, 54,154 are then read in a format cobc accepts.
84
88
  function codePastColumn72(src) {
85
89
  const lines = src.split(/\r?\n/, 2001);
@@ -89,7 +93,9 @@ function codePastColumn72(src) {
89
93
  const { quote, comment } = stateAtColumn72(l);
90
94
  if (comment) continue;
91
95
  if (quote && expandTabs(lines[i + 1] || '')[6] !== '-') return true;
92
- const tag = /^\S{1,8}$/.test(l.slice(72).trim());
96
+ const tail = l.slice(72).trim();
97
+ // A tail closing a parenthesis it did not open finishes code cut at 72, as in `...LENGTH(WS-IDX)`.
98
+ const tag = /^\S{1,8}$/.test(tail) && balanced(tail);
93
99
  if (/\S/.test(l[71]) && /\S/.test(l[72]) && !tag) return true;
94
100
  const code = l.slice(7, 72).trim();
95
101
  if (l.length > 80 && !tag && code && !code.endsWith('.')) return true;
@@ -97,6 +103,15 @@ function codePastColumn72(src) {
97
103
  return false;
98
104
  }
99
105
 
106
+ function balanced(s) {
107
+ let depth = 0;
108
+ for (const c of s) {
109
+ if (c === '(') depth++;
110
+ else if (c === ')' && --depth < 0) return false;
111
+ }
112
+ return depth === 0;
113
+ }
114
+
100
115
  function stateAtColumn72(l) {
101
116
  let quote = null;
102
117
  for (let i = 7; i < 72 && i < l.length; i++) {
@@ -619,7 +634,7 @@ function isProgramFile(p, ctx) {
619
634
  const seen = (ctx.programFiles ||= new Map());
620
635
  if (!seen.has(p)) {
621
636
  let text = '';
622
- try { text = readSource(p).text; } catch { /* an unreadable candidate is not known to be a program */ }
637
+ try { text = ctx.readText(p); } catch { /* an unreadable candidate is not known to be a program */ }
623
638
  seen.set(p, /^[^*\n]{0,6}\s*PROGRAM-ID\s*\./im.test(text));
624
639
  }
625
640
  return seen.get(p);
@@ -772,7 +787,7 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
772
787
  if (!path) return [];
773
788
  if (stack.includes(path) || stack.length > 40) { record.status = 'recursive'; return []; }
774
789
  if (++ctx.inclusions > MAX_INCLUSIONS || ctx.copyTokens > MAX_COPY_TOKENS) { record.status = 'expansion-limit'; return []; }
775
- const src = readSource(path).text;
790
+ const src = ctx.readText(path);
776
791
  // A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
777
792
  // directive earlier in the including file may have changed from the file's starting format.
778
793
  const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
@@ -857,78 +872,6 @@ function stripDirecting(tokens) {
857
872
  return out;
858
873
  }
859
874
 
860
- function picInfo(pic, constants) {
861
- const info = { digits: 0, display: 0, signed: false, alphanumeric: false, national: false };
862
- const re = /(.)\(([A-Za-z0-9_-]+)\)|(.)/g;
863
- let m;
864
- while ((m = re.exec(pic))) {
865
- const ch = (m[1] || m[3]).toUpperCase();
866
- let n = 1;
867
- if (m[1]) {
868
- const raw = m[2];
869
- n = /^\d+$/.test(raw) ? Number(raw) : Number(constants && constants.get(raw.toUpperCase()));
870
- if (!Number.isFinite(n) || n < 0) n = 0;
871
- }
872
- if (ch === 'S') { info.signed = true; continue; }
873
- if (ch === 'V' || ch === 'P') continue;
874
- if (ch === '9') { info.digits += n; info.display += n; continue; }
875
- if (ch === 'N' || ch === 'G') { info.national = true; info.display += 2 * n; continue; }
876
- if (ch === 'X' || ch === 'A') info.alphanumeric = true;
877
- info.display += n;
878
- }
879
- return info;
880
- }
881
-
882
- // Storage bytes of a literal: hexadecimal literals hold one byte per two digits, a Z literal adds a
883
- // terminating null, a national literal holds two bytes per character.
884
- function literalBytes(tok) {
885
- const prefix = tok.prefix || '';
886
- if (prefix === 'X' || prefix === 'BX' || prefix === 'NX') return Math.floor(tok.v.length / (prefix === 'NX' ? 4 : 2)) * (prefix === 'NX' ? 2 : 1);
887
- if (prefix === 'Z') return tok.v.length + 1;
888
- if (prefix === 'N' || prefix === 'NC' || prefix === 'U') return tok.v.length * 2;
889
- return tok.v.length;
890
- }
891
-
892
- function binaryBytes(digits, scheme) {
893
- if (scheme === '2-4-8') return digits <= 4 ? 2 : digits <= 9 ? 4 : 8;
894
- if (scheme === '1--8') return [1, 1, 1, 2, 2, 3, 3, 4, 4, 4, 5, 5, 6, 6, 6, 7, 7, 8, 8][Math.min(digits, 18)] || 8;
895
- return digits <= 2 ? 1 : digits <= 4 ? 2 : digits <= 9 ? 4 : 8;
896
- }
897
-
898
- function elementarySize(item, scheme, constants) {
899
- if (!item.picture && !item.usage) {
900
- const constBytes = constants && constants.textBytes;
901
- const fromValue = item.values.reduce((n, v) => n + (v.t === 'lit' ? literalBytes(v) : v.t === 'word' && constBytes && constBytes.has(v.u) ? constBytes.get(v.u) : 0), 0);
902
- if (item.section === 'SCREEN' && !fromValue && item.screenRefItem) return item.screenRefItem.size || 0;
903
- if (item.section === 'SCREEN') return Math.max(1, fromValue);
904
- if (fromValue) return fromValue;
905
- }
906
- if (item.section === 'SCREEN' && !item.picture) return 1;
907
- const usage = item.effectiveUsage || 'DISPLAY';
908
- const p = item.picture ? picInfo(item.picture, constants) : null;
909
- const digits = p ? p.digits : 0;
910
- const u = usage.replace('COMPUTATIONAL', 'COMP');
911
- if (u === 'COMP-1' || u === 'FLOAT-SHORT') return 4;
912
- if (u === 'COMP-2' || u === 'FLOAT-LONG' || u === 'FLOAT-DECIMAL-16') return 8;
913
- if (u === 'FLOAT-DECIMAL-34') return 16;
914
- if (u === 'INDEX') return 4;
915
- if (u === 'POINTER' || u === 'PROGRAM-POINTER' || u === 'FUNCTION-POINTER' || u === 'PROCEDURE-POINTER') return 8;
916
- if (u === 'BINARY-CHAR') return 1;
917
- if (u === 'BINARY-SHORT' || u === 'SIGNED-SHORT' || u === 'UNSIGNED-SHORT') return 2;
918
- if (u === 'BINARY-LONG' || u === 'BINARY-INT' || u === 'SIGNED-INT' || u === 'UNSIGNED-INT') return 4;
919
- if (u === 'BINARY-DOUBLE' || u === 'BINARY-LONG-LONG' || u === 'BINARY-C-LONG' || u === 'SIGNED-LONG' || u === 'UNSIGNED-LONG') return 8;
920
- if (u === 'COMP-3' || u === 'PACKED-DECIMAL') return Math.floor(digits / 2) + 1;
921
- if (u === 'COMP-6') return Math.ceil(digits / 2);
922
- if (u === 'COMP-X' || u === 'COMP-N' || (u === 'COMP-5' && p && p.alphanumeric)) {
923
- if (p && p.alphanumeric) return p.display;
924
- return Math.max(1, Math.ceil((digits * Math.log(10)) / Math.log(256)));
925
- }
926
- if (u === 'COMP' || u === 'COMP-4' || u === 'COMP-5' || u === 'BINARY') return binaryBytes(digits, scheme);
927
- let len = p ? p.display : 0;
928
- if (item.signSeparate && p && p.signed) len++;
929
- return len;
930
- }
931
-
932
875
  function parseDataEntry(toks, section, file) {
933
876
  const levelTok = toks[0];
934
877
  const level = Number(levelTok.v);
@@ -1096,120 +1039,6 @@ function applyTypes(items) {
1096
1039
  return added;
1097
1040
  }
1098
1041
 
1099
- // SYNCHRONIZED aligns a binary, floating-point, pointer or index item to its own length; packed,
1100
- // COMP-X and display items are not moved.
1101
- const ALIGNED_USAGE = /^(COMP|COMP-[1245]|BINARY(-[A-Z-]+)?|FLOAT-[A-Z0-9-]+|(PROGRAM-|FUNCTION-|PROCEDURE-)?POINTER|INDEX|(UN)?SIGNED-[A-Z]+)$/;
1102
-
1103
- function computeSizes(roots, scheme, constants) {
1104
- // `at` is where the item starts, counted from the start of its record, because the compiler aligns
1105
- // a SYNCHRONIZED item against the record and puts the slack inside the group that holds it. A
1106
- // table's entry is laid out from its own start and rounded up to its widest alignment, so every
1107
- // occurrence aligns alike.
1108
- const visit = (item, inheritedUsage, inheritedSignSeparate, at) => {
1109
- item.effectiveUsage = item.usage || inheritedUsage || null;
1110
- if (inheritedSignSeparate && !item.signExplicit) item.signSeparate = true;
1111
- const structural = item.children.filter(c => c.level !== 88 && c.level !== 66 && c.level !== 78);
1112
- if (!structural.length) {
1113
- const one = elementarySize(item, scheme, constants);
1114
- item.contributes = one * item.occurs;
1115
- // The listing prints the whole table only for POINTER and INDEX; every other usage prints one occurrence.
1116
- const wholeTable = /^(POINTER|INDEX)$/.test(item.effectiveUsage || '');
1117
- item.size = wholeTable ? item.contributes : one;
1118
- const usage = (item.effectiveUsage || '').replace('COMPUTATIONAL', 'COMP');
1119
- item.align = item.sync && ALIGNED_USAGE.test(usage) ? Math.min(one, 8) : 1;
1120
- item.maxAlign = item.align;
1121
- return;
1122
- }
1123
- const base = item.occurs > 1 ? 0 : at;
1124
- let offset = 0;
1125
- let end = 0;
1126
- let maxAlign = 1;
1127
- const startOf = new Map();
1128
- for (const c of structural) {
1129
- if (c.redefines) {
1130
- // REDEFINES names a SIBLING. Resolving it by name across the whole program reached the
1131
- // first item of that name anywhere, which crossed records.
1132
- const known = startOf.has(c.redefines);
1133
- visit(c, item.effectiveUsage, item.signSeparate, base + (known ? startOf.get(c.redefines) : offset));
1134
- const start = known ? startOf.get(c.redefines) : offset - c.contributes;
1135
- startOf.set(c.name, start);
1136
- c.localStart = start;
1137
- end = Math.max(end, start + c.contributes);
1138
- // A REDEFINES larger than the item it redefines pushes the next sibling past its end.
1139
- offset = Math.max(offset, start + c.contributes);
1140
- maxAlign = Math.max(maxAlign, c.maxAlign);
1141
- continue;
1142
- }
1143
- visit(c, item.effectiveUsage, item.signSeparate, base + offset);
1144
- const skew = (base + offset) % c.align;
1145
- if (c.align > 1 && skew) offset += c.align - skew;
1146
- startOf.set(c.name, offset);
1147
- c.localStart = offset;
1148
- offset += c.contributes;
1149
- end = Math.max(end, offset);
1150
- maxAlign = Math.max(maxAlign, c.maxAlign);
1151
- }
1152
- if (item.occurs > 1 && end % maxAlign) end += maxAlign - (end % maxAlign);
1153
- item.size = end * item.occurs;
1154
- item.contributes = item.size;
1155
- item.align = 1;
1156
- item.maxAlign = maxAlign;
1157
- };
1158
- for (const r of roots) visit(r, null, false, 0);
1159
- // Offsets are assigned after sizing: a child's absolute start depends on its parent's, which is
1160
- // only known once the parent's own siblings have been laid out.
1161
- const place = (item, at) => {
1162
- item.offset = at;
1163
- for (const c of item.children) if (c.localStart != null) place(c, at + c.localStart);
1164
- };
1165
- for (const r of roots) place(r, 0);
1166
- }
1167
-
1168
- // A report group is laid out by column, not by adding up its items: a line is as wide as the column
1169
- // its rightmost item ends in, COLUMN PLUS counting on from the end of the item before. A group holds
1170
- // its lines one after another, and every 01 group of a report, like the file the report is written
1171
- // to, is as large as the largest group.
1172
- function layoutReport(rd) {
1173
- const structural = (x) => x.children.filter(c => c.level !== 88 && c.level !== 66 && c.level !== 78);
1174
- const mark = (x) => { x.rd = rd; for (const c of x.children) mark(c); };
1175
- for (const g of rd.groups) mark(g);
1176
- let width = 0;
1177
- // Items printed at the same column under PRESENT WHEN each keep their own storage, so a line is
1178
- // never smaller than its items laid end to end.
1179
- const lineWidth = (line) => {
1180
- let last = 0;
1181
- let right = 0;
1182
- let total = 0;
1183
- const place = (c) => {
1184
- const len = c.contributes || c.size || 0;
1185
- const start = c.rwColumn ? (c.rwColumn.at != null ? c.rwColumn.at : last + c.rwColumn.plus) : last + 1;
1186
- last = start + len - 1;
1187
- right = Math.max(right, last);
1188
- total += len;
1189
- };
1190
- const walk = (x) => { for (const c of structural(x)) { if (structural(c).length) walk(c); else place(c); } };
1191
- if (structural(line).length) walk(line);
1192
- else place(line);
1193
- return Math.max(right, total);
1194
- };
1195
- // The entries that open a line; a group with no LINE clause anywhere in it is one line.
1196
- for (const g of rd.groups) {
1197
- const lines = [];
1198
- const collect = (x) => { if (x.rwLine) { lines.push(x); return; } for (const c of structural(x)) collect(c); };
1199
- collect(g);
1200
- if (!lines.length) lines.push(g);
1201
- let total = 0;
1202
- for (const line of lines) {
1203
- const w = lineWidth(line);
1204
- if (line !== g) { line.size = w; line.contributes = w * (line.occurs || 1); }
1205
- total += w;
1206
- }
1207
- width = Math.max(width, total);
1208
- }
1209
- for (const g of rd.groups) { g.size = width; g.contributes = width; }
1210
- rd.width = width;
1211
- }
1212
-
1213
1042
  const ENV_PARAGRAPHS = new Set(['CONFIGURATION', 'SOURCE-COMPUTER', 'OBJECT-COMPUTER', 'SPECIAL-NAMES', 'REPOSITORY', 'INPUT-OUTPUT', 'FILE-CONTROL', 'I-O-CONTROL']);
1214
1043
 
1215
1044
  const SECTION_NAMES = { 'WORKING-STORAGE': 'WORKING-STORAGE', 'LOCAL-STORAGE': 'LOCAL-STORAGE', LINKAGE: 'LINKAGE', FILE: 'FILE', SCREEN: 'SCREEN', REPORT: 'REPORT', COMMUNICATION: 'COMMUNICATION' };
@@ -1680,13 +1509,19 @@ function indexTokens(seg) {
1680
1509
  if (depth === 1 && t.t === 'op' && t.v === ':' && colonAt < 0) colonAt = j;
1681
1510
  }
1682
1511
  if (host) {
1512
+ // A constant on the other side of the colon: X(I:3) ends 2 past I, and X(5:N) starts at 5.
1513
+ const lone = (from, to) => (to - from === 1 && (seg[from].t === 'word' || seg[from].t === 'num') && /^\d+$/.test(seg[from].v) ? Number(seg[from].v) : null);
1514
+ const refLength = colonAt < 0 ? null : lone(colonAt + 1, close);
1515
+ const refStart = colonAt < 0 ? null : lone(k + 1, colonAt);
1683
1516
  for (let j = k + 1; j < close; j++) {
1684
1517
  const t = seg[j];
1685
1518
  // A number is tokenized as a word, and no name is all digits.
1686
1519
  if (t.t !== 'word' || /^\d+$/.test(t.v)) continue;
1687
1520
  const prev = seg[j - 1];
1688
1521
  if (prev && prev.t === 'word' && (prev.u === 'OF' || prev.u === 'IN' || prev.u === 'FUNCTION')) continue;
1689
- out.push({ host, tok: t, kind: colonAt < 0 ? 'subscript' : j < colonAt ? 'refmod-offset' : 'refmod-length', offset: offsetOf(seg, j, k, close, colonAt) });
1522
+ const kind = colonAt < 0 ? 'subscript' : j < colonAt ? 'refmod-offset' : 'refmod-length';
1523
+ const other = kind === 'refmod-offset' ? refLength : kind === 'refmod-length' ? refStart : null;
1524
+ out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}) });
1690
1525
  }
1691
1526
  }
1692
1527
  lastHost = host;
@@ -1758,7 +1593,30 @@ function markReceiving(verb, seg, set) {
1758
1593
  case 'SET': { const k = stopAt(0, ['TO', 'UP', 'DOWN']); mark(0, k); break; }
1759
1594
  case 'INITIALISE': case 'INITIALIZE': { mark(0, stopAt(0, ['REPLACING', 'WITH', 'ALL', 'TO', 'THEN', 'DEFAULT', 'FILLER', 'ALPHANUMERIC', 'NUMERIC', 'VALUE'])); break; }
1760
1595
  case 'STRING': { const k = at(['INTO']); if (k >= 0) mark(k + 1, stopAt(k + 1, tail)); const p = at(['POINTER']); if (p >= 0) mark(p + 1, p + 2); break; }
1761
- case 'UNSTRING': { const k = at(['INTO']); if (k >= 0) mark(k + 1, stopAt(k + 1, ['DELIMITER', 'COUNT', 'WITH', 'TALLYING', 'ON', 'NOT', 'POINTER'])); for (const w of ['POINTER']) { const p = at([w]); if (p >= 0) mark(p + 1, p + 2); } break; }
1596
+ case 'UNSTRING': {
1597
+ // Every receiver is written, and so is each field a DELIMITER IN or COUNT IN phrase names.
1598
+ const k = at(['INTO']);
1599
+ if (k >= 0) {
1600
+ const end = stopAt(k + 1, ['WITH', 'POINTER', 'TALLYING', 'ON', 'NOT']);
1601
+ // The field after DELIMITER IN or COUNT IN starts its own run, which identifierTokens would read as a qualifier.
1602
+ const own = (j) => { const t = seg[j]; if (t && t.t === 'word') set.set(t, verb); };
1603
+ let from = k + 1;
1604
+ let phrase = false;
1605
+ for (let j = k + 1, depth = 0; j <= end; j++) {
1606
+ const t = seg[j];
1607
+ if (t && t.t === 'sep') { depth += t.v === '(' ? 1 : t.v === ')' ? -1 : 0; continue; }
1608
+ if (j < end && (depth || t.t !== 'word' || (t.u !== 'DELIMITER' && t.u !== 'COUNT'))) continue;
1609
+ if (j > from) { if (phrase) { own(from); if (j > from + 1) mark(from + 1, j); } else mark(from, j); }
1610
+ from = j + 1 + (seg[j + 1] && seg[j + 1].t === 'word' && seg[j + 1].u === 'IN' ? 1 : 0);
1611
+ phrase = true;
1612
+ }
1613
+ }
1614
+ const p = at(['POINTER']);
1615
+ if (p >= 0) mark(p + 1, p + 2);
1616
+ const tl = at(['TALLYING']);
1617
+ if (tl >= 0) { const t = seg[tl + 1] && seg[tl + 1].t === 'word' && seg[tl + 1].u === 'IN' ? seg[tl + 2] : seg[tl + 1]; if (t && t.t === 'word') set.set(t, verb); }
1618
+ break;
1619
+ }
1762
1620
  case 'INSPECT': { if (at(['REPLACING', 'CONVERTING']) >= 0) mark(0, 1); const tl = at(['TALLYING']); if (tl >= 0) mark(tl + 1, tl + 2); break; }
1763
1621
  case 'READ': case 'RETURN': { const k = at(['INTO']); if (k >= 0) mark(k + 1, k + 2); break; }
1764
1622
  case 'WRITE': case 'REWRITE': case 'RELEASE': { if (at(['FROM']) >= 0) mark(0, 1); break; }
@@ -1902,20 +1760,11 @@ const NOT_A_NAME_AFTER = new Set(['OF', 'IN', 'FUNCTION', 'DFHRESP', 'DFHVALUE']
1902
1760
  // by A, and the variables of a SELECT or FETCH INTO list, up to FROM, are written. An INSERT's INTO
1903
1761
  // names a table. The block keeps them, each by the token that resolves to its item.
1904
1762
  function hostVariableRefs(exec, refs, receiving) {
1905
- const toks = exec.toks;
1906
- const into = ['SELECT', 'FETCH'].includes(toks[0]?.u) ? toks.findIndex((x) => x.t === 'word' && x.u === 'INTO') : -1;
1907
- const fromAt = into < 0 ? -1 : toks.findIndex((x, k) => k > into && x.t === 'word' && x.u === 'FROM');
1908
- const intoEnd = fromAt < 0 ? toks.length : fromAt;
1909
1763
  exec.hostVariables = [];
1910
- for (let k = 0; k + 1 < toks.length; k++) {
1911
- if (!(toks[k].t === 'op' && toks[k].v === ':' && toks[k + 1].t === 'word')) continue;
1912
- const path = [toks[k + 1]];
1913
- for (let j = k + 2; j + 1 < toks.length && toks[j].t === 'period' && toks[j].joined && toks[j + 1].t === 'word'; j += 2) path.push(toks[j + 1]);
1914
- const name = path[path.length - 1];
1764
+ for (const { path, tok: name, written } of hostVariablesIn(exec.toks)) {
1915
1765
  name.verb = 'EXEC SQL';
1916
1766
  refs.push({ tok: name, zone: 'sql' });
1917
1767
  for (let q = path.length - 2; q >= 0; q--) refs.push({ tok: { t: 'word', u: 'OF', line: name.line, file: name.file }, zone: 'sql' }, { tok: path[q], zone: 'sql' });
1918
- const written = into >= 0 && k > into && k < intoEnd;
1919
1768
  if (written) receiving.set(name, 'EXEC SQL');
1920
1769
  exec.hostVariables.push({ tok: name, written });
1921
1770
  }
@@ -1999,7 +1848,7 @@ function collectCopybookDefines(src, format, ctx, depth) {
1999
1848
  const path = resolveCopy(name, null, ctx);
2000
1849
  if (!path || seen.has(path)) continue;
2001
1850
  seen.add(path);
2002
- const text = readSource(path).text;
1851
+ const text = ctx.readText(path);
2003
1852
  const fmt = detectFormat(text) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : current;
2004
1853
  normalize(text, fmt, ctx.defines, ctx.std);
2005
1854
  collectCopybookDefines(text, fmt, ctx, depth + 1);
@@ -2031,7 +1880,9 @@ export function parseSource(src, file, opts = {}) {
2031
1880
  const scheme = BINARY_SIZE[opts.std || 'default'] || BINARY_SIZE.default;
2032
1881
  const ctx = { mainDir: opts.mainDir || dirname(file), includeDirs: opts.includeDirs || [], systemDirs: opts.systemDirs || [],
2033
1882
  fileIndex: opts.fileIndex || null, allowAbsoluteCopy: !!opts.allowAbsoluteCopy, cache: new Map(), copies: [], diags: [], copyFormat: opts.copyFormat || format, defines: new Map(), std: opts.std,
2034
- inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0 };
1883
+ inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0,
1884
+ // Where a copybook's text comes from: the disk, or the source tree the program was read from.
1885
+ readText: opts.readText || ((p) => readSource(p).text) };
2035
1886
  collectCopybookDefines(src, format, ctx, 0);
2036
1887
  const norm = normalize(src, format, ctx.defines, opts.std);
2037
1888
  ctx.mainFormat = norm.finalFormat;
@@ -7,6 +7,7 @@
7
7
  // name routines nothing implements: the goal is a program that compiles.
8
8
  import { detectFormat, expandTabs } from './parser.mjs';
9
9
  import { EIB_LAYOUT } from './words.mjs';
10
+ import { sqlTokens, sqlRoles } from './embedded-sql.mjs';
10
11
  import { builtinValue, translateCics, eibCopybook, CONSTANT_COPYBOOKS, constantsCopybook, symbolicMapCopybook } from './precompile-cics.mjs';
11
12
 
12
13
  const isFree = (format) => format === 'free' || format === 'terminal';
@@ -84,20 +85,6 @@ function findBlocks(lines, format) {
84
85
  return { blocks, builtins };
85
86
  }
86
87
 
87
- // The words, host variables and literals of a statement. A host variable is :NAME, qualified as
88
- // :GROUP.NAME, and may carry an indicator variable, :NAME:IND or :NAME INDICATOR :IND.
89
- function tokenize(sql) {
90
- const toks = [];
91
- const re = /'(?:[^']|'')*'|"(?:[^"]|"")*"|:\s*[A-Z0-9_$#@-]+(?:\.[A-Z0-9_$#@-]+)*|[A-Z0-9_$#@-]+|[(),=<>+*/;.]/gi;
92
- for (const m of sql.matchAll(re)) {
93
- const t = m[0];
94
- if (t.startsWith(':')) toks.push({ host: t.slice(1).trim().toUpperCase() });
95
- else if (/^['"]/.test(t)) toks.push({ lit: t });
96
- else toks.push({ word: t.toUpperCase() });
97
- }
98
- return toks;
99
- }
100
-
101
88
  // How many subscripts a host variable needs: one for each OCCURS on it or on a group holding it.
102
89
  function subscriptCounter(items = []) {
103
90
  const byName = new Map();
@@ -116,76 +103,6 @@ function subscriptCounter(items = []) {
116
103
  // where the array starts.
117
104
  const cobolName = (host, subscripts = 0) => host.split('.').reverse().join(' OF ') + (subscripts ? ` (${Array(subscripts).fill(1).join(' ')})` : '');
118
105
 
119
- // Clauses that end an INTO list.
120
- const AFTER_INTO = new Set(['FROM', 'WHERE', 'GROUP', 'HAVING', 'ORDER', 'FETCH', 'FOR', 'WITH', 'OPTIMIZE', 'QUERYNO', 'SKIP', 'UNION', 'USING', 'VALUES']);
121
-
122
- // Which host variables a statement reads and which it writes, as Db2's SQL reference gives them. A
123
- // variable both read and written is in both lists. INTO a host variable writes it (INSERT INTO and
124
- // MERGE INTO name a table); so do SET's targets, GET DIAGNOSTICS' and ASSOCIATE LOCATORS' list. A
125
- // procedure's argument that is a lone variable may be IN, OUT or INOUT, which only the server knows.
126
- function roles(toks) {
127
- const sending = [];
128
- const receiving = [];
129
- const read = (h) => { if (!sending.includes(h)) sending.push(h); };
130
- const write = (h) => { if (!receiving.includes(h)) receiving.push(h); };
131
- const verb = toks.find((t) => t.word)?.word;
132
- if (verb === 'CALL') return callRoles(toks, read, write), { sending, receiving };
133
- // DESCRIBE and PREPARE read the SQLDA's SQLN and write the rest of it.
134
- const intoBoth = verb === 'DESCRIBE' || verb === 'PREPARE';
135
- let into = false;
136
- let assigning = verb === 'SET';
137
- let depth = 0;
138
- for (let i = 0; i < toks.length; i++) {
139
- const t = toks[i];
140
- const next = toks[i + 1];
141
- if (t.word === '(') depth++;
142
- else if (t.word === ')') depth--;
143
- else if (verb === 'SET' && t.word === '=' && depth === 0) assigning = false;
144
- else if (verb === 'SET' && t.word === ',' && depth === 0) assigning = true;
145
- if (t.word === 'INTO' && next && (next.host || next.word === 'DESCRIPTOR')) { into = true; continue; }
146
- if (into && t.word && AFTER_INTO.has(t.word)) into = false;
147
- if (t.word === 'DESCRIPTOR' && next?.host) {
148
- // FETCH ... INTO DESCRIPTOR and USING DESCRIPTOR are synonyms: the program fills the SQLDA and Db2 writes where it points.
149
- if (verb === 'FETCH' || intoBoth) { read(next.host); write(next.host); } else read(next.host);
150
- i++;
151
- continue;
152
- }
153
- if (!t.host) continue;
154
- const written = into || assigning
155
- || (verb === 'GET' && next?.word === '=')
156
- || (verb === 'ASSOCIATE' && !toks.slice(0, i).some((x) => x.word === 'WITH'));
157
- if (!written || (into && intoBoth)) read(t.host);
158
- if (written) write(t.host);
159
- }
160
- return { sending, receiving };
161
- }
162
-
163
- // CALL :name reads the name. Each argument that is a lone variable, with or without its indicator,
164
- // is read and may be written; any other argument is an expression, IN only.
165
- function callRoles(toks, read, write) {
166
- let depth = 0;
167
- let arg = [];
168
- const close = () => {
169
- const hosts = arg.filter((t) => t.host);
170
- const lone = hosts.length && arg.every((t) => t.host || t.word === 'INDICATOR');
171
- for (const h of hosts) { read(h.host); if (lone) write(h.host); }
172
- arg = [];
173
- };
174
- for (let i = 1; i < toks.length; i++) {
175
- const t = toks[i];
176
- if (t.word === 'DESCRIPTOR' && toks[i + 1]?.host) { read(toks[i + 1].host); write(toks[i + 1].host); i++; continue; }
177
- if (t.word === '(') { if (depth++ > 0) arg.push(t); continue; }
178
- if (t.word === ')') { if (--depth > 0) arg.push(t); else close(); continue; }
179
- if (depth === 0) { if (t.host) read(t.host); continue; }
180
- if (depth === 1 && t.word === ',') { close(); continue; }
181
- arg.push(t);
182
- }
183
- }
184
-
185
- export function hostVariableRoles(sql) {
186
- return roles(tokenize(sql));
187
- }
188
-
189
106
  // The WHENEVER tests after each executable statement, from Db2's definitions of the conditions. IBM
190
107
  // does not document which wins when two hold at once, so the order is fixed, not the source's.
191
108
  const WHENEVER_TEST = {
@@ -201,12 +118,12 @@ const declarative = (words) => words[0] === 'DECLARE' && words[1] !== 'GLOBAL';
201
118
 
202
119
  // What one block becomes, as COBOL text on one logical line, or '' to blank it.
203
120
  function translate(block, state) {
204
- const toks = tokenize(block.text);
121
+ const toks = sqlTokens(block.text);
205
122
  const words = toks.filter((t) => t.word).map((t) => t.word);
206
123
  const verb = words[0] || '';
207
124
  if (verb === 'INCLUDE') return words[1] ? `COPY ${words[1]}` : '';
208
125
  if (!state.procedure) {
209
- if (declarative(words) && words.includes('CURSOR')) state.cursors.set(words[1], roles(toks).sending);
126
+ if (declarative(words) && words.includes('CURSOR')) state.cursors.set(words[1], sqlRoles(toks).sending);
210
127
  return '';
211
128
  }
212
129
  if (verb === 'WHENEVER') {
@@ -220,11 +137,11 @@ function translate(block, state) {
220
137
  return 'CONTINUE';
221
138
  }
222
139
  if (declarative(words)) {
223
- if (words.includes('CURSOR')) state.cursors.set(words[1], roles(toks).sending);
140
+ if (words.includes('CURSOR')) state.cursors.set(words[1], sqlRoles(toks).sending);
224
141
  return 'CONTINUE';
225
142
  }
226
143
  if (NOT_EXECUTABLE.has(verb)) return 'CONTINUE';
227
- let { sending, receiving } = roles(toks);
144
+ let { sending, receiving } = sqlRoles(toks);
228
145
  let name = verb;
229
146
  if (verb === 'OPEN' && state.cursors.has(words[1])) sending = [...new Set([...state.cursors.get(words[1]), ...sending])];
230
147
  if (verb === 'EXECUTE' && words[1] === 'IMMEDIATE') name = 'EXECUTE-IMMEDIATE';
package/lib/revision.json CHANGED
@@ -1 +1 @@
1
- {"commit":"b144074728e1b40f1c6a230b06f1a6b8b0758d51"}
1
+ {"commit":"f5995391b095f62da69d501ea7f6bd11a75bdc6e","tag":"v0.2.150"}