@portll/cobolwork 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. package/README.md +89 -26
  2. package/STABILITY.md +76 -0
  3. package/bin/cobolwork.mjs +56 -14
  4. package/lib/baseline.mjs +9 -0
  5. package/lib/bms.mjs +21 -7
  6. package/lib/build.mjs +133 -53
  7. package/lib/capabilities.mjs +13 -11
  8. package/lib/cics-commands.mjs +501 -9
  9. package/lib/compliance.mjs +11 -0
  10. package/lib/consequence.mjs +17 -0
  11. package/lib/control-reuse.mjs +113 -0
  12. package/lib/control-workers.mjs +161 -0
  13. package/lib/control.mjs +77 -4
  14. package/lib/dataflow.mjs +402 -152
  15. package/lib/db2/cursor.mjs +15 -0
  16. package/lib/db2/read.mjs +170 -0
  17. package/lib/db2/rules.mjs +77 -0
  18. package/lib/db2/stmt/alter.mjs +473 -0
  19. package/lib/db2/stmt/grant.mjs +125 -0
  20. package/lib/db2/stmt/index.mjs +141 -0
  21. package/lib/db2/stmt/misc.mjs +325 -0
  22. package/lib/db2/stmt/routine.mjs +564 -0
  23. package/lib/db2/stmt/storage.mjs +146 -0
  24. package/lib/db2/stmt/table.mjs +540 -0
  25. package/lib/db2/stmt/view.mjs +146 -0
  26. package/lib/diff.mjs +17 -5
  27. package/lib/equivalence.mjs +44 -5
  28. package/lib/evidence/cli.mjs +26 -7
  29. package/lib/evidence/record.mjs +16 -6
  30. package/lib/evidence/seal.mjs +17 -0
  31. package/lib/evidence/store.mjs +44 -15
  32. package/lib/evidence/timestamp.mjs +83 -0
  33. package/lib/evidence/verify.mjs +120 -42
  34. package/lib/exec-reading.mjs +51 -0
  35. package/lib/execution.mjs +3 -2
  36. package/lib/explain.mjs +2 -0
  37. package/lib/exploitability.mjs +11 -2
  38. package/lib/hlasm/asm/data.mjs +128 -0
  39. package/lib/hlasm/asm/listing.mjs +43 -0
  40. package/lib/hlasm/asm/output.mjs +45 -0
  41. package/lib/hlasm/asm/sections.mjs +161 -0
  42. package/lib/hlasm/asm/symbols.mjs +78 -0
  43. package/lib/hlasm/exec.mjs +27 -0
  44. package/lib/hlasm/expr.mjs +167 -0
  45. package/lib/hlasm/instr.mjs +73 -0
  46. package/lib/hlasm/locate.mjs +333 -0
  47. package/lib/hlasm/macro/authorization.mjs +146 -0
  48. package/lib/hlasm/macro/datasets.mjs +107 -0
  49. package/lib/hlasm/macro/io.mjs +162 -0
  50. package/lib/hlasm/macro/le.mjs +53 -0
  51. package/lib/hlasm/macro/linkage.mjs +143 -0
  52. package/lib/hlasm/macro/operator.mjs +77 -0
  53. package/lib/hlasm/macro/program.mjs +184 -0
  54. package/lib/hlasm/macro/recovery.mjs +85 -0
  55. package/lib/hlasm/macro/storage.mjs +167 -0
  56. package/lib/hlasm/macro/structured.mjs +131 -0
  57. package/lib/hlasm/model.mjs +97 -0
  58. package/lib/hlasm/mvs38.mjs +47 -0
  59. package/lib/hlasm/operands.mjs +59 -0
  60. package/lib/hlasm/optable.mjs +61 -0
  61. package/lib/hlasm/read.mjs +130 -0
  62. package/lib/hlasm.mjs +44 -7
  63. package/lib/ims/dli.mjs +37 -0
  64. package/lib/ims/macro/dbd.mjs +299 -0
  65. package/lib/ims/macro/psb.mjs +286 -0
  66. package/lib/ims/model.mjs +149 -0
  67. package/lib/ims/operands.mjs +23 -0
  68. package/lib/ims/read.mjs +37 -0
  69. package/lib/ims/rules.mjs +135 -0
  70. package/lib/inventory.mjs +10 -7
  71. package/lib/ironwork-ids.mjs +24 -0
  72. package/lib/ironwork.mjs +20 -13
  73. package/lib/kernel/pds-archive.mjs +256 -0
  74. package/lib/kernel/registry.mjs +32 -23
  75. package/lib/kernel/shared-pass.mjs +163 -0
  76. package/lib/kernel/source-tree.mjs +104 -36
  77. package/lib/kernel/version-key.mjs +17 -0
  78. package/lib/layout.mjs +26 -31
  79. package/lib/options.mjs +26 -4
  80. package/lib/parser.mjs +136 -17
  81. package/lib/pli/cursor.mjs +15 -0
  82. package/lib/pli/expr.mjs +101 -0
  83. package/lib/pli/include.mjs +82 -0
  84. package/lib/pli/layout.mjs +125 -0
  85. package/lib/pli/lex.mjs +198 -0
  86. package/lib/pli/program.mjs +280 -0
  87. package/lib/pli/rules/based.mjs +95 -0
  88. package/lib/pli/rules/conditions.mjs +68 -0
  89. package/lib/pli/rules/entry.mjs +130 -0
  90. package/lib/pli/rules/index.mjs +24 -0
  91. package/lib/pli/rules/preprocessor.mjs +55 -0
  92. package/lib/pli/statements.mjs +130 -0
  93. package/lib/pli/stmt/alloc.mjs +45 -0
  94. package/lib/pli/stmt/assignment.mjs +56 -0
  95. package/lib/pli/stmt/call.mjs +104 -0
  96. package/lib/pli/stmt/conditions.mjs +94 -0
  97. package/lib/pli/stmt/control.mjs +219 -0
  98. package/lib/pli/stmt/declare.mjs +149 -0
  99. package/lib/pli/stmt/exec.mjs +55 -0
  100. package/lib/pli/stmt/io.mjs +239 -0
  101. package/lib/pli/stmt/misc.mjs +4 -0
  102. package/lib/pli/stmt/preprocessor.mjs +242 -0
  103. package/lib/pli/stmt/procedure.mjs +258 -0
  104. package/lib/pli/stmt/stream.mjs +283 -0
  105. package/lib/pli/storage.mjs +129 -0
  106. package/lib/policy.mjs +6 -0
  107. package/lib/precompile-check.mjs +124 -0
  108. package/lib/precompile-cics.mjs +8 -4
  109. package/lib/reach.mjs +11 -2
  110. package/lib/revision.json +1 -1
  111. package/lib/sarif.mjs +44 -3
  112. package/lib/scan.mjs +28 -12
  113. package/lib/sets/abend.mjs +82 -8
  114. package/lib/sets/build.mjs +43 -37
  115. package/lib/sets/cics.mjs +20 -38
  116. package/lib/sets/compile.mjs +46 -18
  117. package/lib/sets/copybook.mjs +5 -3
  118. package/lib/sets/crypto.mjs +5 -3
  119. package/lib/sets/ddl.mjs +36 -0
  120. package/lib/sets/flow.mjs +32 -8
  121. package/lib/sets/hidden.mjs +5 -3
  122. package/lib/sets/hlasm.mjs +124 -8
  123. package/lib/sets/ims.mjs +139 -0
  124. package/lib/sets/log.mjs +11 -9
  125. package/lib/sets/opaque.mjs +27 -7
  126. package/lib/sets/pli.mjs +40 -0
  127. package/lib/sets/priv.mjs +5 -3
  128. package/lib/sets/recon.mjs +5 -3
  129. package/lib/sets/secrets.mjs +112 -0
  130. package/lib/sets/semantics.mjs +8 -3
  131. package/lib/sets/web.mjs +39 -23
  132. package/lib/site.mjs +10 -0
  133. package/lib/sources.mjs +80 -20
  134. package/lib/statement-cursor.mjs +67 -0
  135. package/lib/verify.mjs +3 -2
  136. package/lib/version.mjs +6 -0
  137. package/package.json +3 -2
  138. package/rules/compliance-cobit2019.json +520 -5
  139. package/rules/compliance-dora.json +509 -5
  140. package/rules/compliance-ffiec.json +505 -1
  141. package/rules/compliance-nist80053.json +557 -1
  142. package/rules/gitleaks-mainframe.toml +46 -15
  143. package/rules/hlasm-instructions.json +2396 -0
  144. package/rules/hlasm-optables.json +8024 -0
  145. package/schema/cobolwork-baseline.schema.json +36 -0
  146. package/schema/cobolwork-build-provenance.schema.json +187 -0
  147. package/schema/cobolwork-build.schema.json +382 -0
  148. package/schema/cobolwork-capabilities.schema.json +239 -0
  149. package/schema/cobolwork-diff.schema.json +217 -0
  150. package/schema/cobolwork-evidence.schema.json +161 -0
  151. package/schema/cobolwork-execution.schema.json +53 -0
  152. package/schema/cobolwork-explain.schema.json +360 -0
  153. package/schema/cobolwork-finding.schema.json +465 -0
  154. package/schema/cobolwork-flow.schema.json +465 -0
  155. package/schema/cobolwork-gate.schema.json +211 -0
  156. package/schema/cobolwork-inventory.schema.json +206 -0
  157. package/schema/cobolwork-parse.schema.json +105 -0
  158. package/schema/cobolwork-reach.schema.json +74 -0
  159. package/schema/cobolwork-report.schema.json +559 -0
  160. package/schema/cobolwork-witness.schema.json +107 -0
  161. package/schema/cobolwork.baseline.schema.json +101 -0
  162. package/schema/cobolwork.policy.schema.json +7 -0
  163. package/schema/cobolwork.site.schema.json +116 -0
package/lib/parser.mjs CHANGED
@@ -210,7 +210,8 @@ function evaluateCondition(text, defines) {
210
210
  return null;
211
211
  }
212
212
 
213
- export function normalize(src, format, defines, std) {
213
+ // A debugging line (D in the indicator) is source under WITH DEBUGGING MODE and a comment otherwise.
214
+ export function normalize(src, format, defines, std, debugging = false) {
214
215
  // Micro Focus reads a free-form line with * or / in column 1 as a comment, as cobc -std=mf does.
215
216
  const columnOneComments = std === 'mf' || std === 'mf-strict';
216
217
  const phys = src.split(/\r?\n/);
@@ -310,7 +311,7 @@ export function normalize(src, format, defines, std) {
310
311
  text = l.slice(7, 7 + width);
311
312
  if (!FIXED_INDICATORS.has(indicator)) { diags.push({ kind: 'invalid-indicator', line }); indicator = ' '; }
312
313
  }
313
- if (indicator === '*' || indicator === '/' || indicator === 'D' || indicator === 'd') continue;
314
+ if (indicator === '*' || indicator === '/' || ((indicator === 'D' || indicator === 'd') && !debugging)) continue;
314
315
  if (fmt === 'free' && (/^\s*\*>/.test(text) || (columnOneComments && /^[*/]/.test(text)))) continue;
315
316
  // AUTHOR, INSTALLATION, DATE-WRITTEN, DATE-COMPILED, SECURITY and REMARKS take a comment-entry:
316
317
  // any text, a COPY or an apostrophe included. It runs to the end of the header's line, and in
@@ -409,8 +410,9 @@ export function tokenize(norm, file) {
409
410
  if (c === ' ' || c === '\t' || c === '\f' || c === '\r') { i++; continue; }
410
411
  if (exec && exec.kind === 'SQL' && c === '-' && s[i + 1] === '-') break;
411
412
  if (picPending) {
413
+ // = is no PICTURE symbol, so == closes the pseudo-text a picture string ends.
412
414
  let j = i;
413
- while (j < n && s[j] !== ' ') j++;
415
+ while (j < n && s[j] !== ' ' && !(s[j] === '=' && s[j + 1] === '=')) j++;
414
416
  let pic = s.slice(i, j);
415
417
  if (pic.toUpperCase() === 'IS') { emit(tok('word', pic, line, file)); i = j; continue; }
416
418
  let trailingPeriod = false;
@@ -853,7 +855,7 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
853
855
  // A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
854
856
  // directive earlier in the including file may have changed from the file's starting format.
855
857
  const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
856
- const norm = normalize(src, fmt, ctx.defines, ctx.std);
858
+ const norm = normalize(src, fmt, ctx.defines, ctx.std, ctx.debugging);
857
859
  // A tag written against other text - :TAG:-FIELD, FS-(), 'X'-CLE - is replaced in the text, as the
858
860
  // compiler does, so the text around it joins the replacement into one word. A literal is taken as a
859
861
  // tag only where it touches a word; elsewhere it is replaced token for token like any operand.
@@ -896,6 +898,53 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
896
898
  return out;
897
899
  }
898
900
 
901
+ // REPLACE ==:TAG:== BY ==text== replaces the colon-delimited operand wherever the program text holds
902
+ // it, inside a word or a PICTURE as much as standing alone (IBM Enterprise COBOL, REPLACE statement,
903
+ // replacement rules). Like a tag in COPY REPLACING it is replaced in the text, from the statement on
904
+ // until REPLACE OFF or a REPLACE that does not name it again; a REPLACE itself is left as written.
905
+ const TAG = /^:[A-Za-z0-9_-]+:$/;
906
+
907
+ function replaceStatementAt(text, at) {
908
+ const rest = text.slice(at);
909
+ const off = /^REPLACE\s+OFF\s*\./i.exec(rest);
910
+ if (off) return { end: at + off[0].length, off: true, tags: [] };
911
+ const head = /^REPLACE(\s+ALSO)?\s*/i.exec(rest);
912
+ let k = head[0].length;
913
+ const tags = [];
914
+ let pairs = 0;
915
+ for (;;) {
916
+ const pair = /^(?:(?:LEADING|TRAILING)\s+)?==([\s\S]*?)==\s*BY\s*==([\s\S]*?)==\s*/i.exec(rest.slice(k));
917
+ if (!pair) break;
918
+ pairs++;
919
+ if (TAG.test(pair[1].trim())) tags.push([pair[1].trim(), pair[2].trim().replace(/\s+/g, ' ')]);
920
+ k += pair[0].length;
921
+ }
922
+ if (!pairs) return null;
923
+ if (rest[k] === '.') k++;
924
+ return { end: at + k, also: !!head[1], tags };
925
+ }
926
+
927
+ function replaceTags(norm) {
928
+ const text = norm.entries.map((e) => e.text).join('\n');
929
+ if (!/==\s*:[A-Za-z0-9_-]+:\s*==/.test(text)) return;
930
+ let active = [];
931
+ let out = '';
932
+ let from = 0;
933
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- TAG admits no metacharacter
934
+ const swap = (chunk) => active.reduce((t, [tag, to]) => t.replace(new RegExp(tag, 'gi'), to), chunk);
935
+ for (const m of text.matchAll(/\bREPLACE\b/gi)) {
936
+ if (m.index < from) continue;
937
+ const st = replaceStatementAt(text, m.index);
938
+ if (!st) continue;
939
+ out += swap(text.slice(from, m.index)) + text.slice(m.index, st.end);
940
+ active = st.off ? [] : st.also ? [...active, ...st.tags] : st.tags;
941
+ from = st.end;
942
+ }
943
+ out += swap(text.slice(from));
944
+ const lines = out.split('\n');
945
+ if (lines.length === norm.entries.length) norm.entries.forEach((e, i) => { e.text = lines[i]; });
946
+ }
947
+
899
948
  // A REPLACE governs the source up to the next one, which replaces it unless ALSO; OFF ends it; output is not rescanned.
900
949
  function applyReplaceStatements(tokens, ctx) {
901
950
  let active = [];
@@ -976,6 +1025,8 @@ function parseDataEntry(toks, section, file) {
976
1025
  continue;
977
1026
  }
978
1027
  if (u === 'PIC' || u === 'PICTURE') { j++; if (toks[j] && toks[j].u === 'IS') j++; if (toks[j]) item.picture = toks[j].v; j++; continue; }
1028
+ // PIC U BYTE-LENGTH n: a UTF-8 item of n bytes holding as many characters as fit.
1029
+ if (u === 'BYTE-LENGTH') { j++; if (toks[j] && toks[j].u === 'IS') j++; if (toks[j] && /^\d+$/.test(toks[j].v)) { item.byteLength = Number(toks[j].v); j++; } continue; }
979
1030
  if (u === 'USAGE') { j++; if (toks[j] && toks[j].u === 'IS') j++; if (toks[j]) { item.usage = toks[j].u; j++; } continue; }
980
1031
  if (USAGE_WORDS.has(u)) { item.usage = u; j++; continue; }
981
1032
  // OCCURS DYNAMIC [CAPACITY IN name] [FROM n] [TO m]: sized at its most, none without TO; the
@@ -1048,6 +1099,9 @@ function parseDataEntry(toks, section, file) {
1048
1099
  if (u === 'CONSTANT') { item.constant = true; j++; if (toks[j] && toks[j].u === 'AS') j++; while (toks[j] && (toks[j].t === 'num' || toks[j].t === 'lit' || (toks[j].t === 'word' && /^\d+$/.test(toks[j].v)))) item.values.push(toks[j++]); continue; }
1049
1100
  if (u === 'VALUE' || u === 'VALUES') {
1050
1101
  j++;
1102
+ // A screen item's VALUE is one literal; what follows it is the item's other clauses, and a later
1103
+ // VALUE (two entries run together by a missing period) replaces it.
1104
+ if (section === 'SCREEN') { if (toks[j] && toks[j].u === 'IS') j++; if (toks[j]) { item.values = [toks[j]]; j++; } continue; }
1051
1105
  while (toks[j] && (toks[j].t !== 'word' || !(DATA_CLAUSE_WORDS.has(toks[j].u) || (report && REPORT_CLAUSE_WORDS.has(toks[j].u))) || toks[j].u === 'IS')) { item.values.push(toks[j]); j++; }
1052
1106
  continue;
1053
1107
  }
@@ -1073,6 +1127,25 @@ function debugItem(at) {
1073
1127
  return [root, ...root.children];
1074
1128
  }
1075
1129
 
1130
+ // DEBUG-CONTENTS is X(n), n left to the compiler (IBM, DEBUG-ITEM). GnuCOBOL makes it as long as the
1131
+ // longest item USE FOR DEBUGGING names, a CD's records included, and never shorter than 30.
1132
+ function sizeDebugContents(prog, tokens, from, to, scheme, constants) {
1133
+ const named = [];
1134
+ for (let i = from; i < to - 3; i++) {
1135
+ if (tokens[i].u !== 'USE' || tokens[i + 1].u !== 'FOR' || tokens[i + 2].u !== 'DEBUGGING') continue;
1136
+ for (let k = i + 3; k < to && tokens[k].t !== 'period'; k++) if (tokens[k].t === 'word') named.push(tokens[k].u);
1137
+ }
1138
+ const sizes = named.flatMap((n) => prog.items.filter((it) => it.name === n && it.level < 50 && !it.implicit).map((it) => it.size || 0)
1139
+ .concat(prog.items.filter((it) => it.cd === n).map((it) => it.size || 0)));
1140
+ const longest = Math.max(30, ...sizes);
1141
+ if (longest === 30) return;
1142
+ const root = prog.items.find((it) => it.implicit && it.name === 'DEBUG-ITEM');
1143
+ const contents = root && root.children.find((c) => c.name === 'DEBUG-CONTENTS');
1144
+ if (!contents) return;
1145
+ contents.picture = `X(${longest})`;
1146
+ computeSizes([root], scheme, constants);
1147
+ }
1148
+
1076
1149
  // TYPE gives an item the picture, usage and subordinate items of the TYPEDEF it names. The listing
1077
1150
  // prints the typed item alone; what it holds is implied, reachable by qualification (RE OF WS-Z),
1078
1151
  // so the copies are returned as items marked typeClone.
@@ -1169,6 +1242,7 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1169
1242
  if (s[m]) f.assign = { t: s[m].t, v: s[m].t === 'word' ? s[m].u : s[m].v };
1170
1243
  }
1171
1244
  for (let m = n + 1; m < s.length; m++) if (s[m].t === 'word') f.envRefs.push(s[m]);
1245
+ f.lineSequential = s.some((t, m) => t.u === 'LINE' && s[m + 1] && s[m + 1].u === 'SEQUENTIAL');
1172
1246
  prog.files.push(f);
1173
1247
  } else {
1174
1248
  for (const t of s) if (t.t === 'word') prog.refs.push({ tok: t, zone: 'env' });
@@ -1182,6 +1256,7 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1182
1256
  let stack = [];
1183
1257
  let currentFd = null;
1184
1258
  let currentRd = null;
1259
+ let currentCd = null;
1185
1260
  const dataSentences = sentences(tokens, dataStart, dataEnd);
1186
1261
  for (let s of dataSentences) {
1187
1262
  // What an EXEC SQL INCLUDE brings in follows it in the same sentence, up to its own first period.
@@ -1196,6 +1271,7 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1196
1271
  section = first.u === 'RD' ? 'REPORT' : 'COMMUNICATION';
1197
1272
  currentFd = null;
1198
1273
  currentRd = first.u === 'RD' ? { name: s[1].u, line: first.line, file: first.file, groups: [] } : null;
1274
+ currentCd = first.u === 'CD' ? s[1].u : null;
1199
1275
  if (currentRd) (prog.reports ||= []).push(currentRd);
1200
1276
  else (prog.cds ||= []).push(s[1].u);
1201
1277
  stack = [];
@@ -1204,12 +1280,7 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1204
1280
  if (first.t === 'word' && (first.u === 'FD' || first.u === 'SD') && s[1]) {
1205
1281
  section = 'FILE';
1206
1282
  currentRd = null;
1207
- currentFd = { kind: first.u, name: s[1].u, line: first.line, file: first.file, records: [], fdTok: s[1], declaredMax: 0, varying: false };
1208
- const recAt = s.findIndex(x => x.t === 'word' && x.u === 'RECORD');
1209
- if (recAt >= 0) for (let m = recAt + 1; m < s.length && !['LABEL', 'BLOCK', 'DATA', 'VALUE', 'RECORDING', 'CODE-SET', 'LINAGE', 'REPORT', 'REPORTS', 'DEPENDING'].includes(s[m].u); m++) {
1210
- if (s[m].u === 'VARYING') currentFd.varying = true;
1211
- if (/^\d+$/.test(s[m].v)) currentFd.declaredMax = Math.max(currentFd.declaredMax, Number(s[m].v));
1212
- }
1283
+ currentFd = { kind: first.u, name: s[1].u, line: first.line, file: first.file, records: [], fdTok: s[1], ...recordClause(s) };
1213
1284
  const repAt = s.findIndex(x => x.t === 'word' && (x.u === 'REPORT' || x.u === 'REPORTS'));
1214
1285
  if (repAt >= 0) {
1215
1286
  currentFd.reports = [];
@@ -1225,6 +1296,7 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1225
1296
  }
1226
1297
  if ((first.t === 'word' || first.t === 'num') && /^\d{1,2}$/.test(first.v)) {
1227
1298
  const item = parseDataEntry(s, section, first.file);
1299
+ if (section === 'COMMUNICATION' && item.level === 1) item.cd = currentCd;
1228
1300
  if (item.level === 88 || item.level === 66) {
1229
1301
  const parent = item.level === 88 ? stack[stack.length - 1] : null;
1230
1302
  if (parent) { parent.children.push(item); item.parent = parent; }
@@ -1283,6 +1355,7 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1283
1355
  || siblings.find(x => x !== it && x.name === it.redefines) || null;
1284
1356
  }
1285
1357
  computeSizes(roots, scheme, constants);
1358
+ if (prog.debuggingMode && procAt >= 0) sizeDebugContents(prog, tokens, procAt, procEnd, scheme, constants);
1286
1359
  const subtreeCache = new Map();
1287
1360
  const subtree = (record) => {
1288
1361
  if (subtreeCache.has(record)) return subtreeCache.get(record);
@@ -1319,12 +1392,18 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1319
1392
  it.renamesSpan = all.filter(x => x.offset >= from && x.offset + x.contributes <= to && x !== it);
1320
1393
  }
1321
1394
  for (const rd of prog.reports || []) layoutReport(rd);
1322
- // A file is as large as the longer of its declared length and its record description; the
1323
- // compiler warns when a record exceeds the declared maximum but still uses the record. A report
1324
- // file's record is as wide as the widest line of the reports written to it.
1395
+ // A file's record is as long as its RECORD clause allows, or its longest record description
1396
+ // where that is longer or there is no clause; a record longer than the clause still wins, as the
1397
+ // compiler warns and uses it. The FD of a line-sequential file, which GnuCOBOL has and IBM does
1398
+ // not, is sized by its records whatever RECORD CONTAINS says, though RECORD IS VARYING still sets
1399
+ // its maximum. A report file's record is as wide as the widest line of the reports written to it.
1325
1400
  for (const fd of prog.fds || []) {
1326
1401
  const reports = (fd.reports || []).map(n => (prog.reports || []).find(r => r.name === n)).filter(Boolean);
1327
- fd.size = Math.max(fd.declaredMax || 0, 0, ...fd.records.map(r => r.size || 0), ...reports.map(r => r.width || 0));
1402
+ const described = Math.max(0, ...fd.records.map(recordLength), ...reports.map(r => r.width || 0));
1403
+ const select = prog.files.find(f => f.name === fd.name);
1404
+ const ignored = select && select.lineSequential && fd.kind === 'FD' && fd.recordFormat !== 3;
1405
+ const clause = ignored ? 0 : fd.fixedLength || fd.varyingMax || fd.rangeMax || 0;
1406
+ fd.size = Math.max(clause, described);
1328
1407
  }
1329
1408
  }
1330
1409
 
@@ -1334,6 +1413,34 @@ function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
1334
1413
  return prog;
1335
1414
  }
1336
1415
 
1416
+ const FD_CLAUSES = new Set(['LABEL', 'BLOCK', 'DATA', 'VALUE', 'RECORDING', 'CODE-SET', 'LINAGE', 'REPORT', 'REPORTS', 'DEPENDING', 'RECORD']);
1417
+
1418
+ // The RECORD clause of an FD in IBM's three formats, CONTAINS n, CONTAINS n TO m and IS VARYING ...
1419
+ // TO m, each giving the longest record the file holds. LABEL RECORD and DATA RECORD are other clauses
1420
+ // that share the word.
1421
+ function recordClause(s) {
1422
+ for (let at = 1; at < s.length; at++) {
1423
+ if (s[at].u !== 'RECORD' || ['LABEL', 'DATA'].includes(s[at - 1].u)) continue;
1424
+ let varying = false;
1425
+ let to = false;
1426
+ const nums = [];
1427
+ let toNum = null;
1428
+ for (let m = at + 1; m < s.length && !FD_CLAUSES.has(s[m].u); m++) {
1429
+ if (s[m].u === 'VARYING') varying = true;
1430
+ else if (s[m].u === 'TO') to = true;
1431
+ else if (/^\d+$/.test(s[m].v)) { nums.push(Number(s[m].v)); if (to && toNum == null) toNum = Number(s[m].v); }
1432
+ }
1433
+ if (varying) return { recordFormat: 3, varyingMax: toNum };
1434
+ if (nums.length === 1 && !to) return { recordFormat: 1, fixedLength: nums[0] };
1435
+ if (nums.length) return { recordFormat: 2, rangeMax: nums.at(-1) };
1436
+ }
1437
+ return { recordFormat: null };
1438
+ }
1439
+
1440
+ // A record's length in its file. OCCURS on a level-01 record is a GnuCOBOL extension IBM refuses;
1441
+ // GnuCOBOL sizes the file by one occurrence.
1442
+ const recordLength = (r) => (r.occurs > 1 ? r.size / r.occurs : r.size || 0);
1443
+
1337
1444
  export function segmentEnd(tokens, i, to) {
1338
1445
  let depth = 0;
1339
1446
  for (let k = i + 1; k < to; k++) {
@@ -1485,10 +1592,15 @@ function parseProcedure(tokens, procAt, to, prog, hostVariables) {
1485
1592
  const constText = target.t === 'word' && prog.constants && prog.constants.text
1486
1593
  ? prog.constants.text.get(target.u) : undefined;
1487
1594
  const resolved = target.t === 'lit' ? target.v : constText;
1595
+ // A literal program name is padded to its field; the compiler drops the trailing spaces. A
1596
+ // literal holding a path calls the program its last component names, as the runtime loads it.
1597
+ const literal = resolved === undefined ? null : String(resolved).trimEnd();
1598
+ const name = literal === null ? target.u : literal.split(/[\\/]/).pop();
1488
1599
  prog.calls.push({
1489
- // A literal program name is padded to its field; the compiler drops the trailing spaces.
1490
1600
  kind: resolved === undefined ? 'I' : 'L',
1491
- name: resolved === undefined ? target.u : String(resolved).trimEnd(),
1601
+ name,
1602
+ ...(literal !== null && name !== literal ? { calledAs: literal } : {}),
1603
+ ...(target.t === 'lit' && target.prefix === 'X' ? { hex: true } : {}),
1492
1604
  ...(constText !== undefined ? { viaConstant: target.u } : {}),
1493
1605
  line: t.line, file: t.file, targetTok: target, using: args, stmtIndex: prog.statements.length - 1,
1494
1606
  });
@@ -1953,8 +2065,15 @@ export function parseSource(src, file, opts = {}) {
1953
2065
  // Where a copybook's text comes from: the disk, or the source tree the program was read from.
1954
2066
  readText: opts.readText || ((p) => readSource(p).text) };
1955
2067
  collectCopybookDefines(src, format, ctx, 0);
1956
- const norm = normalize(src, format, ctx.defines, opts.std);
2068
+ let norm = normalize(src, format, ctx.defines, opts.std);
2069
+ // WITH DEBUGGING MODE in SOURCE-COMPUTER makes the debugging lines of the program and its
2070
+ // copybooks source; it is read from the text with those lines still comments.
2071
+ if (/\bSOURCE-COMPUTER\s*\.[^.]*\bDEBUGGING\s+MODE\b/i.test(norm.entries.map(e => e.text).join(' '))) {
2072
+ ctx.debugging = true;
2073
+ norm = normalize(src, format, ctx.defines, opts.std, true);
2074
+ }
1957
2075
  ctx.mainFormat = norm.finalFormat;
2076
+ replaceTags(norm);
1958
2077
  const { tokens: raw, diags } = tokenize(norm, file);
1959
2078
  ctx.diags.push(...norm.diags.map(d => ({ ...d, file })), ...diags);
1960
2079
  const expanded = stripDirecting(applyReplaceStatements(expand(raw, ctx, [resolve(file)], norm.finalFormat), ctx));
@@ -0,0 +1,15 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // The PL/I statement cursor: the shared token cursor, failing with PliSyntax.
3
+ import { tokenCursor, split } from '../statement-cursor.mjs';
4
+
5
+ export class PliSyntax extends Error {
6
+ constructor(message, tok) {
7
+ super(message);
8
+ this.name = 'PliSyntax';
9
+ this.line = tok ? tok.line : null;
10
+ this.col = tok ? tok.col : null;
11
+ }
12
+ }
13
+
14
+ export const cursor = (toks) => tokenCursor(toks, PliSyntax);
15
+ export { split };
@@ -0,0 +1,101 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // PL/I expressions and references, as the statement parsers consume them: an operator tree by
3
+ // PL/I precedence, every token consumed, and the references the expression reads.
4
+ import { cursor } from './cursor.mjs';
5
+
6
+ // Higher binds tighter. Infix ¬ is exclusive-or, at the level of |.
7
+ const PREC = {
8
+ '**': 7, '*': 6, '/': 6, '+': 5, '-': 5, '||': 4,
9
+ '=': 3, '¬=': 3, '<': 3, '>': 3, '<=': 3, '>=': 3, '¬<': 3, '¬>': 3, '<>': 3,
10
+ '&': 2, '|': 1, '¬': 1,
11
+ };
12
+ const PREFIX = new Set(['+', '-', '¬']);
13
+
14
+ // A reference: a name, then any mix of (arguments), '.' name, and '->' or '=>' locator
15
+ // qualification. args holds one list per parenthesised group, each argument an expression or
16
+ // { t: 'star' } for '*'.
17
+ export function parseReference(c) {
18
+ const start = c.pos;
19
+ const first = c.word();
20
+ if (!first) c.fail('a reference');
21
+ const path = [first.u];
22
+ const locators = [];
23
+ const args = [];
24
+ for (;;) {
25
+ if (c.isOp('(')) { args.push(c.items().map(argument)); continue; }
26
+ if (c.isOp('.') && c.isWord(undefined, 1)) { c.next(); path.push(c.next().u); continue; }
27
+ if (c.isOp(['->', '=>']) && c.isWord(undefined, 1)) { c.next(); locators.push(path.length); path.push(c.next().u); continue; }
28
+ break;
29
+ }
30
+ return { t: 'ref', name: path[path.length - 1], path, locators, args, toks: c.slice(start) };
31
+ }
32
+
33
+ function argument(toks) {
34
+ if (toks.length === 1 && toks[0].t === 'op' && toks[0].v === '*') return { t: 'star', toks };
35
+ const s = cursor(toks);
36
+ const e = parseExpression(s);
37
+ if (!s.done()) s.fail('the end of the argument');
38
+ return e;
39
+ }
40
+
41
+ function primary(c) {
42
+ const t = c.peek();
43
+ if (!t) c.fail('an operand');
44
+ if (t.t === 'num') return { t: 'num', tok: c.next() };
45
+ if (t.t === 'lit') return { t: 'lit', tok: c.next() };
46
+ if (t.t === 'word') return parseReference(c);
47
+ if (t.t === 'op' && t.v === '(') {
48
+ c.next();
49
+ const inner = parseExpression(c);
50
+ c.expectOp(')');
51
+ // (3)'AB' and (N)'0'B repeat the literal: a parenthesis is never otherwise followed by one.
52
+ if (c.peek() && c.peek().t === 'lit') return { t: 'lit', tok: c.next(), factor: inner };
53
+ return { t: 'paren', expr: inner };
54
+ }
55
+ return c.fail('an operand');
56
+ }
57
+
58
+ // A prefix operator applies to its operand's ** chain: -A**2 is -(A**2).
59
+ function unary(c) {
60
+ const t = c.peek();
61
+ if (t && t.t === 'op' && PREFIX.has(t.v)) {
62
+ c.next();
63
+ return { op: `prefix${t.v}`, args: [binary(c, 7)] };
64
+ }
65
+ return primary(c);
66
+ }
67
+
68
+ function binary(c, min, stopOps = []) {
69
+ let left = unary(c);
70
+ for (;;) {
71
+ const t = c.peek();
72
+ if (!t || t.t !== 'op' || !(t.v in PREC) || stopOps.includes(t.v)) return left;
73
+ const p = PREC[t.v];
74
+ if (p < min) return left;
75
+ c.next();
76
+ const right = binary(c, t.v === '**' ? p : p + 1, stopOps);
77
+ left = { op: t.v, args: [left, right] };
78
+ }
79
+ }
80
+
81
+ function collect(node, out) {
82
+ if (!node) return;
83
+ if (node.t === 'ref') {
84
+ for (const at of node.locators) out.push(node.path.slice(0, at).join('.'));
85
+ out.push(node.path.join('.'));
86
+ for (const list of node.args) for (const a of list) collect(a.tree ?? a, out);
87
+ } else if (node.t === 'paren') collect(node.expr.tree, out);
88
+ else if (node.t === 'lit' && node.factor) collect(node.factor.tree, out);
89
+ else if (node.args) for (const a of node.args) collect(a, out);
90
+ }
91
+
92
+ // An expression, stopping before the first token that cannot continue it: ',' or ')', a word after a
93
+ // complete operand (so a caller's stopWords need no test), an assignment operator, or an op in
94
+ // stopOps. { t: 'expr', tree, toks, refs }.
95
+ export function parseExpression(c, { stopOps = [] } = {}) {
96
+ const start = c.pos;
97
+ const tree = binary(c, 1, stopOps);
98
+ const refs = [];
99
+ collect(tree, refs);
100
+ return { t: 'expr', tree, toks: c.slice(start), refs: [...new Set(refs)] };
101
+ }
@@ -0,0 +1,82 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // %INCLUDE expansion at the token level, as the preprocessor does it before statements exist: a
3
+ // member can hold the rest of a DECLARE (`DCL 1 REC, %INCLUDE RECFLDS;`), so a member's tokens
4
+ // replace the directive in the token stream and statements are split afterwards. Every token keeps
5
+ // the file and line it was read from.
6
+ import { basename, dirname, extname } from 'node:path';
7
+ import { tokenize, statements } from './lex.mjs';
8
+
9
+ // Upper-case member name -> paths of the files that could hold it, in path order.
10
+ export function membersOf(paths) {
11
+ const out = new Map();
12
+ for (const p of [...paths].sort()) {
13
+ const name = basename(p, extname(p)).toUpperCase();
14
+ if (!out.has(name)) out.set(name, []);
15
+ out.get(name).push(p);
16
+ }
17
+ return out;
18
+ }
19
+
20
+ // The member a directive in `from` means: one in the same directory if there is one, else the first.
21
+ export function chooseMember(candidates, from) {
22
+ if (!candidates || !candidates.length) return null;
23
+ return candidates.find((p) => dirname(p) === dirname(from)) || candidates[0];
24
+ }
25
+
26
+ // The members a %INCLUDE or %XINCLUDE directive names: name, 'name' or ddname(name), comma-separated.
27
+ function directiveMembers(toks) {
28
+ const names = [];
29
+ for (let k = 2; k < toks.length; k++) {
30
+ const t = toks[k];
31
+ if (t.t === 'op' && t.v === ',') continue;
32
+ if (t.t === 'word' && toks[k + 1]?.t === 'op' && toks[k + 1].v === '(' && toks[k + 3]?.t === 'op' && toks[k + 3].v === ')') {
33
+ const m = toks[k + 2];
34
+ names.push(String(m.u ?? m.v).toUpperCase());
35
+ k += 3;
36
+ } else if (t.t === 'word') names.push(t.u);
37
+ else if (t.t === 'lit') names.push(t.v.toUpperCase());
38
+ }
39
+ return names;
40
+ }
41
+
42
+ // readMember(name, fromFile) -> { path, text } or null.
43
+ export function expandIncludes(tokens, { file, readMember, maxDepth = 16 }) {
44
+ const included = [];
45
+ const unresolved = [];
46
+ const cycles = [];
47
+ const seen = new Set();
48
+ const walk = (toks, from, chain) => {
49
+ const out = [];
50
+ for (let i = 0; i < toks.length; i++) {
51
+ const t = toks[i];
52
+ const d = toks[i + 1];
53
+ if (!(t.t === 'op' && t.v === '%' && d && d.t === 'word' && (d.u === 'INCLUDE' || d.u === 'XINCLUDE'))) { out.push(t); continue; }
54
+ let end = i;
55
+ while (end < toks.length && toks[end].t !== 'semi') end++;
56
+ const directive = toks.slice(i, end + 1);
57
+ const keep = [];
58
+ for (const name of directiveMembers(directive)) {
59
+ if (d.u === 'XINCLUDE' && seen.has(name)) continue;
60
+ if (chain.includes(name)) { cycles.push({ chain: [...chain, name] }); continue; }
61
+ if (chain.length >= maxDepth) { unresolved.push({ name, file: t.file ?? from, line: t.line, why: 'too deep' }); continue; }
62
+ const m = readMember(name, from);
63
+ if (!m) { unresolved.push({ name, file: t.file ?? from, line: t.line }); keep.push(name); continue; }
64
+ seen.add(name);
65
+ included.push({ name, path: m.path, from: { file: t.file ?? from, line: t.line } });
66
+ out.push(...walk(tokenize(m.text, { file: m.path }).tokens, m.path, [...chain, name]));
67
+ }
68
+ // An unresolved member keeps its directive, so the statement is still there to be counted.
69
+ if (keep.length) out.push(...directive);
70
+ i = end;
71
+ }
72
+ return out;
73
+ };
74
+ return { tokens: walk(tokens, file, []), included, unresolved, cycles };
75
+ }
76
+
77
+ // readPli with %INCLUDE members spliced in.
78
+ export function readPliExpanded(text, { file, readMember }) {
79
+ const { tokens, diags, process, margins } = tokenize(text, { file });
80
+ const ex = expandIncludes(tokens, { file, readMember });
81
+ return { statements: statements(ex.tokens), diags, process, margins, included: ex.included, unresolved: ex.unresolved, cycles: ex.cycles };
82
+ }
@@ -0,0 +1,125 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // PL/I structure mapping: DECLARE items into structure trees, and each member's offset and size by the
3
+ // Enterprise PL/I Language Reference's rules, innermost minor structures first, every pair of units
4
+ // placed with the first shifted toward the second as far as its alignment allows.
5
+ import { storageOf } from './storage.mjs';
6
+
7
+ const DW = 64;
8
+ const mod = (a, m) => ((a % m) + m) % m;
9
+
10
+ // The trees a DECLARE's items describe. A member's logical level is one deeper than the structure
11
+ // that holds it, whatever level numbers the source wrote.
12
+ export function structuresOf(items) {
13
+ const roots = [];
14
+ const stack = [];
15
+ for (const it of items) {
16
+ const node = { name: it.name, level: it.level, dims: it.dims || [], attributes: it.attributes || [], line: it.line, ...(it.file ? { file: it.file } : {}), children: [] };
17
+ if (it.level == null || it.level <= 1) { roots.push(node); stack.length = 0; if (it.level != null) stack.push(node); node.logical = 1; continue; }
18
+ while (stack.length && stack[stack.length - 1].level >= it.level) stack.pop();
19
+ const parent = stack[stack.length - 1];
20
+ if (!parent) { roots.push(node); node.logical = 1; node.orphan = true; stack.push(node); continue; }
21
+ node.logical = parent.logical + 1;
22
+ parent.children.push(node);
23
+ stack.push(node);
24
+ }
25
+ return roots;
26
+ }
27
+
28
+ // The number of elements a dimension list gives, or null when a bound is not a constant.
29
+ function extent(dims) {
30
+ let n = 1;
31
+ for (const d of dims) {
32
+ const parts = [[]];
33
+ for (const t of d) { if (t.t === 'op' && t.v === ':') parts.push([]); else parts[parts.length - 1].push(t); }
34
+ const num = (p) => {
35
+ if (p.length === 1 && p[0].t === 'num' && /^\d+$/.test(p[0].v)) return Number(p[0].v);
36
+ if (p.length === 2 && p[0].t === 'op' && (p[0].v === '-' || p[0].v === '+') && p[1].t === 'num' && /^\d+$/.test(p[1].v)) return (p[0].v === '-' ? -1 : 1) * Number(p[1].v);
37
+ return null;
38
+ };
39
+ const [lo, hi] = parts.length === 2 ? [num(parts[0]), num(parts[1])] : [1, num(parts[0])];
40
+ if (lo == null || hi == null) return null;
41
+ n *= Math.max(0, hi - lo + 1);
42
+ }
43
+ return n;
44
+ }
45
+
46
+ const explicitAlignment = (node) => (node.attributes.some((a) => a.name === 'ALIGNED') ? true : node.attributes.some((a) => a.name === 'UNALIGNED') ? false : null);
47
+
48
+ // Pairs two units: the first begins at its offset from a doubleword boundary, the second at the
49
+ // first position after it that its alignment allows, then the first moves toward the second by
50
+ // whole multiples of its own alignment. Positions are in bits.
51
+ function pair(a, b) {
52
+ const end = a.offset + a.bits;
53
+ const start = b.structure ? end + mod(b.offset - end, b.align) : Math.ceil(end / b.align) * b.align;
54
+ const shift = Math.floor((start - end) / a.align) * a.align;
55
+ const first = a.offset + shift;
56
+ return { first, second: start, unit: { offset: mod(first, DW), bits: start + b.bits - first, align: Math.max(a.align, b.align), structure: true } };
57
+ }
58
+
59
+ // Maps one node: an element takes its storage; a structure maps its members into one unit, a union
60
+ // overlays them. Every node gets `at`, its start in bits within the unit of its parent.
61
+ function mapNode(node, inherited, problems) {
62
+ const own = explicitAlignment(node);
63
+ const inherit = own ?? inherited;
64
+ const count = node.dims.length ? extent(node.dims) : 1;
65
+ let unit;
66
+ if (!node.children.length) {
67
+ const s = storageOf(node.attributes, { inherited, name: node.name });
68
+ node.storage = s;
69
+ if (!s.known) problems.push({ name: node.name, line: node.line, why: s.type === 'TYPE' ? `type ${s.typeName} is defined by a DEFINE this reading does not resolve` : s.type ? `the extent of ${s.type} is not a constant` : 'no data attributes' });
70
+ unit = { offset: 0, bits: s.bits ?? 0, align: s.align, structure: false };
71
+ } else if (node.attributes.some((a) => a.name === 'UNION')) {
72
+ const members = node.children.map((ch) => mapNode(ch, inherit, problems));
73
+ let len = 0;
74
+ node.children.forEach((ch, k) => { ch.at = mod(members[k].offset, members[k].align); len = Math.max(len, ch.at + members[k].bits); });
75
+ unit = { offset: 0, bits: len, align: Math.max(...members.map((m) => m.align)), structure: true };
76
+ } else {
77
+ const members = node.children.map((ch) => mapNode(ch, inherit, problems));
78
+ let acc = { ...members[0] };
79
+ const starts = [members[0].offset];
80
+ for (let k = 1; k < members.length; k++) {
81
+ const { first, second, unit: u } = pair(acc, members[k]);
82
+ const moved = first - acc.offset;
83
+ for (let j = 0; j < k; j++) starts[j] += moved;
84
+ starts.push(starts[0] + (second - first));
85
+ acc = u;
86
+ }
87
+ const base = starts[0];
88
+ node.children.forEach((ch, k) => { ch.at = starts[k] - base; });
89
+ unit = { offset: acc.offset, bits: acc.bits, align: acc.align, structure: true };
90
+ }
91
+ if (count == null) problems.push({ name: node.name, line: node.line, why: 'a dimension bound is not a constant' });
92
+ node.count = count;
93
+ // Each element of an array starts on the same boundary, so an element is padded to its alignment.
94
+ const stride = count > 1 ? Math.ceil(unit.bits / unit.align) * unit.align : unit.bits;
95
+ node.strideBits = stride;
96
+ return count > 1 ? { ...unit, bits: stride * count } : unit;
97
+ }
98
+
99
+ // Byte and bit offsets from the start of the major structure, and sizes, set on every node.
100
+ function place(node, atBits) {
101
+ node.offsetBits = atBits;
102
+ node.offset = Math.floor(atBits / 8);
103
+ if (atBits % 8) node.bitOffset = atBits % 8;
104
+ node.size = Math.ceil(node.strideBits / 8);
105
+ node.occurs = node.count ?? null;
106
+ for (const ch of node.children) place(ch, atBits + ch.at);
107
+ }
108
+
109
+ // Lays out every structure a DECLARE describes. Returns the trees, each node carrying offset (bytes
110
+ // from its major structure), bitOffset when not on a byte, size (bytes of one element), occurs, and
111
+ // storage for an element; problems lists what could not be sized.
112
+ export function layout(items) {
113
+ const roots = structuresOf(items);
114
+ const problems = [];
115
+ for (const r of roots) {
116
+ const u = mapNode(r, null, problems);
117
+ // Storage begins at the byte holding the first bit; unaligned bits shifted toward their
118
+ // successor leave their padding there.
119
+ const lead = u.offset % 8;
120
+ r.totalBits = u.bits;
121
+ place(r, lead);
122
+ r.total = Math.ceil((lead + u.bits) / 8);
123
+ }
124
+ return { roots, problems };
125
+ }