@portll/cobolwork 0.0.1 → 0.2.76
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/LICENSING.md +93 -0
- package/NOTICE +9 -0
- package/README.md +325 -3
- package/THIRD-PARTY-NOTICES.md +118 -0
- package/bin/cobolwork.mjs +354 -0
- package/lib/advisories.mjs +133 -0
- package/lib/baseline.mjs +154 -0
- package/lib/bms.mjs +453 -0
- package/lib/build.mjs +402 -0
- package/lib/capabilities.mjs +79 -0
- package/lib/cics-commands.mjs +281 -0
- package/lib/compliance.mjs +81 -0
- package/lib/consequence.mjs +139 -0
- package/lib/control.mjs +1515 -0
- package/lib/csd.mjs +77 -0
- package/lib/dataflow.mjs +1506 -0
- package/lib/diff.mjs +344 -0
- package/lib/explain.mjs +145 -0
- package/lib/gate.mjs +383 -0
- package/lib/index.mjs +6 -0
- package/lib/inventory.mjs +79 -0
- package/lib/jcl.mjs +478 -0
- package/lib/kernel/findings.mjs +94 -0
- package/lib/kernel/identity.mjs +216 -0
- package/lib/kernel/memory.mjs +217 -0
- package/lib/kernel/printable.mjs +6 -0
- package/lib/kernel/registry.mjs +79 -0
- package/lib/kernel/ruleset.mjs +72 -0
- package/lib/kernel/source-tree.mjs +159 -0
- package/lib/kev.mjs +27 -0
- package/lib/options.mjs +512 -0
- package/lib/packs.mjs +148 -0
- package/lib/parser.mjs +2055 -0
- package/lib/policy.mjs +163 -0
- package/lib/precompile-cics.mjs +169 -0
- package/lib/precompile.mjs +544 -0
- package/lib/reach.mjs +122 -0
- package/lib/revision.json +1 -0
- package/lib/revision.mjs +89 -0
- package/lib/sarif.mjs +222 -0
- package/lib/scan.mjs +272 -0
- package/lib/sets/build.mjs +234 -0
- package/lib/sets/cics.mjs +306 -0
- package/lib/sets/compile.mjs +187 -0
- package/lib/sets/copybook.mjs +174 -0
- package/lib/sets/flow.mjs +487 -0
- package/lib/sets/hidden.mjs +216 -0
- package/lib/sets/jcl.mjs +440 -0
- package/lib/sets/log.mjs +406 -0
- package/lib/sets/opaque.mjs +102 -0
- package/lib/sets/priv.mjs +322 -0
- package/lib/sets/recon.mjs +267 -0
- package/lib/sets/vendor.mjs +117 -0
- package/lib/sets/web.mjs +327 -0
- package/lib/site.mjs +164 -0
- package/lib/sources.mjs +156 -0
- package/lib/tui/app.mjs +325 -0
- package/lib/tui/keys.mjs +39 -0
- package/lib/tui/model.mjs +96 -0
- package/lib/tui/run.mjs +38 -0
- package/lib/tui/screen.mjs +59 -0
- package/lib/tui/terminal.mjs +46 -0
- package/lib/utilities.mjs +296 -0
- package/lib/version.mjs +15 -0
- package/lib/words.mjs +318 -0
- package/package.json +45 -6
- package/rules/advisories.json +264 -0
- package/rules/compliance-dora.json +2151 -0
- package/rules/compliance-ffiec.json +2134 -0
- package/rules/compliance-nist80053.json +2134 -0
- package/rules/gitleaks-mainframe.toml +57 -0
- package/rules/kev-ids.json +1729 -0
- package/rules/packs/broadcom.json +124 -0
- package/rules/packs/connectdirect.json +116 -0
- package/rules/packs/controlm.json +114 -0
- package/rules/system-layouts.json +28 -0
- package/schema/cobolwork-coverage.schema.json +65 -0
- package/schema/cobolwork.policy.schema.json +54 -0
package/lib/parser.mjs
ADDED
|
@@ -0,0 +1,2055 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-or-later
|
|
2
|
+
import { readdirSync, realpathSync, statSync } from 'node:fs';
|
|
3
|
+
import { readSource } from './sources.mjs';
|
|
4
|
+
import { optionTokens } from './options.mjs';
|
|
5
|
+
import {
|
|
6
|
+
RESERVED_WORDS, SPECIAL_REGISTERS, SYSTEM_NAMES, INTRINSIC_FUNCTIONS,
|
|
7
|
+
CONTEXT_SENSITIVE_WORDS, DIRECTIVE_WORDS, EXCEPTION_CONDITIONS,
|
|
8
|
+
EIB_FIELDS, DIB_FIELDS, SQLCA_FIELDS,
|
|
9
|
+
} from './words.mjs';
|
|
10
|
+
import { CICS_COMMANDS, CICS_EVERY_COMMAND } from './cics-commands.mjs';
|
|
11
|
+
import { cicsCommand } from './precompile-cics.mjs';
|
|
12
|
+
import { join, dirname, basename, resolve, isAbsolute, sep, delimiter } from 'node:path';
|
|
13
|
+
|
|
14
|
+
const FIXED_INDICATORS = new Set([' ', '*', '/', '-', 'D', 'd', '$']);
|
|
15
|
+
const COMMENT_ENTRY = /^\s*(AUTHOR|INSTALLATION|DATE-WRITTEN|DATE-COMPILED|SECURITY|REMARKS)\s*\./i;
|
|
16
|
+
const COPY_EXTS = ['', '.CPY', '.CBL', '.COB', '.cpy', '.cbl', '.cob', '.copy', '.COPY', '.inc', '.INC'];
|
|
17
|
+
// Caps on one parse's COPY expansion: nested COPYs double per level, and the corpus maximum is 86 inclusions.
|
|
18
|
+
const MAX_INCLUSIONS = 2000;
|
|
19
|
+
const MAX_COPY_TOKENS = 2000000;
|
|
20
|
+
// Caps on what REPLACING and REPLACE may add per parse; nested REPLACING doubles text at every level.
|
|
21
|
+
const MAX_REPLACED_GROWTH = 500000;
|
|
22
|
+
const MAX_REPLACED_CHARS = 16000000;
|
|
23
|
+
// Bounded: a mainframe member name is at most eight characters, and the loose prefix form counted
|
|
24
|
+
// an absent ATTRIBUTES.cpy as a system copybook, which made an incomplete tree read as covered.
|
|
25
|
+
const SYSTEM_COPY = /^(?:(?:DFH|DSN|CEE|IGZ|EZA|BPX|CSQ|CMQ|DLI)[A-Z0-9$#@]{0,5}|SQLCA|SQLDA|ATTRIB)$/i;
|
|
26
|
+
const EXEC_KINDS = new Set(['SQL', 'CICS', 'DLI', 'SQLIMS', 'ADO', 'HTML', 'ORACLE', 'TP', 'IDMS', 'XML']);
|
|
27
|
+
const BINARY_SIZE = { default: '1-2-4-8', ibm: '2-4-8', mf: '1--8' };
|
|
28
|
+
|
|
29
|
+
export const VERBS = new Set(`ACCEPT ADD ALLOCATE ALTER CALL CANCEL CLOSE COMMIT COMPUTE CONTINUE DELETE DISABLE DISPLAY DIVIDE
|
|
30
|
+
ENABLE ENTRY EVALUATE EXHIBIT EXIT FREE GENERATE GO GOBACK IF INITIALIZE INITIALISE INITIATE INSPECT INVOKE JSON MERGE MOVE MULTIPLY
|
|
31
|
+
OPEN PERFORM PURGE RAISE READ READY RECEIVE RELEASE RESET RESUME RETURN REWRITE ROLLBACK SEARCH SEND SET SORT START STOP
|
|
32
|
+
STRING SUBTRACT SUPPRESS TERMINATE TRANSFORM UNLOCK UNSTRING VALIDATE WRITE XML`.split(/\s+/));
|
|
33
|
+
export const NOT_LABELS = new Set(['EXIT', 'GOBACK', 'CONTINUE', 'DECLARATIVES', 'STOP', 'ELSE', 'NEXT', 'END']);
|
|
34
|
+
const STATEMENT_BREAKS = new Set(['ELSE', 'WHEN', 'THEN']);
|
|
35
|
+
export const SCOPE_TERMINATORS = new Set(`END-ACCEPT END-ADD END-CALL END-CHAIN END-COMPUTE END-DELETE END-DISPLAY END-DIVIDE
|
|
36
|
+
END-EVALUATE END-EXEC END-IF END-JSON END-MULTIPLY END-OF-PAGE END-PERFORM END-READ END-RECEIVE END-RETURN END-REWRITE
|
|
37
|
+
END-SEARCH END-SEND END-START END-STRING END-SUBTRACT END-UNSTRING END-WRITE END-XML END-INVOKE END-SET`.split(/\s+/));
|
|
38
|
+
const USAGE_WORDS = new Set(`BINARY BINARY-CHAR BINARY-SHORT BINARY-LONG BINARY-DOUBLE BINARY-C-LONG COMP COMPUTATIONAL
|
|
39
|
+
COMP-1 COMP-2 COMP-3 COMP-4 COMP-5 COMP-6 COMP-X COMP-N COMPUTATIONAL-1 COMPUTATIONAL-2 COMPUTATIONAL-3 COMPUTATIONAL-4
|
|
40
|
+
COMPUTATIONAL-5 COMPUTATIONAL-6 COMPUTATIONAL-X COMPUTATIONAL-N DISPLAY FLOAT-SHORT FLOAT-LONG FLOAT-DECIMAL-16
|
|
41
|
+
FLOAT-DECIMAL-34 INDEX NATIONAL PACKED-DECIMAL POINTER PROGRAM-POINTER FUNCTION-POINTER SIGNED-SHORT SIGNED-INT
|
|
42
|
+
SIGNED-LONG UNSIGNED-SHORT UNSIGNED-INT UNSIGNED-LONG BINARY-INT BINARY-LONG-LONG PROCEDURE-POINTER`.split(/\s+/));
|
|
43
|
+
const DATA_CLAUSE_WORDS = new Set([...USAGE_WORDS, 'PIC', 'PICTURE', 'USAGE', 'OCCURS', 'REDEFINES', 'VALUE', 'VALUES',
|
|
44
|
+
'SIGN', 'LEADING', 'TRAILING', 'SYNC', 'SYNCHRONIZED', 'JUST', 'JUSTIFIED', 'BLANK', 'EXTERNAL', 'GLOBAL', 'BASED',
|
|
45
|
+
'RENAMES', 'TYPEDEF', 'LIKE', 'CONSTANT', 'IS', 'FILLER']);
|
|
46
|
+
// The clauses of a report group description, so an entry that opens with one is a FILLER.
|
|
47
|
+
const REPORT_CLAUSE_WORDS = new Set(`TYPE LINE LINES COLUMN COLUMNS COL COLS SOURCE SUM VALUE VALUES GROUP INDICATE NEXT PRESENT
|
|
48
|
+
ABSENT JUST JUSTIFIED BLANK RESET UPON OCCURS VARYING SIGN USAGE PIC PICTURE`.split(/\s+/));
|
|
49
|
+
const SCREEN_CLAUSE_WORDS = new Set(`LINE COL COLUMN BLANK BELL BEEP BLINK HIGHLIGHT LOWLIGHT REVERSE-VIDEO REVERSE UNDERLINE
|
|
50
|
+
FOREGROUND-COLOR BACKGROUND-COLOR FOREGROUND-COLOUR BACKGROUND-COLOUR FROM TO USING AUTO AUTO-SKIP SECURE NO-ECHO REQUIRED
|
|
51
|
+
FULL PROMPT ERASE EOL EOS NUMBER PLUS MINUS SIZE LEFTLINE OVERLINE GRID UPPER LOWER SCROLL TIME-OUT CONTROL`.split(/\s+/));
|
|
52
|
+
|
|
53
|
+
export function expandTabs(line, width = 8) {
|
|
54
|
+
if (!line.includes('\t')) return line;
|
|
55
|
+
let out = '';
|
|
56
|
+
for (const ch of line) out += ch === '\t' ? ' '.repeat(width - (out.length % width)) : ch;
|
|
57
|
+
return out;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function detectFormat(src) {
|
|
61
|
+
let nonblank = 0, badIndicator = 0, fixedMarkers = 0, starCol1Bad = 0, codeCol2to7 = 0, seqDigits = 0;
|
|
62
|
+
for (const raw of src.split(/\r?\n/, 600)) {
|
|
63
|
+
const l = expandTabs(raw);
|
|
64
|
+
if (!l.trim()) continue;
|
|
65
|
+
nonblank++;
|
|
66
|
+
if (/^\*(?!>)/.test(l) && l.length >= 7 && !FIXED_INDICATORS.has(l[6])) starCol1Bad++;
|
|
67
|
+
if (/^ {1,6}[^\s*]/.test(l)) codeCol2to7++;
|
|
68
|
+
if (/^\d{6}/.test(l)) seqDigits++;
|
|
69
|
+
if (/^(IDENTIFICATION|ID|PROCEDURE|DATA|ENVIRONMENT)\s+DIVISION/i.test(l) || /^(WORKING-STORAGE|LINKAGE|FILE|LOCAL-STORAGE)\s+SECTION/i.test(l)) return 'free';
|
|
70
|
+
if (l.length >= 7 && !FIXED_INDICATORS.has(l[6])) badIndicator++;
|
|
71
|
+
if (/^\d{6}/.test(l) || /^.{6}[*\/]/.test(l)) fixedMarkers++;
|
|
72
|
+
}
|
|
73
|
+
if (!nonblank) return 'fixed';
|
|
74
|
+
if (starCol1Bad > 0 && codeCol2to7 > 0 && seqDigits === 0) return 'terminal';
|
|
75
|
+
if (badIndicator / nonblank > 0.1 && badIndicator > fixedMarkers) return 'free';
|
|
76
|
+
return seqDigits === 0 && codePastColumn72(src) ? 'free' : 'fixed';
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Code a fixed-form reading would cut off at column 72, in a program written for -free: a literal
|
|
80
|
+
// still open there that the next line does not continue, a token running across the boundary, or a
|
|
81
|
+
// line past 80 whose code does not end its sentence by 72. An inline comment, a comment entry and a
|
|
82
|
+
// lone tag of up to eight characters are what fixed-form programs keep there. Of the 54,487 programs
|
|
83
|
+
// cobc compiled in a 3,184-repository corpus, 54,154 are then read in a format cobc accepts.
|
|
84
|
+
function codePastColumn72(src) {
|
|
85
|
+
const lines = src.split(/\r?\n/, 2001);
|
|
86
|
+
for (let i = 0; i < Math.min(lines.length, 2000); i++) {
|
|
87
|
+
const l = expandTabs(lines[i]).replace(/\s+$/, '');
|
|
88
|
+
if (l.length <= 72 || l[6] === '*' || l[6] === '/' || /^\s*\*>/.test(l) || COMMENT_ENTRY.test(l.slice(7))) continue;
|
|
89
|
+
const { quote, comment } = stateAtColumn72(l);
|
|
90
|
+
if (comment) continue;
|
|
91
|
+
if (quote && expandTabs(lines[i + 1] || '')[6] !== '-') return true;
|
|
92
|
+
const tag = /^\S{1,8}$/.test(l.slice(72).trim());
|
|
93
|
+
if (/\S/.test(l[71]) && /\S/.test(l[72]) && !tag) return true;
|
|
94
|
+
const code = l.slice(7, 72).trim();
|
|
95
|
+
if (l.length > 80 && !tag && code && !code.endsWith('.')) return true;
|
|
96
|
+
}
|
|
97
|
+
return false;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
function stateAtColumn72(l) {
|
|
101
|
+
let quote = null;
|
|
102
|
+
for (let i = 7; i < 72 && i < l.length; i++) {
|
|
103
|
+
const c = l[i];
|
|
104
|
+
if (quote) { if (c === quote) { if (l[i + 1] === quote) i++; else quote = null; } }
|
|
105
|
+
else if (c === '"' || c === "'") quote = c;
|
|
106
|
+
else if (c === '*' && l[i + 1] === '>') return { quote: null, comment: true };
|
|
107
|
+
}
|
|
108
|
+
return { quote, comment: false };
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
function scanQuotes(text, open) {
|
|
112
|
+
let q = open;
|
|
113
|
+
let closedLast = null;
|
|
114
|
+
for (let i = 0; i < text.length; i++) {
|
|
115
|
+
const c = text[i];
|
|
116
|
+
if (q) {
|
|
117
|
+
if (c === q) { if (text[i + 1] === q) i++; else { if (i === text.length - 1) closedLast = q; q = null; } }
|
|
118
|
+
} else if (c === '"' || c === "'") q = c;
|
|
119
|
+
else if (c === '*' && text[i + 1] === '>') return { text: text.slice(0, i), open: null };
|
|
120
|
+
}
|
|
121
|
+
return { text, open: q, closedLast };
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
const PREDEFINED = new Map([['P64', 'SET']]);
|
|
125
|
+
|
|
126
|
+
function defineDirective(directive, defines) {
|
|
127
|
+
const body = directive.replace(/\*>.*$/, '').replace(/^>>\s*DEFINE\s+/i, '').trim();
|
|
128
|
+
const m = body.match(/^(?:CONSTANT\s+)?([A-Za-z0-9_-]+)\s+(?:AS\s+)?(.*?)\s*(OVERRIDE)?\s*$/i);
|
|
129
|
+
if (!m) return;
|
|
130
|
+
const name = m[1].toUpperCase();
|
|
131
|
+
const raw = m[2].trim();
|
|
132
|
+
if (/^(OFF|PARAMETER)$/i.test(raw)) { if (/^OFF$/i.test(raw)) defines.delete(name); return; }
|
|
133
|
+
const value = raw.replace(/^(['"])(.*)\1$/, '$2');
|
|
134
|
+
if (m[3] || !defines.has(name)) defines.set(name, value);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// $SET and >>SET: the source format it names, and with CONSTANT a name the source may use as a value.
|
|
138
|
+
function setDirective(text, defines) {
|
|
139
|
+
const c = /\bCONSTANT\s+([A-Za-z0-9_-]+)\s+(?:(["'])(.*?)\2|(\S+))/i.exec(text);
|
|
140
|
+
if (c && defines) defines.set(c[1].toUpperCase(), c[3] !== undefined ? c[3] : c[4]);
|
|
141
|
+
const m = text.match(/SOURCEFORMAT\s*\(?\s*["']?(FREE|FIXED|VARIABLE)/i);
|
|
142
|
+
return m ? m[1].toLowerCase() : null;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function evaluateCondition(text, defines) {
|
|
146
|
+
const t = text.replace(/\*>.*$/, '').trim();
|
|
147
|
+
const known = (n) => (defines && defines.has(n)) || PREDEFINED.has(n);
|
|
148
|
+
const valueOf = (n) => (defines && defines.has(n) ? defines.get(n) : PREDEFINED.get(n));
|
|
149
|
+
let m = t.match(/^([A-Za-z0-9_-]+)\s+(?:IS\s+)?(NOT\s+)?(DEFINED|SET)$/i);
|
|
150
|
+
if (m) { const r = known(m[1].toUpperCase()); return m[2] ? !r : r; }
|
|
151
|
+
m = t.match(/^([A-Za-z0-9_-]+)\s*(<=|>=|<>|=|<|>)\s*(['"]?)([^'"]*)\3$/);
|
|
152
|
+
if (m) {
|
|
153
|
+
const name = m[1].toUpperCase();
|
|
154
|
+
if (!known(name)) return null;
|
|
155
|
+
const left = valueOf(name);
|
|
156
|
+
const right = m[4];
|
|
157
|
+
const numeric = /^-?\d+(\.\d+)?$/.test(left) && /^-?\d+(\.\d+)?$/.test(right);
|
|
158
|
+
const [x, y] = numeric ? [Number(left), Number(right)] : [String(left), String(right)];
|
|
159
|
+
switch (m[2]) { case '<': return x < y; case '>': return x > y; case '<=': return x <= y; case '>=': return x >= y; case '=': return x === y; default: return x !== y; }
|
|
160
|
+
}
|
|
161
|
+
return null;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
export function normalize(src, format, defines, std) {
|
|
165
|
+
// Micro Focus reads a free-form line with * or / in column 1 as a comment, as cobc -std=mf does.
|
|
166
|
+
const columnOneComments = std === 'mf' || std === 'mf-strict';
|
|
167
|
+
const phys = src.split(/\r?\n/);
|
|
168
|
+
const entries = [];
|
|
169
|
+
const diags = [];
|
|
170
|
+
let fmt = format;
|
|
171
|
+
let last = null;
|
|
172
|
+
let seenCode = false;
|
|
173
|
+
const condStack = [];
|
|
174
|
+
let openLiteralLines = 0;
|
|
175
|
+
const options = [];
|
|
176
|
+
let identification = false;
|
|
177
|
+
let commentEntry = false;
|
|
178
|
+
for (let i = 0; i < phys.length; i++) {
|
|
179
|
+
const line = i + 1;
|
|
180
|
+
const l = expandTabs(phys[i].replace(/\x1a/g, ''));
|
|
181
|
+
const entry = { line, text: '', fmt };
|
|
182
|
+
entries.push(entry);
|
|
183
|
+
const trimmed = l.trim();
|
|
184
|
+
if (!trimmed) continue;
|
|
185
|
+
// A directive may follow a sequence area in fixed or variable format ("GC0712 >>IF ...").
|
|
186
|
+
const afterSequence = fmt !== 'free' && fmt !== 'terminal' && l.length > 7 ? l.slice(6).trim() : '';
|
|
187
|
+
const directive = trimmed.startsWith('>>') ? trimmed : afterSequence.startsWith('>>') ? afterSequence : null;
|
|
188
|
+
if (directive) {
|
|
189
|
+
const skipping = condStack.some(f => f.state !== 'taking');
|
|
190
|
+
const m = directive.match(/^>>\s*SOURCE\s+(?:FORMAT\s+)?(?:IS\s+)?(FREE|FIXED|VARIABLE|TERMINAL)/i);
|
|
191
|
+
if (m) { if (!skipping) fmt = m[1].toLowerCase(); }
|
|
192
|
+
else if (/^>>\s*DEFINE\b/i.test(directive)) {
|
|
193
|
+
if (!skipping && defines) defineDirective(directive, defines);
|
|
194
|
+
}
|
|
195
|
+
else if (/^>>\s*IF\b/i.test(directive)) {
|
|
196
|
+
if (skipping) condStack.push({ state: 'done' });
|
|
197
|
+
else {
|
|
198
|
+
const v = evaluateCondition(directive.replace(/^>>\s*IF\s+/i, ''), defines);
|
|
199
|
+
// Unknown conditions keep the first branch: a compiler keeps exactly one, and the first
|
|
200
|
+
// is the configuration the source was written for more often than not.
|
|
201
|
+
condStack.push({ state: v === false ? 'waiting' : 'taking' });
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
else if (/^>>\s*ELIF\b|^>>\s*ELSE\s+IF\b/i.test(directive)) {
|
|
205
|
+
const f = condStack[condStack.length - 1];
|
|
206
|
+
if (f) {
|
|
207
|
+
if (f.state === 'taking' || f.state === 'done') f.state = 'done';
|
|
208
|
+
else {
|
|
209
|
+
const v = evaluateCondition(directive.replace(/^>>\s*(?:ELIF|ELSE\s+IF)\s+/i, ''), defines);
|
|
210
|
+
f.state = v === false ? 'waiting' : 'taking';
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
else if (/^>>\s*ELSE\b/i.test(directive)) {
|
|
215
|
+
const f = condStack[condStack.length - 1];
|
|
216
|
+
if (f) f.state = f.state === 'waiting' ? 'taking' : 'done';
|
|
217
|
+
}
|
|
218
|
+
else if (/^>>\s*END-IF\b/i.test(directive)) condStack.pop();
|
|
219
|
+
else if (/^>>\s*SET\b/i.test(directive) && !skipping) fmt = setDirective(directive, defines) || fmt;
|
|
220
|
+
continue;
|
|
221
|
+
}
|
|
222
|
+
entry.fmt = fmt;
|
|
223
|
+
// Micro Focus directives: $IF, $ELSE and $END select source as >>IF does, and $SET sets the
|
|
224
|
+
// source format or, with CONSTANT, a name the source may use as a value.
|
|
225
|
+
const dollar = trimmed.startsWith('$') ? trimmed : fmt !== 'free' && fmt !== 'terminal' && l[6] === '$' ? l.slice(6).trim() : null;
|
|
226
|
+
if (dollar) {
|
|
227
|
+
const skipping = condStack.some(f => f.state !== 'taking');
|
|
228
|
+
if (/^\$\s*IF\b/i.test(dollar)) {
|
|
229
|
+
if (skipping) condStack.push({ state: 'done' });
|
|
230
|
+
else condStack.push({ state: evaluateCondition(dollar.replace(/^\$\s*IF\s+/i, ''), defines) === false ? 'waiting' : 'taking' });
|
|
231
|
+
} else if (/^\$\s*ELSE\b/i.test(dollar)) {
|
|
232
|
+
const f = condStack[condStack.length - 1];
|
|
233
|
+
if (f) f.state = f.state === 'waiting' ? 'taking' : 'done';
|
|
234
|
+
} else if (/^\$\s*END\b/i.test(dollar)) condStack.pop();
|
|
235
|
+
else if (!skipping && /^\$\s*SET\b/i.test(dollar)) fmt = setDirective(dollar, defines) || fmt;
|
|
236
|
+
continue;
|
|
237
|
+
}
|
|
238
|
+
if (condStack.some(f => f.state !== 'taking')) continue;
|
|
239
|
+
// Compiler options the source sets for itself. The card is not code, but what it says about how
|
|
240
|
+
// the program is compiled is kept: SSRANGE decides whether a bad subscript overwrites or abends.
|
|
241
|
+
if (!seenCode && (/^(CBL|PROCESS)\b/i.test(trimmed) || /^(CBL|PROCESS)\b/i.test(l.slice(7).trim()))) {
|
|
242
|
+
const card = /^(CBL|PROCESS)\b/i.test(trimmed) ? trimmed : l.slice(7).trim();
|
|
243
|
+
options.push(...optionTokens(card.replace(/^(CBL|PROCESS)\b/i, '')));
|
|
244
|
+
continue;
|
|
245
|
+
}
|
|
246
|
+
let indicator = ' ';
|
|
247
|
+
let text;
|
|
248
|
+
let width;
|
|
249
|
+
if (fmt === 'free') {
|
|
250
|
+
text = l;
|
|
251
|
+
width = l.length;
|
|
252
|
+
} else if (fmt === 'terminal') {
|
|
253
|
+
indicator = l[0];
|
|
254
|
+
const isIndicator = indicator === ' ' || indicator === '*' || indicator === '/' || (indicator === '-' && l[1] === ' ') || ((indicator === 'D' || indicator === 'd') && l[1] === ' ');
|
|
255
|
+
if (!isIndicator) { indicator = ' '; text = l.slice(0, 320); } else text = l.slice(1, 320);
|
|
256
|
+
width = 319;
|
|
257
|
+
} else {
|
|
258
|
+
if (l.length < 7) continue;
|
|
259
|
+
indicator = l[6];
|
|
260
|
+
width = fmt === 'variable' ? 243 : 65;
|
|
261
|
+
text = l.slice(7, 7 + width);
|
|
262
|
+
if (!FIXED_INDICATORS.has(indicator)) { diags.push({ kind: 'invalid-indicator', line }); indicator = ' '; }
|
|
263
|
+
}
|
|
264
|
+
if (indicator === '*' || indicator === '/' || indicator === 'D' || indicator === 'd') continue;
|
|
265
|
+
if (fmt === 'free' && (/^\s*\*>/.test(text) || (columnOneComments && /^[*/]/.test(text)))) continue;
|
|
266
|
+
// AUTHOR, INSTALLATION, DATE-WRITTEN, DATE-COMPILED, SECURITY and REMARKS take a comment-entry:
|
|
267
|
+
// any text, a COPY or an apostrophe included. It runs to the end of the header's line, and in
|
|
268
|
+
// fixed form on through every line with Area A blank, as GnuCOBOL reads it.
|
|
269
|
+
if (commentEntry) {
|
|
270
|
+
if (!text.slice(0, 4).trim()) continue;
|
|
271
|
+
commentEntry = false;
|
|
272
|
+
}
|
|
273
|
+
if (/^\s*(IDENTIFICATION|ID)\s+DIVISION\b/i.test(text)) identification = true;
|
|
274
|
+
else if (/^\s*(ENVIRONMENT|DATA|PROCEDURE)\s+DIVISION\b/i.test(text)) identification = false;
|
|
275
|
+
const entryHeader = identification && COMMENT_ENTRY.exec(text);
|
|
276
|
+
if (entryHeader) {
|
|
277
|
+
text = entryHeader[0].padEnd(text.length, ' ');
|
|
278
|
+
commentEntry = fmt !== 'free';
|
|
279
|
+
}
|
|
280
|
+
// Free format carries an unterminated literal onto the next line, with or without a leading
|
|
281
|
+
// hyphen. Bounded, so a literal that is unterminated by mistake cannot swallow the file.
|
|
282
|
+
if (fmt === 'free' && last && last.open && openLiteralLines < 5) {
|
|
283
|
+
let t = text.replace(/^\s*-\s*/, '');
|
|
284
|
+
if (/^\s*-/.test(text) && (t[0] === '"' || t[0] === "'")) t = t.slice(1);
|
|
285
|
+
const cont = scanQuotes(t, last.open);
|
|
286
|
+
last.entry.text += cont.text;
|
|
287
|
+
last.open = cont.open;
|
|
288
|
+
openLiteralLines = cont.open ? openLiteralLines + 1 : 0;
|
|
289
|
+
continue;
|
|
290
|
+
}
|
|
291
|
+
if (indicator === '-' && last) {
|
|
292
|
+
// A literal that closes in the last column, continued by a line opening with its quote, was
|
|
293
|
+
// not closed: the two quotes are one quote written across the break.
|
|
294
|
+
const lead = text.replace(/^\s+/, '');
|
|
295
|
+
if (!last.open && last.closedLast && last.full && lead[0] === last.closedLast) {
|
|
296
|
+
const s = scanQuotes(last.closedLast + lead.slice(1), last.closedLast);
|
|
297
|
+
last.entry.text += s.text.slice(1);
|
|
298
|
+
last.open = s.open;
|
|
299
|
+
last.closedLast = s.closedLast;
|
|
300
|
+
continue;
|
|
301
|
+
}
|
|
302
|
+
if (last.open) {
|
|
303
|
+
let t = text.replace(/^\s+/, '');
|
|
304
|
+
if (t[0] === '"' || t[0] === "'") t = t.slice(1);
|
|
305
|
+
const s = scanQuotes(t, last.open);
|
|
306
|
+
last.entry.text = last.entry.text.padEnd(last.width, ' ') + s.text;
|
|
307
|
+
last.open = s.open;
|
|
308
|
+
last.closedLast = s.closedLast;
|
|
309
|
+
} else {
|
|
310
|
+
const s = scanQuotes(text.replace(/^\s+/, ''), null);
|
|
311
|
+
last.entry.text = last.entry.text.trimEnd() + s.text;
|
|
312
|
+
last.open = s.open;
|
|
313
|
+
last.closedLast = s.closedLast;
|
|
314
|
+
}
|
|
315
|
+
last.full = text.length === width;
|
|
316
|
+
continue;
|
|
317
|
+
}
|
|
318
|
+
const s = scanQuotes(text, null);
|
|
319
|
+
entry.text = s.text;
|
|
320
|
+
if (s.text.trim()) { seenCode = true; last = { entry, open: s.open, width, closedLast: s.closedLast, full: text.length === width }; openLiteralLines = s.open ? 1 : 0; }
|
|
321
|
+
}
|
|
322
|
+
return { entries, diags, finalFormat: fmt, options };
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
function tok(t, v, line, file, extra) { return { t, v, u: t === 'word' ? v.toUpperCase() : v, line, file, ...extra }; }
|
|
326
|
+
|
|
327
|
+
// A floating-point literal's exponent: 1.5E3, 2.5E-2, +1.0E+05. Only after a mantissa with a
|
|
328
|
+
// decimal point, since 1E3 is a word a program may declare.
|
|
329
|
+
function exponentAt(s, j) {
|
|
330
|
+
const m = /^[Ee][+-]?\d{1,3}(?![A-Za-z0-9_$#@-])/.exec(s.slice(j, j + 6));
|
|
331
|
+
return m ? m[0].length : 0;
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
export function tokenize(norm, file) {
|
|
335
|
+
const out = [];
|
|
336
|
+
const diags = [];
|
|
337
|
+
let exec = null;
|
|
338
|
+
const emit = (tk) => {
|
|
339
|
+
if (exec) {
|
|
340
|
+
if (tk.t === 'word' && tk.u === 'END-EXEC') { out.push(exec); exec = null; return; }
|
|
341
|
+
exec.toks.push(tk);
|
|
342
|
+
return;
|
|
343
|
+
}
|
|
344
|
+
if (tk.t === 'word' && EXEC_KINDS.has(tk.u)) {
|
|
345
|
+
const prev = out[out.length - 1];
|
|
346
|
+
if (prev && prev.t === 'word' && (prev.u === 'EXEC' || prev.u === 'EXECUTE')) {
|
|
347
|
+
out.pop();
|
|
348
|
+
exec = tok('exec', tk.u, prev.line, file, { kind: tk.u, toks: [] });
|
|
349
|
+
return;
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
out.push(tk);
|
|
353
|
+
};
|
|
354
|
+
let picPending = false;
|
|
355
|
+
for (const { line, text: s } of norm.entries) {
|
|
356
|
+
let i = 0;
|
|
357
|
+
const n = s.length;
|
|
358
|
+
while (i < n) {
|
|
359
|
+
const c = s[i];
|
|
360
|
+
if (c === ' ' || c === '\t' || c === '\f' || c === '\r') { i++; continue; }
|
|
361
|
+
if (exec && exec.kind === 'SQL' && c === '-' && s[i + 1] === '-') break;
|
|
362
|
+
if (picPending) {
|
|
363
|
+
let j = i;
|
|
364
|
+
while (j < n && s[j] !== ' ') j++;
|
|
365
|
+
let pic = s.slice(i, j);
|
|
366
|
+
if (pic.toUpperCase() === 'IS') { emit(tok('word', pic, line, file)); i = j; continue; }
|
|
367
|
+
let trailingPeriod = false;
|
|
368
|
+
if (/[.,;]$/.test(pic) && pic.length > 1) { trailingPeriod = pic.endsWith('.'); pic = pic.slice(0, -1); }
|
|
369
|
+
emit(tok('pic', pic, line, file));
|
|
370
|
+
if (trailingPeriod) emit(tok('period', '.', line, file));
|
|
371
|
+
picPending = false;
|
|
372
|
+
i = j;
|
|
373
|
+
continue;
|
|
374
|
+
}
|
|
375
|
+
// A separator comma or semicolon, with or without the space the standard asks for after it:
|
|
376
|
+
// C(I,J) and USING A,B compile, and a comma inside a number was read with the number above.
|
|
377
|
+
if (c === ',' || c === ';') { i++; continue; }
|
|
378
|
+
if (c === '=' && s[i + 1] === '=') { emit(tok('pseudo', '==', line, file)); i += 2; continue; }
|
|
379
|
+
const prefixed = /^(?:NX|NC|BX|[XxNnZzGgBbUuHh]|nx|nc|bx)(?=["'])/.exec(s.slice(i, i + 3));
|
|
380
|
+
if (c === '"' || c === "'" || prefixed) {
|
|
381
|
+
const start = prefixed ? i + prefixed[0].length : i;
|
|
382
|
+
const q = s[start];
|
|
383
|
+
let j = start + 1, v = '';
|
|
384
|
+
let closed = false;
|
|
385
|
+
while (j < n) {
|
|
386
|
+
if (s[j] === q) { if (s[j + 1] === q) { v += q; j += 2; continue; } closed = true; j++; break; }
|
|
387
|
+
v += s[j++];
|
|
388
|
+
}
|
|
389
|
+
if (!closed) diags.push({ kind: 'unterminated-literal', line, file });
|
|
390
|
+
emit(tok('lit', v, line, file, { prefix: prefixed ? prefixed[0].toUpperCase() : '' }));
|
|
391
|
+
i = j;
|
|
392
|
+
continue;
|
|
393
|
+
}
|
|
394
|
+
if (c === '.') {
|
|
395
|
+
if (i + 1 >= n || s[i + 1] === ' ') { emit(tok('period', '.', line, file)); i++; continue; }
|
|
396
|
+
if (/[0-9]/.test(s[i + 1] || '')) {
|
|
397
|
+
let j = i + 1; while (j < n && /[0-9]/.test(s[j])) j++;
|
|
398
|
+
j += exponentAt(s, j);
|
|
399
|
+
emit(tok('num', s.slice(i, j), line, file)); i = j; continue;
|
|
400
|
+
}
|
|
401
|
+
// Not a separator: a separator period is followed by a space or the end of the line.
|
|
402
|
+
emit(tok('period', '.', line, file, { joined: /[A-Za-z0-9_$#@-]/.test(s[i + 1]) })); i++; continue;
|
|
403
|
+
}
|
|
404
|
+
if (/[A-Za-z0-9_$#@\u0080-\u00ff]/.test(c)) {
|
|
405
|
+
let j = i + 1;
|
|
406
|
+
while (j < n && /[A-Za-z0-9_\-$#@\u0080-\u00ff]/.test(s[j])) j++;
|
|
407
|
+
let v = s.slice(i, j);
|
|
408
|
+
if (/^\d+$/.test(v) && (s[j] === '.' || s[j] === ',') && /[0-9]/.test(s[j + 1] || '')) {
|
|
409
|
+
let k = j + 1; while (k < n && /[0-9]/.test(s[k])) k++;
|
|
410
|
+
k += exponentAt(s, k);
|
|
411
|
+
v = s.slice(i, k); j = k;
|
|
412
|
+
emit(tok('num', v, line, file));
|
|
413
|
+
} else {
|
|
414
|
+
emit(tok('word', v, line, file, i < 4 ? { areaA: true } : undefined));
|
|
415
|
+
const u = v.toUpperCase();
|
|
416
|
+
if ((u === 'PIC' || u === 'PICTURE') && !exec) picPending = true;
|
|
417
|
+
}
|
|
418
|
+
i = j;
|
|
419
|
+
continue;
|
|
420
|
+
}
|
|
421
|
+
const two = s.slice(i, i + 2);
|
|
422
|
+
if (two === '**' || two === '>=' || two === '<=' || two === '<>') { emit(tok('op', two, line, file)); i += 2; continue; }
|
|
423
|
+
if ('()'.includes(c)) { emit(tok('sep', c, line, file)); i++; continue; }
|
|
424
|
+
if (':+-*/=<>&'.includes(c)) { emit(tok('op', c, line, file)); i++; continue; }
|
|
425
|
+
diags.push({ kind: 'unexpected-char', line, file });
|
|
426
|
+
i++;
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
if (exec) { diags.push({ kind: 'unterminated-exec', line: exec.line, file }); out.push(exec); }
|
|
430
|
+
for (const tk of out) { const e = norm.entries[tk.line - 1]; if (e) tk.fmt = e.fmt; }
|
|
431
|
+
return { tokens: out, diags };
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
// Every rule set walks the tree through here. A directory it may not read is recorded, not treated
|
|
435
|
+
// as empty: only ENOENT means absent. Symlinks are followed while they stay inside the tree, so a
|
|
436
|
+
// symlinked copy library is read; one pointing outside is counted and never followed, because a
|
|
437
|
+
// scan reads the tree it was given and nothing else.
|
|
438
|
+
export function buildFileIndex(root) {
|
|
439
|
+
const index = new Map();
|
|
440
|
+
const dirs = new Set();
|
|
441
|
+
const unreadableDirs = [];
|
|
442
|
+
const symlinks = { followed: 0, outside: 0, broken: 0 };
|
|
443
|
+
let top;
|
|
444
|
+
try { top = realpathSync(root); } catch (e) { if (e.code === 'ENOENT') return { index, copyDirs: [], unreadableDirs, symlinks }; throw e; }
|
|
445
|
+
const inside = (p) => p === top || p.startsWith(top + sep);
|
|
446
|
+
const visited = new Set();
|
|
447
|
+
const addFile = (p, name, d) => { index.set(p.toLowerCase(), p); if (/\.(cpy|copy|inc|cbl|cob)$/i.test(name)) dirs.add(d); };
|
|
448
|
+
// Each real directory is walked once, however many links lead to it, or its programs count twice.
|
|
449
|
+
const walk = (d, real) => {
|
|
450
|
+
if (visited.has(real)) return;
|
|
451
|
+
visited.add(real);
|
|
452
|
+
let es;
|
|
453
|
+
try { es = readdirSync(d, { withFileTypes: true }); } catch (e) {
|
|
454
|
+
if (e.code === 'ENOENT') return;
|
|
455
|
+
if (e.code === 'EACCES' || e.code === 'EPERM') { unreadableDirs.push(d); return; }
|
|
456
|
+
throw e;
|
|
457
|
+
}
|
|
458
|
+
for (const e of es.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))) {
|
|
459
|
+
if (e.name === '.git' || e.name.startsWith('._')) continue;
|
|
460
|
+
const p = join(d, e.name);
|
|
461
|
+
if (e.isDirectory()) walk(p, join(real, e.name));
|
|
462
|
+
else if (e.isFile()) addFile(p, e.name, d);
|
|
463
|
+
else if (e.isSymbolicLink()) {
|
|
464
|
+
let target;
|
|
465
|
+
let st;
|
|
466
|
+
try { target = realpathSync(p); st = statSync(target); } catch { symlinks.broken++; continue; }
|
|
467
|
+
if (!inside(target)) { symlinks.outside++; continue; }
|
|
468
|
+
symlinks.followed++;
|
|
469
|
+
if (st.isDirectory()) walk(p, target);
|
|
470
|
+
else if (st.isFile()) addFile(p, e.name, d);
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
};
|
|
474
|
+
walk(root, top);
|
|
475
|
+
index.root = root;
|
|
476
|
+
return { index, copyDirs: [...dirs].sort(), unreadableDirs, symlinks };
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
function readOperand(tokens, j) {
|
|
480
|
+
const t = tokens[j];
|
|
481
|
+
if (!t) return [null, j];
|
|
482
|
+
if (t.t === 'pseudo') {
|
|
483
|
+
const toks = [];
|
|
484
|
+
let k = j + 1;
|
|
485
|
+
while (k < tokens.length && tokens[k].t !== 'pseudo') toks.push(tokens[k++]);
|
|
486
|
+
return [{ pseudo: true, toks }, k + 1];
|
|
487
|
+
}
|
|
488
|
+
if (t.t === 'lit' || t.t === 'num' || t.t === 'pic') return [{ pseudo: false, toks: [t] }, j + 1];
|
|
489
|
+
if (t.t !== 'word') return [null, j];
|
|
490
|
+
// An identifier operand keeps its qualifiers and subscripts: VAR BY IN-A IN INPUT-REC.
|
|
491
|
+
const toks = [t];
|
|
492
|
+
let k = j + 1;
|
|
493
|
+
while (tokens[k] && tokens[k].t === 'word' && (tokens[k].u === 'IN' || tokens[k].u === 'OF') && tokens[k + 1] && tokens[k + 1].t === 'word') toks.push(tokens[k++], tokens[k++]);
|
|
494
|
+
if (tokens[k] && tokens[k].t === 'sep' && tokens[k].v === '(') {
|
|
495
|
+
let depth = 0, e = k;
|
|
496
|
+
for (; e < tokens.length && tokens[e].t !== 'period' && tokens[e].t !== 'pseudo'; e++) {
|
|
497
|
+
if (tokens[e].t === 'sep') depth += tokens[e].v === '(' ? 1 : -1;
|
|
498
|
+
if (depth === 0) break;
|
|
499
|
+
}
|
|
500
|
+
if (depth === 0 && e < tokens.length) { toks.push(...tokens.slice(k, e + 1)); k = e + 1; }
|
|
501
|
+
}
|
|
502
|
+
return [{ pseudo: false, toks }, k];
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
function sameToken(a, b) {
|
|
506
|
+
if (a.t === 'word' || b.t === 'word') return a.t === b.t && a.u === b.u;
|
|
507
|
+
return a.t === b.t && a.v === b.v;
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
function readReplacingPairs(tokens, j) {
|
|
511
|
+
const pairs = [];
|
|
512
|
+
while (j < tokens.length && tokens[j].t !== 'period') {
|
|
513
|
+
let mode = null;
|
|
514
|
+
if (tokens[j].t === 'word' && (tokens[j].u === 'LEADING' || tokens[j].u === 'TRAILING')) { mode = tokens[j].u; j++; }
|
|
515
|
+
if (tokens[j] && tokens[j].t === 'word' && tokens[j].u === 'ALSO') { j++; continue; }
|
|
516
|
+
const [from, j2] = readOperand(tokens, j);
|
|
517
|
+
if (!from || !tokens[j2] || tokens[j2].u !== 'BY') break;
|
|
518
|
+
const [to, j3] = readOperand(tokens, j2 + 1);
|
|
519
|
+
if (!to) break;
|
|
520
|
+
pairs.push({ mode, from, to });
|
|
521
|
+
j = j3;
|
|
522
|
+
}
|
|
523
|
+
return [pairs, j];
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
// A COPY's REPLACING reaches the text of the copybooks nested in it, except what their own REPLACING
|
|
527
|
+
// produced: cobc names `05 PFX-ID` AA-TWO-ID under COPY LEAF REPLACING LEADING ==PFX== BY ==AA-TWO==
|
|
528
|
+
// inside a COPY REPLACING LEADING ==AA== BY ==XX==. A REPLACE statement still sees that text.
|
|
529
|
+
function applyReplacing(tokens, pairs, maxGrowth = Infinity, byCopy = false) {
|
|
530
|
+
if (!pairs.length) return tokens;
|
|
531
|
+
let grown = 0;
|
|
532
|
+
const out = [];
|
|
533
|
+
for (let i = 0; i < tokens.length;) {
|
|
534
|
+
if (byCopy && tokens[i].copyReplaced) { out.push(tokens[i++]); continue; }
|
|
535
|
+
let hit = false;
|
|
536
|
+
for (const p of pairs) {
|
|
537
|
+
if (p.mode || p.partial) continue;
|
|
538
|
+
const f = p.from.toks;
|
|
539
|
+
if (!f.length || i + f.length > tokens.length) continue;
|
|
540
|
+
let ok = true;
|
|
541
|
+
for (let k = 0; k < f.length; k++) if (!sameToken(tokens[i + k], f[k]) || (byCopy && tokens[i + k].copyReplaced)) { ok = false; break; }
|
|
542
|
+
if (ok) {
|
|
543
|
+
grown += p.to.toks.length - f.length;
|
|
544
|
+
if (grown > maxGrowth) return null;
|
|
545
|
+
for (const r of p.to.toks) out.push({ ...r, line: tokens[i].line, file: tokens[i].file, ...(byCopy ? { copyReplaced: true } : {}) });
|
|
546
|
+
i += f.length;
|
|
547
|
+
hit = true;
|
|
548
|
+
break;
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
if (hit) continue;
|
|
552
|
+
const was = tokens[i];
|
|
553
|
+
let tk = was;
|
|
554
|
+
if (tk.t === 'pic') {
|
|
555
|
+
// Parentheses delimit text words, so a pattern word inside X(NAME) is replaced like any other.
|
|
556
|
+
for (const p of pairs) {
|
|
557
|
+
if (p.mode || p.partial || p.from.toks.length !== 1 || p.to.toks.length > 1) continue;
|
|
558
|
+
const pat = p.from.toks[0].v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
559
|
+
const rep = p.to.toks[0] ? p.to.toks[0].v : '';
|
|
560
|
+
const v = tk.v.replace(new RegExp(`\\(${pat}\\)`, 'gi'), `(${rep})`);
|
|
561
|
+
if (v !== tk.v) tk = { ...tk, v, u: v };
|
|
562
|
+
}
|
|
563
|
+
}
|
|
564
|
+
if (tk.t === 'word') {
|
|
565
|
+
for (const p of pairs) {
|
|
566
|
+
if (!p.mode || !p.from.toks[0]) continue;
|
|
567
|
+
const pat = p.from.toks[0].u;
|
|
568
|
+
const rep = p.to.toks[0] ? p.to.toks[0].v : '';
|
|
569
|
+
if (p.mode === 'LEADING' && tk.u.startsWith(pat)) { const v = rep + tk.v.slice(pat.length); tk = { ...tk, v, u: v.toUpperCase() }; break; }
|
|
570
|
+
if (p.mode === 'TRAILING' && tk.u.endsWith(pat)) { const v = tk.v.slice(0, tk.v.length - pat.length) + rep; tk = { ...tk, v, u: v.toUpperCase() }; break; }
|
|
571
|
+
}
|
|
572
|
+
}
|
|
573
|
+
out.push(byCopy && tk !== was ? { ...tk, copyReplaced: true } : tk);
|
|
574
|
+
i++;
|
|
575
|
+
}
|
|
576
|
+
return out;
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
// A COPY resolves inside the tree through the index, which IS the tree, or in a declared library
|
|
580
|
+
// outside it (the system copybooks, COBCPY, COBOLWORK_COPYPATH), which may be read from disk. A
|
|
581
|
+
// name that is absolute, or that climbs out of the directory it was joined to, never reaches the
|
|
582
|
+
// filesystem: `COPY "/etc/hosts"` resolved before this, and a scan reads the tree it was given.
|
|
583
|
+
export function copyRefusal(name, ctx = {}) {
|
|
584
|
+
if (isAbsolute(name)) return ctx.allowAbsoluteCopy ? null : 'refused-absolute';
|
|
585
|
+
// A relative name may climb to a sibling directory (`../copybooks/X.cpy` is ordinary); what it may
|
|
586
|
+
// not do is leave the tree. Without an index there is no tree to leave.
|
|
587
|
+
const root = ctx.fileIndex && ctx.fileIndex.root;
|
|
588
|
+
if (!root || !ctx.mainDir) return null;
|
|
589
|
+
const p = resolve(ctx.mainDir, name);
|
|
590
|
+
const top = resolve(root);
|
|
591
|
+
return p === top || p.startsWith(top + sep) ? null : 'refused-outside';
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
// Files in the tree by lower-case file name, built once per index. A COPY of a plain name looks its
|
|
595
|
+
// candidates up here and takes the one in the highest-priority directory, instead of trying every
|
|
596
|
+
// directory in turn — which is what made callers cap the include list at 120 directories and
|
|
597
|
+
// report the copybooks beyond it as missing.
|
|
598
|
+
const byNameCache = new WeakMap();
|
|
599
|
+
function filesByName(index) {
|
|
600
|
+
let m = byNameCache.get(index);
|
|
601
|
+
if (!m) {
|
|
602
|
+
m = new Map();
|
|
603
|
+
for (const p of index.values()) { const k = basename(p).toLowerCase(); if (!m.has(k)) m.set(k, []); m.get(k).push(p); }
|
|
604
|
+
byNameCache.set(index, m);
|
|
605
|
+
}
|
|
606
|
+
return m;
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
// A build's copy library holds copybooks, not programs, so a COPY prefers a file that is not itself a
|
|
610
|
+
// program: IBM's Bank of Z keeps the program INQACCCU.cbl beside BNK1CCA.cbl and the copybook of the
|
|
611
|
+
// same name in cics/copy, and cobc run from the source's directory pastes the program in. Only a
|
|
612
|
+
// name a program could carry is read to find out. A COPY with nothing else to take still takes the
|
|
613
|
+
// program, as cobc does - that is how a nested program is copied in - so a program cobc compiles
|
|
614
|
+
// resolves as it did.
|
|
615
|
+
const PROGRAM_EXT = /\.(cbl|cob)$|^[^.]*$/i;
|
|
616
|
+
const DIVISION_WORDS = new Set(['IDENTIFICATION', 'ID', 'ENVIRONMENT', 'DATA', 'PROCEDURE']);
|
|
617
|
+
function isProgramFile(p, ctx) {
|
|
618
|
+
if (!PROGRAM_EXT.test(basename(p))) return false;
|
|
619
|
+
const seen = (ctx.programFiles ||= new Map());
|
|
620
|
+
if (!seen.has(p)) {
|
|
621
|
+
let text = '';
|
|
622
|
+
try { text = readSource(p).text; } catch { /* an unreadable candidate is not known to be a program */ }
|
|
623
|
+
seen.set(p, /^[^*\n]{0,6}\s*PROGRAM-ID\s*\./im.test(text));
|
|
624
|
+
}
|
|
625
|
+
return seen.get(p);
|
|
626
|
+
}
|
|
627
|
+
|
|
628
|
+
function resolveInTree(name, ctx) {
|
|
629
|
+
const byName = filesByName(ctx.fileIndex);
|
|
630
|
+
if (!ctx.dirRank) {
|
|
631
|
+
ctx.dirRank = new Map();
|
|
632
|
+
[ctx.mainDir, ...ctx.includeDirs].forEach((d, i) => { const k = resolve(d).toLowerCase(); if (!ctx.dirRank.has(k)) ctx.dirRank.set(k, i); });
|
|
633
|
+
}
|
|
634
|
+
const hits = [];
|
|
635
|
+
COPY_EXTS.forEach((ext, e) => {
|
|
636
|
+
for (const p of byName.get((name + ext).toLowerCase()) || []) {
|
|
637
|
+
const d = ctx.dirRank.get(dirname(p).toLowerCase());
|
|
638
|
+
if (d !== undefined) hits.push({ p, rank: d * COPY_EXTS.length + e });
|
|
639
|
+
}
|
|
640
|
+
});
|
|
641
|
+
// In a data division nothing can be a program: a tree that holds the program and not the copybook
|
|
642
|
+
// of its name (IBM's CICS Bank Sample, flattened into one directory) is missing the copybook.
|
|
643
|
+
const inData = ctx.division === 'DATA';
|
|
644
|
+
if (!inData && hits.length < 2) return hits.length ? hits[0].p : null;
|
|
645
|
+
hits.sort((a, b) => a.rank - b.rank);
|
|
646
|
+
const copybook = hits.find((h) => !isProgramFile(h.p, ctx));
|
|
647
|
+
if (copybook) return copybook.p;
|
|
648
|
+
return inData || !hits.length ? null : hits[0].p;
|
|
649
|
+
}
|
|
650
|
+
|
|
651
|
+
function resolveCopy(name, lib, ctx) {
|
|
652
|
+
if (copyRefusal(name, ctx)) return null;
|
|
653
|
+
if (ctx.fileIndex && !lib && !/[\\/]/.test(name)) {
|
|
654
|
+
const hit = resolveInTree(name, ctx);
|
|
655
|
+
if (hit) return hit;
|
|
656
|
+
return resolveDeclared(name, lib, ctx);
|
|
657
|
+
}
|
|
658
|
+
return resolveByDirs(name, lib, ctx, true);
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
function resolveDeclared(name, lib, ctx) {
|
|
662
|
+
return resolveByDirs(name, lib, ctx, false);
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
// The extensions under which some file in the index carries this name's last segment. A tree
|
|
666
|
+
// candidate hits only a path the index holds, so any other extension cannot hit in any directory.
|
|
667
|
+
// null when the segment is one join and resolve would rewrite.
|
|
668
|
+
function indexedExts(name, index) {
|
|
669
|
+
const byName = filesByName(index);
|
|
670
|
+
const out = new Set();
|
|
671
|
+
for (const ext of COPY_EXTS) {
|
|
672
|
+
const leaf = basename(name + ext);
|
|
673
|
+
if (!leaf || leaf === '.' || leaf === '..') return null;
|
|
674
|
+
if (byName.has(leaf.toLowerCase())) out.add(ext);
|
|
675
|
+
}
|
|
676
|
+
return out;
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
function resolveByDirs(name, lib, ctx, withTree) {
|
|
680
|
+
// Read at call time, so a caller that sets the variable after importing this module is honoured.
|
|
681
|
+
const envDirs = (process.env.COBOLWORK_COPYPATH || process.env.COBCPY || '').split(delimiter).filter(Boolean);
|
|
682
|
+
const treeExts = withTree && ctx.fileIndex && !isAbsolute(name) ? indexedExts(name, ctx.fileIndex) : null;
|
|
683
|
+
const candidates = [];
|
|
684
|
+
if (isAbsolute(name)) candidates.push({ base: name, dir: null, declared: true });
|
|
685
|
+
else {
|
|
686
|
+
const lists = [[[...ctx.systemDirs, ...envDirs], true]];
|
|
687
|
+
if (withTree && (!treeExts || treeExts.size)) lists.unshift([[ctx.mainDir, ...ctx.includeDirs], false]);
|
|
688
|
+
for (const [list, declared] of lists) {
|
|
689
|
+
for (const d of list) for (const b of (lib ? [join(d, lib, name), join(d, name)] : [join(d, name)])) candidates.push({ base: b, dir: resolve(d), declared });
|
|
690
|
+
}
|
|
691
|
+
}
|
|
692
|
+
const home = ctx.mainDir && resolve(ctx.mainDir);
|
|
693
|
+
const inData = ctx.division === 'DATA';
|
|
694
|
+
let program = null;
|
|
695
|
+
for (const c of candidates) for (const ext of COPY_EXTS) {
|
|
696
|
+
if (treeExts && !c.declared && !treeExts.has(ext)) continue;
|
|
697
|
+
const p = resolve(c.base + ext);
|
|
698
|
+
// Inside the tree the index is the confinement: only a file it holds can be a hit. A declared
|
|
699
|
+
// library outside the tree is confined to itself.
|
|
700
|
+
if (c.declared && c.dir && p !== c.dir && !p.startsWith(c.dir + sep)) continue;
|
|
701
|
+
let hit = null;
|
|
702
|
+
if (ctx.cache.has(p)) hit = ctx.cache.get(p);
|
|
703
|
+
else {
|
|
704
|
+
if (ctx.fileIndex && !c.declared) hit = ctx.fileIndex.get(p.toLowerCase()) || null;
|
|
705
|
+
else {
|
|
706
|
+
try { if (statSync(p).isFile()) hit = p; } catch (e) { if (e.code !== 'ENOENT' && e.code !== 'ENOTDIR') throw e; }
|
|
707
|
+
}
|
|
708
|
+
ctx.cache.set(p, hit);
|
|
709
|
+
}
|
|
710
|
+
if (!hit) continue;
|
|
711
|
+
if (!c.declared && (c.dir === home || inData) && isProgramFile(hit, ctx)) { if (!inData) program ||= hit; continue; }
|
|
712
|
+
return hit;
|
|
713
|
+
}
|
|
714
|
+
return program;
|
|
715
|
+
}
|
|
716
|
+
|
|
717
|
+
function expand(tokens, ctx, stack, format) {
|
|
718
|
+
const out = [];
|
|
719
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
720
|
+
const tk = tokens[i];
|
|
721
|
+
// Where a COPY sits decides what it may take (resolveInTree); copybooks it brings in inherit it.
|
|
722
|
+
if (tk.t === 'word' && tokens[i + 1] && tokens[i + 1].u === 'DIVISION' && DIVISION_WORDS.has(tk.u)) ctx.division = tk.u === 'ID' ? 'IDENTIFICATION' : tk.u;
|
|
723
|
+
if (tk.t === 'exec' && tk.kind === 'SQL') {
|
|
724
|
+
out.push(tk);
|
|
725
|
+
const w = tk.toks;
|
|
726
|
+
if (w[0] && w[0].u === 'INCLUDE' && w[1]) {
|
|
727
|
+
const name = w[1].v;
|
|
728
|
+
if (/^(SQLCA|SQLDA)$/i.test(name)) { ctx.copies.push({ name, via: 'sql-include', status: 'system', file: tk.file, line: tk.line }); continue; }
|
|
729
|
+
for (const t of includeCopy(name, null, [], tk, 'sql-include', ctx, stack, format)) out.push(t);
|
|
730
|
+
}
|
|
731
|
+
continue;
|
|
732
|
+
}
|
|
733
|
+
// `INCLUDE member.` opening a sentence is a COPY to cobc, which pastes the member in.
|
|
734
|
+
if (tk.t === 'word' && tk.u === 'INCLUDE' && (i === 0 || tokens[i - 1].t === 'period')
|
|
735
|
+
&& tokens[i + 1] && (tokens[i + 1].t === 'word' || tokens[i + 1].t === 'lit') && tokens[i + 2] && tokens[i + 2].t === 'period') {
|
|
736
|
+
for (const t of includeCopy(tokens[i + 1].v, null, [], tk, 'include', ctx, stack, format)) out.push(t);
|
|
737
|
+
i += 2;
|
|
738
|
+
continue;
|
|
739
|
+
}
|
|
740
|
+
if (tk.t !== 'word' || tk.u !== 'COPY') { out.push(tk); continue; }
|
|
741
|
+
let j = i + 1;
|
|
742
|
+
const nameTok = tokens[j];
|
|
743
|
+
if (!nameTok || (nameTok.t !== 'word' && nameTok.t !== 'lit')) { out.push(tk); continue; }
|
|
744
|
+
j++;
|
|
745
|
+
// `COPY CSS00100.cpy.` and `COPY WS.WS.` name the members CSS00100.cpy and WS.WS: a period with
|
|
746
|
+
// text straight after it is part of the name, and only the one followed by a space ends the
|
|
747
|
+
// statement. Without this the tail is left in the stream and reads as a paragraph named CPY.
|
|
748
|
+
let name = nameTok.v;
|
|
749
|
+
while (nameTok.t === 'word' && tokens[j] && tokens[j].t === 'period' && tokens[j].joined
|
|
750
|
+
&& tokens[j + 1] && tokens[j + 1].t === 'word' && tokens[j + 1].line === tokens[j].line) {
|
|
751
|
+
name = `${name}.${tokens[j + 1].v}`;
|
|
752
|
+
j += 2;
|
|
753
|
+
}
|
|
754
|
+
let lib = null;
|
|
755
|
+
if (tokens[j] && tokens[j].t === 'word' && (tokens[j].u === 'OF' || tokens[j].u === 'IN') && tokens[j + 1]) { lib = tokens[j + 1].v; j += 2; }
|
|
756
|
+
if (tokens[j] && tokens[j].u === 'SUPPRESS') { j++; if (tokens[j] && tokens[j].u === 'PRINTING') j++; }
|
|
757
|
+
let pairs = [];
|
|
758
|
+
if (tokens[j] && tokens[j].u === 'REPLACING') [pairs, j] = readReplacingPairs(tokens, j + 1);
|
|
759
|
+
if (tokens[j] && tokens[j].t === 'period') j++;
|
|
760
|
+
else ctx.diags.push({ kind: 'copy-without-period', file: tk.file, line: tk.line });
|
|
761
|
+
i = j - 1;
|
|
762
|
+
for (const t of includeCopy(name, lib, pairs, tk, 'copy', ctx, stack, format)) out.push(t);
|
|
763
|
+
}
|
|
764
|
+
return out;
|
|
765
|
+
}
|
|
766
|
+
|
|
767
|
+
function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
|
|
768
|
+
const path = resolveCopy(name, lib, ctx);
|
|
769
|
+
const refused = copyRefusal(name, ctx);
|
|
770
|
+
const record = { name, lib, via, file: at.file, line: at.line, status: path ? 'resolved' : refused || (SYSTEM_COPY.test(name) ? 'system' : 'missing'), path };
|
|
771
|
+
ctx.copies.push(record);
|
|
772
|
+
if (!path) return [];
|
|
773
|
+
if (stack.includes(path) || stack.length > 40) { record.status = 'recursive'; return []; }
|
|
774
|
+
if (++ctx.inclusions > MAX_INCLUSIONS || ctx.copyTokens > MAX_COPY_TOKENS) { record.status = 'expansion-limit'; return []; }
|
|
775
|
+
const src = readSource(path).text;
|
|
776
|
+
// A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
|
|
777
|
+
// directive earlier in the including file may have changed from the file's starting format.
|
|
778
|
+
const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
|
|
779
|
+
const norm = normalize(src, fmt, ctx.defines, ctx.std);
|
|
780
|
+
// A tag written against other text - :TAG:-FIELD, FS-(), 'X'-CLE - is replaced in the text, as the
|
|
781
|
+
// compiler does, so the text around it joins the replacement into one word. A literal is taken as a
|
|
782
|
+
// tag only where it touches a word; elsewhere it is replaced token for token like any operand.
|
|
783
|
+
const escape = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
784
|
+
const partial = [];
|
|
785
|
+
for (const p of pairs) {
|
|
786
|
+
const f = p.from.toks;
|
|
787
|
+
if (p.mode || !f.length) continue;
|
|
788
|
+
const pat = f.map(t => t.v).join('');
|
|
789
|
+
if (p.from.pseudo && /^[:(][A-Za-z0-9_-]*[:)]$/.test(pat)) { partial.push([p, new RegExp(escape(pat), 'gi')]); continue; }
|
|
790
|
+
if (p.from.pseudo || f.length !== 1 || f[0].t !== 'lit') continue;
|
|
791
|
+
const lit = `(['"])${escape(f[0].v)}\\1`;
|
|
792
|
+
const touching = new RegExp(`${lit}(?=[A-Za-z0-9-])|(?<=[A-Za-z0-9-])${lit}`);
|
|
793
|
+
if (norm.entries.some(e => touching.test(e.text))) partial.push([p, new RegExp(lit, 'g')]);
|
|
794
|
+
}
|
|
795
|
+
if (partial.length) {
|
|
796
|
+
for (const e of norm.entries) {
|
|
797
|
+
for (const [p, re] of partial) {
|
|
798
|
+
const rep = p.to.toks.map(t => (t.t === 'lit' ? `'${t.v}'` : t.v)).join(' ');
|
|
799
|
+
let hits = 0;
|
|
800
|
+
const text = e.text.replace(re, () => { hits++; return rep; });
|
|
801
|
+
ctx.replacedChars += Math.max(0, hits * (rep.length - p.from.toks.map(t => t.v).join('').length));
|
|
802
|
+
if (ctx.replacedChars > MAX_REPLACED_CHARS) { record.status = 'expansion-limit'; return []; }
|
|
803
|
+
e.text = text;
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
for (const [p] of partial) p.partial = true;
|
|
807
|
+
}
|
|
808
|
+
const { tokens, diags } = tokenize(norm, path);
|
|
809
|
+
ctx.copyTokens += tokens.length;
|
|
810
|
+
ctx.diags.push(...diags, ...norm.diags.map(d => ({ ...d, file: path })));
|
|
811
|
+
const expanded = expand(tokens, ctx, [...stack, path], norm.finalFormat);
|
|
812
|
+
const out = applyReplacing(expanded, pairs, MAX_REPLACED_GROWTH - ctx.replacedGrowth, true);
|
|
813
|
+
if (!out) { record.status = 'expansion-limit'; return []; }
|
|
814
|
+
ctx.replacedGrowth += out.length - expanded.length;
|
|
815
|
+
return out;
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
// A REPLACE governs the source up to the next one, which replaces it unless ALSO; OFF ends it; output is not rescanned.
|
|
819
|
+
function applyReplaceStatements(tokens, ctx) {
|
|
820
|
+
let active = [];
|
|
821
|
+
const out = [];
|
|
822
|
+
let from = 0;
|
|
823
|
+
const flush = (to) => {
|
|
824
|
+
if (from >= to) return;
|
|
825
|
+
const text = tokens.slice(from, to);
|
|
826
|
+
let done = applyReplacing(text, active, MAX_REPLACED_GROWTH - ctx.replacedGrowth);
|
|
827
|
+
if (!done) {
|
|
828
|
+
ctx.copies.push({ name: 'REPLACE', lib: null, via: 'replace', file: text[0].file, line: text[0].line, status: 'expansion-limit', path: null });
|
|
829
|
+
done = text;
|
|
830
|
+
} else ctx.replacedGrowth += done.length - text.length;
|
|
831
|
+
for (const t of done) out.push(t);
|
|
832
|
+
};
|
|
833
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
834
|
+
const tk = tokens[i];
|
|
835
|
+
const next = tokens[i + 1];
|
|
836
|
+
if (tk.t === 'word' && tk.u === 'REPLACE' && next && (next.t === 'pseudo' || next.u === 'OFF' || next.u === 'ALSO' || next.u === 'LEADING' || next.u === 'TRAILING')) {
|
|
837
|
+
flush(i);
|
|
838
|
+
if (next.u === 'OFF') { active = []; i += tokens[i + 2] && tokens[i + 2].t === 'period' ? 2 : 1; from = i + 1; continue; }
|
|
839
|
+
const [pairs, j] = readReplacingPairs(tokens, next.u === 'ALSO' ? i + 2 : i + 1);
|
|
840
|
+
active = next.u === 'ALSO' ? [...active, ...pairs] : pairs;
|
|
841
|
+
i = tokens[j] && tokens[j].t === 'period' ? j : j - 1;
|
|
842
|
+
from = i + 1;
|
|
843
|
+
}
|
|
844
|
+
}
|
|
845
|
+
flush(tokens.length);
|
|
846
|
+
return out;
|
|
847
|
+
}
|
|
848
|
+
|
|
849
|
+
function stripDirecting(tokens) {
|
|
850
|
+
const out = [];
|
|
851
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
852
|
+
const tk = tokens[i];
|
|
853
|
+
if (tk.t === 'word' && /^(EJECT|SKIP1|SKIP2|SKIP3)$/.test(tk.u)) { if (tokens[i + 1] && tokens[i + 1].t === 'period') i++; continue; }
|
|
854
|
+
if (tk.t === 'word' && tk.u === 'TITLE' && tokens[i + 1] && tokens[i + 1].t === 'lit') { i++; if (tokens[i + 1] && tokens[i + 1].t === 'period') i++; continue; }
|
|
855
|
+
out.push(tk);
|
|
856
|
+
}
|
|
857
|
+
return out;
|
|
858
|
+
}
|
|
859
|
+
|
|
860
|
+
function picInfo(pic, constants) {
|
|
861
|
+
const info = { digits: 0, display: 0, signed: false, alphanumeric: false, national: false };
|
|
862
|
+
const re = /(.)\(([A-Za-z0-9_-]+)\)|(.)/g;
|
|
863
|
+
let m;
|
|
864
|
+
while ((m = re.exec(pic))) {
|
|
865
|
+
const ch = (m[1] || m[3]).toUpperCase();
|
|
866
|
+
let n = 1;
|
|
867
|
+
if (m[1]) {
|
|
868
|
+
const raw = m[2];
|
|
869
|
+
n = /^\d+$/.test(raw) ? Number(raw) : Number(constants && constants.get(raw.toUpperCase()));
|
|
870
|
+
if (!Number.isFinite(n) || n < 0) n = 0;
|
|
871
|
+
}
|
|
872
|
+
if (ch === 'S') { info.signed = true; continue; }
|
|
873
|
+
if (ch === 'V' || ch === 'P') continue;
|
|
874
|
+
if (ch === '9') { info.digits += n; info.display += n; continue; }
|
|
875
|
+
if (ch === 'N' || ch === 'G') { info.national = true; info.display += 2 * n; continue; }
|
|
876
|
+
if (ch === 'X' || ch === 'A') info.alphanumeric = true;
|
|
877
|
+
info.display += n;
|
|
878
|
+
}
|
|
879
|
+
return info;
|
|
880
|
+
}
|
|
881
|
+
|
|
882
|
+
// Storage bytes of a literal: hexadecimal literals hold one byte per two digits, a Z literal adds a
|
|
883
|
+
// terminating null, a national literal holds two bytes per character.
|
|
884
|
+
function literalBytes(tok) {
|
|
885
|
+
const prefix = tok.prefix || '';
|
|
886
|
+
if (prefix === 'X' || prefix === 'BX' || prefix === 'NX') return Math.floor(tok.v.length / (prefix === 'NX' ? 4 : 2)) * (prefix === 'NX' ? 2 : 1);
|
|
887
|
+
if (prefix === 'Z') return tok.v.length + 1;
|
|
888
|
+
if (prefix === 'N' || prefix === 'NC' || prefix === 'U') return tok.v.length * 2;
|
|
889
|
+
return tok.v.length;
|
|
890
|
+
}
|
|
891
|
+
|
|
892
|
+
function binaryBytes(digits, scheme) {
|
|
893
|
+
if (scheme === '2-4-8') return digits <= 4 ? 2 : digits <= 9 ? 4 : 8;
|
|
894
|
+
if (scheme === '1--8') return [1, 1, 1, 2, 2, 3, 3, 4, 4, 4, 5, 5, 6, 6, 6, 7, 7, 8, 8][Math.min(digits, 18)] || 8;
|
|
895
|
+
return digits <= 2 ? 1 : digits <= 4 ? 2 : digits <= 9 ? 4 : 8;
|
|
896
|
+
}
|
|
897
|
+
|
|
898
|
+
function elementarySize(item, scheme, constants) {
|
|
899
|
+
if (!item.picture && !item.usage) {
|
|
900
|
+
const constBytes = constants && constants.textBytes;
|
|
901
|
+
const fromValue = item.values.reduce((n, v) => n + (v.t === 'lit' ? literalBytes(v) : v.t === 'word' && constBytes && constBytes.has(v.u) ? constBytes.get(v.u) : 0), 0);
|
|
902
|
+
if (item.section === 'SCREEN' && !fromValue && item.screenRefItem) return item.screenRefItem.size || 0;
|
|
903
|
+
if (item.section === 'SCREEN') return Math.max(1, fromValue);
|
|
904
|
+
if (fromValue) return fromValue;
|
|
905
|
+
}
|
|
906
|
+
if (item.section === 'SCREEN' && !item.picture) return 1;
|
|
907
|
+
const usage = item.effectiveUsage || 'DISPLAY';
|
|
908
|
+
const p = item.picture ? picInfo(item.picture, constants) : null;
|
|
909
|
+
const digits = p ? p.digits : 0;
|
|
910
|
+
const u = usage.replace('COMPUTATIONAL', 'COMP');
|
|
911
|
+
if (u === 'COMP-1' || u === 'FLOAT-SHORT') return 4;
|
|
912
|
+
if (u === 'COMP-2' || u === 'FLOAT-LONG' || u === 'FLOAT-DECIMAL-16') return 8;
|
|
913
|
+
if (u === 'FLOAT-DECIMAL-34') return 16;
|
|
914
|
+
if (u === 'INDEX') return 4;
|
|
915
|
+
if (u === 'POINTER' || u === 'PROGRAM-POINTER' || u === 'FUNCTION-POINTER' || u === 'PROCEDURE-POINTER') return 8;
|
|
916
|
+
if (u === 'BINARY-CHAR') return 1;
|
|
917
|
+
if (u === 'BINARY-SHORT' || u === 'SIGNED-SHORT' || u === 'UNSIGNED-SHORT') return 2;
|
|
918
|
+
if (u === 'BINARY-LONG' || u === 'BINARY-INT' || u === 'SIGNED-INT' || u === 'UNSIGNED-INT') return 4;
|
|
919
|
+
if (u === 'BINARY-DOUBLE' || u === 'BINARY-LONG-LONG' || u === 'BINARY-C-LONG' || u === 'SIGNED-LONG' || u === 'UNSIGNED-LONG') return 8;
|
|
920
|
+
if (u === 'COMP-3' || u === 'PACKED-DECIMAL') return Math.floor(digits / 2) + 1;
|
|
921
|
+
if (u === 'COMP-6') return Math.ceil(digits / 2);
|
|
922
|
+
if (u === 'COMP-X' || u === 'COMP-N' || (u === 'COMP-5' && p && p.alphanumeric)) {
|
|
923
|
+
if (p && p.alphanumeric) return p.display;
|
|
924
|
+
return Math.max(1, Math.ceil((digits * Math.log(10)) / Math.log(256)));
|
|
925
|
+
}
|
|
926
|
+
if (u === 'COMP' || u === 'COMP-4' || u === 'COMP-5' || u === 'BINARY') return binaryBytes(digits, scheme);
|
|
927
|
+
let len = p ? p.display : 0;
|
|
928
|
+
if (item.signSeparate && p && p.signed) len++;
|
|
929
|
+
return len;
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
function parseDataEntry(toks, section, file) {
|
|
933
|
+
const levelTok = toks[0];
|
|
934
|
+
const level = Number(levelTok.v);
|
|
935
|
+
const item = { level, name: 'FILLER', section, line: levelTok.line, file: levelTok.file, picture: null, usage: null, occurs: 1,
|
|
936
|
+
redefines: null, signSeparate: false, sync: false, values: [], children: [], parent: null, refsInData: [] };
|
|
937
|
+
const report = section === 'REPORT';
|
|
938
|
+
let j = 1;
|
|
939
|
+
if (toks[j] && toks[j].t === 'word' && !DATA_CLAUSE_WORDS.has(toks[j].u) && !(section === 'SCREEN' && SCREEN_CLAUSE_WORDS.has(toks[j].u))
|
|
940
|
+
&& !(report && REPORT_CLAUSE_WORDS.has(toks[j].u))) { item.name = toks[j].u; j++; }
|
|
941
|
+
else if (toks[j] && toks[j].u === 'FILLER') j++;
|
|
942
|
+
while (j < toks.length) {
|
|
943
|
+
const t = toks[j];
|
|
944
|
+
const u = t.u;
|
|
945
|
+
if (report && (u === 'LINE' || u === 'LINES')) {
|
|
946
|
+
item.rwLine = true;
|
|
947
|
+
j++;
|
|
948
|
+
while (toks[j] && ['NUMBER', 'NUMBERS', 'IS', 'ARE', 'ON', 'NEXT', 'PAGE', 'PLUS', '+'].includes(toks[j].u)) j++;
|
|
949
|
+
if (toks[j] && /^\d+$/.test(toks[j].v)) j++;
|
|
950
|
+
continue;
|
|
951
|
+
}
|
|
952
|
+
if (report && (u === 'COLUMN' || u === 'COLUMNS' || u === 'COL' || u === 'COLS')) {
|
|
953
|
+
j++;
|
|
954
|
+
let plus = false;
|
|
955
|
+
while (toks[j] && ['NUMBER', 'NUMBERS', 'IS', 'ARE', 'LEFT', 'RIGHT', 'CENTER', 'CENTRE', 'PLUS', '+'].includes(toks[j].u)) { if (toks[j].u === 'PLUS' || toks[j].u === '+') plus = true; j++; }
|
|
956
|
+
if (toks[j] && /^\d+$/.test(toks[j].v)) { item.rwColumn = plus ? { plus: Number(toks[j].v) } : { at: Number(toks[j].v) }; j++; }
|
|
957
|
+
continue;
|
|
958
|
+
}
|
|
959
|
+
// What a report line prints, and the condition that decides whether it prints, come from these;
|
|
960
|
+
// they are references like any other.
|
|
961
|
+
if (report && (u === 'SOURCE' || u === 'SUM' || u === 'UPON' || u === 'RESET' || ((u === 'PRESENT' || u === 'ABSENT') && toks[j + 1]?.u === 'WHEN'))) {
|
|
962
|
+
j += u === 'PRESENT' || u === 'ABSENT' ? 2 : 1;
|
|
963
|
+
while (toks[j] && (toks[j].t !== 'word' || !(REPORT_CLAUSE_WORDS.has(toks[j].u) || DATA_CLAUSE_WORDS.has(toks[j].u)))) {
|
|
964
|
+
if (toks[j].t === 'word' && !['IS', 'ARE', 'ON'].includes(toks[j].u) && /[A-Z]/.test(toks[j].u)) item.refsInData.push(toks[j]);
|
|
965
|
+
j++;
|
|
966
|
+
}
|
|
967
|
+
continue;
|
|
968
|
+
}
|
|
969
|
+
if (u === 'PIC' || u === 'PICTURE') { j++; if (toks[j] && toks[j].u === 'IS') j++; if (toks[j]) item.picture = toks[j].v; j++; continue; }
|
|
970
|
+
if (u === 'USAGE') { j++; if (toks[j] && toks[j].u === 'IS') j++; if (toks[j]) { item.usage = toks[j].u; j++; } continue; }
|
|
971
|
+
if (USAGE_WORDS.has(u)) { item.usage = u; j++; continue; }
|
|
972
|
+
// OCCURS DYNAMIC [CAPACITY IN name] [FROM n] [TO m]: sized at its most, none without TO; the
|
|
973
|
+
// capacity name is declared by the clause.
|
|
974
|
+
if (u === 'OCCURS' && toks[j + 1] && toks[j + 1].u === 'DYNAMIC') {
|
|
975
|
+
j += 2;
|
|
976
|
+
item.occurs = 0;
|
|
977
|
+
for (;;) {
|
|
978
|
+
const w = toks[j] && toks[j].u;
|
|
979
|
+
if (w === 'CAPACITY') { j++; if (toks[j] && toks[j].u === 'IN') j++; if (toks[j] && toks[j].t === 'word') item.capacityName = toks[j++].u; continue; }
|
|
980
|
+
if (w === 'FROM' || w === 'TO') { j++; if (toks[j] && /^\d+$/.test(toks[j].v)) { if (w === 'TO') item.occurs = Number(toks[j].v); j++; } continue; }
|
|
981
|
+
if (w === 'INITIALIZED') { j++; continue; }
|
|
982
|
+
break;
|
|
983
|
+
}
|
|
984
|
+
continue;
|
|
985
|
+
}
|
|
986
|
+
if (u === 'OCCURS') {
|
|
987
|
+
j++;
|
|
988
|
+
item.occursDeclared = true;
|
|
989
|
+
const readCount = () => {
|
|
990
|
+
const t2 = toks[j];
|
|
991
|
+
if (!t2) return null;
|
|
992
|
+
j++;
|
|
993
|
+
if (/^\d+$/.test(t2.v)) return { n: Number(t2.v) };
|
|
994
|
+
if (t2.t === 'word') return { name: t2.u };
|
|
995
|
+
return null;
|
|
996
|
+
};
|
|
997
|
+
const first = readCount();
|
|
998
|
+
let last = first;
|
|
999
|
+
if (toks[j] && toks[j].u === 'TO') { j++; last = readCount() || first; }
|
|
1000
|
+
if (last && last.n != null) item.occurs = last.n;
|
|
1001
|
+
else if (last && last.name) item.occursName = last.name;
|
|
1002
|
+
continue;
|
|
1003
|
+
}
|
|
1004
|
+
// dependingOn marks the table as variable: its size above is its largest, not what it holds.
|
|
1005
|
+
if (u === 'DEPENDING') { j++; if (toks[j] && toks[j].u === 'ON') j++; if (toks[j] && toks[j].t === 'word') { item.refsInData.push(toks[j]); item.dependingOn = toks[j].u; } j++; continue; }
|
|
1006
|
+
if (u === 'KEY') { j++; if (toks[j] && toks[j].u === 'IS') j++; while (toks[j] && toks[j].t === 'word' && !DATA_CLAUSE_WORDS.has(toks[j].u) && !['INDEXED', 'ASCENDING', 'DESCENDING'].includes(toks[j].u)) item.refsInData.push(toks[j++]); continue; }
|
|
1007
|
+
if (u === 'INDEXED') { j++; if (toks[j] && toks[j].u === 'BY') j++; item.indexNames = []; while (toks[j] && toks[j].t === 'word' && !DATA_CLAUSE_WORDS.has(toks[j].u) && !['ASCENDING', 'DESCENDING'].includes(toks[j].u)) item.indexNames.push(toks[j++].u); continue; }
|
|
1008
|
+
if (u === 'REDEFINES') { item.redefines = toks[j + 1] ? toks[j + 1].u : null; j += 2; continue; }
|
|
1009
|
+
if (u === 'RENAMES') {
|
|
1010
|
+
j++;
|
|
1011
|
+
item.renames = [];
|
|
1012
|
+
let wantName = true;
|
|
1013
|
+
while (toks[j] && toks[j].t === 'word') {
|
|
1014
|
+
const w = toks[j].u;
|
|
1015
|
+
item.refsInData.push(toks[j]);
|
|
1016
|
+
if (w === 'THRU' || w === 'THROUGH') { wantName = true; j++; continue; }
|
|
1017
|
+
if (w === 'OF' || w === 'IN') {
|
|
1018
|
+
if (toks[j + 1] && item.renames.length) { item.refsInData.push(toks[j + 1]); item.renames[item.renames.length - 1].quals.push(toks[j + 1].u); }
|
|
1019
|
+
j += 2;
|
|
1020
|
+
continue;
|
|
1021
|
+
}
|
|
1022
|
+
if (wantName) { item.renames.push({ name: w, quals: [] }); wantName = false; }
|
|
1023
|
+
j++;
|
|
1024
|
+
}
|
|
1025
|
+
continue;
|
|
1026
|
+
}
|
|
1027
|
+
if (u === 'TYPEDEF') { item.typedef = true; j++; if (toks[j] && toks[j].u === 'STRONG') j++; continue; }
|
|
1028
|
+
if (u === 'TYPE' && !report) {
|
|
1029
|
+
j++;
|
|
1030
|
+
if (toks[j] && toks[j].u === 'TO') j++;
|
|
1031
|
+
if (toks[j] && toks[j].t === 'word') { item.typeName = toks[j].u; item.refsInData.push(toks[j]); }
|
|
1032
|
+
j++;
|
|
1033
|
+
continue;
|
|
1034
|
+
}
|
|
1035
|
+
if (u === 'SEPARATE') { item.signSeparate = true; j++; continue; }
|
|
1036
|
+
if (u === 'SIGN' || u === 'LEADING' || u === 'TRAILING') { item.signExplicit = true; j++; continue; }
|
|
1037
|
+
if (u === 'SYNC' || u === 'SYNCHRONIZED') { item.sync = true; j++; continue; }
|
|
1038
|
+
if (u === 'CONSTANT') { item.constant = true; j++; if (toks[j] && toks[j].u === 'AS') j++; while (toks[j] && (toks[j].t === 'num' || toks[j].t === 'lit' || (toks[j].t === 'word' && /^\d+$/.test(toks[j].v)))) item.values.push(toks[j++]); continue; }
|
|
1039
|
+
if (u === 'VALUE' || u === 'VALUES') {
|
|
1040
|
+
j++;
|
|
1041
|
+
while (toks[j] && (toks[j].t !== 'word' || !(DATA_CLAUSE_WORDS.has(toks[j].u) || (report && REPORT_CLAUSE_WORDS.has(toks[j].u))) || toks[j].u === 'IS')) { item.values.push(toks[j]); j++; }
|
|
1042
|
+
continue;
|
|
1043
|
+
}
|
|
1044
|
+
if (section === 'SCREEN' && (u === 'FROM' || u === 'TO' || u === 'USING')) { if (toks[j + 1] && toks[j + 1].t === 'word') { item.refsInData.push(toks[j + 1]); if (!item.screenRef) item.screenRef = toks[j + 1].u; } j += 2; continue; }
|
|
1045
|
+
j++;
|
|
1046
|
+
}
|
|
1047
|
+
return item;
|
|
1048
|
+
}
|
|
1049
|
+
|
|
1050
|
+
// WITH DEBUGGING MODE declares DEBUG-ITEM, laid out as the debug module lays it out, with the
|
|
1051
|
+
// thirty-character DEBUG-CONTENTS the compiler gives it.
|
|
1052
|
+
function debugItem(at) {
|
|
1053
|
+
const entry = (level, name, picture, sign) => ({ level, name, section: 'WORKING-STORAGE', line: at.line, file: at.file, picture, usage: null,
|
|
1054
|
+
occurs: 1, redefines: null, signSeparate: !!sign, signExplicit: !!sign, sync: false, values: [], children: [], parent: null, refsInData: [], implicit: true });
|
|
1055
|
+
const root = entry(1, 'DEBUG-ITEM', null);
|
|
1056
|
+
for (const [name, picture, sign] of [['DEBUG-LINE', 'X(6)'], ['FILLER', 'X'], ['DEBUG-NAME', 'X(30)'], ['FILLER', 'X'],
|
|
1057
|
+
['DEBUG-SUB-1', 'S9(4)', true], ['FILLER', 'X'], ['DEBUG-SUB-2', 'S9(4)', true], ['FILLER', 'X'], ['DEBUG-SUB-3', 'S9(4)', true],
|
|
1058
|
+
['FILLER', 'X'], ['DEBUG-CONTENTS', 'X(30)']]) {
|
|
1059
|
+
const c = entry(3, name, picture, sign);
|
|
1060
|
+
c.parent = root;
|
|
1061
|
+
root.children.push(c);
|
|
1062
|
+
}
|
|
1063
|
+
return [root, ...root.children];
|
|
1064
|
+
}
|
|
1065
|
+
|
|
1066
|
+
// TYPE gives an item the picture, usage and subordinate items of the TYPEDEF it names. The listing
|
|
1067
|
+
// prints the typed item alone; what it holds is implied, reachable by qualification (RE OF WS-Z),
|
|
1068
|
+
// so the copies are returned as items marked typeClone.
|
|
1069
|
+
function applyTypes(items) {
|
|
1070
|
+
const typedefs = new Map(items.filter(i => i.typedef).map(i => [i.name, i]));
|
|
1071
|
+
const added = [];
|
|
1072
|
+
if (!typedefs.size) return added;
|
|
1073
|
+
const clone = (x, parent, at) => {
|
|
1074
|
+
const c = { ...x, parent, section: at.section, line: at.line, file: at.file, children: [], refsInData: [], typedef: false, typeClone: true };
|
|
1075
|
+
c.children = x.children.map(k => clone(k, c, at));
|
|
1076
|
+
added.push(c);
|
|
1077
|
+
return c;
|
|
1078
|
+
};
|
|
1079
|
+
// A TYPEDEF may itself be declared with TYPE, so a type is applied once its own type has been.
|
|
1080
|
+
for (let pass = 0; pass < 8; pass++) {
|
|
1081
|
+
let changed = false;
|
|
1082
|
+
for (const it of items) {
|
|
1083
|
+
if (!it.typeName || it.typeApplied) continue;
|
|
1084
|
+
const td = typedefs.get(it.typeName);
|
|
1085
|
+
if (!td || td === it || (td.typeName && !td.typeApplied)) continue;
|
|
1086
|
+
it.typeApplied = true;
|
|
1087
|
+
changed = true;
|
|
1088
|
+
if (!it.picture) it.picture = td.picture;
|
|
1089
|
+
if (!it.usage) it.usage = td.usage;
|
|
1090
|
+
if (td.signSeparate) it.signSeparate = true;
|
|
1091
|
+
if (!it.children.length) it.children = td.children.map(k => clone(k, it, it));
|
|
1092
|
+
}
|
|
1093
|
+
if (!changed) break;
|
|
1094
|
+
}
|
|
1095
|
+
return added;
|
|
1096
|
+
}
|
|
1097
|
+
|
|
1098
|
+
// SYNCHRONIZED aligns a binary, floating-point, pointer or index item to its own length; packed,
|
|
1099
|
+
// COMP-X and display items are not moved.
|
|
1100
|
+
const ALIGNED_USAGE = /^(COMP|COMP-[1245]|BINARY(-[A-Z-]+)?|FLOAT-[A-Z0-9-]+|(PROGRAM-|FUNCTION-|PROCEDURE-)?POINTER|INDEX|(UN)?SIGNED-[A-Z]+)$/;
|
|
1101
|
+
|
|
1102
|
+
function computeSizes(roots, scheme, constants) {
|
|
1103
|
+
// `at` is where the item starts, counted from the start of its record, because the compiler aligns
|
|
1104
|
+
// a SYNCHRONIZED item against the record and puts the slack inside the group that holds it. A
|
|
1105
|
+
// table's entry is laid out from its own start and rounded up to its widest alignment, so every
|
|
1106
|
+
// occurrence aligns alike.
|
|
1107
|
+
const visit = (item, inheritedUsage, inheritedSignSeparate, at) => {
|
|
1108
|
+
item.effectiveUsage = item.usage || inheritedUsage || null;
|
|
1109
|
+
if (inheritedSignSeparate && !item.signExplicit) item.signSeparate = true;
|
|
1110
|
+
const structural = item.children.filter(c => c.level !== 88 && c.level !== 66 && c.level !== 78);
|
|
1111
|
+
if (!structural.length) {
|
|
1112
|
+
const one = elementarySize(item, scheme, constants);
|
|
1113
|
+
item.contributes = one * item.occurs;
|
|
1114
|
+
// The listing prints the whole table only for POINTER and INDEX; every other usage prints one occurrence.
|
|
1115
|
+
const wholeTable = /^(POINTER|INDEX)$/.test(item.effectiveUsage || '');
|
|
1116
|
+
item.size = wholeTable ? item.contributes : one;
|
|
1117
|
+
const usage = (item.effectiveUsage || '').replace('COMPUTATIONAL', 'COMP');
|
|
1118
|
+
item.align = item.sync && ALIGNED_USAGE.test(usage) ? Math.min(one, 8) : 1;
|
|
1119
|
+
item.maxAlign = item.align;
|
|
1120
|
+
return;
|
|
1121
|
+
}
|
|
1122
|
+
const base = item.occurs > 1 ? 0 : at;
|
|
1123
|
+
let offset = 0;
|
|
1124
|
+
let end = 0;
|
|
1125
|
+
let maxAlign = 1;
|
|
1126
|
+
const startOf = new Map();
|
|
1127
|
+
for (const c of structural) {
|
|
1128
|
+
if (c.redefines) {
|
|
1129
|
+
// REDEFINES names a SIBLING. Resolving it by name across the whole program reached the
|
|
1130
|
+
// first item of that name anywhere, which crossed records.
|
|
1131
|
+
const known = startOf.has(c.redefines);
|
|
1132
|
+
visit(c, item.effectiveUsage, item.signSeparate, base + (known ? startOf.get(c.redefines) : offset));
|
|
1133
|
+
const start = known ? startOf.get(c.redefines) : offset - c.contributes;
|
|
1134
|
+
startOf.set(c.name, start);
|
|
1135
|
+
c.localStart = start;
|
|
1136
|
+
end = Math.max(end, start + c.contributes);
|
|
1137
|
+
// A REDEFINES larger than the item it redefines pushes the next sibling past its end.
|
|
1138
|
+
offset = Math.max(offset, start + c.contributes);
|
|
1139
|
+
maxAlign = Math.max(maxAlign, c.maxAlign);
|
|
1140
|
+
continue;
|
|
1141
|
+
}
|
|
1142
|
+
visit(c, item.effectiveUsage, item.signSeparate, base + offset);
|
|
1143
|
+
const skew = (base + offset) % c.align;
|
|
1144
|
+
if (c.align > 1 && skew) offset += c.align - skew;
|
|
1145
|
+
startOf.set(c.name, offset);
|
|
1146
|
+
c.localStart = offset;
|
|
1147
|
+
offset += c.contributes;
|
|
1148
|
+
end = Math.max(end, offset);
|
|
1149
|
+
maxAlign = Math.max(maxAlign, c.maxAlign);
|
|
1150
|
+
}
|
|
1151
|
+
if (item.occurs > 1 && end % maxAlign) end += maxAlign - (end % maxAlign);
|
|
1152
|
+
item.size = end * item.occurs;
|
|
1153
|
+
item.contributes = item.size;
|
|
1154
|
+
item.align = 1;
|
|
1155
|
+
item.maxAlign = maxAlign;
|
|
1156
|
+
};
|
|
1157
|
+
for (const r of roots) visit(r, null, false, 0);
|
|
1158
|
+
// Offsets are assigned after sizing: a child's absolute start depends on its parent's, which is
|
|
1159
|
+
// only known once the parent's own siblings have been laid out.
|
|
1160
|
+
const place = (item, at) => {
|
|
1161
|
+
item.offset = at;
|
|
1162
|
+
for (const c of item.children) if (c.localStart != null) place(c, at + c.localStart);
|
|
1163
|
+
};
|
|
1164
|
+
for (const r of roots) place(r, 0);
|
|
1165
|
+
}
|
|
1166
|
+
|
|
1167
|
+
// A report group is laid out by column, not by adding up its items: a line is as wide as the column
|
|
1168
|
+
// its rightmost item ends in, COLUMN PLUS counting on from the end of the item before. A group holds
|
|
1169
|
+
// its lines one after another, and every 01 group of a report, like the file the report is written
|
|
1170
|
+
// to, is as large as the largest group.
|
|
1171
|
+
function layoutReport(rd) {
|
|
1172
|
+
const structural = (x) => x.children.filter(c => c.level !== 88 && c.level !== 66 && c.level !== 78);
|
|
1173
|
+
const mark = (x) => { x.rd = rd; for (const c of x.children) mark(c); };
|
|
1174
|
+
for (const g of rd.groups) mark(g);
|
|
1175
|
+
let width = 0;
|
|
1176
|
+
// Items printed at the same column under PRESENT WHEN each keep their own storage, so a line is
|
|
1177
|
+
// never smaller than its items laid end to end.
|
|
1178
|
+
const lineWidth = (line) => {
|
|
1179
|
+
let last = 0;
|
|
1180
|
+
let right = 0;
|
|
1181
|
+
let total = 0;
|
|
1182
|
+
const place = (c) => {
|
|
1183
|
+
const len = c.contributes || c.size || 0;
|
|
1184
|
+
const start = c.rwColumn ? (c.rwColumn.at != null ? c.rwColumn.at : last + c.rwColumn.plus) : last + 1;
|
|
1185
|
+
last = start + len - 1;
|
|
1186
|
+
right = Math.max(right, last);
|
|
1187
|
+
total += len;
|
|
1188
|
+
};
|
|
1189
|
+
const walk = (x) => { for (const c of structural(x)) { if (structural(c).length) walk(c); else place(c); } };
|
|
1190
|
+
if (structural(line).length) walk(line);
|
|
1191
|
+
else place(line);
|
|
1192
|
+
return Math.max(right, total);
|
|
1193
|
+
};
|
|
1194
|
+
// The entries that open a line; a group with no LINE clause anywhere in it is one line.
|
|
1195
|
+
for (const g of rd.groups) {
|
|
1196
|
+
const lines = [];
|
|
1197
|
+
const collect = (x) => { if (x.rwLine) { lines.push(x); return; } for (const c of structural(x)) collect(c); };
|
|
1198
|
+
collect(g);
|
|
1199
|
+
if (!lines.length) lines.push(g);
|
|
1200
|
+
let total = 0;
|
|
1201
|
+
for (const line of lines) {
|
|
1202
|
+
const w = lineWidth(line);
|
|
1203
|
+
if (line !== g) { line.size = w; line.contributes = w * (line.occurs || 1); }
|
|
1204
|
+
total += w;
|
|
1205
|
+
}
|
|
1206
|
+
width = Math.max(width, total);
|
|
1207
|
+
}
|
|
1208
|
+
for (const g of rd.groups) { g.size = width; g.contributes = width; }
|
|
1209
|
+
rd.width = width;
|
|
1210
|
+
}
|
|
1211
|
+
|
|
1212
|
+
const ENV_PARAGRAPHS = new Set(['CONFIGURATION', 'SOURCE-COMPUTER', 'OBJECT-COMPUTER', 'SPECIAL-NAMES', 'REPOSITORY', 'INPUT-OUTPUT', 'FILE-CONTROL', 'I-O-CONTROL']);
|
|
1213
|
+
|
|
1214
|
+
const SECTION_NAMES = { 'WORKING-STORAGE': 'WORKING-STORAGE', 'LOCAL-STORAGE': 'LOCAL-STORAGE', LINKAGE: 'LINKAGE', FILE: 'FILE', SCREEN: 'SCREEN', REPORT: 'REPORT', COMMUNICATION: 'COMMUNICATION' };
|
|
1215
|
+
|
|
1216
|
+
function sentences(tokens, from, to) {
|
|
1217
|
+
const out = [];
|
|
1218
|
+
let cur = [];
|
|
1219
|
+
for (let i = from; i < to; i++) {
|
|
1220
|
+
const t = tokens[i];
|
|
1221
|
+
if (t.t === 'period') { if (cur.length) out.push(cur); cur = []; continue; }
|
|
1222
|
+
cur.push(t);
|
|
1223
|
+
}
|
|
1224
|
+
if (cur.length) out.push(cur);
|
|
1225
|
+
return out;
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1228
|
+
function findDivision(tokens, from, to, name) {
|
|
1229
|
+
for (let i = from; i < to - 1; i++) if (tokens[i].t === 'word' && tokens[i].u === name && tokens[i + 1].t === 'word' && tokens[i + 1].u === 'DIVISION') return i;
|
|
1230
|
+
return -1;
|
|
1231
|
+
}
|
|
1232
|
+
|
|
1233
|
+
function parseProgram(tokens, from, to, scheme, defines, hostVariables = true) {
|
|
1234
|
+
const prog = { id: null, line: tokens[from] ? tokens[from].line : 0, items: [], files: [], labels: [], calls: [], execs: [], refs: [], accepts: [], diags: [], statements: [], resolved: new Map() };
|
|
1235
|
+
for (let i = from; i < to - 1; i++) {
|
|
1236
|
+
if (tokens[i].t === 'word' && (tokens[i].u === 'PROGRAM-ID' || tokens[i].u === 'FUNCTION-ID')) {
|
|
1237
|
+
let j = i + 1;
|
|
1238
|
+
if (tokens[j] && tokens[j].t === 'period') j++;
|
|
1239
|
+
if (tokens[j]) prog.id = tokens[j].v.toUpperCase();
|
|
1240
|
+
break;
|
|
1241
|
+
}
|
|
1242
|
+
}
|
|
1243
|
+
const dataAt = findDivision(tokens, from, to, 'DATA');
|
|
1244
|
+
let procAt = findDivision(tokens, from, to, 'PROCEDURE');
|
|
1245
|
+
// A program without an ENVIRONMENT DIVISION header can still open with its paragraphs.
|
|
1246
|
+
let envAt = findDivision(tokens, from, to, 'ENVIRONMENT');
|
|
1247
|
+
if (envAt < 0) {
|
|
1248
|
+
const stop = [dataAt, procAt, to].filter(x => x >= 0).reduce((a, b) => Math.min(a, b));
|
|
1249
|
+
for (let i = from; i < stop; i++) if (tokens[i].t === 'word' && ENV_PARAGRAPHS.has(tokens[i].u) && tokens[i + 1] && (tokens[i + 1].t === 'period' || tokens[i + 1].u === 'SECTION')) { envAt = i; break; }
|
|
1250
|
+
}
|
|
1251
|
+
const procEnd = to;
|
|
1252
|
+
let wsAt = -1;
|
|
1253
|
+
if (dataAt < 0) {
|
|
1254
|
+
for (let i = from; i < to - 1; i++) if (tokens[i].t === 'word' && SECTION_NAMES[tokens[i].u] && tokens[i + 1].u === 'SECTION') { wsAt = i; break; }
|
|
1255
|
+
}
|
|
1256
|
+
const dataStart = dataAt >= 0 ? dataAt : wsAt;
|
|
1257
|
+
const dataEnd = procAt >= 0 ? procAt : to;
|
|
1258
|
+
const envEnd = dataStart >= 0 ? dataStart : dataEnd;
|
|
1259
|
+
|
|
1260
|
+
if (envAt >= 0) {
|
|
1261
|
+
for (const s of sentences(tokens, envAt, envEnd)) {
|
|
1262
|
+
const dbg = s.findIndex((t, k) => t.u === 'DEBUGGING' && s[k + 1] && s[k + 1].u === 'MODE');
|
|
1263
|
+
if (dbg >= 0) prog.debuggingMode = s[dbg];
|
|
1264
|
+
const k = s.findIndex(t => t.u === 'SELECT');
|
|
1265
|
+
if (k >= 0 && s[k + 1]) {
|
|
1266
|
+
let n = k + 1;
|
|
1267
|
+
if (s[n].u === 'OPTIONAL') n++;
|
|
1268
|
+
const f = { name: s[n] ? s[n].u : null, line: s[k].line, file: s[k].file, assign: null, envRefs: [] };
|
|
1269
|
+
const a = s.findIndex(t => t.u === 'ASSIGN');
|
|
1270
|
+
if (a >= 0) {
|
|
1271
|
+
let m = a + 1;
|
|
1272
|
+
while (s[m] && (s[m].u === 'TO' || s[m].u === 'USING' || s[m].u === 'DYNAMIC' || s[m].u === 'EXTERNAL')) m++;
|
|
1273
|
+
if (s[m]) f.assign = { t: s[m].t, v: s[m].t === 'word' ? s[m].u : s[m].v };
|
|
1274
|
+
}
|
|
1275
|
+
for (let m = n + 1; m < s.length; m++) if (s[m].t === 'word') f.envRefs.push(s[m]);
|
|
1276
|
+
prog.files.push(f);
|
|
1277
|
+
} else {
|
|
1278
|
+
for (const t of s) if (t.t === 'word') prog.refs.push({ tok: t, zone: 'env' });
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
|
|
1283
|
+
const roots = [];
|
|
1284
|
+
if (dataStart >= 0) {
|
|
1285
|
+
let section = null;
|
|
1286
|
+
let stack = [];
|
|
1287
|
+
let currentFd = null;
|
|
1288
|
+
let currentRd = null;
|
|
1289
|
+
const dataSentences = sentences(tokens, dataStart, dataEnd);
|
|
1290
|
+
for (let s of dataSentences) {
|
|
1291
|
+
// What an EXEC SQL INCLUDE brings in follows it in the same sentence, up to its own first period.
|
|
1292
|
+
while (s.length && s[0].t === 'exec') { prog.execs.push(s[0]); s = s.slice(1); }
|
|
1293
|
+
if (!s.length) continue;
|
|
1294
|
+
const first = s[0];
|
|
1295
|
+
if (first.u === 'DATA' && s[1] && s[1].u === 'DIVISION') { if (s.length > 2 && s[2].u === 'SECTION') {} continue; }
|
|
1296
|
+
if (first.t === 'word' && SECTION_NAMES[first.u] && s[1] && s[1].u === 'SECTION') { section = SECTION_NAMES[first.u]; stack = []; currentFd = null; currentRd = null; continue; }
|
|
1297
|
+
// A report description and a communication description open their own sections and are not
|
|
1298
|
+
// files. An FD or SD needs no FILE SECTION header before it: the compiler reads one without.
|
|
1299
|
+
if (first.t === 'word' && (first.u === 'RD' || first.u === 'CD') && s[1]) {
|
|
1300
|
+
section = first.u === 'RD' ? 'REPORT' : 'COMMUNICATION';
|
|
1301
|
+
currentFd = null;
|
|
1302
|
+
currentRd = first.u === 'RD' ? { name: s[1].u, line: first.line, file: first.file, groups: [] } : null;
|
|
1303
|
+
if (currentRd) (prog.reports ||= []).push(currentRd);
|
|
1304
|
+
else (prog.cds ||= []).push(s[1].u);
|
|
1305
|
+
stack = [];
|
|
1306
|
+
continue;
|
|
1307
|
+
}
|
|
1308
|
+
if (first.t === 'word' && (first.u === 'FD' || first.u === 'SD') && s[1]) {
|
|
1309
|
+
section = 'FILE';
|
|
1310
|
+
currentRd = null;
|
|
1311
|
+
currentFd = { kind: first.u, name: s[1].u, line: first.line, file: first.file, records: [], fdTok: s[1], declaredMax: 0, varying: false };
|
|
1312
|
+
const recAt = s.findIndex(x => x.t === 'word' && x.u === 'RECORD');
|
|
1313
|
+
if (recAt >= 0) for (let m = recAt + 1; m < s.length && !['LABEL', 'BLOCK', 'DATA', 'VALUE', 'RECORDING', 'CODE-SET', 'LINAGE', 'REPORT', 'REPORTS', 'DEPENDING'].includes(s[m].u); m++) {
|
|
1314
|
+
if (s[m].u === 'VARYING') currentFd.varying = true;
|
|
1315
|
+
if (/^\d+$/.test(s[m].v)) currentFd.declaredMax = Math.max(currentFd.declaredMax, Number(s[m].v));
|
|
1316
|
+
}
|
|
1317
|
+
const repAt = s.findIndex(x => x.t === 'word' && (x.u === 'REPORT' || x.u === 'REPORTS'));
|
|
1318
|
+
if (repAt >= 0) {
|
|
1319
|
+
currentFd.reports = [];
|
|
1320
|
+
for (let m = repAt + 1; m < s.length && s[m].t === 'word' && !['LABEL', 'BLOCK', 'DATA', 'VALUE', 'RECORD', 'RECORDING', 'CODE-SET', 'LINAGE'].includes(s[m].u); m++) {
|
|
1321
|
+
if (s[m].u !== 'IS' && s[m].u !== 'ARE') currentFd.reports.push(s[m].u);
|
|
1322
|
+
}
|
|
1323
|
+
}
|
|
1324
|
+
for (let m = 2; m < s.length; m++) if (s[m].t === 'word' && s[m - 1] && ['ON', 'DEPENDING', 'IS'].includes(s[m - 1].u)) prog.refs.push({ tok: s[m], zone: 'data' });
|
|
1325
|
+
prog.fds = prog.fds || [];
|
|
1326
|
+
prog.fds.push(currentFd);
|
|
1327
|
+
stack = [];
|
|
1328
|
+
continue;
|
|
1329
|
+
}
|
|
1330
|
+
if ((first.t === 'word' || first.t === 'num') && /^\d{1,2}$/.test(first.v)) {
|
|
1331
|
+
const item = parseDataEntry(s, section, first.file);
|
|
1332
|
+
if (item.level === 88 || item.level === 66) {
|
|
1333
|
+
const parent = item.level === 88 ? stack[stack.length - 1] : null;
|
|
1334
|
+
if (parent) { parent.children.push(item); item.parent = parent; }
|
|
1335
|
+
else if (item.level === 66 && roots.length) { roots[roots.length - 1].children.push(item); item.parent = roots[roots.length - 1]; }
|
|
1336
|
+
prog.items.push(item);
|
|
1337
|
+
// RENAMES and VALUE THRU operands are references like any other.
|
|
1338
|
+
for (const r of item.refsInData) prog.refs.push({ tok: r, zone: 'data' });
|
|
1339
|
+
continue;
|
|
1340
|
+
}
|
|
1341
|
+
if (item.level === 78) { prog.items.push(item); continue; }
|
|
1342
|
+
if (item.level === 1 || item.level === 77) {
|
|
1343
|
+
stack = [item];
|
|
1344
|
+
roots.push(item);
|
|
1345
|
+
if (currentFd && section === 'FILE') { currentFd.records.push(item); item.fd = currentFd; }
|
|
1346
|
+
if (currentRd && section === 'REPORT') { currentRd.groups.push(item); item.rd = currentRd; }
|
|
1347
|
+
} else {
|
|
1348
|
+
while (stack.length && stack[stack.length - 1].level >= item.level) stack.pop();
|
|
1349
|
+
const parent = stack[stack.length - 1];
|
|
1350
|
+
if (parent) { parent.children.push(item); item.parent = parent; } else roots.push(item);
|
|
1351
|
+
stack.push(item);
|
|
1352
|
+
}
|
|
1353
|
+
prog.items.push(item);
|
|
1354
|
+
for (const r of item.refsInData) prog.refs.push({ tok: r, zone: 'data' });
|
|
1355
|
+
continue;
|
|
1356
|
+
}
|
|
1357
|
+
prog.diags.push({ kind: 'unrecognised-data-sentence', line: first.line, file: first.file });
|
|
1358
|
+
}
|
|
1359
|
+
prog.items.push(...applyTypes(prog.items));
|
|
1360
|
+
if (prog.debuggingMode) { const d = debugItem(prog.debuggingMode); roots.unshift(d[0]); prog.items.unshift(...d); }
|
|
1361
|
+
const constants = new Map();
|
|
1362
|
+
constants.textBytes = new Map();
|
|
1363
|
+
constants.text = new Map();
|
|
1364
|
+
for (const [k, v] of defines || []) { if (/^-?\d+$/.test(v)) constants.set(k, Number(v)); else constants.textBytes.set(k, v.length); }
|
|
1365
|
+
for (const it of prog.items) {
|
|
1366
|
+
if (it.level !== 78 && !it.constant) continue;
|
|
1367
|
+
const textLit = it.values.find(v => v.t === 'lit');
|
|
1368
|
+
if (textLit && !constants.textBytes.has(it.name)) constants.textBytes.set(it.name, literalBytes(textLit));
|
|
1369
|
+
if (textLit && !constants.text.has(it.name)) constants.text.set(it.name, textLit.v);
|
|
1370
|
+
const lit = it.values.find(v => v.t === 'num' || (v.t === 'word' && /^\d+$/.test(v.v)));
|
|
1371
|
+
if (lit) constants.set(it.name, Number(lit.v));
|
|
1372
|
+
}
|
|
1373
|
+
for (const it of prog.items) if (it.occursName) it.occurs = constants.get(it.occursName) ?? 1;
|
|
1374
|
+
prog.renamesPending = prog.items.filter(it => it.level === 66 && it.renames && it.renames.length);
|
|
1375
|
+
prog.constants = constants;
|
|
1376
|
+
for (const it of prog.items) if (it.screenRef) it.screenRefItem = prog.items.find(x => x.name === it.screenRef && x.section !== 'SCREEN') || null;
|
|
1377
|
+
// REDEFINES names a SIBLING: another item at the same level under the same parent, or another
|
|
1378
|
+
// 01 in the same section when there is no parent. Resolving it by name across the whole program
|
|
1379
|
+
// reached the first item of that name anywhere, which crossed records in both directions;
|
|
1380
|
+
// resolving it only within a group missed 01 REDEFINES 01, which is how a message buffer and
|
|
1381
|
+
// the record laid over it are usually written.
|
|
1382
|
+
for (const it of prog.items) {
|
|
1383
|
+
if (!it.redefines) continue;
|
|
1384
|
+
const siblings = it.parent ? it.parent.children : roots.filter(x => x.section === it.section);
|
|
1385
|
+
const before = siblings.slice(0, Math.max(0, siblings.indexOf(it)));
|
|
1386
|
+
it.redefinesItem = [...before].reverse().find(x => x.name === it.redefines)
|
|
1387
|
+
|| siblings.find(x => x !== it && x.name === it.redefines) || null;
|
|
1388
|
+
}
|
|
1389
|
+
computeSizes(roots, scheme, constants);
|
|
1390
|
+
const subtreeCache = new Map();
|
|
1391
|
+
const subtree = (record) => {
|
|
1392
|
+
if (subtreeCache.has(record)) return subtreeCache.get(record);
|
|
1393
|
+
const all = [];
|
|
1394
|
+
(function walk(x) { for (const c of x.children) { if (c.level === 88 || c.level === 66 || c.level === 78) continue; all.push(c); walk(c); } })(record);
|
|
1395
|
+
subtreeCache.set(record, all);
|
|
1396
|
+
return all;
|
|
1397
|
+
};
|
|
1398
|
+
// RENAMES spans storage from the start of the first item to the end of the last, and either
|
|
1399
|
+
// end may be a group, so the span comes from offsets rather than a list of leaves.
|
|
1400
|
+
for (const it of prog.renamesPending || []) {
|
|
1401
|
+
const record = it.parent;
|
|
1402
|
+
if (!record) continue;
|
|
1403
|
+
const all = subtree(record);
|
|
1404
|
+
const match = (spec) => {
|
|
1405
|
+
const cands = all.filter(x => x.name === spec.name);
|
|
1406
|
+
if (cands.length < 2 || !spec.quals.length) return cands[0];
|
|
1407
|
+
return cands.find(c => {
|
|
1408
|
+
const anc = [];
|
|
1409
|
+
for (let q = c.parent; q; q = q.parent) anc.push(q.name);
|
|
1410
|
+
let pos = 0;
|
|
1411
|
+
for (const q of spec.quals) { const k = anc.indexOf(q, pos); if (k < 0) return false; pos = k + 1; }
|
|
1412
|
+
return true;
|
|
1413
|
+
}) || cands[0];
|
|
1414
|
+
};
|
|
1415
|
+
const a = match(it.renames[0]);
|
|
1416
|
+
const b = it.renames[1] ? match(it.renames[1]) : a;
|
|
1417
|
+
if (!a || !b) continue;
|
|
1418
|
+
const from = Math.min(a.offset, b.offset);
|
|
1419
|
+
const to = Math.max(a.offset + a.contributes, b.offset + b.contributes);
|
|
1420
|
+
it.size = to - from;
|
|
1421
|
+
it.contributes = 0;
|
|
1422
|
+
// The items the alias covers, so a value reaching one of them is known to reach the alias.
|
|
1423
|
+
it.renamesSpan = all.filter(x => x.offset >= from && x.offset + x.contributes <= to && x !== it);
|
|
1424
|
+
}
|
|
1425
|
+
for (const rd of prog.reports || []) layoutReport(rd);
|
|
1426
|
+
// A file is as large as the longer of its declared length and its record description; the
|
|
1427
|
+
// compiler warns when a record exceeds the declared maximum but still uses the record. A report
|
|
1428
|
+
// file's record is as wide as the widest line of the reports written to it.
|
|
1429
|
+
for (const fd of prog.fds || []) {
|
|
1430
|
+
const reports = (fd.reports || []).map(n => (prog.reports || []).find(r => r.name === n)).filter(Boolean);
|
|
1431
|
+
fd.size = Math.max(fd.declaredMax || 0, 0, ...fd.records.map(r => r.size || 0), ...reports.map(r => r.width || 0));
|
|
1432
|
+
}
|
|
1433
|
+
}
|
|
1434
|
+
|
|
1435
|
+
if (procAt >= 0) parseProcedure(tokens, procAt, procEnd, prog, hostVariables);
|
|
1436
|
+
else prog.diags.push({ kind: 'no-procedure-division' });
|
|
1437
|
+
resolveReferences(prog);
|
|
1438
|
+
return prog;
|
|
1439
|
+
}
|
|
1440
|
+
|
|
1441
|
+
export function segmentEnd(tokens, i, to) {
|
|
1442
|
+
let depth = 0;
|
|
1443
|
+
for (let k = i + 1; k < to; k++) {
|
|
1444
|
+
const t = tokens[k];
|
|
1445
|
+
if (t.t === 'sep') { depth += t.v === '(' ? 1 : -1; continue; }
|
|
1446
|
+
if (depth > 0) continue;
|
|
1447
|
+
if (t.t === 'period' || t.t === 'exec') return k;
|
|
1448
|
+
if (t.t === 'word' && (VERBS.has(t.u) || SCOPE_TERMINATORS.has(t.u) || STATEMENT_BREAKS.has(t.u))) {
|
|
1449
|
+
const prev = tokens[k - 1];
|
|
1450
|
+
if (prev && prev.t === 'word' && ((prev.u === 'EXIT' && t.u === 'PERFORM') || prev.u === 'USAGE' || prev.u === 'UPON')) continue;
|
|
1451
|
+
if (t.u === 'EXIT' && prev && prev.t === 'word' && prev.u === 'UNTIL') continue;
|
|
1452
|
+
if (t.u === 'DISPLAY' && prev && prev.u === 'IS') continue;
|
|
1453
|
+
return k;
|
|
1454
|
+
}
|
|
1455
|
+
}
|
|
1456
|
+
return to;
|
|
1457
|
+
}
|
|
1458
|
+
|
|
1459
|
+
function parseProcedure(tokens, procAt, to, prog, hostVariables) {
|
|
1460
|
+
let i = procAt + 2;
|
|
1461
|
+
const headerRefs = [];
|
|
1462
|
+
while (i < to && tokens[i].t !== 'period') { if (tokens[i].t === 'word') headerRefs.push(tokens[i]); i++; }
|
|
1463
|
+
const receiving = new Map();
|
|
1464
|
+
let inUsing = false;
|
|
1465
|
+
let paramMode = 'REFERENCE';
|
|
1466
|
+
prog.paramTokens = [];
|
|
1467
|
+
for (let k = 0; k < headerRefs.length; k++) {
|
|
1468
|
+
const t = headerRefs[k];
|
|
1469
|
+
// WITH C LINKAGE, WITH PASCAL LINKAGE: a calling convention, not data.
|
|
1470
|
+
if (t.u === 'WITH' && headerRefs[k + 2] && headerRefs[k + 2].u === 'LINKAGE') { k += 2; continue; }
|
|
1471
|
+
if (t.u === 'USING' || t.u === 'CHAINING') { inUsing = true; continue; }
|
|
1472
|
+
if (t.u === 'RETURNING') { inUsing = false; continue; }
|
|
1473
|
+
if (inUsing && ['REFERENCE', 'VALUE', 'CONTENT'].includes(t.u)) { paramMode = t.u; continue; }
|
|
1474
|
+
if (inUsing && !['BY', 'OPTIONAL'].includes(t.u)) { receiving.set(t, 'PROCEDURE-USING'); prog.paramTokens.push({ tok: t, mode: paramMode }); }
|
|
1475
|
+
prog.refs.push({ tok: t, zone: 'proc' });
|
|
1476
|
+
}
|
|
1477
|
+
i++;
|
|
1478
|
+
// Where the statements begin, so a reader of control flow can walk the same tokens in order.
|
|
1479
|
+
prog.proc = { tokens, from: i, to };
|
|
1480
|
+
let sentenceStart = true;
|
|
1481
|
+
let currentVerb = null;
|
|
1482
|
+
let openMode = null;
|
|
1483
|
+
for (; i < to; i++) {
|
|
1484
|
+
const t = tokens[i];
|
|
1485
|
+
if (t.t === 'period') { sentenceStart = true; continue; }
|
|
1486
|
+
if (sentenceStart && t.t === 'word') {
|
|
1487
|
+
const next = tokens[i + 1];
|
|
1488
|
+
if (next && next.t === 'word' && next.u === 'SECTION' && !NOT_LABELS.has(t.u)) {
|
|
1489
|
+
prog.labels.push({ kind: 'S', name: t.u, line: t.line, file: t.file, at: i });
|
|
1490
|
+
i++;
|
|
1491
|
+
if (tokens[i + 1] && tokens[i + 1].t !== 'period') i++;
|
|
1492
|
+
continue;
|
|
1493
|
+
}
|
|
1494
|
+
if (next && next.t === 'period' && !NOT_LABELS.has(t.u) && !VERBS.has(t.u) && !SCOPE_TERMINATORS.has(t.u)) {
|
|
1495
|
+
prog.labels.push({ kind: 'P', name: t.u, line: t.line, file: t.file, at: i });
|
|
1496
|
+
continue;
|
|
1497
|
+
}
|
|
1498
|
+
if (t.u === 'END' && next && next.u === 'DECLARATIVES') { i++; continue; }
|
|
1499
|
+
}
|
|
1500
|
+
// What an INCLUDE brings in follows it, and may open with a paragraph header.
|
|
1501
|
+
if (t.t === 'exec' && t.kind === 'SQL' && t.toks[0]?.u === 'INCLUDE') { prog.execs.push(t); continue; }
|
|
1502
|
+
sentenceStart = false;
|
|
1503
|
+
if (t.t === 'exec') {
|
|
1504
|
+
prog.execs.push(t);
|
|
1505
|
+
if (t.kind === 'SQL' && hostVariables) hostVariableRefs(t, prog.refs, receiving);
|
|
1506
|
+
if (t.kind === 'CICS' && hostVariables) cicsArgumentRefs(t, prog.refs, receiving);
|
|
1507
|
+
continue;
|
|
1508
|
+
}
|
|
1509
|
+
if (t.t !== 'word') continue;
|
|
1510
|
+
// WHEN is not a verb, so its condition is kept as a statement of its own for whatever reads
|
|
1511
|
+
// conditions. It moves no data, and the tokens around it are read as they are anywhere else.
|
|
1512
|
+
if (t.u === 'WHEN' && currentVerb) {
|
|
1513
|
+
const cond = tokens.slice(i + 1, segmentEnd(tokens, i, to));
|
|
1514
|
+
prog.statements.push({ verb: 'WHEN', line: t.line, file: t.file, at: i, targets: [], sources: identifierTokens(cond, 0, cond.length),
|
|
1515
|
+
literals: cond.filter((x) => x.t === 'lit' || x.t === 'num'), ops: cond.filter((x) => x.t === 'op').map((x) => x.v) });
|
|
1516
|
+
}
|
|
1517
|
+
if (!VERBS.has(t.u)) {
|
|
1518
|
+
t.verb = currentVerb;
|
|
1519
|
+
if (currentVerb === 'OPEN') { if (['INPUT', 'OUTPUT', 'EXTEND', 'I-O'].includes(t.u)) openMode = t.u; else t.openMode = openMode; }
|
|
1520
|
+
prog.refs.push({ tok: t, zone: 'proc' });
|
|
1521
|
+
continue;
|
|
1522
|
+
}
|
|
1523
|
+
const prev = tokens[i - 1];
|
|
1524
|
+
if (prev && prev.t === 'word' && prev.u === 'EXIT' && t.u === 'PERFORM') continue;
|
|
1525
|
+
if (t.u === 'EXIT' && prev && prev.t === 'word' && prev.u === 'UNTIL') continue;
|
|
1526
|
+
currentVerb = t.u;
|
|
1527
|
+
openMode = null;
|
|
1528
|
+
const end = segmentEnd(tokens, i, to);
|
|
1529
|
+
const seg = tokens.slice(i + 1, end);
|
|
1530
|
+
// Collect this statement's targets in their own map: snapshotting the program-wide map per
|
|
1531
|
+
// statement is quadratic, and one generated file has tens of thousands of statements.
|
|
1532
|
+
const stmtReceiving = new Map();
|
|
1533
|
+
markReceiving(t.u, seg, stmtReceiving);
|
|
1534
|
+
const targets = [...stmtReceiving.keys()];
|
|
1535
|
+
for (const [k, v] of stmtReceiving) receiving.set(k, v);
|
|
1536
|
+
const targetSet = new Set(targets);
|
|
1537
|
+
const stmt = { verb: t.u, line: t.line, file: t.file, at: i, end, targets, sources: identifierTokens(seg, 0, seg.length).filter(x => !targetSet.has(x)), literals: seg.filter(x => x.t === 'lit' || x.t === 'num') };
|
|
1538
|
+
// A condition's operators, which say whether it bounds a value or only compares it for equality.
|
|
1539
|
+
if (t.u === 'IF' || t.u === 'EVALUATE') stmt.ops = seg.filter(x => x.t === 'op').map(x => x.v);
|
|
1540
|
+
const indexes = indexTokens(seg);
|
|
1541
|
+
if (indexes.length) stmt.indexes = indexes;
|
|
1542
|
+
const fns = seg.filter((x, k) => x.t === 'word' && k > 0 && seg[k - 1].t === 'word' && seg[k - 1].u === 'FUNCTION').map((x) => x.u);
|
|
1543
|
+
if (fns.length) stmt.fns = fns;
|
|
1544
|
+
// INSPECT ... TALLYING writes a count, not the bytes it inspected.
|
|
1545
|
+
if (t.u === 'INSPECT' && seg.some((x) => x.t === 'word' && x.u === 'TALLYING') && !seg.some((x) => x.t === 'word' && (x.u === 'REPLACING' || x.u === 'CONVERTING'))) stmt.counts = true;
|
|
1546
|
+
// A PERFORM's UNTIL test is a condition, not data moved into the VARYING variable.
|
|
1547
|
+
// SEARCH moves nothing out of the table it searches: its index counts, and each WHEN is a
|
|
1548
|
+
// condition kept as a statement of its own.
|
|
1549
|
+
if (t.u === 'SEARCH') stmt.sources = [];
|
|
1550
|
+
if (t.u === 'MOVE' && seg[0] && seg[0].t === 'word' && (seg[0].u === 'CORRESPONDING' || seg[0].u === 'CORR')) stmt.corresponding = true;
|
|
1551
|
+
if (t.u === 'PERFORM') {
|
|
1552
|
+
const cond = untilTokens(seg);
|
|
1553
|
+
if (cond.size) stmt.sources = stmt.sources.filter((x) => !cond.has(x));
|
|
1554
|
+
const loops = loopsOf(seg);
|
|
1555
|
+
if (loops.length) stmt.loops = loops;
|
|
1556
|
+
}
|
|
1557
|
+
prog.statements.push(stmt);
|
|
1558
|
+
if (t.u === 'CALL') {
|
|
1559
|
+
const target = seg[0] && seg[0].t === 'word' && seg[1] && seg[1].t === 'lit' ? seg[1] : seg[0];
|
|
1560
|
+
if (target) {
|
|
1561
|
+
const usingAt = seg.findIndex(x => x.t === 'word' && x.u === 'USING');
|
|
1562
|
+
const args = [];
|
|
1563
|
+
if (usingAt >= 0) {
|
|
1564
|
+
let mode = 'REFERENCE';
|
|
1565
|
+
let depth = 0;
|
|
1566
|
+
for (let k = usingAt + 1; k < seg.length; k++) {
|
|
1567
|
+
const x = seg[k];
|
|
1568
|
+
if (x.t === 'sep') { depth += x.v === '(' ? 1 : -1; continue; }
|
|
1569
|
+
if (depth > 0) continue;
|
|
1570
|
+
if (x.t === 'word' && (x.u === 'BY' || x.u === 'OPTIONAL')) continue;
|
|
1571
|
+
if (x.t === 'word' && ['REFERENCE', 'CONTENT', 'VALUE'].includes(x.u)) { mode = x.u; continue; }
|
|
1572
|
+
if (x.t === 'word' && ['RETURNING', 'GIVING', 'ON', 'NOT', 'EXCEPTION', 'OVERFLOW'].includes(x.u)) break;
|
|
1573
|
+
// `X OF Y` is ONE argument, X. Taking the qualifier as an argument of its own put the
|
|
1574
|
+
// sink on the whole record — so any field of it looked tainted — and shifted every
|
|
1575
|
+
// later argument by one, which broke the mapping onto the callee's parameters.
|
|
1576
|
+
if (x.t === 'word' && (x.u === 'OF' || x.u === 'IN')) { k++; continue; }
|
|
1577
|
+
// LENGTH OF X and ADDRESS OF X are arguments in their own right: a placeholder keeps
|
|
1578
|
+
// the positions of the arguments after them.
|
|
1579
|
+
if (x.t === 'word' && (x.u === 'LENGTH' || x.u === 'ADDRESS') && seg[k + 1] && seg[k + 1].u === 'OF') { args.push({ word: null, of: x.u, mode }); k += 2; continue; }
|
|
1580
|
+
if (x.t === 'lit' || x.t === 'num') { args.push({ lit: x.v, mode }); continue; }
|
|
1581
|
+
if (x.t === 'word') args.push({ word: x.u, mode, tok: x });
|
|
1582
|
+
}
|
|
1583
|
+
}
|
|
1584
|
+
// A CALL naming a level-78 constant is a call to the program that constant spells, and the
|
|
1585
|
+
// compiler records it as a literal. Read as an identifier it loses the callee — so the
|
|
1586
|
+
// cross-program edge disappears — and it manufactures a dynamic-program-load finding out of
|
|
1587
|
+
// a name fixed at compile time. Measured on the 500-repository corpus: 3,092 of 3,180 call
|
|
1588
|
+
// disagreements were this one shape.
|
|
1589
|
+
const constText = target.t === 'word' && prog.constants && prog.constants.text
|
|
1590
|
+
? prog.constants.text.get(target.u) : undefined;
|
|
1591
|
+
const resolved = target.t === 'lit' ? target.v : constText;
|
|
1592
|
+
prog.calls.push({
|
|
1593
|
+
// A literal program name is padded to its field; the compiler drops the trailing spaces.
|
|
1594
|
+
kind: resolved === undefined ? 'I' : 'L',
|
|
1595
|
+
name: resolved === undefined ? target.u : String(resolved).trimEnd(),
|
|
1596
|
+
...(constText !== undefined ? { viaConstant: target.u } : {}),
|
|
1597
|
+
line: t.line, file: t.file, targetTok: target, using: args, stmtIndex: prog.statements.length - 1,
|
|
1598
|
+
});
|
|
1599
|
+
}
|
|
1600
|
+
}
|
|
1601
|
+
if (t.u === 'ENTRY') {
|
|
1602
|
+
const usingAt = seg.findIndex(x => x.t === 'word' && x.u === 'USING');
|
|
1603
|
+
if (usingAt >= 0) for (const x of seg.slice(usingAt + 1)) if (x.t === 'word' && !['BY', 'REFERENCE', 'VALUE', 'CONTENT', 'OPTIONAL'].includes(x.u)) receiving.set(x, 'ENTRY-USING');
|
|
1604
|
+
}
|
|
1605
|
+
if (t.u === 'ACCEPT') {
|
|
1606
|
+
const fromAt = seg.findIndex(x => x.t === 'word' && x.u === 'FROM');
|
|
1607
|
+
// The token, not just the name: ACCEPT F-DATA OF REC-B names one of two same-named fields.
|
|
1608
|
+
prog.accepts.push({ target: seg[0] ? seg[0].u : null, targetTok: seg[0] || null, from: fromAt >= 0 && seg[fromAt + 1] ? seg[fromAt + 1].u : null, line: t.line, file: t.file });
|
|
1609
|
+
}
|
|
1610
|
+
}
|
|
1611
|
+
prog.receivingTokens = receiving;
|
|
1612
|
+
}
|
|
1613
|
+
|
|
1614
|
+
// Identifiers a statement reads. Parentheses usually hold a subscript, whose value is not the data
|
|
1615
|
+
// being moved — but the arguments of an intrinsic function are: `MOVE FUNCTION TRIM(WS-IN) TO X`
|
|
1616
|
+
// carries WS-IN into X, and skipping everything in parentheses lost that whole class of flows.
|
|
1617
|
+
function identifierTokens(seg, from, to) {
|
|
1618
|
+
const out = [];
|
|
1619
|
+
const inFunction = [];
|
|
1620
|
+
for (let k = from; k < to; k++) {
|
|
1621
|
+
const t = seg[k];
|
|
1622
|
+
if (t.t === 'sep') {
|
|
1623
|
+
if (t.v === '(') {
|
|
1624
|
+
const opener = seg[k - 1];
|
|
1625
|
+
const before = seg[k - 2];
|
|
1626
|
+
inFunction.push(!!(opener && opener.t === 'word' && before && before.t === 'word' && before.u === 'FUNCTION') || (inFunction.length > 0 && inFunction[inFunction.length - 1]));
|
|
1627
|
+
} else inFunction.pop();
|
|
1628
|
+
continue;
|
|
1629
|
+
}
|
|
1630
|
+
if (t.t !== 'word') continue;
|
|
1631
|
+
if (inFunction.length && !inFunction[inFunction.length - 1]) continue;
|
|
1632
|
+
const prev = seg[k - 1];
|
|
1633
|
+
if (prev && prev.t === 'word' && (prev.u === 'OF' || prev.u === 'IN')) continue;
|
|
1634
|
+
// The function's own name is not data.
|
|
1635
|
+
if (prev && prev.t === 'word' && prev.u === 'FUNCTION') continue;
|
|
1636
|
+
out.push(t);
|
|
1637
|
+
}
|
|
1638
|
+
return out;
|
|
1639
|
+
}
|
|
1640
|
+
|
|
1641
|
+
// What a statement indexes with. identifierTokens leaves subscripts out, because a subscript is not
|
|
1642
|
+
// the data a statement moves; a rule about where a statement reads or writes needs exactly them.
|
|
1643
|
+
// For each parenthesised group after a data name - reached back over OF and IN, so ELEM OF TAB (I)
|
|
1644
|
+
// indexes ELEM - every name inside it, marked as a subscript, or as the offset or the length of a
|
|
1645
|
+
// reference modification. A second group straight after the first, X(I)(1:3), indexes the same
|
|
1646
|
+
// name. A function's arguments are not an index. A name that turns out not to be data - IF (A > B)
|
|
1647
|
+
// - is left for the caller, which resolves names and drops what does not resolve.
|
|
1648
|
+
function indexTokens(seg) {
|
|
1649
|
+
const out = [];
|
|
1650
|
+
let lastHost = null, lastClose = -1;
|
|
1651
|
+
for (let k = 1; k < seg.length; k++) {
|
|
1652
|
+
const open = seg[k];
|
|
1653
|
+
if (open.t !== 'sep' || open.v !== '(') continue;
|
|
1654
|
+
let host = null;
|
|
1655
|
+
if (seg[k - 1].t === 'sep' && seg[k - 1].v === ')' && lastClose === k - 1) host = lastHost;
|
|
1656
|
+
else if (seg[k - 1].t === 'word') {
|
|
1657
|
+
let h = k - 1;
|
|
1658
|
+
while (h >= 2 && seg[h - 1].t === 'word' && (seg[h - 1].u === 'OF' || seg[h - 1].u === 'IN') && seg[h - 2].t === 'word') h -= 2;
|
|
1659
|
+
if (!(seg[h - 1] && seg[h - 1].t === 'word' && seg[h - 1].u === 'FUNCTION')) host = seg[h];
|
|
1660
|
+
}
|
|
1661
|
+
let depth = 0, close = k, colonAt = -1;
|
|
1662
|
+
for (let j = k; j < seg.length; j++) {
|
|
1663
|
+
const t = seg[j];
|
|
1664
|
+
if (t.t === 'sep') { depth += t.v === '(' ? 1 : -1; if (depth === 0) { close = j; break; } continue; }
|
|
1665
|
+
if (depth === 1 && t.t === 'op' && t.v === ':' && colonAt < 0) colonAt = j;
|
|
1666
|
+
}
|
|
1667
|
+
if (host) {
|
|
1668
|
+
for (let j = k + 1; j < close; j++) {
|
|
1669
|
+
const t = seg[j];
|
|
1670
|
+
// A number is tokenized as a word, and no name is all digits.
|
|
1671
|
+
if (t.t !== 'word' || /^\d+$/.test(t.v)) continue;
|
|
1672
|
+
const prev = seg[j - 1];
|
|
1673
|
+
if (prev && prev.t === 'word' && (prev.u === 'OF' || prev.u === 'IN' || prev.u === 'FUNCTION')) continue;
|
|
1674
|
+
out.push({ host, tok: t, kind: colonAt < 0 ? 'subscript' : j < colonAt ? 'refmod-offset' : 'refmod-length' });
|
|
1675
|
+
}
|
|
1676
|
+
}
|
|
1677
|
+
lastHost = host;
|
|
1678
|
+
lastClose = close;
|
|
1679
|
+
k = close;
|
|
1680
|
+
}
|
|
1681
|
+
return out;
|
|
1682
|
+
}
|
|
1683
|
+
|
|
1684
|
+
// The tokens of a PERFORM's UNTIL conditions, each running to the next AFTER or VARYING.
|
|
1685
|
+
function untilTokens(seg) {
|
|
1686
|
+
const out = new Set();
|
|
1687
|
+
let depth = 0, inUntil = false;
|
|
1688
|
+
for (const x of seg) {
|
|
1689
|
+
if (x.t === 'sep') { depth += x.v === '(' ? 1 : -1; if (inUntil) out.add(x); continue; }
|
|
1690
|
+
if (depth === 0 && x.t === 'word') {
|
|
1691
|
+
if (x.u === 'UNTIL') { inUntil = true; continue; }
|
|
1692
|
+
if (x.u === 'AFTER' || x.u === 'VARYING') inUntil = false;
|
|
1693
|
+
}
|
|
1694
|
+
if (inUntil) out.add(x);
|
|
1695
|
+
}
|
|
1696
|
+
return out;
|
|
1697
|
+
}
|
|
1698
|
+
|
|
1699
|
+
// Each counter a PERFORM varies, with the condition that stops it: VARYING I ... UNTIL c, and every
|
|
1700
|
+
// AFTER J ... UNTIL c after it. A bare PERFORM UNTIL has a condition and no counter.
|
|
1701
|
+
function loopsOf(seg) {
|
|
1702
|
+
const out = [];
|
|
1703
|
+
let depth = 0, cur = null, inUntil = false;
|
|
1704
|
+
for (let k = 0; k < seg.length; k++) {
|
|
1705
|
+
const x = seg[k];
|
|
1706
|
+
if (x.t === 'sep') { depth += x.v === '(' ? 1 : -1; if (inUntil && cur) cur.until.push(x); continue; }
|
|
1707
|
+
if (depth === 0 && x.t === 'word' && (x.u === 'VARYING' || x.u === 'AFTER')) {
|
|
1708
|
+
cur = { counter: seg[k + 1] && seg[k + 1].t === 'word' ? seg[k + 1] : null, until: [] };
|
|
1709
|
+
out.push(cur);
|
|
1710
|
+
inUntil = false;
|
|
1711
|
+
continue;
|
|
1712
|
+
}
|
|
1713
|
+
if (depth === 0 && x.t === 'word' && x.u === 'UNTIL') {
|
|
1714
|
+
if (!cur) { cur = { counter: null, until: [] }; out.push(cur); }
|
|
1715
|
+
inUntil = true;
|
|
1716
|
+
continue;
|
|
1717
|
+
}
|
|
1718
|
+
if (inUntil && cur) cur.until.push(x);
|
|
1719
|
+
}
|
|
1720
|
+
return out.filter((l) => l.until.length);
|
|
1721
|
+
}
|
|
1722
|
+
|
|
1723
|
+
function markReceiving(verb, seg, set) {
|
|
1724
|
+
const at = (w, start = 0) => { let depth = 0; for (let k = start; k < seg.length; k++) { const t = seg[k]; if (t.t === 'sep') { depth += t.v === '(' ? 1 : -1; continue; } if (depth === 0 && t.t === 'word' && w.includes(t.u)) return k; } return -1; };
|
|
1725
|
+
const stopAt = (start, words) => { const k = at(words, start); return k < 0 ? seg.length : k; };
|
|
1726
|
+
// A keyword with nothing after it (`ACCEPT.`, `WITH POINTER.`) names no target.
|
|
1727
|
+
const mark = (from, to) => { for (const t of identifierTokens(seg, from, Math.min(to, seg.length))) set.set(t, verb); };
|
|
1728
|
+
const tail = ['ON', 'NOT', 'SIZE', 'ERROR', 'OVERFLOW', 'EXCEPTION', 'INVALID', 'AT', 'END', 'ROUNDED', 'WITH', 'DELIMITER', 'COUNT', 'TALLYING', 'POINTER', 'REMAINDER', 'CORRESPONDING', 'CORR'];
|
|
1729
|
+
switch (verb) {
|
|
1730
|
+
case 'MOVE': { const k = at(['TO']); if (k >= 0) mark(k + 1, stopAt(k + 1, tail)); break; }
|
|
1731
|
+
case 'ADD': case 'SUBTRACT': { const g = at(['GIVING']); const k = g >= 0 ? g : at([verb === 'ADD' ? 'TO' : 'FROM']); if (k >= 0) mark(k + 1, stopAt(k + 1, tail)); break; }
|
|
1732
|
+
case 'MULTIPLY': { const g = at(['GIVING']); const k = g >= 0 ? g : at(['BY']); if (k >= 0) mark(k + 1, stopAt(k + 1, tail)); break; }
|
|
1733
|
+
case 'DIVIDE': { const g = at(['GIVING']); const k = g >= 0 ? g : at(['INTO']); if (k >= 0) mark(k + 1, stopAt(k + 1, tail)); const r = at(['REMAINDER']); if (r >= 0) mark(r + 1, stopAt(r + 1, tail)); break; }
|
|
1734
|
+
case 'COMPUTE': { const k = seg.findIndex(t => (t.t === 'op' && t.v === '=') || (t.t === 'word' && t.u === 'EQUAL')); if (k >= 0) mark(0, k); break; }
|
|
1735
|
+
case 'ACCEPT': { mark(0, 1); break; }
|
|
1736
|
+
case 'SET': { const k = stopAt(0, ['TO', 'UP', 'DOWN']); mark(0, k); break; }
|
|
1737
|
+
case 'INITIALISE': case 'INITIALIZE': { mark(0, stopAt(0, ['REPLACING', 'WITH', 'ALL', 'TO', 'THEN', 'DEFAULT', 'FILLER', 'ALPHANUMERIC', 'NUMERIC', 'VALUE'])); break; }
|
|
1738
|
+
case 'STRING': { const k = at(['INTO']); if (k >= 0) mark(k + 1, stopAt(k + 1, tail)); const p = at(['POINTER']); if (p >= 0) mark(p + 1, p + 2); break; }
|
|
1739
|
+
case 'UNSTRING': { const k = at(['INTO']); if (k >= 0) mark(k + 1, stopAt(k + 1, ['DELIMITER', 'COUNT', 'WITH', 'TALLYING', 'ON', 'NOT', 'POINTER'])); for (const w of ['POINTER']) { const p = at([w]); if (p >= 0) mark(p + 1, p + 2); } break; }
|
|
1740
|
+
case 'INSPECT': { if (at(['REPLACING', 'CONVERTING']) >= 0) mark(0, 1); const tl = at(['TALLYING']); if (tl >= 0) mark(tl + 1, tl + 2); break; }
|
|
1741
|
+
case 'READ': case 'RETURN': { const k = at(['INTO']); if (k >= 0) mark(k + 1, k + 2); break; }
|
|
1742
|
+
case 'WRITE': case 'REWRITE': case 'RELEASE': { if (at(['FROM']) >= 0) mark(0, 1); break; }
|
|
1743
|
+
case 'CALL': { const k = at(['RETURNING', 'GIVING']); if (k >= 0) mark(k + 1, k + 2); break; }
|
|
1744
|
+
case 'PERFORM': { const v = at(['VARYING']); if (v >= 0) mark(v + 1, v + 2); break; }
|
|
1745
|
+
case 'SEARCH': { const v = at(['VARYING']); if (v >= 0) mark(v + 1, v + 2); break; }
|
|
1746
|
+
default: break;
|
|
1747
|
+
}
|
|
1748
|
+
}
|
|
1749
|
+
|
|
1750
|
+
// The pairs a CORRESPONDING phrase matches: subordinate items of one name under the same names, one
|
|
1751
|
+
// of each pair elementary. FILLER, levels 66 and 88, and an item that REDEFINES, OCCURS or is an
|
|
1752
|
+
// index take no part, nor does anything under one.
|
|
1753
|
+
function correspondingPairs(from, to) {
|
|
1754
|
+
const takesPart = (x) => x.name !== 'FILLER' && x.level !== 66 && x.level !== 88 && !x.redefines && !x.occursDeclared && x.usage !== 'INDEX';
|
|
1755
|
+
const elementary = (x) => !x.children.some((c) => c.level !== 88);
|
|
1756
|
+
const pairs = [];
|
|
1757
|
+
const walk = (a, b) => {
|
|
1758
|
+
for (const cb of b.children) {
|
|
1759
|
+
if (!takesPart(cb)) continue;
|
|
1760
|
+
const ca = a.children.find((x) => takesPart(x) && x.name === cb.name);
|
|
1761
|
+
if (!ca) continue;
|
|
1762
|
+
if (elementary(ca) || elementary(cb)) pairs.push([ca, cb]);
|
|
1763
|
+
else walk(ca, cb);
|
|
1764
|
+
}
|
|
1765
|
+
};
|
|
1766
|
+
if (from.children) walk(from, to);
|
|
1767
|
+
return pairs;
|
|
1768
|
+
}
|
|
1769
|
+
|
|
1770
|
+
function resolveReferences(prog) {
|
|
1771
|
+
const byName = new Map();
|
|
1772
|
+
const add = (name, target) => { if (!byName.has(name)) byName.set(name, []); byName.get(name).push(target); };
|
|
1773
|
+
for (const it of prog.items) if (it.name !== 'FILLER') add(it.name, it);
|
|
1774
|
+
for (const fd of prog.fds || []) add(fd.name, fd);
|
|
1775
|
+
const ancestors = (x) => { const a = []; let p = x.parent || x.fd; while (p) { a.push(p.name); p = p.parent || p.fd; } return a; };
|
|
1776
|
+
for (const it of prog.items) { it.directRefs = 0; it.receiving = false; it.receivingVia = new Set(); it.refVerbs = new Set(); }
|
|
1777
|
+
for (const fd of prog.fds || []) { fd.directRefs = 1; fd.receiving = false; }
|
|
1778
|
+
const receiving = prog.receivingTokens || new Map();
|
|
1779
|
+
const toks = prog.refs;
|
|
1780
|
+
const qualifiersOf = (i) => {
|
|
1781
|
+
const q = [];
|
|
1782
|
+
for (let k = i + 1; k + 1 < toks.length; k += 2) {
|
|
1783
|
+
const ofTok = toks[k].tok, qTok = toks[k + 1].tok;
|
|
1784
|
+
if (!(ofTok.u === 'OF' || ofTok.u === 'IN') || qTok.line !== ofTok.line && false) break;
|
|
1785
|
+
q.push(qTok.u);
|
|
1786
|
+
}
|
|
1787
|
+
return q;
|
|
1788
|
+
};
|
|
1789
|
+
for (let i = 0; i < toks.length; i++) {
|
|
1790
|
+
const { tok, zone } = toks[i];
|
|
1791
|
+
if (tok.u === 'OF' || tok.u === 'IN') continue;
|
|
1792
|
+
const prev = toks[i - 1];
|
|
1793
|
+
// What ADDRESS OF and LENGTH OF name is their operand, not a qualifier.
|
|
1794
|
+
if (prev && (prev.tok.u === 'OF' || prev.tok.u === 'IN') && !(prev.tok.u === 'OF' && ['ADDRESS', 'LENGTH'].includes(toks[i - 2]?.tok.u))) continue;
|
|
1795
|
+
if (prev && prev.tok.u === 'FUNCTION') continue;
|
|
1796
|
+
const cands = byName.get(tok.u);
|
|
1797
|
+
if (!cands) continue;
|
|
1798
|
+
const quals = qualifiersOf(i);
|
|
1799
|
+
let pick = cands;
|
|
1800
|
+
if (quals.length) pick = cands.filter(c => { const a = ancestors(c); let pos = 0; for (const q of quals) { const k = a.indexOf(q, pos); if (k < 0) return false; pos = k + 1; } return true; });
|
|
1801
|
+
if (!pick.length) pick = cands;
|
|
1802
|
+
const target = pick[0];
|
|
1803
|
+
target.directRefs++;
|
|
1804
|
+
prog.resolved.set(tok, target);
|
|
1805
|
+
if (target.kind && tok.verb === 'OPEN' && ['OUTPUT', 'EXTEND', 'I-O'].includes(tok.openMode)) target.receiving = true;
|
|
1806
|
+
if (receiving.has(tok)) { target.receiving = true; if (target.receivingVia) target.receivingVia.add(receiving.get(tok)); }
|
|
1807
|
+
if (target.refVerbs && tok.verb) target.refVerbs.add(tok.verb);
|
|
1808
|
+
// A qualifier is a reference only where the name needs it, being declared more than once, as
|
|
1809
|
+
// cobc counts it: U OF J with U unique leaves J referenced by its child alone.
|
|
1810
|
+
let cursor = target;
|
|
1811
|
+
for (const q of cands.length > 1 ? quals : []) {
|
|
1812
|
+
let p = cursor.parent || cursor.fd;
|
|
1813
|
+
while (p && p.name !== q) p = p.parent || p.fd;
|
|
1814
|
+
if (p) { p.directRefs++; cursor = p; }
|
|
1815
|
+
}
|
|
1816
|
+
void zone;
|
|
1817
|
+
}
|
|
1818
|
+
// MOVE CORRESPONDING writes each receiving item it pairs, and each is a reference of its own; the
|
|
1819
|
+
// sending items are read through their group.
|
|
1820
|
+
for (const st of prog.statements) {
|
|
1821
|
+
if (!st.corresponding) continue;
|
|
1822
|
+
const from = st.sources.map((x) => prog.resolved.get(x)).find(Boolean);
|
|
1823
|
+
for (const tok of from ? st.targets : []) {
|
|
1824
|
+
const to = prog.resolved.get(tok);
|
|
1825
|
+
if (!to || !to.children) continue;
|
|
1826
|
+
for (const [, item] of correspondingPairs(from, to)) {
|
|
1827
|
+
item.directRefs++;
|
|
1828
|
+
item.receiving = true;
|
|
1829
|
+
item.receivingVia.add('MOVE');
|
|
1830
|
+
item.refVerbs.add('MOVE');
|
|
1831
|
+
}
|
|
1832
|
+
}
|
|
1833
|
+
}
|
|
1834
|
+
for (const f of prog.files) {
|
|
1835
|
+
const fd = (prog.fds || []).find(x => x.name === f.name);
|
|
1836
|
+
// A file status field is written by the statements that write the file, so it receives
|
|
1837
|
+
// exactly when its file does.
|
|
1838
|
+
if (fd && fd.receiving) {
|
|
1839
|
+
for (let k = 0; k < f.envRefs.length; k++) {
|
|
1840
|
+
if (f.envRefs[k].u !== 'STATUS') continue;
|
|
1841
|
+
const target = f.envRefs.slice(k + 1).find(x => !['IS', 'ARE'].includes(x.u));
|
|
1842
|
+
const cand = target && byName.get(target.u);
|
|
1843
|
+
if (cand && cand[0] && cand[0].receivingVia) { cand[0].receiving = true; cand[0].receivingVia.add('FILE-STATUS'); }
|
|
1844
|
+
}
|
|
1845
|
+
}
|
|
1846
|
+
if (fd) { fd.select = f; fd.directRefs += 0; }
|
|
1847
|
+
for (const r of f.envRefs) { const c = byName.get(r.u); if (c && c[0] && !(prog.fds || []).includes(c[0])) c[0].directRefs++; }
|
|
1848
|
+
if (fd) fd.directRefs += 0;
|
|
1849
|
+
}
|
|
1850
|
+
const hasRefDesc = (x) => x.children.some(c => c.directRefs > 0 || hasRefDesc(c));
|
|
1851
|
+
for (const it of prog.items) {
|
|
1852
|
+
let refParent = false;
|
|
1853
|
+
// cobc carries a group's reference down to its items but not through an unnamed FILLER group.
|
|
1854
|
+
if (it.level !== 88) for (let p = it.parent; p && p.name !== 'FILLER'; p = p.parent) if (p.directRefs > 0) { refParent = true; break; }
|
|
1855
|
+
it.refState = it.directRefs > 0 ? 'refs' : refParent ? 'parent' : hasRefDesc(it) ? 'child' : 'none';
|
|
1856
|
+
}
|
|
1857
|
+
}
|
|
1858
|
+
|
|
1859
|
+
// Everything the ENVIRONMENT DIVISION holds counts as declared: SPECIAL-NAMES declares mnemonic,
|
|
1860
|
+
// alphabet, class, symbolic-character and switch-status names, REPOSITORY functions and classes,
|
|
1861
|
+
// and Micro Focus declares an ASSIGN operand no item declares as PIC X(255).
|
|
1862
|
+
function declaredNames(prog) {
|
|
1863
|
+
const names = new Set();
|
|
1864
|
+
for (const it of prog.items) { names.add(it.name); for (const n of it.indexNames || []) names.add(n); if (it.capacityName) names.add(it.capacityName); }
|
|
1865
|
+
for (const fd of prog.fds || []) names.add(fd.name);
|
|
1866
|
+
for (const rd of prog.reports || []) names.add(rd.name);
|
|
1867
|
+
for (const cd of prog.cds || []) names.add(cd);
|
|
1868
|
+
for (const f of prog.files) { if (f.name) names.add(f.name); for (const t of f.envRefs) names.add(t.u); }
|
|
1869
|
+
for (const l of prog.labels) names.add(l.name);
|
|
1870
|
+
for (const r of prog.refs) if (r.zone === 'env') names.add(r.tok.u);
|
|
1871
|
+
return names;
|
|
1872
|
+
}
|
|
1873
|
+
|
|
1874
|
+
// Skipped as resolveReferences skips a qualifier and a function's name; the condition inside
|
|
1875
|
+
// DFHRESP() or DFHVALUE() is the translator's to resolve.
|
|
1876
|
+
const NOT_A_NAME_AFTER = new Set(['OF', 'IN', 'FUNCTION', 'DFHRESP', 'DFHVALUE']);
|
|
1877
|
+
|
|
1878
|
+
// FD clauses are left out: a word after IS there is as often a recording mode as a name.
|
|
1879
|
+
// An EXEC SQL block's host variables are references to the program's data: :A.B names B qualified
|
|
1880
|
+
// by A, and the variables of a SELECT or FETCH INTO list, up to FROM, are written. An INSERT's INTO
|
|
1881
|
+
// names a table. The block keeps them, each by the token that resolves to its item.
|
|
1882
|
+
function hostVariableRefs(exec, refs, receiving) {
|
|
1883
|
+
const toks = exec.toks;
|
|
1884
|
+
const into = ['SELECT', 'FETCH'].includes(toks[0]?.u) ? toks.findIndex((x) => x.t === 'word' && x.u === 'INTO') : -1;
|
|
1885
|
+
const fromAt = into < 0 ? -1 : toks.findIndex((x, k) => k > into && x.t === 'word' && x.u === 'FROM');
|
|
1886
|
+
const intoEnd = fromAt < 0 ? toks.length : fromAt;
|
|
1887
|
+
exec.hostVariables = [];
|
|
1888
|
+
for (let k = 0; k + 1 < toks.length; k++) {
|
|
1889
|
+
if (!(toks[k].t === 'op' && toks[k].v === ':' && toks[k + 1].t === 'word')) continue;
|
|
1890
|
+
const path = [toks[k + 1]];
|
|
1891
|
+
for (let j = k + 2; j + 1 < toks.length && toks[j].t === 'period' && toks[j].joined && toks[j + 1].t === 'word'; j += 2) path.push(toks[j + 1]);
|
|
1892
|
+
const name = path[path.length - 1];
|
|
1893
|
+
name.verb = 'EXEC SQL';
|
|
1894
|
+
refs.push({ tok: name, zone: 'sql' });
|
|
1895
|
+
for (let q = path.length - 2; q >= 0; q--) refs.push({ tok: { t: 'word', u: 'OF', line: name.line, file: name.file }, zone: 'sql' }, { tok: path[q], zone: 'sql' });
|
|
1896
|
+
const written = into >= 0 && k > into && k < intoEnd;
|
|
1897
|
+
if (written) receiving.set(name, 'EXEC SQL');
|
|
1898
|
+
exec.hostVariables.push({ tok: name, written });
|
|
1899
|
+
}
|
|
1900
|
+
}
|
|
1901
|
+
|
|
1902
|
+
// An EXEC CICS command's option arguments are references to the program's data, written where the
|
|
1903
|
+
// option receives by IBM's direction for that command (lib/cics-commands.mjs); a literal, LENGTH OF,
|
|
1904
|
+
// FUNCTION or DFHRESP(...) is only sent. RECEIVE MAP without INTO writes the map's input record and
|
|
1905
|
+
// SEND MAP without FROM reads its output one, as the translator names them.
|
|
1906
|
+
function cicsArgumentRefs(exec, refs, receiving) {
|
|
1907
|
+
const toks = exec.toks;
|
|
1908
|
+
const options = [];
|
|
1909
|
+
for (let k = 0; k < toks.length; k++) {
|
|
1910
|
+
if (toks[k].t !== 'word') continue;
|
|
1911
|
+
if (!(toks[k + 1] && toks[k + 1].t === 'sep' && toks[k + 1].v === '(')) { options.push({ word: toks[k].u }); continue; }
|
|
1912
|
+
let depth = 0;
|
|
1913
|
+
let e = k + 1;
|
|
1914
|
+
for (; e < toks.length; e++) if (toks[e].t === 'sep' && (depth += toks[e].v === '(' ? 1 : -1) === 0) break;
|
|
1915
|
+
options.push({ word: toks[k].u, args: toks.slice(k + 2, e) });
|
|
1916
|
+
k = e;
|
|
1917
|
+
}
|
|
1918
|
+
const name = cicsCommand(options);
|
|
1919
|
+
const command = name && CICS_COMMANDS[name];
|
|
1920
|
+
for (const o of options.slice(1)) {
|
|
1921
|
+
if (!o.args || !o.args.length) continue;
|
|
1922
|
+
const direction = command ? (command.options[o.word] || CICS_EVERY_COMMAND[o.word] || command.anyOption) : null;
|
|
1923
|
+
const first = o.args[0];
|
|
1924
|
+
if (direction === 'label' || first.t !== 'word' || /^DFH(RESP|VALUE)$/.test(first.u) || first.u === 'FUNCTION') continue;
|
|
1925
|
+
const computed = first.u === 'LENGTH' && o.args[1] && o.args[1].u === 'OF';
|
|
1926
|
+
for (const x of o.args) if (x.t === 'word') { x.verb = 'EXEC CICS'; refs.push({ tok: x, zone: 'cics' }); }
|
|
1927
|
+
if (!computed && (direction === 'receives' || direction === 'both')) receiving.set(first, 'EXEC CICS');
|
|
1928
|
+
}
|
|
1929
|
+
const given = new Set(options.map((o) => o.word));
|
|
1930
|
+
for (const [option, d] of Object.entries(command?.defaults || {})) {
|
|
1931
|
+
const map = options.find((o) => o.word === d.name)?.args;
|
|
1932
|
+
if (!map || map.length !== 1 || map[0].t !== 'lit' || given.has(option) || d.unless.some((w) => given.has(w))) continue;
|
|
1933
|
+
const tok = { t: 'word', v: `${map[0].v}${d.suffix}`, u: `${map[0].v}${d.suffix}`.toUpperCase(), line: exec.line, file: exec.file, verb: 'EXEC CICS' };
|
|
1934
|
+
refs.push({ tok, zone: 'cics' });
|
|
1935
|
+
if (command.options[option] === 'receives') receiving.set(tok, 'EXEC CICS');
|
|
1936
|
+
}
|
|
1937
|
+
}
|
|
1938
|
+
|
|
1939
|
+
function unresolvedReferences(prog, declared, defines, sql) {
|
|
1940
|
+
const fromItems = new Set(prog.items.flatMap((it) => it.refsInData));
|
|
1941
|
+
const out = [];
|
|
1942
|
+
for (let i = 0; i < prog.refs.length; i++) {
|
|
1943
|
+
const { tok, zone } = prog.refs[i];
|
|
1944
|
+
if (zone === 'env' || zone === 'sql' || zone === 'cics' || (zone === 'data' && !fromItems.has(tok)) || prog.resolved.has(tok)) continue;
|
|
1945
|
+
const u = tok.u;
|
|
1946
|
+
const prev = prog.refs[i - 1];
|
|
1947
|
+
// A word without a letter is a number or a paragraph number, or bytes an encoding left behind.
|
|
1948
|
+
if (u === 'OF' || u === 'IN' || (prev && NOT_A_NAME_AFTER.has(prev.tok.u)) || !/[A-Z]/.test(u)) continue;
|
|
1949
|
+
if (declared.has(u) || defines.has(u) || PREDEFINED.has(u)) continue;
|
|
1950
|
+
if (RESERVED_WORDS.has(u) || SPECIAL_REGISTERS.has(u) || SYSTEM_NAMES.has(u) || INTRINSIC_FUNCTIONS.has(u)) continue;
|
|
1951
|
+
// A context-sensitive word, a compiler-directive word and an exception-condition name are not
|
|
1952
|
+
// reserved, so a program may use one as its own data name. This check cannot tell which construct
|
|
1953
|
+
// it is in, so it credits all three: naming a word the program never declared is the claim being
|
|
1954
|
+
// made here, and these are words the program did not have to declare.
|
|
1955
|
+
if (CONTEXT_SENSITIVE_WORDS.has(u) || DIRECTIVE_WORDS.has(u) || EXCEPTION_CONDITIONS.has(u)) continue;
|
|
1956
|
+
if (u.startsWith('DFH') || EIB_FIELDS.has(u) || DIB_FIELDS.has(u) || (sql && SQLCA_FIELDS.has(u))) continue;
|
|
1957
|
+
out.push({ name: u, file: tok.file, line: tok.line });
|
|
1958
|
+
}
|
|
1959
|
+
return out;
|
|
1960
|
+
}
|
|
1961
|
+
|
|
1962
|
+
// A >>DEFINE inside a copybook can decide a >>IF in the file that copies it, but copybooks are only
|
|
1963
|
+
// expanded after the file is normalized. Read the defines from every copybook the file names first.
|
|
1964
|
+
function collectCopybookDefines(src, format, ctx, depth) {
|
|
1965
|
+
if (depth > 8) return;
|
|
1966
|
+
const seen = ctx.defineScan || (ctx.defineScan = new Set());
|
|
1967
|
+
let current = format;
|
|
1968
|
+
for (const raw of src.split(/\r?\n/)) {
|
|
1969
|
+
const l = expandTabs(raw);
|
|
1970
|
+
const sw = /(?:^|\s)>>\s*SOURCE\s+(?:FORMAT\s+)?(?:IS\s+)?(FREE|FIXED|VARIABLE|TERMINAL)/i.exec(l);
|
|
1971
|
+
if (sw) { current = sw[1].toLowerCase(); continue; }
|
|
1972
|
+
if (current !== 'free' && current !== 'terminal' && l.length > 6 && '*/'.includes(l[6])) continue;
|
|
1973
|
+
const code = l.replace(/\*>.*$/, '');
|
|
1974
|
+
const m = /(?:^|[\s.])COPY\s+("[^"]+"|'[^']+'|[A-Za-z0-9_-]+)/i.exec(code);
|
|
1975
|
+
if (!m) continue;
|
|
1976
|
+
const name = m[1].replace(/^["']|["']$/g, '');
|
|
1977
|
+
const path = resolveCopy(name, null, ctx);
|
|
1978
|
+
if (!path || seen.has(path)) continue;
|
|
1979
|
+
seen.add(path);
|
|
1980
|
+
const text = readSource(path).text;
|
|
1981
|
+
const fmt = detectFormat(text) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : current;
|
|
1982
|
+
normalize(text, fmt, ctx.defines, ctx.std);
|
|
1983
|
+
collectCopybookDefines(text, fmt, ctx, depth + 1);
|
|
1984
|
+
}
|
|
1985
|
+
}
|
|
1986
|
+
|
|
1987
|
+
// A paragraph or section header in Area A of the procedure division ends the sentence before it,
|
|
1988
|
+
// period or not ("optional period used"): in variable form under any dialect, and in fixed form under
|
|
1989
|
+
// these, as cobc reads each. The period is supplied, so every reader of the tokens sees the sentence end.
|
|
1990
|
+
const OPTIONAL_PERIOD = new Set(['ibm', 'ibm-strict', 'mvs', 'mvs-strict', 'rm', 'rm-strict', 'bs2000', 'gcos', 'realia']);
|
|
1991
|
+
function supplyHeaderPeriods(tokens, fixedToo) {
|
|
1992
|
+
const out = [];
|
|
1993
|
+
let inProcedure = false;
|
|
1994
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
1995
|
+
const t = tokens[i];
|
|
1996
|
+
const next = tokens[i + 1];
|
|
1997
|
+
if (t.t === 'word' && next && next.u === 'DIVISION') inProcedure = t.u === 'PROCEDURE';
|
|
1998
|
+
const header = inProcedure && t.t === 'word' && t.areaA && (t.fmt === 'variable' || (fixedToo && t.fmt === 'fixed')) && next
|
|
1999
|
+
&& (next.t === 'period' || (next.t === 'word' && next.u === 'SECTION'))
|
|
2000
|
+
&& !VERBS.has(t.u) && !NOT_LABELS.has(t.u) && !SCOPE_TERMINATORS.has(t.u);
|
|
2001
|
+
if (header && out.length && out[out.length - 1].t !== 'period') out.push(tok('period', '.', t.line, t.file, { implied: true, fmt: t.fmt }));
|
|
2002
|
+
out.push(t);
|
|
2003
|
+
}
|
|
2004
|
+
return out;
|
|
2005
|
+
}
|
|
2006
|
+
|
|
2007
|
+
export function parseSource(src, file, opts = {}) {
|
|
2008
|
+
const format = opts.format && opts.format !== 'auto' ? opts.format : detectFormat(src);
|
|
2009
|
+
const scheme = BINARY_SIZE[opts.std || 'default'] || BINARY_SIZE.default;
|
|
2010
|
+
const ctx = { mainDir: opts.mainDir || dirname(file), includeDirs: opts.includeDirs || [], systemDirs: opts.systemDirs || [],
|
|
2011
|
+
fileIndex: opts.fileIndex || null, allowAbsoluteCopy: !!opts.allowAbsoluteCopy, cache: new Map(), copies: [], diags: [], copyFormat: opts.copyFormat || format, defines: new Map(), std: opts.std,
|
|
2012
|
+
inclusions: 0, copyTokens: 0, replacedGrowth: 0, replacedChars: 0 };
|
|
2013
|
+
collectCopybookDefines(src, format, ctx, 0);
|
|
2014
|
+
const norm = normalize(src, format, ctx.defines, opts.std);
|
|
2015
|
+
ctx.mainFormat = norm.finalFormat;
|
|
2016
|
+
const { tokens: raw, diags } = tokenize(norm, file);
|
|
2017
|
+
ctx.diags.push(...norm.diags.map(d => ({ ...d, file })), ...diags);
|
|
2018
|
+
const expanded = stripDirecting(applyReplaceStatements(expand(raw, ctx, [resolve(file)], norm.finalFormat), ctx));
|
|
2019
|
+
const tokens = supplyHeaderPeriods(expanded, OPTIONAL_PERIOD.has(opts.std));
|
|
2020
|
+
const starts = [];
|
|
2021
|
+
for (let i = 0; i < tokens.length - 1; i++) {
|
|
2022
|
+
const t = tokens[i];
|
|
2023
|
+
if (t.t === 'word' && (t.u === 'IDENTIFICATION' || t.u === 'ID') && tokens[i + 1].u === 'DIVISION') starts.push(i);
|
|
2024
|
+
else if (t.t === 'word' && (t.u === 'PROGRAM-ID' || t.u === 'FUNCTION-ID') && !(starts.length && i - starts[starts.length - 1] <= 3)) {
|
|
2025
|
+
const prevStart = starts[starts.length - 1];
|
|
2026
|
+
const hasProc = prevStart !== undefined && tokens.slice(prevStart, i).some((x, k, arr) => x.u === 'PROCEDURE' && arr[k + 1] && arr[k + 1].u === 'DIVISION');
|
|
2027
|
+
if (prevStart === undefined || hasProc) starts.push(i);
|
|
2028
|
+
}
|
|
2029
|
+
}
|
|
2030
|
+
// Micro Focus lets a program open at its environment or data division with no header at all, and
|
|
2031
|
+
// the compiler reads it as one program named for its file. A copybook has no procedure division.
|
|
2032
|
+
let unnamed = null;
|
|
2033
|
+
if (!starts.length && tokens.some((t, i) => t.u === 'PROCEDURE' && tokens[i + 1] && tokens[i + 1].u === 'DIVISION')) {
|
|
2034
|
+
starts.push(0);
|
|
2035
|
+
unnamed = basename(file).replace(/\.[^.]*$/, '').toUpperCase();
|
|
2036
|
+
}
|
|
2037
|
+
const programs = [];
|
|
2038
|
+
for (let k = 0; k < starts.length; k++) {
|
|
2039
|
+
let end = k + 1 < starts.length ? starts[k + 1] : tokens.length;
|
|
2040
|
+
for (let i = starts[k]; i < end - 1; i++) if (tokens[i].u === 'END' && (tokens[i + 1].u === 'PROGRAM' || tokens[i + 1].u === 'FUNCTION')) { end = i; break; }
|
|
2041
|
+
const prog = parseProgram(tokens, starts[k], end, scheme, ctx.defines, opts.hostVariables !== false);
|
|
2042
|
+
if (!prog.id && unnamed) prog.id = unnamed;
|
|
2043
|
+
programs.push(prog);
|
|
2044
|
+
}
|
|
2045
|
+
// A nested program sees what its container declares GLOBAL, and the parse does not record which
|
|
2046
|
+
// those are, so a name any program in the source declares counts for all of them.
|
|
2047
|
+
const declared = new Set(programs.flatMap((p) => [...declaredNames(p)]));
|
|
2048
|
+
const sqlca = ctx.copies.some((c) => /^SQLCA$/i.test(c.name));
|
|
2049
|
+
for (const p of programs) p.unresolvedRefs = unresolvedReferences(p, declared, ctx.defines, sqlca || p.execs.some((e) => e.kind === 'SQL'));
|
|
2050
|
+
return { file, format, finalFormat: norm.finalFormat, options: norm.options, programs, copies: ctx.copies, diags: ctx.diags, tokenCount: tokens.length };
|
|
2051
|
+
}
|
|
2052
|
+
|
|
2053
|
+
export function parseFile(file, opts = {}) {
|
|
2054
|
+
return parseSource(readSource(file).text, file, opts);
|
|
2055
|
+
}
|