@portll/cobolwork 0.2.150 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,12 +6,13 @@
6
6
  // running system confirms one, so a verdict says confirmed only where the estate brings the result
7
7
  // of its own test as a witness feed. See docs/spec/reach.md §9.
8
8
  import { readFileSync, realpathSync } from 'node:fs';
9
- import { basename, sep, delimiter } from 'node:path';
9
+ import { basename, dirname, join, resolve, sep, delimiter } from 'node:path';
10
10
  import { ALL_RULES } from './kernel/registry.mjs';
11
11
  import { EXPLOITABILITY } from './kernel/findings.mjs';
12
12
  import { kindsOf } from './consequence.mjs';
13
13
  import { WHO } from './sets/flow.mjs';
14
14
  import { entryName, accessOfEntry, accessFromEntry, privilegedFromEntry } from './reach.mjs';
15
+ import { verifyEvidence } from './evidence/verify.mjs';
15
16
 
16
17
  export const HANDLING = 'This report joins routes an attacker can drive with the estate\'s own access facts. It holds no source '
17
18
  + 'text, but it is a list of what to attack first: keep it where the access records it was built from may go, and do not '
@@ -62,9 +63,22 @@ const WITHOUT_A_TEST = {
62
63
  // response code is not input, so no test on it clears a route; the fix is not to send it.
63
64
  function fixAt(f, kinds) {
64
65
  const last = f.trace?.length ? f.trace[f.trace.length - 1] : null;
66
+ const instead = WITHOUT_A_TEST[kinds.sink];
67
+ // A value a STRING built holds the input as one part among literals, where no allow-list or class
68
+ // test describes the whole. The test belongs on the field folded in, before the STRING.
69
+ const built = f.trace ? f.trace.findLastIndex((h) => /^STRING at /.test(h.via || '')) : -1;
70
+ const part = built > 0 ? f.trace[built - 1] : null;
71
+ const where = part?.item && /\bat (.+):(\d+)$/.exec(f.trace[built].via);
72
+ if (where && kinds.source !== 'system-response' && TESTS[kinds.sink]) {
73
+ const [, path, line] = where;
74
+ return {
75
+ program: f.trace[built].program || f.program || null, path, line: Number(line), item: part.item,
76
+ builtInto: { item: last.item, path: f.path, line: f.line },
77
+ test: `before the STRING at ${path}:${line} that builds ${last.item}, on every route to it, test ${TESTS[kinds.sink](part.item)}, and let only a value that passes reach it${instead ? `; ${instead} needs no test` : ''}`,
78
+ };
79
+ }
65
80
  const item = last?.item || 'the value';
66
81
  const test = kinds.source === 'system-response' ? null : TESTS[kinds.sink]?.(item) || null;
67
- const instead = WITHOUT_A_TEST[kinds.sink];
68
82
  return {
69
83
  program: f.program || null, path: f.path, line: f.line, item,
70
84
  ...(test ? { test: `before the operation at ${f.path}:${f.line}, on every route to it, test ${test}, and let only a value that passes reach it${instead ? `; ${instead} needs no test` : ''}` }
@@ -156,6 +170,28 @@ const OUTCOMES = new Set(['reproduced', 'not-reproduced']);
156
170
  const str = (v) => typeof v === 'string' && v.length > 0;
157
171
  const isoDate = (v) => typeof v === 'string' && /^\d{4}-\d{2}-\d{2}$/.test(v);
158
172
 
173
+ // What an ironwork run a result cites shows: each operation the marker reached and the abend the run
174
+ // ended with, from a journal in an evidence directory that verifies. A string says why it shows nothing.
175
+ function witnessedRun(ev, base, verdicts) {
176
+ if (!ev || typeof ev !== 'object' || !str(ev.dir) || !str(ev.run)) return 'cites evidence without a directory and a run';
177
+ if (!/^[0-9TZ]+-[0-9a-f]{16}$/.test(ev.run)) return `cites run ${ev.run}, which is not a run id`;
178
+ const dir = resolve(base, ev.dir);
179
+ if (!verdicts.has(dir)) verdicts.set(dir, verifyEvidence(dir));
180
+ const v = verdicts.get(dir);
181
+ if (!v.verified) return `cites evidence in ${ev.dir} that does not verify`;
182
+ if (v.unrecorded.includes(ev.run)) return `cites run ${ev.run}, which the ledger in ${ev.dir} does not record`;
183
+ let text;
184
+ try { text = readFileSync(join(dir, 'runs', `${ev.run}.jsonl`), 'utf8'); } catch { return `cites run ${ev.run}, which ${ev.dir} does not hold`; }
185
+ const records = [];
186
+ for (const line of text.split('\n')) {
187
+ if (!line) continue;
188
+ const rec = JSON.parse(line);
189
+ if (rec.kind === 'sink' && rec.reached === true) records.push({ kind: 'sink', sink: rec.sink, file: rec.file, line: rec.line });
190
+ else if (rec.kind === 'abend' && str(rec.file) && Number.isInteger(rec.line)) records.push({ kind: 'abend', code: rec.code, file: rec.file, line: rec.line });
191
+ }
192
+ return records.length ? { run: ev.run, records } : `cites run ${ev.run}, whose journal shows no operation the marker reached and no abend at a line`;
193
+ }
194
+
159
195
  // The results of the estate's own tests, keyed by the fingerprint of the finding each tried, loaded
160
196
  // for one scan and refused from inside the scanned tree the way a reach feed is: a list of what was
161
197
  // reproduced is a list of what to attack. Shape:
@@ -176,6 +212,7 @@ export function loadWitnessFeed(path, { root = null } = {}) {
176
212
  if (!doc.results || typeof doc.results !== 'object' || Array.isArray(doc.results)) { feed.problem = 'holds no results'; return feed; }
177
213
  feed.witness = doc.witness;
178
214
  feed.recorded = doc.recorded;
215
+ const verdicts = new Map();
179
216
  for (const [fingerprint, r] of Object.entries(doc.results)) {
180
217
  const why = !r || typeof r !== 'object' ? 'is not an object'
181
218
  : !OUTCOMES.has(r.outcome) ? `says outcome ${JSON.stringify(r.outcome)}, not "reproduced" or "not-reproduced"`
@@ -183,7 +220,9 @@ export function loadWitnessFeed(path, { root = null } = {}) {
183
220
  : !isoDate(r.on) ? 'does not say when the test ran'
184
221
  : !str(r.system) ? 'does not say which system it ran on' : null;
185
222
  if (why) { feed.refused.push({ fingerprint, why: `the result for ${fingerprint} ${why}` }); continue; }
186
- feed.results.set(fingerprint, { outcome: r.outcome, by: r.by, on: r.on, system: r.system, ...(str(r.reference) ? { reference: r.reference } : {}) });
223
+ const evidence = r.evidence === undefined ? null : witnessedRun(r.evidence, dirname(real), verdicts);
224
+ if (typeof evidence === 'string') { feed.refused.push({ fingerprint, why: `the result for ${fingerprint} ${evidence}` }); continue; }
225
+ feed.results.set(fingerprint, { outcome: r.outcome, by: r.by, on: r.on, system: r.system, ...(str(r.reference) ? { reference: r.reference } : {}), ...(evidence ? { evidence } : {}) });
187
226
  }
188
227
  return feed;
189
228
  }
@@ -214,8 +253,11 @@ export function applyWitness(findings, checked, feeds) {
214
253
  const x = f.exploitability;
215
254
  const hit = x && f.fingerprint ? newest(good, f.fingerprint) : null;
216
255
  if (!hit) return false;
217
- matched.add(f.fingerprint);
218
256
  const { r, feed } = hit;
257
+ // A cited run witnesses this finding only where its journal places the operation or the abend at the finding's line.
258
+ const at = r.evidence ? r.evidence.records.find((e) => e.line === f.line && basename(e.file) === basename(f.path)) : null;
259
+ if (r.evidence && !at) return false;
260
+ matched.add(f.fingerprint);
219
261
  const cite = `${feed.file} (${feed.witness}, recorded ${feed.recorded})`;
220
262
  x.witness = { ...r, from: cite };
221
263
  byWitness[r.outcome]++;
@@ -227,6 +269,9 @@ export function applyWitness(findings, checked, feeds) {
227
269
  const was = x.verdict;
228
270
  x.verdict = 'confirmed';
229
271
  x.because.push(`reproduced on ${r.system} by ${r.by} on ${r.on}${r.reference ? ` (${r.reference})` : ''}, per ${cite}`);
272
+ if (at) x.because.push(at.kind === 'sink'
273
+ ? `run ${r.evidence.run}'s verified journal records the marker reaching the ${at.sink} operation at ${at.file}:${at.line}`
274
+ : `run ${r.evidence.run}'s verified journal records the run ending with ${at.code} at ${at.file}:${at.line}`);
230
275
  delete x.unknown;
231
276
  if (!x.fixAt) x.fixAt = fixAt(f, kindsOf(f.rule));
232
277
  if (f.guardedFrom) f.sev = f.guardedFrom;
package/lib/gate.mjs CHANGED
@@ -20,7 +20,9 @@ const MAX_EDIT = 2000;
20
20
  const MAX_LINES = 200000;
21
21
  const MAX_COMPILES = 50;
22
22
  const COMPILE_BUDGET_MS = 10 * 60 * 1000;
23
- const PASSING = new Set(['cleared-by-check', 'statement-removed', 'source-removed', 'no-longer-fires']);
23
+ // A fix keeps the program's statements and stops the route: deleting the flagged statement or the
24
+ // source statement removes what the program did, and fails.
25
+ const PASSING = new Set(['cleared-by-check', 'no-longer-fires']);
24
26
  const UNDECIDED = new Set(['lowered-by-check', 'gone-unexplained']);
25
27
 
26
28
  const pathText = (p) => printable(String(p ?? ''), 120);
@@ -115,15 +117,18 @@ export function findCobc({ explicit = null, repo, env = process.env, platform =
115
117
  return found.path ? found : { path: null, why: `--cobc ${pathText(resolve(explicit))} is not a file outside the repository` };
116
118
  }
117
119
 
118
- // A compiler run as `cobc -fsyntax-only`, from outside the tree, for a bounded time.
120
+ // A compiler run as `cobc -fsyntax-only`, from outside the tree, for a bounded time. Its error lines
121
+ // are kept so a drafter whose patch stopped a program compiling is told where.
119
122
  export function cobcCompiler(cobc) {
120
123
  return (file, { includeDirs = [], free = false } = {}) => {
121
124
  const r = spawnSync(cobc, ['-fsyntax-only', ...(free ? ['-free'] : []), ...includeDirs.flatMap((d) => ['-I', d]), file],
122
- { cwd: tmpdir(), timeout: 60000, windowsHide: true, stdio: 'ignore' });
123
- return r.status === 0;
125
+ { cwd: tmpdir(), timeout: 60000, windowsHide: true, stdio: ['ignore', 'ignore', 'pipe'], encoding: 'latin1', maxBuffer: 1 << 20 });
126
+ return { ok: r.status === 0, messages: String(r.stderr || '').split(/\r?\n/).filter((l) => /: error: /.test(l)) };
124
127
  };
125
128
  }
126
129
 
130
+ const MAX_COMPILER_LINES = 3;
131
+
127
132
  // Level 1 matches across both lists first; each looser level pairs only what the level above left.
128
133
  export function pairLists(base, head, baseKeys, headKeys) {
129
134
  const headOf = new Array(base.length).fill(-1);
@@ -222,6 +227,10 @@ function targetReason(t, r) {
222
227
  return `the target is still reported, lowered from ${t.sev} to ${r.head.sev} by a check this patch added at ${where({ path: r.guard.file, line: r.guard.line })}; a check that lowers rather than stops may not turn away everything it should, so a person decides`;
223
228
  case 'program-removed':
224
229
  return `the patch removes ${pathText(t.path)}${t.program ? ` or its program ${printable(t.program, 40)}` : ''}, which held the target; removing the program is not a fix the gate accepts`;
230
+ case 'statement-removed':
231
+ return `the patch deletes the target's statement at ${where(t)}; a fix keeps the statement and stops the route to it, so deleting it is not a fix the gate accepts`;
232
+ case 'source-removed':
233
+ return `the patch deletes the statement the target's input comes from; a fix keeps it and stops the route from it, so deleting it is not a fix the gate accepts`;
225
234
  case 'gone-unexplained':
226
235
  return `the target, ${at}, is gone, but the patch neither removed its statement or its source nor added a check that stops its route; the route may be cut between its ends, or sent through something the engine does not follow, and a person decides`;
227
236
  default:
@@ -246,16 +255,29 @@ function compileCheck({ compiler, baseRoot, headRoot, baseProgs, headProgs, chan
246
255
  try { free = detectFormat(readFileSync(join(root, file), 'latin1')) === 'free'; } catch { /* unreadable: compiled as fixed */ }
247
256
  return { includeDirs: [...dirs].sort(), free };
248
257
  };
258
+ // A compiler answers true or false, or { ok, messages }.
259
+ const compile = (root, progs, file) => {
260
+ const r = compiler.run(join(root, file), optsFor(root, progs, file));
261
+ return r && typeof r === 'object' ? { ok: !!r.ok, messages: r.messages || [] } : { ok: !!r, messages: [] };
262
+ };
249
263
  const regressed = [], neverCompiled = [];
250
264
  const deadline = Date.now() + (compiler.budgetMs ?? COMPILE_BUDGET_MS);
251
265
  for (const file of [...files].sort()) {
252
266
  if (Date.now() > deadline) return { ok: null, compiled: `not compiled: the compiler ran past the gate's ${Math.round((compiler.budgetMs ?? COMPILE_BUDGET_MS) / 60000)}-minute budget`, reasons: [] };
253
- const now = compiler.run(join(headRoot, file), optsFor(headRoot, headProgs, file));
254
- if (now) continue;
255
- const had = existsSync(join(baseRoot, file)) && compiler.run(join(baseRoot, file), optsFor(baseRoot, baseProgs, file));
256
- (had ? regressed : neverCompiled).push(file);
267
+ const now = compile(headRoot, headProgs, file);
268
+ if (now.ok) continue;
269
+ const had = existsSync(join(baseRoot, file)) && compile(baseRoot, baseProgs, file).ok;
270
+ if (had) regressed.push({ file, messages: now.messages });
271
+ else neverCompiled.push(file);
272
+ }
273
+ // The compiler names the file by where the head was written; the reader knows it by its path.
274
+ const asInRepo = (m) => m.split(join(headRoot, sep)).join('');
275
+ if (regressed.length) {
276
+ return { ok: false, compiled: `${regressed.length} program(s) compiled before the patch and do not after`, reasons: regressed.flatMap(({ file, messages }) => [
277
+ `${pathText(file)} compiled before the patch and does not after`,
278
+ ...messages.slice(0, MAX_COMPILER_LINES).map((m) => `the compiler: ${asInRepo(m).trim()}`),
279
+ ]) };
257
280
  }
258
- if (regressed.length) return { ok: false, compiled: `${regressed.length} program(s) compiled before the patch and do not after`, reasons: regressed.map((f) => `${pathText(f)} compiled before the patch and does not after`) };
259
281
  if (neverCompiled.length) return { ok: null, compiled: `${neverCompiled.length} program(s) do not compile before or after the patch, or are new and do not compile, which the gate cannot tell from a missing copy library`, reasons: [] };
260
282
  return { ok: true, compiled: `${files.size} program(s) compile`, reasons: [] };
261
283
  }
@@ -38,6 +38,7 @@ export const EVIDENCE = Object.freeze({
38
38
  advisory: 'a component version matches a published advisory against it',
39
39
  exposure: 'information about the estate is written into the source',
40
40
  change: 'a change moves an interface or adds a call target, which a reviewer has to look at',
41
+ execution: 'a run of the program on a recorded input ended this way, and the run\'s verified journal is the record of it',
41
42
  coverage: 'the analysis stopped following here; a limit of the tool, not a defect in the code',
42
43
  context: 'describes the estate - an entry point, a product in use - and asserts no defect',
43
44
  });
@@ -60,6 +61,7 @@ export const WHO_ACTS = Object.freeze({
60
61
  advisory: 'whoever owns the build',
61
62
  exposure: 'the owner',
62
63
  change: 'the reviewer of that change',
64
+ execution: "the program's owner",
63
65
  coverage: "nobody's code: read more, or accept the limit",
64
66
  context: 'nobody: it asserts no defect',
65
67
  });
@@ -27,6 +27,7 @@ import { scanLog, LOG_RULES } from '../sets/log.mjs';
27
27
  import { scanCompile, COMPILE_RULES } from '../sets/compile.mjs';
28
28
  import { scanSemantics, SEMANTICS_RULES } from '../sets/semantics.mjs';
29
29
  import { scanZowe, ZOWE_RULES } from '../sets/zowe.mjs';
30
+ import { scanAbend, ABEND_RULES } from '../sets/abend.mjs';
30
31
  import { toolName } from './ruleset.mjs';
31
32
 
32
33
  // The order is the order a scan runs them and the order --only lists them. It is not significant
@@ -47,6 +48,7 @@ export const REGISTRY = [
47
48
  { name: 'log', scan: scanLog, rules: LOG_RULES },
48
49
  { name: 'semantics', scan: scanSemantics, rules: SEMANTICS_RULES },
49
50
  { name: 'zowe', scan: scanZowe, rules: ZOWE_RULES },
51
+ { name: 'abend', scan: scanAbend, rules: ABEND_RULES },
50
52
  ];
51
53
 
52
54
  // Re-exported from the kernel's ruleset module, which is where it has to live: this file imports
@@ -154,6 +154,7 @@ function heldTree({ kind, root, store, rel, systemDirs = [], copyDirs = null, cl
154
154
  // Buffer. Paths are used exactly as given, so a test reads the way it writes.
155
155
  export function memoryTree(files, { root = '/memory', systemDirs = [] } = {}) {
156
156
  const store = new Map(Object.entries(files).map(([p, v]) => [p, Buffer.isBuffer(v) ? v : Buffer.from(v, 'latin1')]));
157
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- root is escaped
157
158
  const rel = (p) => String(p).replace(/\\/g, '/').replace(new RegExp('^' + root.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') + '/?'), '');
158
159
  return heldTree({ kind: 'memory', root, store, rel, systemDirs });
159
160
  }
package/lib/options.mjs CHANGED
@@ -433,7 +433,11 @@ export function compilerTasks(text) {
433
433
  };
434
434
  const valueOf = (v) => (v && typeof v === 'object' && !Array.isArray(v) ? v.value : v);
435
435
  const folder = (v) => String(v).replace(/\$\{(?:workspaceFolder|workspaceRoot)(?::[^}]*)?\}/g, '.').replace(/\$\{(?:\/|pathSeparator)\}/g, '/');
436
- const quoted = (a) => { const s = String(valueOf(a) ?? ''); return /[\s"]/.test(s) || s === '' ? `"${s.replace(/"/g, '\\"')}"` : s; };
436
+ const quoted = (a, dialect) => {
437
+ const s = String(valueOf(a) ?? '');
438
+ const esc = ESCAPE[dialect];
439
+ return /[\s"]/.test(s) || s === '' ? `"${s.replace(/./gs, (c) => (c === '"' || c === esc ? esc + c : c))}"` : s;
440
+ };
437
441
  const out = [];
438
442
  const tasks = [doc, ...(Array.isArray(doc && doc.tasks) ? doc.tasks : [])];
439
443
  for (const task of tasks) {
@@ -448,7 +452,7 @@ export function compilerTasks(text) {
448
452
  const line = lineOf(JSON.stringify(command).slice(1, 41), task.label && JSON.stringify(task.label));
449
453
  const state = { dir };
450
454
  const found = [];
451
- commandLine(folder([command, ...args.map(quoted)].join(' ')), dialect, state, line, found);
455
+ commandLine(folder([command, ...args.map((a) => quoted(a, dialect))].join(' ')), dialect, state, line, found);
452
456
  out.push(...found);
453
457
  }
454
458
  }
package/lib/packs.mjs CHANGED
@@ -113,8 +113,10 @@ export function packProblems(pack) {
113
113
  const scopes = r.appliesTo || ['jcl-instream'];
114
114
  for (const sc of scopes) if (!SCOPES.includes(sc)) problems.push(`${pack.name}:${r.id}: unknown scope '${sc}'`);
115
115
  if (!scopes.length) problems.push(`${pack.name}:${r.id}: no scope, so it can never match anything`);
116
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- a shipped pack's pattern, checked here to compile
116
117
  try { new RegExp(r.pattern, r.flags || 'i'); } catch (e) { problems.push(`${pack.name}:${r.id}: ${e.message}`); }
117
118
  if (r.setsContext && !r.setsContextPattern) problems.push(`${pack.name}:${r.id}: setsContext without a pattern that sets it`);
119
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- a shipped pack's pattern, checked here to compile
118
120
  if (r.setsContextPattern) { try { new RegExp(r.setsContextPattern, 'i'); } catch (e) { problems.push(`${pack.name}:${r.id}: setsContextPattern ${e.message}`); } }
119
121
  // A rule that waits for a context nothing in the pack sets can never fire, which is the same
120
122
  // failure as an unknown scope and is caught the same way.
@@ -139,10 +141,12 @@ export function compilePack(pack) {
139
141
  pack: pack.name,
140
142
  vendor: pack.vendor,
141
143
  product: pack.product,
144
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- a shipped pack's pattern; test/hostile times each one
142
145
  re: new RegExp(r.pattern, r.flags || 'i'),
143
146
  appliesTo: r.appliesTo || ['jcl-instream'],
144
147
  // A rule may declare that it only means what it says once an earlier line has put the stream
145
148
  // into a mode, and a rule may be the thing that sets that mode.
149
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- a shipped pack's pattern; test/hostile times each one
146
150
  setsContextRe: r.setsContext && r.setsContextPattern ? new RegExp(r.setsContextPattern, 'i') : null,
147
151
  }));
148
152
  }
package/lib/parser.mjs CHANGED
@@ -8,7 +8,8 @@ import {
8
8
  EIB_FIELDS, DIB_FIELDS, SQLCA_FIELDS,
9
9
  } from './words.mjs';
10
10
  import { CICS_COMMANDS, CICS_EVERY_COMMAND } from './cics-commands.mjs';
11
- import { cicsCommand } from './precompile-cics.mjs';
11
+ import { cicsCommand, symbolicMapCopybook } from './precompile-cics.mjs';
12
+ import { parseBms } from './bms.mjs';
12
13
  import { BINARY_SIZE, literalBytes, computeSizes, layoutReport } from './layout.mjs';
13
14
  import { hostVariablesIn } from './embedded-sql.mjs';
14
15
  import { join, dirname, basename, resolve, isAbsolute, sep, delimiter } from 'node:path';
@@ -88,7 +89,7 @@ export function detectFormat(src) {
88
89
  function codePastColumn72(src) {
89
90
  const lines = src.split(/\r?\n/, 2001);
90
91
  for (let i = 0; i < Math.min(lines.length, 2000); i++) {
91
- const l = expandTabs(lines[i]).replace(/\s+$/, '');
92
+ const l = expandTabs(lines[i]).trimEnd();
92
93
  if (l.length <= 72 || l[6] === '*' || l[6] === '/' || /^\s*\*>/.test(l) || COMMENT_ENTRY.test(l.slice(7))) continue;
93
94
  const { quote, comment } = stateAtColumn72(l);
94
95
  if (comment) continue;
@@ -138,9 +139,42 @@ function scanQuotes(text, open) {
138
139
 
139
140
  const PREDEFINED = new Map([['P64', 'SET']]);
140
141
 
142
+ // Cuts a floating comment where /\*>.*$/ would: at the first *> after the last line break.
143
+ function withoutInlineComment(s) {
144
+ let from = 0;
145
+ for (const b of ['\n', '\r', '\u2028', '\u2029']) from = Math.max(from, s.lastIndexOf(b) + 1);
146
+ const at = s.indexOf('*>', from);
147
+ return at < 0 ? s : s.slice(0, at);
148
+ }
149
+
150
+ // The name, value and OVERRIDE of a >>DEFINE body as /^(?:CONSTANT\s+)?([A-Za-z0-9_-]+)\s+(?:AS\s+)?(.*?)\s*(OVERRIDE)?\s*$/i
151
+ // reads them, in time linear in the body.
152
+ function defineParts(body) {
153
+ const space = (c) => c !== undefined && /\s/.test(c);
154
+ const skip = (i) => { while (space(body[i])) i++; return i; };
155
+ const from = (start) => {
156
+ let j = start;
157
+ while (j < body.length && /[A-Za-z0-9_-]/.test(body[j])) j++;
158
+ if (j === start || !space(body[j])) return null;
159
+ let v = skip(j);
160
+ if (/^AS$/i.test(body.slice(v, v + 2)) && space(body[v + 2])) v = skip(v + 2);
161
+ let end = body.length;
162
+ while (end > v && space(body[end - 1])) end--;
163
+ let override;
164
+ if (end - 8 >= v && /^OVERRIDE$/i.test(body.slice(end - 8, end))) {
165
+ override = body.slice(end - 8, end);
166
+ end -= 8;
167
+ while (end > v && space(body[end - 1])) end--;
168
+ }
169
+ const value = body.slice(v, end);
170
+ return /[\n\r\u2028\u2029]/.test(value) ? null : [body, body.slice(start, j), value, override];
171
+ };
172
+ return (/^CONSTANT$/i.test(body.slice(0, 8)) && space(body[8]) && from(skip(8))) || from(0);
173
+ }
174
+
141
175
  function defineDirective(directive, defines) {
142
- const body = directive.replace(/\*>.*$/, '').replace(/^>>\s*DEFINE\s+/i, '').trim();
143
- const m = body.match(/^(?:CONSTANT\s+)?([A-Za-z0-9_-]+)\s+(?:AS\s+)?(.*?)\s*(OVERRIDE)?\s*$/i);
176
+ const body = withoutInlineComment(directive).replace(/^>>\s*DEFINE\s+/i, '').trim();
177
+ const m = defineParts(body);
144
178
  if (!m) return;
145
179
  const name = m[1].toUpperCase();
146
180
  const raw = m[2].trim();
@@ -153,17 +187,17 @@ function defineDirective(directive, defines) {
153
187
  function setDirective(text, defines) {
154
188
  const c = /\bCONSTANT\s+([A-Za-z0-9_-]+)\s+(?:(["'])(.*?)\2|(\S+))/i.exec(text);
155
189
  if (c && defines) defines.set(c[1].toUpperCase(), c[3] !== undefined ? c[3] : c[4]);
156
- const m = text.match(/SOURCEFORMAT\s*\(?\s*["']?(FREE|FIXED|VARIABLE)/i);
190
+ const m = text.match(/SOURCEFORMAT\s*(?:\(\s*)?["']?(FREE|FIXED|VARIABLE)/i);
157
191
  return m ? m[1].toLowerCase() : null;
158
192
  }
159
193
 
160
194
  function evaluateCondition(text, defines) {
161
- const t = text.replace(/\*>.*$/, '').trim();
195
+ const t = withoutInlineComment(text).trim();
162
196
  const known = (n) => (defines && defines.has(n)) || PREDEFINED.has(n);
163
197
  const valueOf = (n) => (defines && defines.has(n) ? defines.get(n) : PREDEFINED.get(n));
164
198
  let m = t.match(/^([A-Za-z0-9_-]+)\s+(?:IS\s+)?(NOT\s+)?(DEFINED|SET)$/i);
165
199
  if (m) { const r = known(m[1].toUpperCase()); return m[2] ? !r : r; }
166
- m = t.match(/^([A-Za-z0-9_-]+)\s*(<=|>=|<>|=|<|>)\s*(['"]?)([^'"]*)\3$/);
200
+ m = t.match(/^([A-Za-z0-9_-]+)\s*(<=|>=|<>|=|<|>)\s*(?!\s)(['"]?)([^'"]*)\3$/);
167
201
  if (m) {
168
202
  const name = m[1].toUpperCase();
169
203
  if (!known(name)) return null;
@@ -446,11 +480,18 @@ export function tokenize(norm, file) {
446
480
  return { tokens: out, diags };
447
481
  }
448
482
 
449
- // Every rule set walks the tree through here. A directory it may not read is recorded, not treated
450
- // as empty: only ENOENT means absent. Symlinks are followed while they stay inside the tree, so a
451
- // symlinked copy library is read; one pointing outside is counted and never followed, because a
452
- // scan reads the tree it was given and nothing else.
453
- export function buildFileIndex(root) {
483
+ // A drive that stalls or drops for a moment answers a listing with ENOENT or an I/O error. Such a
484
+ // listing is tried again after a pause; the pauses add up to about 1.3 seconds.
485
+ const LISTING_RETRIED = new Set(['ENOENT', 'EIO', 'ETIMEDOUT', 'ENXIO', 'EBUSY', 'EAGAIN', 'EINTR', 'ESTALE', 'ENOTCONN']);
486
+ const LISTING_PAUSES_MS = [50, 250, 1000];
487
+ const pause = (ms) => Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
488
+
489
+ // Every rule set walks the tree through here. A directory that cannot be listed is recorded, not
490
+ // treated as empty: a scan over it is incomplete, not a scan of a smaller tree. Only the root's own
491
+ // ENOENT means absent; a directory the walk found in its parent was there. Symlinks are followed
492
+ // while they stay inside the tree, so a symlinked copy library is read; one pointing outside is
493
+ // counted and never followed, because a scan reads the tree it was given and nothing else.
494
+ export function buildFileIndex(root, { readdir = readdirSync, wait = pause } = {}) {
454
495
  const index = new Map();
455
496
  const dirs = new Set();
456
497
  const unreadableDirs = [];
@@ -460,16 +501,20 @@ export function buildFileIndex(root) {
460
501
  const inside = (p) => p === top || p.startsWith(top + sep);
461
502
  const visited = new Set();
462
503
  const addFile = (p, name, d) => { index.set(p.toLowerCase(), p); if (/\.(cpy|copy|inc|cbl|cob)$/i.test(name)) dirs.add(d); };
504
+ const list = (d) => {
505
+ for (let i = 0; ; i++) {
506
+ try { return readdir(d, { withFileTypes: true }); } catch (e) {
507
+ if (i >= LISTING_PAUSES_MS.length || !LISTING_RETRIED.has(e.code)) throw e;
508
+ wait(LISTING_PAUSES_MS[i]);
509
+ }
510
+ }
511
+ };
463
512
  // Each real directory is walked once, however many links lead to it, or its programs count twice.
464
513
  const walk = (d, real) => {
465
514
  if (visited.has(real)) return;
466
515
  visited.add(real);
467
516
  let es;
468
- try { es = readdirSync(d, { withFileTypes: true }); } catch (e) {
469
- if (e.code === 'ENOENT') return;
470
- if (e.code === 'EACCES' || e.code === 'EPERM') { unreadableDirs.push(d); return; }
471
- throw e;
472
- }
517
+ try { es = list(d); } catch { unreadableDirs.push(d); return; }
473
518
  for (const e of es.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0))) {
474
519
  if (e.name === '.git' || e.name.startsWith('._')) continue;
475
520
  const p = join(d, e.name);
@@ -572,7 +617,8 @@ function applyReplacing(tokens, pairs, maxGrowth = Infinity, byCopy = false) {
572
617
  if (p.mode || p.partial || p.from.toks.length !== 1 || p.to.toks.length > 1) continue;
573
618
  const pat = p.from.toks[0].v.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
574
619
  const rep = p.to.toks[0] ? p.to.toks[0].v : '';
575
- const v = tk.v.replace(new RegExp(`\\(${pat}\\)`, 'gi'), `(${rep})`);
620
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- pat is escaped
621
+ const v = tk.v.replace(new RegExp(`\\(${pat}\\)`, 'gi'), () => `(${rep})`);
576
622
  if (v !== tk.v) tk = { ...tk, v, u: v };
577
623
  }
578
624
  }
@@ -779,15 +825,31 @@ function expand(tokens, ctx, stack, format) {
779
825
  return out;
780
826
  }
781
827
 
828
+ // The symbolic map BMS would generate for a COBOL mapset the tree holds only as BMS source, each
829
+ // line placed at the BMS statement it comes from.
830
+ function symbolicMapFor(name, ctx) {
831
+ const want = name.toUpperCase();
832
+ for (const p of filesByName(ctx.fileIndex).get(`${name}.bms`.toLowerCase()) || []) {
833
+ let mapsets;
834
+ try { ({ mapsets } = parseBms(ctx.readText(p))); } catch { continue; }
835
+ const mapset = mapsets.find((m) => m.name?.toUpperCase() === want && String(m.options?.lang || 'COBOL').toUpperCase() === 'COBOL');
836
+ const copybook = mapset && symbolicMapCopybook(mapset);
837
+ if (copybook) return { path: p, ...copybook };
838
+ }
839
+ return null;
840
+ }
841
+
782
842
  function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
783
- const path = resolveCopy(name, lib, ctx);
784
843
  const refused = copyRefusal(name, ctx);
785
- const record = { name, lib, via, file: at.file, line: at.line, status: path ? 'resolved' : refused || (SYSTEM_COPY.test(name) ? 'system' : 'missing'), path };
844
+ let path = resolveCopy(name, lib, ctx);
845
+ const generated = !path && !refused && !lib && via === 'copy' && ctx.fileIndex ? symbolicMapFor(name, ctx) : null;
846
+ if (generated) path = generated.path;
847
+ const record = { name, lib, via, file: at.file, line: at.line, status: path ? 'resolved' : refused || (SYSTEM_COPY.test(name) ? 'system' : 'missing'), path, ...(generated ? { generatedFrom: 'bms' } : {}) };
786
848
  ctx.copies.push(record);
787
849
  if (!path) return [];
788
850
  if (stack.includes(path) || stack.length > 40) { record.status = 'recursive'; return []; }
789
851
  if (++ctx.inclusions > MAX_INCLUSIONS || ctx.copyTokens > MAX_COPY_TOKENS) { record.status = 'expansion-limit'; return []; }
790
- const src = ctx.readText(path);
852
+ const src = generated ? generated.text : ctx.readText(path);
791
853
  // A copybook is read in the format in force where the COPY statement sits, which a >>SOURCE
792
854
  // directive earlier in the including file may have changed from the file's starting format.
793
855
  const fmt = detectFormat(src) === 'terminal' && ctx.copyFormat === 'auto' ? 'terminal' : (at.fmt || (ctx.copyFormat !== 'auto' ? ctx.copyFormat : (inheritedFormat || ctx.mainFormat)));
@@ -801,10 +863,13 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
801
863
  const f = p.from.toks;
802
864
  if (p.mode || !f.length) continue;
803
865
  const pat = f.map(t => t.v).join('');
866
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- escaped
804
867
  if (p.from.pseudo && /^[:(][A-Za-z0-9_-]*[:)]$/.test(pat)) { partial.push([p, new RegExp(escape(pat), 'gi')]); continue; }
805
868
  if (p.from.pseudo || f.length !== 1 || f[0].t !== 'lit') continue;
806
869
  const lit = `(['"])${escape(f[0].v)}\\1`;
870
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- lit is escaped
807
871
  const touching = new RegExp(`${lit}(?=[A-Za-z0-9-])|(?<=[A-Za-z0-9-])${lit}`);
872
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- lit is escaped
808
873
  if (norm.entries.some(e => touching.test(e.text))) partial.push([p, new RegExp(lit, 'g')]);
809
874
  }
810
875
  if (partial.length) {
@@ -821,6 +886,7 @@ function includeCopy(name, lib, pairs, at, via, ctx, stack, inheritedFormat) {
821
886
  for (const [p] of partial) p.partial = true;
822
887
  }
823
888
  const { tokens, diags } = tokenize(norm, path);
889
+ if (generated) for (const t of tokens) t.line = generated.lines[t.line - 1] ?? t.line;
824
890
  ctx.copyTokens += tokens.length;
825
891
  ctx.diags.push(...diags, ...norm.diags.map(d => ({ ...d, file: path })));
826
892
  const expanded = expand(tokens, ctx, [...stack, path], norm.finalFormat);
@@ -1513,6 +1579,9 @@ function indexTokens(seg) {
1513
1579
  const lone = (from, to) => (to - from === 1 && (seg[from].t === 'word' || seg[from].t === 'num') && /^\d+$/.test(seg[from].v) ? Number(seg[from].v) : null);
1514
1580
  const refLength = colonAt < 0 ? null : lone(colonAt + 1, close);
1515
1581
  const refStart = colonAt < 0 ? null : lone(k + 1, colonAt);
1582
+ // A length is judged against where its reference starts: a name, a name moved by a constant, or neither.
1583
+ const named = (from, to) => (seg[from].t === 'word' && !/^\d+$/.test(seg[from].v) && (to - from === 1 || (to - from === 3 && offsetOf(seg, from, k, close, colonAt))) ? { tok: seg[from], offset: offsetOf(seg, from, k, close, colonAt) } : null);
1584
+ const refFrom = colonAt < 0 || refStart != null ? null : named(k + 1, colonAt) || { tok: null, offset: 0 };
1516
1585
  for (let j = k + 1; j < close; j++) {
1517
1586
  const t = seg[j];
1518
1587
  // A number is tokenized as a word, and no name is all digits.
@@ -1521,7 +1590,7 @@ function indexTokens(seg) {
1521
1590
  if (prev && prev.t === 'word' && (prev.u === 'OF' || prev.u === 'IN' || prev.u === 'FUNCTION')) continue;
1522
1591
  const kind = colonAt < 0 ? 'subscript' : j < colonAt ? 'refmod-offset' : 'refmod-length';
1523
1592
  const other = kind === 'refmod-offset' ? refLength : kind === 'refmod-length' ? refStart : null;
1524
- out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}) });
1593
+ out.push({ host, tok: t, kind, offset: offsetOf(seg, j, k, close, colonAt), ...(other ? { span: other } : {}), ...(kind === 'refmod-length' && refFrom ? { from: refFrom } : {}) });
1525
1594
  }
1526
1595
  }
1527
1596
  lastHost = host;
@@ -1841,7 +1910,7 @@ function collectCopybookDefines(src, format, ctx, depth) {
1841
1910
  const sw = /(?:^|\s)>>\s*SOURCE\s+(?:FORMAT\s+)?(?:IS\s+)?(FREE|FIXED|VARIABLE|TERMINAL)/i.exec(l);
1842
1911
  if (sw) { current = sw[1].toLowerCase(); continue; }
1843
1912
  if (current !== 'free' && current !== 'terminal' && l.length > 6 && '*/'.includes(l[6])) continue;
1844
- const code = l.replace(/\*>.*$/, '');
1913
+ const code = withoutInlineComment(l);
1845
1914
  const m = /(?:^|[\s.])COPY\s+("[^"]+"|'[^']+'|[A-Za-z0-9_-]+)/i.exec(code);
1846
1915
  if (!m) continue;
1847
1916
  const name = m[1].replace(/^["']|["']$/g, '');
@@ -122,12 +122,14 @@ export const constantsCopybook = (names) => `${names.map((n) => ` 01 ${n}
122
122
  // prefix when TIOAPFX=YES, then for each named field its length, flag or attribute byte, extended
123
123
  // attribute bytes and data, the output record redefining the input. Grouped (GRPNAME) and OCCURS
124
124
  // fields are laid out differently, and a mapset holding either is not written. Returns the copybook
125
- // text and the names it declares, or null.
125
+ // text, the names it declares and, for each of its lines, the BMS line it comes from; or null.
126
126
  const ATTRIBUTE_LETTER = { COLOR: 'C', PS: 'P', HILIGHT: 'H', VALIDN: 'V', OUTLINE: 'U', SOSI: 'M', TRANSP: 'T' };
127
127
  export function symbolicMapCopybook(mapset) {
128
128
  const out = [];
129
129
  const names = [];
130
- const line = (s) => out.push(` ${s}`);
130
+ const lines = [];
131
+ let from = mapset.line;
132
+ const line = (s) => { out.push(` ${s}`); lines.push(from); };
131
133
  const declare = (n, s) => { names.push(n); line(s); };
132
134
  // The assembler writes FILLER; each gets a name here, the layout unchanged, so the grade can tell
133
135
  // the stand-in's items from the program's.
@@ -138,6 +140,7 @@ export function symbolicMapCopybook(mapset) {
138
140
  if (fields.some((f) => f.grpname || f.occurs > 1)) return null;
139
141
  const { input, output, attributes } = map.symbolic;
140
142
  if (!input && !output) continue;
143
+ from = map.line;
141
144
  const prefix = String(map.operands?.get?.('TIOAPFX') ?? mapset.options.tioapfx ?? '').toUpperCase() === 'YES';
142
145
  const pic = (f, which) => f[which] || `X(${f.length})`;
143
146
  if (input) {
@@ -145,6 +148,7 @@ export function symbolicMapCopybook(mapset) {
145
148
  if (prefix) line(` 02 ${filler(input)} PIC X(12).`);
146
149
  for (const f of fields) {
147
150
  const n = f.name.toUpperCase();
151
+ from = f.line;
148
152
  declare(`${n}L`, ` 02 ${n}L COMP PIC S9(4).`);
149
153
  declare(`${n}F`, ` 02 ${n}F PIC X.`);
150
154
  line(` 02 ${filler(input)} REDEFINES ${n}F.`);
@@ -154,10 +158,12 @@ export function symbolicMapCopybook(mapset) {
154
158
  }
155
159
  }
156
160
  if (output) {
161
+ from = map.line;
157
162
  declare(output, input ? `01 ${output} REDEFINES ${input}.` : `01 ${output}.`);
158
163
  if (prefix) line(` 02 ${filler(output)} PIC X(12).`);
159
164
  for (const f of fields) {
160
165
  const n = f.name.toUpperCase();
166
+ from = f.line;
161
167
  if (input) line(` 02 ${filler(output)} PIC X(3).`);
162
168
  else { line(` 02 ${filler(output)} PIC X(2).`); declare(`${n}A`, ` 02 ${n}A PIC X.`); }
163
169
  for (const a of attributes) declare(`${n}${ATTRIBUTE_LETTER[a]}`, ` 02 ${n}${ATTRIBUTE_LETTER[a]} PIC X.`);
@@ -165,5 +171,5 @@ export function symbolicMapCopybook(mapset) {
165
171
  }
166
172
  }
167
173
  }
168
- return out.length ? { text: `${out.join('\n')}\n`, names } : null;
174
+ return out.length ? { text: `${out.join('\n')}\n`, names, lines } : null;
169
175
  }
@@ -188,7 +188,7 @@ function place(lines, block, format, text) {
188
188
  while (w < words.length && writeFrom + out.length + (out ? 1 : 0) + words[w].length <= to) out += (out ? ' ' : '') + words[w++];
189
189
  const line = lines[li].padEnd(to);
190
190
  lines[li] = line.slice(0, blankFrom) + ' '.repeat(Math.max(0, writeFrom - blankFrom)) + out.padEnd(Math.max(0, to - Math.max(writeFrom, blankFrom))) + line.slice(to);
191
- if (li !== block.endLine) lines[li] = lines[li].replace(/\s+$/, '');
191
+ if (li !== block.endLine) lines[li] = lines[li].trimEnd();
192
192
  }
193
193
  return words.slice(w);
194
194
  }
@@ -208,6 +208,7 @@ function spill(words, format) {
208
208
  }
209
209
 
210
210
  // A division or section header, or a word, standing alone: not part of a longer name.
211
+ // nosemgrep: javascript.lang.security.audit.detect-non-literal-regexp.detect-non-literal-regexp -- callers pass literal patterns
211
212
  const word = (re) => new RegExp(`(?<![A-Z0-9-])${re}(?![A-Z0-9-])`, 'i');
212
213
  const HEADERS = [['procedure', word('PROCEDURE\\s+DIVISION')], ['linkage', word('LINKAGE\\s+SECTION')], ['data', word('DATA\\s+DIVISION')], ['after', word('(?:REPORT|SCREEN)\\s+SECTION')]];
213
214
  const PROGRAM_ID = word('PROGRAM-ID');
@@ -373,7 +374,7 @@ function translateText(t, shared, ctx, before = new Map(), after = new Map()) {
373
374
  }
374
375
  const rest = place(lines, b, format, text);
375
376
  if (periodOnly) {
376
- lines[b.endLine] = lines[b.endLine].replace(/\s+$/, '');
377
+ lines[b.endLine] = lines[b.endLine].trimEnd();
377
378
  if (rest.length) { add(after, b.endLine, spill(rest, format)); state.spilled++; }
378
379
  } else if (rest.length) {
379
380
  // What followed END-EXEC - a period, or the rest of a sentence - follows the whole translation.
package/lib/revision.json CHANGED
@@ -1 +1 @@
1
- {"commit":"f5995391b095f62da69d501ea7f6bd11a75bdc6e","tag":"v0.2.150"}
1
+ {"commit":"371041fbffe625df6b0a872249cf071268fcb6a6"}
package/lib/sarif.mjs CHANGED
@@ -32,8 +32,9 @@ const LEVEL = { crit: 'error', high: 'error', med: 'warning', low: 'note', info:
32
32
  // Locations are URI references: each segment is encoded, lone surrogates from Windows names replaced first.
33
33
  const LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/g;
34
34
  const uriOf = (p) => String(p).split('/').map((seg) => encodeURIComponent(seg.replace(LONE_SURROGATE, '\uFFFD'))).join('/');
35
- // Square brackets in a SARIF message are link syntax, and a finding's text can quote the tree.
36
- const textOf = (s) => (typeof s === 'string' ? s.replace(/[[\]]/g, '\\$&') : s);
35
+ // Square brackets in a SARIF message are link syntax, and a finding's text can quote the tree. A
36
+ // backslash escapes the character after it, so one before a bracket or a backslash is escaped too.
37
+ const textOf = (s) => (typeof s === 'string' ? s.replace(/\\(?=[\\[\]])|[[\]]/g, '\\$&') : s);
37
38
  // SARIF's three levels fold critical into high and info into low, so the severity also travels as
38
39
  // the number GitHub code scanning ranks by. Info is coverage and context, which rank as nothing.
39
40
  const SECURITY_SEVERITY = { crit: '9.5', high: '8.0', med: '5.5', low: '3.0' };
package/lib/scan.mjs CHANGED
@@ -74,7 +74,8 @@ function scanEach(root, opts) {
74
74
  const reports = opts.repos.map(repo => ({ repo, r: scanAll(join(root, repo), { ...opts, repos: null, repoName: repo, feedRoot: opts.feedRoot || root }) }));
75
75
  const findings = [];
76
76
  const checked = [];
77
- const fixedAt = (repo, x) => (x?.fixAt ? { exploitability: { ...x, fixAt: { ...x.fixAt, path: underRepo(repo, x.fixAt.path) } } } : {});
77
+ const fixedAt = (repo, x) => (x?.fixAt ? { exploitability: { ...x, fixAt: { ...x.fixAt, path: underRepo(repo, x.fixAt.path),
78
+ ...(x.fixAt.builtInto ? { builtInto: { ...x.fixAt.builtInto, path: underRepo(repo, x.fixAt.builtInto.path) } } : {}) } } } : {});
78
79
  const prefixed = (repo, f) => ({ ...f, path: underRepo(repo, f.path), ...(f.related ? { related: f.related.map(x => ({ ...x, path: underRepo(repo, x.path) })) } : {}), ...fixedAt(repo, f.exploitability) });
79
80
  for (const { repo, r } of reports) {
80
81
  for (const f of r.findings) findings.push(prefixed(repo, f));