@portll/cobolwork 0.2.140 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/dataflow.mjs CHANGED
@@ -1,5 +1,5 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
- import { inScope, isProgram, isBms, readSource, relPath } from './sources.mjs';
2
+ import { inScope, isProgram, isBms, relPath } from './sources.mjs';
3
3
  import { parseBms, symbolicNames } from './bms.mjs';
4
4
  import { dirname, join } from 'node:path';
5
5
  import { parseSource, buildFileIndex } from './parser.mjs';
@@ -15,7 +15,7 @@ import { eachWithinMemory, watchMemoryBuffer } from './kernel/memory.mjs';
15
15
  // A flow finding is located by source and sink rather than by one path and line, so it keeps its
16
16
  // own comparator. The text comparison underneath it is the shared one.
17
17
  import { byText } from './kernel/findings.mjs';
18
- import { buildControl, creditOf, hasFact, within } from './control.mjs';
18
+ import { buildControl, creditOf, hasFact, topOf, within, wholeTop } from './control.mjs';
19
19
  import { directoryTree } from './kernel/source-tree.mjs';
20
20
 
21
21
  const OS_COMMAND_ROUTINE = /^(SYSTEM|C\$SYSTEM|CBL_EXEC_RUN_UNIT|CBL_GC_HOSTED|BXPSYSTM)$/i;
@@ -197,6 +197,7 @@ export function analyze(root, opts = {}) {
197
197
  const nodes = [];
198
198
  const programs = [];
199
199
  const jclPending = [];
200
+ const jclTrees = new Map();
200
201
  // CICS definitions, from CSD extracts and from DFHCSDUP job input, and the estate's own statement
201
202
  // of which DDs and queues reach the internal reader.
202
203
  const csd = { tdqueues: new Map(), transactions: new Map(), urimaps: new Map() };
@@ -209,7 +210,7 @@ export function analyze(root, opts = {}) {
209
210
  const jobSteps = [];
210
211
  // The maps of the repository being read, by MAPSET/MAP and by MAP alone.
211
212
  const bmsMaps = new Map();
212
- const site = loadSite(root, opts.site || null);
213
+ const site = loadSite(root, opts.site || null, opts.tree);
213
214
  const relOf = new Map();
214
215
  const rel = (p) => { let r = relOf.get(p); if (r === undefined) { r = relPath(root, p); relOf.set(p, r); } return r; };
215
216
  // `at` is the statement an edge that is not a statement's own is read at: the CALL or LINK that
@@ -405,6 +406,13 @@ export function analyze(root, opts = {}) {
405
406
  // A whole number no sign is written on, which no value put in it can leave below 0.
406
407
  const unsignedWhole = (it) => it.level !== 88 && !it.index && !(it.children || []).length
407
408
  && /^9+$/.test(String(it.picture || '').toUpperCase().replace(/(\w)\((\d+)\)/g, (_, ch, n) => ch.repeat(Number(n))));
409
+ // Where a length's reference starts: the start's node, a constant it is moved by, and the span
410
+ // facts that keep start and length together inside the item.
411
+ const startOf = (from, len, size) => {
412
+ const s = from.tok ? itemOfToken(from.tok) : null;
413
+ const spans = ctl && ctl.spans && s && len ? ctl.spans.filter((f) => ((f.a === s && f.b === len) || (f.a === len && f.b === s)) && f.k + from.offset <= size + 1).map((f) => f.t) : [];
414
+ return { node: from.tok ? nodeOfToken(from.tok) : null, offset: from.offset, size, spans };
415
+ };
408
416
  const indexed = new Set();
409
417
  const subscripting = new Map();
410
418
  // Each subscript use of a name, with the other indices of the same reference: the loop-bound
@@ -437,8 +445,9 @@ export function analyze(root, opts = {}) {
437
445
  const use = `${key}|${point ?? `s${st.at}`}|${offset}`;
438
446
  if (indexed.has(use)) continue;
439
447
  indexed.add(use);
448
+ const start = kind === 'reference-modification' && x.from && base != null ? startOf(x.from, idx, base) : null;
440
449
  at.sinks.push({
441
- kind, onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point, group: `${pk}|${key}`, limit, ...(offset ? { offset } : {}), ...(offset > 0 && idx && unsignedWhole(idx) ? { nonNeg: true } : {}), ...(ssrange ? { ssrange } : {}),
450
+ kind, onlyFrom: FROM_OUTSIDE, file: st.file, line: st.line, point, group: `${pk}|${key}`, limit, ...(start ? { start } : {}), ...(offset ? { offset } : {}), ...(offset > 0 && idx && unsignedWhole(idx) ? { nonNeg: true } : {}), ...(ssrange ? { ssrange } : {}),
442
451
  detail: kind === 'subscript'
443
452
  ? `${at.name} subscripts ${host.name}, a table of ${table.occurs}`
444
453
  : `${at.name} sets the ${x.kind === 'refmod-offset' ? 'start' : 'length'} of a reference to ${host.name}, which is ${host.size} bytes`,
@@ -959,7 +968,7 @@ export function analyze(root, opts = {}) {
959
968
  const tree = (!repo && opts.tree) ? opts.tree : directoryTree(base, opts);
960
969
  const idx = tree.index;
961
970
  const files = tree.list().filter(isProgram).filter(inScope(opts));
962
- for (const f of tree.list().filter(isJcl).filter(inScope(opts))) jclPending.push(f);
971
+ for (const f of tree.list().filter(isJcl).filter(inScope(opts))) { jclPending.push(f); jclTrees.set(f, tree); }
963
972
  for (const f of tree.list().filter((p) => /\.csd$/i.test(p)).filter(inScope(opts))) {
964
973
  try { addCsd(tree.text(f).text, f); } catch (e) { stats.unreadable++; unread.push(`${tree.rel(f)}: ${e.code || e.name}`); }
965
974
  }
@@ -1088,7 +1097,7 @@ export function analyze(root, opts = {}) {
1088
1097
  // ACCEPT FROM COMMAND-LINE is untrusted.
1089
1098
  for (const f of jclPending) {
1090
1099
  let src;
1091
- try { src = readSource(f).text; } catch (e) { stats.unreadable++; unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
1100
+ try { src = jclTrees.get(f).text(f).text; } catch (e) { stats.unreadable++; unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
1092
1101
  let job;
1093
1102
  try { job = parseJcl(src, f); } catch (e) { stats.threw++; unparsed.push(`${rel(f)}: ${e.code || e.name}`); continue; }
1094
1103
  stats.jclFiles++;
@@ -1282,6 +1291,18 @@ export function analyze(root, opts = {}) {
1282
1291
  const bits = p.facts.get(point);
1283
1292
  return bits ? creditOf(n.checks, bits, kind, limit, offset, nonNeg) : NONE;
1284
1293
  }
1294
+ // A length is judged with its start: the last byte, S + L - 1, stays inside the item where the
1295
+ // start at its highest leaves room for the length, or where a fact bounds the two together.
1296
+ function spanned(c, sink, n) {
1297
+ if (c.level < 2 || !sink.start) return c;
1298
+ const p = programs[n.pk];
1299
+ const bits = p && p.ordered && sink.point != null ? p.facts.get(sink.point) : null;
1300
+ const s = sink.start;
1301
+ if (bits && s.spans.some((t) => hasFact(bits, t))) return c;
1302
+ const top = bits && s.node ? topOf(s.node.checks, bits) : null;
1303
+ const len = c.cons ? wholeTop(c.cons) : null;
1304
+ return top != null && len != null && top + s.offset + len - 1 <= s.size ? c : { level: 1, check: c.check };
1305
+ }
1285
1306
  // A loop bound cannot push its counter out of the table when the loop's own condition or the
1286
1307
  // body's checks keep every subscript the counter makes, and each index beside it, in range.
1287
1308
  // That holds whatever the bound is, so it does not depend on the route the input took.
@@ -1311,7 +1332,7 @@ export function analyze(root, opts = {}) {
1311
1332
  let point = sink.point;
1312
1333
  for (let h = hops.length - 1; h >= 0; h--) {
1313
1334
  const n = hops[h].node;
1314
- let c = creditAt(n, point, sink.kind, sink.limit, sink.offset, sink.nonNeg);
1335
+ let c = spanned(creditAt(n, point, sink.kind, sink.limit, sink.offset, sink.nonNeg), sink, n);
1315
1336
  // A count checked against a number above the table's maximum has been checked, not kept in range.
1316
1337
  if (c.level === 2 && sink.top != null && !within(c.cons, sink.top, false)) c = { level: 1, check: c.check };
1317
1338
  if (c.level > best.level) { best = c; by = n; }
@@ -1333,7 +1354,10 @@ export function analyze(root, opts = {}) {
1333
1354
  const at = new Map();
1334
1355
  for (const c of want) { if (!at.has(c.node.id)) at.set(c.node.id, []); at.get(c.node.id).push(c); }
1335
1356
  // Along the way no sink's limit applies, so a bound that holds only against one blocks nothing.
1336
- const holds = (since, point, k, limit = null, offset = 0, nonNeg = false) => since.some((n) => creditAt(n, point, k, limit, offset, nonNeg).level >= level);
1357
+ const holds = (since, point, k, limit = null, offset = 0, nonNeg = false, sink = null) => since.some((n) => {
1358
+ const c = creditAt(n, point, k, limit, offset, nonNeg);
1359
+ return (sink ? spanned(c, sink, n) : c).level >= level;
1360
+ });
1337
1361
  const lost = new Set();
1338
1362
  const dirty = new Map();
1339
1363
  const parts = new Map();
@@ -1342,7 +1366,7 @@ export function analyze(root, opts = {}) {
1342
1366
  const began = edgesWalked;
1343
1367
  for (let i = 0; i < open.length; i++) {
1344
1368
  const st = open[i];
1345
- for (const c of at.get(st.node.id) || []) if (!holds(st.since, c.sink.point, c.sink.kind, c.sink.limit, c.sink.offset, c.sink.nonNeg)) lost.add(c);
1369
+ for (const c of at.get(st.node.id) || []) if (!holds(st.since, c.sink.point, c.sink.kind, c.sink.limit, c.sink.offset, c.sink.nonNeg, c.sink)) lost.add(c);
1346
1370
  for (const e of st.node.edgesOut) {
1347
1371
  edgesWalked++;
1348
1372
  const to = e.to;
package/lib/diff.mjs CHANGED
@@ -1,16 +1,15 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
2
  import { existsSync, mkdirSync, mkdtempSync, realpathSync, rmSync, writeFileSync } from 'node:fs';
3
- import { spawnSync } from 'node:child_process';
4
3
  import { tmpdir } from 'node:os';
5
- import { dirname, join, resolve, sep } from 'node:path';
6
- import { parseSource } from './parser.mjs';
7
- import { inScope, isProgram, readSource, relPath } from './sources.mjs';
4
+ import { dirname, join, resolve } from 'node:path';
5
+ import { git, revisionBlobs, treeOf, treePathParts } from './kernel/git.mjs';
6
+ import { inScope, isProgram } from './sources.mjs';
8
7
  import { scanAll } from './scan.mjs';
9
8
  import { FLOW_MODEL, SCHEMA_VERSION, TOOL_VERSION } from './version.mjs';
10
9
  // A diff finding is located by program rather than by line, so it orders on its own key. The text
11
10
  // comparison underneath it is the shared one.
12
11
  import { byText, evidenceMap } from './kernel/findings.mjs';
13
- import { directoryTree } from './kernel/source-tree.mjs';
12
+ import { directoryTree, gitTree } from './kernel/source-tree.mjs';
14
13
  import { printable } from './kernel/printable.mjs';
15
14
  import { loadSite, SITE_FILE } from './site.mjs';
16
15
  import { loadBaseline, BASELINE_FILE, SUPPRESSING } from './baseline.mjs';
@@ -30,56 +29,18 @@ export const DIFF_RULES = {
30
29
 
31
30
  const MAX_LISTED = 8;
32
31
 
33
- // A reviewed repository's .git/config can name commands; with fsmonitor off and plumbing only, git runs none.
34
- const GIT_ENV = { ...process.env, GIT_OPTIONAL_LOCKS: '0', GIT_TERMINAL_PROMPT: '0' };
35
- const git = (repo, args, opts = {}) => spawnSync('git', ['-c', 'core.fsmonitor=false', '-C', repo, ...args], { env: GIT_ENV, maxBuffer: 256 * 1024 * 1024, ...opts });
36
- const refused = (what, ref, why) => Object.assign(new Error(`${what} ${ref} failed: ${why}`), { code: 'EDIFFREF' });
37
-
38
- function treeOf(repo, ref) {
39
- if (!ref || String(ref).startsWith('-')) throw refused('git rev-parse', ref, 'a revision cannot be empty or start with "-"');
40
- const r = git(repo, ['rev-parse', '--verify', '--quiet', '--end-of-options', `${ref}^{tree}`], { encoding: 'utf8' });
41
- const oid = String(r.stdout || '').trim();
42
- if (r.status !== 0 || !/^[0-9a-f]{40}([0-9a-f]{24})?$/.test(oid)) throw refused('git rev-parse', ref, String(r.stderr || '').trim() || 'not a revision');
43
- return oid;
44
- }
45
-
46
- // A tree path as a file under `dir`, or null if a part is empty, climbs, names a drive or stream, or is .git.
47
- function placeIn(dir, path) {
48
- const parts = path.split('/');
49
- if (parts.some((p) => !p || p === '.' || p === '..' || /[\\:\0]/.test(p) || p.toLowerCase() === '.git')) return null;
50
- const to = resolve(dir, ...parts);
51
- return to.startsWith(dir + sep) ? to : null;
52
- }
53
-
54
- // The revision as committed: blobs as stored, with no filter or line-ending conversion; links and submodules left out.
32
+ // The revision as committed, written into a temporary directory, for a caller that hands the tree
33
+ // to a program outside this process: build's compiler and ironwork read files, not blobs.
55
34
  function materialise(repo, ref) {
56
- const tree = treeOf(repo, ref);
57
- const listed = git(repo, ['ls-tree', '-r', '-z', '--full-tree', tree]);
58
- if (listed.status !== 0) throw refused('git ls-tree', ref, String(listed.stderr || '').trim() || `exit ${listed.status}`);
59
- const blobs = [];
60
- for (const entry of listed.stdout.toString('utf8').split('\0')) {
61
- const m = /^(\d{6}) blob ([0-9a-f]+)\t(.+)$/s.exec(entry);
62
- if (m && m[1] !== '120000') blobs.push({ oid: m[2], path: m[3] });
63
- }
35
+ const { blobs } = revisionBlobs(repo, treeOf(repo, ref), ref);
64
36
  const dir = realpathSync(mkdtempSync(join(tmpdir(), 'cobolwork-diff-')));
65
37
  try {
66
- for (let i = 0; i < blobs.length; i += 500) {
67
- const batch = blobs.slice(i, i + 500);
68
- const r = git(repo, ['cat-file', '--batch'], { input: batch.map((b) => b.oid).join('\n') + '\n' });
69
- if (r.status !== 0) throw refused('git cat-file', ref, String(r.stderr || '').trim() || `exit ${r.status}`);
70
- let at = 0;
71
- for (const b of batch) {
72
- const nl = r.stdout.indexOf(0x0a, at);
73
- const head = r.stdout.toString('utf8', at, nl).split(' ');
74
- if (head[1] !== 'blob') throw refused('git cat-file', ref, `${b.oid} is ${head[1] || 'missing'}`);
75
- const size = Number(head[2]);
76
- const to = placeIn(dir, b.path);
77
- if (to) {
78
- mkdirSync(dirname(to), { recursive: true });
79
- writeFileSync(to, r.stdout.subarray(nl + 1, nl + 1 + size));
80
- }
81
- at = nl + 1 + size + 1;
82
- }
38
+ for (const b of blobs) {
39
+ const parts = treePathParts(b.path);
40
+ if (!parts) continue;
41
+ const to = resolve(dir, ...parts);
42
+ mkdirSync(dirname(to), { recursive: true });
43
+ writeFileSync(to, b.bytes);
83
44
  }
84
45
  } catch (e) {
85
46
  rmSync(dir, { recursive: true, force: true });
@@ -102,9 +63,9 @@ const qualified = (it) => { const names = []; for (let x = it; x; x = x.parent)
102
63
 
103
64
  // A program that cannot be read or parsed is named in `unread`, never dropped: dropped, it would
104
65
  // read as removed from the tree, and every change inside it would go unreported.
105
- function facts(root, opts = {}) {
106
- // Each side of a diff is its own tree: the base is a revision materialised into a temporary
107
- // directory, the head is the working tree, and they are walked separately.
66
+ function facts(side, opts = {}) {
67
+ // Each side of a diff is its own tree: a revision read out of git, or the working tree, and they
68
+ // are walked separately.
108
69
  //
109
70
  // This used to build its own parse options with systemDirs: [], where every other caller passed
110
71
  // the caller's. Nobody decided that - it is what five copies of one object literal do. The
@@ -112,24 +73,24 @@ function facts(root, opts = {}) {
112
73
  // change inside one was invisible to the command whose whole job is what a change reaches. Both
113
74
  // sides now take the same configuration, which for a caller that passes no systemDirs is
114
75
  // exactly the old behaviour.
115
- const tree = directoryTree(root, opts);
116
- const idx = tree.index;
76
+ const tree = side.tree || directoryTree(side.root, opts);
77
+ const rel = (p) => tree.rel(p);
117
78
  const out = new Map();
118
79
  out.unread = [];
119
80
  for (const f of tree.list().filter(isProgram).filter(inScope(opts))) {
120
81
  let src;
121
- try { src = readSource(f).text; } catch (e) { out.unread.push(`${relPath(root, f)}: ${e.code || e.name}`); continue; }
82
+ try { src = tree.text(f).text; } catch (e) { out.unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
122
83
  if (!/PROCEDURE\s+DIVISION|PROGRAM-ID/i.test(src)) continue;
123
84
  let r;
124
85
  try {
125
86
  r = tree.parse(f, src);
126
- } catch (e) { out.unread.push(`${relPath(root, f)}: ${e.code || e.name}`); continue; }
127
- const file = relPath(root, f);
87
+ } catch (e) { out.unread.push(`${rel(f)}: ${e.code || e.name}`); continue; }
88
+ const file = rel(f);
128
89
  for (const p of r.programs) {
129
90
  const items = new Map();
130
91
  for (const it of p.items) {
131
92
  if (!it.name || it.name === 'FILLER' || it.level === 88 || it.level === 66 || it.level === 78) continue;
132
- items.set(qualified(it), { size: it.size, offset: it.offset, from: it.file ? relPath(root, it.file) : file, section: it.section, level: it.level });
93
+ items.set(qualified(it), { size: it.size, offset: it.offset, from: it.file ? rel(it.file) : file, section: it.section, level: it.level });
133
94
  }
134
95
  const calls = new Set();
135
96
  const dynamic = new Set();
@@ -149,7 +110,7 @@ function facts(root, opts = {}) {
149
110
  if (arg) commareas.add(arg.u);
150
111
  }
151
112
  const using = new Set(p.calls.flatMap(c => c.using.filter(a => a.word).map(a => a.word)));
152
- const copied = new Set(r.copies.filter((c) => c.status === 'resolved' && c.path).map((c) => relPath(root, c.path)));
113
+ const copied = new Set(r.copies.filter((c) => c.status === 'resolved' && c.path).map((c) => rel(c.path)));
153
114
  out.set(`${file}#${p.id}`, { id: p.id, file, text: src, items, calls, dynamic, commareas, using, copied });
154
115
  }
155
116
  }
@@ -168,12 +129,17 @@ function countByKey(findings) {
168
129
  return m;
169
130
  }
170
131
 
171
- export function diffTrees(baseRoot, headRoot, opts = {}) {
172
- // The allow list describes the working tree, so it never applies to a revision checked out into
173
- // a temporary directory: applied there it would match nothing and the base would read as empty.
132
+ // Each side is a directory, or a source tree: a git revision read without writing it out.
133
+ const sideOf = (x) => (typeof x === 'string' ? { root: x, tree: null } : { root: x.root, tree: x });
134
+
135
+ export function diffTrees(baseSide, headSide, opts = {}) {
136
+ const baseAt = sideOf(baseSide);
137
+ const headAt = sideOf(headSide);
138
+ // The allow list describes the working tree, so it never applies to a revision: applied there it
139
+ // would match nothing and the base would read as empty.
174
140
  const baseOpts = { ...opts, allow: null };
175
- const base = facts(baseRoot, baseOpts);
176
- const head = facts(headRoot, opts);
141
+ const base = facts(baseAt, baseOpts);
142
+ const head = facts(headAt, opts);
177
143
  const findings = [];
178
144
  const reach = { programsCompared: 0, programsAdded: 0, programsRemoved: 0, programsWithLayoutChange: 0, reachedOnlyThroughCopybooks: 0, copybooksImplicated: new Set() };
179
145
 
@@ -214,8 +180,9 @@ export function diffTrees(baseRoot, headRoot, opts = {}) {
214
180
 
215
181
  const listing = { listSinks: opts.listSinks === true, listSources: opts.listSources === true };
216
182
  const shared = { only: opts.only, fullTrace: opts.fullTrace, systemDirs: opts.systemDirs || [], ...listing };
217
- const before = scanAll(baseRoot, shared);
218
- const after = scanAll(headRoot, { ...shared, allow: opts.allow });
183
+ const withTree = (at) => (at.tree ? { tree: at.tree } : {});
184
+ const before = scanAll(baseAt.root, { ...shared, ...withTree(baseAt) });
185
+ const after = scanAll(headAt.root, { ...shared, ...withTree(headAt), allow: opts.allow });
219
186
  // A file neither tree could read cannot be compared. Its findings are neither introduced nor
220
187
  // resolved, and calling them resolved would report a scan failure as a fix.
221
188
  const unreadable = new Set([...base.unread, ...head.unread].map(u => u.split(': ')[0]));
@@ -237,7 +204,7 @@ export function diffTrees(baseRoot, headRoot, opts = {}) {
237
204
  return n > (afterCounts.get(k) || 0);
238
205
  });
239
206
  const uncomparable = [...new Set([...before.findings, ...after.findings].filter(f => !comparable(f)).map(f => f.path))];
240
- const configurationChanged = configurationChanges(baseRoot, headRoot);
207
+ const configurationChanged = configurationChanges(baseSide, headSide);
241
208
 
242
209
  for (const f of findings) { const m = DIFF_RULES[f.rule]; f.sev = m.sev; f.cwe = m.cwe; f.evidence = m.evidence; }
243
210
  findings.sort((a, b) => byText(a.rule, b.rule) || byText(a.path, b.path) || byText(String(a.program), String(b.program)));
@@ -272,13 +239,16 @@ export function diffTrees(baseRoot, headRoot, opts = {}) {
272
239
  const SITE_LISTS = ['productionQualifiers', 'productionJobPaths', 'nonProductionJobPaths', 'systemNames', 'vendorPacks',
273
240
  'internalReaderDds', 'internalReaderQueues', 'compilerOptions', 'apfLibraries', 'restrictedDatasets', 'surrogateUsers'];
274
241
 
275
- function siteChange(baseRoot, headRoot) {
276
- const was = existsSync(join(baseRoot, SITE_FILE));
277
- const now = existsSync(join(headRoot, SITE_FILE));
242
+ // Whether a side holds a file at its top: in the source tree it was read from, or on disk.
243
+ const holds = (at, name) => (at.tree && at.tree.kind !== 'directory' ? at.tree.contains(resolve(at.tree.root, name)) : existsSync(join(at.root, name)));
244
+
245
+ function siteChange(baseAt, headAt) {
246
+ const was = holds(baseAt, SITE_FILE);
247
+ const now = holds(headAt, SITE_FILE);
278
248
  if (!was && !now) return null;
279
249
  if (!now) return { file: SITE_FILE, change: 'removed, so the rules that need it do not run' };
280
- const b = loadSite(baseRoot);
281
- const h = loadSite(headRoot);
250
+ const b = loadSite(baseAt.root, null, baseAt.tree);
251
+ const h = loadSite(headAt.root, null, headAt.tree);
282
252
  const parts = was ? [] : ['added'];
283
253
  for (const k of SITE_LISTS) {
284
254
  const gone = b[k].filter((x) => !h[k].includes(x)).map((x) => `-${x}`);
@@ -291,16 +261,19 @@ function siteChange(baseRoot, headRoot) {
291
261
  return parts.length ? { file: SITE_FILE, change: printable(parts.join('; '), 400) } : null;
292
262
  }
293
263
 
294
- export const configurationChanges = (baseRoot, headRoot) =>
295
- [siteChange(baseRoot, headRoot), baselineChange(baseRoot, headRoot)].filter(Boolean);
264
+ export const configurationChanges = (baseSide, headSide) => {
265
+ const baseAt = sideOf(baseSide);
266
+ const headAt = sideOf(headSide);
267
+ return [siteChange(baseAt, headAt), baselineChange(baseAt, headAt)].filter(Boolean);
268
+ };
296
269
 
297
- function baselineChange(baseRoot, headRoot) {
298
- const was = existsSync(join(baseRoot, BASELINE_FILE));
299
- const now = existsSync(join(headRoot, BASELINE_FILE));
270
+ function baselineChange(baseAt, headAt) {
271
+ const was = holds(baseAt, BASELINE_FILE);
272
+ const now = holds(headAt, BASELINE_FILE);
300
273
  if (!was && !now) return null;
301
- const held = (root, there) => new Map((there ? loadBaseline(root)?.entries || [] : []).map((e) => [`${e.fingerprint}|${e.rule}`, e]));
302
- const b = held(baseRoot, was);
303
- const h = held(headRoot, now);
274
+ const held = (at, there) => new Map((there ? loadBaseline(at.root, { tree: at.tree })?.entries || [] : []).map((e) => [`${e.fingerprint}|${e.rule}`, e]));
275
+ const b = held(baseAt, was);
276
+ const h = held(headAt, now);
304
277
  const added = [...h.keys()].filter((k) => !b.has(k));
305
278
  const suppressing = added.filter((k) => SUPPRESSING.includes(h.get(k).action)).length;
306
279
  const removed = [...b.keys()].filter((k) => !h.has(k)).length;
@@ -334,11 +307,14 @@ export function withRefs(repo, baseRef, headRef, opts, fn) {
334
307
  }
335
308
  }
336
309
 
310
+ // Reads each revision out of git without writing it anywhere; the head is the working tree when
311
+ // headRef is null.
337
312
  export function diffRefs(repo, baseRef, headRef = null, opts = {}) {
338
- return withRefs(repo, baseRef, headRef, opts, (baseDir, headDir, scoped) => {
339
- const res = diffTrees(baseDir, headDir, scoped);
340
- res.summary.base = baseRef;
341
- res.summary.head = headRef || WORKING_TREE;
342
- return res;
343
- });
313
+ const trees = { systemDirs: opts.systemDirs || [] };
314
+ const base = gitTree(repo, baseRef, trees);
315
+ const head = headRef ? gitTree(repo, headRef, trees) : repo;
316
+ const res = diffTrees(base, head, headRef ? opts : { ...opts, allow: gitScope(repo) });
317
+ res.summary.base = baseRef;
318
+ res.summary.head = headRef || WORKING_TREE;
319
+ return res;
344
320
  }
@@ -50,12 +50,18 @@ function closure(root, file, idx, systemDirs) {
50
50
  return out;
51
51
  }
52
52
 
53
- // Programs the change edits or adds, or whose copybooks it edits: [{ path, base, head, via }] with
54
- // each side's source digest and the copybooks that changed under an unchanged program.
53
+ // Programs the change edits, adds or deletes, or whose copybooks it edits: [{ path, base, head, via }]
54
+ // with each side's source digest (head null for a deleted program) and the copybooks that changed
55
+ // under an unchanged program.
55
56
  export function changedPrograms(baseDir, headDir, allow = null, systemDirs = []) {
56
57
  const out = [];
57
58
  const headTree = directoryTree(headDir);
58
- const baseIdx = directoryTree(baseDir).index;
59
+ const baseTree = directoryTree(baseDir);
60
+ const baseIdx = baseTree.index;
61
+ for (const base of baseTree.list().filter(isProgram).sort()) {
62
+ const path = relPath(baseDir, base);
63
+ if (!existsSync(join(headDir, path))) out.push({ path, base: sha256(readFileSync(base)), head: null, deleted: true });
64
+ }
59
65
  for (const head of headTree.list().filter((p) => isProgram(p) && (!allow || allow.has(p))).sort()) {
60
66
  const path = relPath(headDir, head);
61
67
  const h = sha256(readFileSync(head));
@@ -74,12 +80,17 @@ export function changedPrograms(baseDir, headDir, allow = null, systemDirs = [])
74
80
  // Whether a commit in base..head reads as authored by a tool, from its identities and trailers.
75
81
  const TOOL = /\[bot\]|(^|[^a-z])bot@|copilot|claude|anthropic|openai|codex|devin|cursor|gemini|aider|sweep-ai|dependabot|renovate/i;
76
82
  export function machineAuthored(repo, base, head, env = process.env) {
77
- const r = spawnSync('git', ['-C', repo, 'rev-list', '--format=%x1e%an%x00%ae%x00%cn%x00%ce%x00%B', `${base}..${head}`], { encoding: 'utf8', timeout: 30000, maxBuffer: 16 << 20, env: cleanGitEnv(env) });
83
+ // NUL ends every field and every commit (-z), and a commit message cannot hold one.
84
+ const r = spawnSync('git', ['-C', repo, 'log', '-z', '--format=%an%x00%ae%x00%cn%x00%ce%x00%B', `${base}..${head}`], { encoding: 'utf8', timeout: 30000, maxBuffer: 16 << 20, env: cleanGitEnv(env) });
78
85
  if (r.status !== 0) return { known: false, commits: [] };
86
+ const fields = r.stdout.split('\x00');
87
+ if (fields.at(-1) === '') fields.pop();
88
+ if (fields.length % 5 !== 0) return { known: false, commits: [] };
79
89
  const commits = [];
80
- for (const chunk of r.stdout.split('\x1e').slice(1)) {
81
- const [an, ae, cn, ce, body = ''] = chunk.split('\x00');
82
- const trailers = body.split('\n').filter((l) => /^(co-authored-by|generated-by|assisted-by):/i.test(l.trim()));
90
+ for (let i = 0; i < fields.length; i += 5) {
91
+ const [an, ae, cn, ce, body = ''] = fields.slice(i, i + 5);
92
+ // Control characters are dropped first, so one placed before a trailer does not hide it.
93
+ const trailers = body.split('\n').map((l) => l.replace(/[\x00-\x1f\x7f]/g, '').trim()).filter((l) => /^(co-authored-by|generated-by|assisted-by):/i.test(l));
83
94
  const signals = [an, ae, cn, ce, ...trailers, ...(/generated with/i.test(body) ? ['generated with'] : [])].filter((x) => x && TOOL.test(x));
84
95
  if (signals.length) commits.push({ signals: [...new Set(signals.map((s) => s.trim().slice(0, 80)))] });
85
96
  }
@@ -101,13 +112,20 @@ export function checkEquivalence({ baseDir, headDir, allow, files = [], allowed
101
112
  problems.push(`${s.file} names head ${h ? h.slice(0, 12) : 'nothing'} and base ${b ? b.slice(0, 12) : 'nothing'}, which match no program this change edits`);
102
113
  }
103
114
  }
115
+ // Every statement that names a program is judged, and the program passes only if each does.
104
116
  const programs = changed.map((c) => {
105
- const s = statements.find((x) => x.statement && subjectDigest(x.statement, 'head') === c.head && subjectDigest(x.statement, 'base') === c.base);
117
+ if (c.deleted) return { path: c.path, deleted: true, statement: null, ok: false, because: 'the change deletes this program, and no equivalence statement can show what its callers do now' };
118
+ const matching = statements.filter((x) => x.statement && subjectDigest(x.statement, 'head') === c.head && subjectDigest(x.statement, 'base') === c.base);
119
+ const via = c.via ? { via: c.via } : {};
120
+ if (!matching.length) return { path: c.path, ...via, statement: null, ok: false, because: c.via ? `no equivalence statement names this program, whose copybook${c.via.length > 1 ? 's' : ''} ${c.via.join(', ')} the change edits` : 'no equivalence statement names this change' };
121
+ const judged = matching.map((s) => judge(c, s));
122
+ return judged.find((j) => !j.ok) || judged[0];
123
+ });
124
+ function judge(c, s) {
106
125
  const via = c.via ? { via: c.via } : {};
107
- if (!s) return { path: c.path, ...via, statement: null, ok: false, because: c.via ? `no equivalence statement names this program, whose copybook${c.via.length > 1 ? 's' : ''} ${c.via.join(', ')} the change edits` : 'no equivalence statement names this change' };
108
126
  const p = s.statement.predicate || {};
109
127
  const verdict = p.verdict || null;
110
- const covered = p.coverage !== null && p.coverage !== undefined;
128
+ const covered = !!p.coverage && typeof p.coverage === 'object' && Array.isArray(p.coverage.unreached) && p.coverage.unreached.every((u) => typeof u === 'string');
111
129
  const ranOn = new Set(((p.closure && p.closure.head) || []).map((x) => x.sha256));
112
130
  const missing = (c.viaDigests || []).filter((d) => !ranOn.has(d));
113
131
  const because = !PASSING.has(verdict) ? `the statement's verdict is ${verdict}${Array.isArray(p.inconclusive) && p.inconclusive.length ? ` (${p.inconclusive.join('; ')})` : ''}`
@@ -115,9 +133,9 @@ export function checkEquivalence({ baseDir, headDir, allow, files = [], allowed
115
133
  : !covered ? 'the statement measured no coverage of the changed paragraphs, so it is inconclusive'
116
134
  : allowed && !s.signed ? `the statement is ${s.reason}`
117
135
  : null;
118
- const unreached = covered && Array.isArray(p.coverage.unreached) ? p.coverage.unreached : [];
136
+ const unreached = covered ? p.coverage.unreached : [];
119
137
  return { path: c.path, ...via, statement: s.file, verdict, signed: s.signed, signer: s.signer, coverage: covered, ...(unreached.length ? { unreached } : {}), ok: because === null && !unreached.length, ...(because || unreached.length ? { because: because || `the inputs never reached ${unreached.join(', ')}` } : {}) };
120
- });
138
+ }
121
139
  const authorship = mode === 'machineAuthored' && repo && base ? machineAuthored(repo, base, head || 'HEAD', env) : null;
122
140
  const required = mode === 'always' ? true : mode === 'machineAuthored' ? (authorship && authorship.known ? authorship.commits.length > 0 : null) : false;
123
141
  return { mode, required, ...(authorship ? { machineAuthored: authorship } : {}), programs, problems };
@@ -36,12 +36,14 @@ export const KINDS = {
36
36
  dd: { fields: { dd: str, event: (x) => ['open', 'close', 'end'].includes(x), mode: str, sha256: hex64, bytes: int }, required: ['dd', 'event'] },
37
37
  call: { fields: { program: str, from: str, sha256: hex64 }, required: ['program'] },
38
38
  abend: { fields: { code: str, file: str, line: int }, required: ['code'] },
39
+ step: { fields: { step: str, pgm: str, outcome: str }, required: ['step', 'pgm', 'outcome'] },
40
+ sink: { fields: { sink: str, file: str, line: int, marker: str, reached: (x) => typeof x === 'boolean' }, required: ['sink', 'file', 'line', 'marker', 'reached'] },
39
41
  genesis: { fields: { createdAt: str, rotatedFrom: any }, required: ['createdAt'] },
40
42
  run: { fields: { run: str, runChain: hex32, runLength: int, runTip: hex64 }, required: ['run', 'runChain', 'runLength', 'runTip'] },
41
43
  'lock-broken': { fields: { holderPid: int, ageMs: int }, required: ['holderPid', 'ageMs'] },
42
44
  };
43
45
 
44
- export const JOURNAL_KINDS = new Set(['open', 'input', 'finding', 'suppressed', 'baseline-write', 'witness', 'verdict', 'output', 'close', 'dd', 'call', 'abend']);
46
+ export const JOURNAL_KINDS = new Set(['open', 'input', 'finding', 'suppressed', 'baseline-write', 'witness', 'verdict', 'output', 'close', 'dd', 'call', 'abend', 'step', 'sink']);
45
47
  export const LEDGER_KINDS = new Set(['genesis', 'run', 'lock-broken']);
46
48
 
47
49
  export const newChain = () => randomBytes(16).toString('hex');
@@ -20,9 +20,24 @@ function named(path, root) {
20
20
  return abs.startsWith(r + sep) ? relative(r, abs).split(sep).join('/') : basename(abs);
21
21
  }
22
22
 
23
+ // The compiler's arguments with every value replaced: an option keeps its name, a path or other value
24
+ // becomes <value>. The digest of the arguments as run lets a holder of them check the statement.
25
+ function redacted(argv) {
26
+ return argv.map((a) => {
27
+ const s = String(a);
28
+ if (!s.startsWith('-')) return '<value>';
29
+ const cut = s.search(/[=/\\]/);
30
+ return cut < 0 ? s : `${s.slice(0, cut)}${s[cut] === '=' ? '=' : ''}<value>`;
31
+ });
32
+ }
33
+
23
34
  export function slsaStatement({ provenance, docBytes, root, artifacts = [], runId = null, runTip = null, builderId = null, startedOn = null, finishedOn = null }) {
24
35
  const subject = [{ name: 'build.json', digest: { sha256: sha256(docBytes) } },
25
- ...artifacts.map((p) => ({ name: named(p, root), digest: { sha256: sha256(readFileSync(p)) } }))];
36
+ ...artifacts.map((p) => {
37
+ let bytes;
38
+ try { bytes = readFileSync(p); } catch (e) { throw new Error(`the artifact ${basename(p)} could not be read (${e.code || e.message})`); }
39
+ return { name: named(p, root), digest: { sha256: sha256(bytes) } };
40
+ })];
26
41
 
27
42
  const resolvedDependencies = (provenance.sources || []).map((s) => ({ uri: `file:${s.path}`, name: s.path, digest: { sha256: s.sha256 } }));
28
43
  if (provenance.compiler && provenance.compiler.sha256) {
@@ -36,7 +51,7 @@ export function slsaStatement({ provenance, docBytes, root, artifacts = [], runI
36
51
  revisions: provenance.revisions,
37
52
  policy: { sha256: provenance.policy.sha256, setBy: provenance.policy.setBy },
38
53
  ...(provenance.copylibs && provenance.copylibs.length ? { copylibs: provenance.copylibs.map((d) => basename(d)) } : {}),
39
- ...(provenance.compiler && provenance.compiler.argv ? { compilerArguments: provenance.compiler.argv } : {}),
54
+ ...(provenance.compiler && provenance.compiler.argv ? { compilerArguments: redacted(provenance.compiler.argv), compilerArgumentsSha256: sha256(provenance.compiler.argv.map(String).join('\0')) } : {}),
40
55
  };
41
56
  const internalParameters = {
42
57
  tool: 'cobolwork', toolVersion: provenance.toolVersion, flowModel: provenance.flowModel,
@@ -0,0 +1,77 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // Execution: whether the estate's own runs entered the paragraph a finding is in, from the
3
+ // reports `ironwork run --coverage` writes. A finding in a paragraph a run entered is code the tests
4
+ // exercise; one in a paragraph no run entered is code they never reach. See docs/spec/evidence.md §13.4 (execution coverage).
5
+ import { readFileSync } from 'node:fs';
6
+ import { basename, delimiter, join } from 'node:path';
7
+
8
+ export function executionFeedPaths(opts = {}) {
9
+ if (opts.executionFeeds) return opts.executionFeeds;
10
+ return (process.env.COBOLWORK_EXECUTION || '').split(delimiter).filter(Boolean);
11
+ }
12
+
13
+ // One report: per PROGRAM-ID, each paragraph's source line, name and how often a run entered it.
14
+ // Shape (ironwork's): { programs: [{ program, detail: [{ name, line, entered }] }] }.
15
+ export function loadExecutionFeed(path) {
16
+ const feed = { file: basename(path), programs: new Map(), problem: null };
17
+ let doc;
18
+ try { doc = JSON.parse(readFileSync(path, 'utf8')); } catch (e) { feed.problem = e.code ? `could not be read (${e.code})` : `is not JSON (${e.message})`; return feed; }
19
+ if (!doc || !Array.isArray(doc.programs)) { feed.problem = 'holds no programs'; return feed; }
20
+ for (const p of doc.programs) {
21
+ const detail = Array.isArray(p?.detail) ? p.detail : null;
22
+ if (typeof p?.program !== 'string' || !detail || !detail.every((d) => typeof d?.name === 'string' && Number.isInteger(d.line) && Number.isInteger(d.entered) && d.entered >= 0)) {
23
+ feed.problem = `a program's paragraphs are not { name, line, entered }`;
24
+ return feed;
25
+ }
26
+ feed.programs.set(p.program.toUpperCase(), detail.map((d) => ({ name: d.name, line: d.line, entered: d.entered })));
27
+ }
28
+ return feed;
29
+ }
30
+
31
+ const PROGRAM_ID = /^.{0,6}[ \d]?\s*PROGRAM-ID\.?\s+['"]?([A-Za-z0-9#$@-]+)/;
32
+
33
+ // The programs a source holds, by PROGRAM-ID and the line it starts on.
34
+ function programsOf(text) {
35
+ const out = [];
36
+ text.split(/\r?\n/).forEach((line, i) => {
37
+ const m = PROGRAM_ID.exec(line);
38
+ if (m) out.push({ id: m[1].toUpperCase(), line: i + 1 });
39
+ });
40
+ return out;
41
+ }
42
+
43
+ // Each finding inside a paragraph of a covered program gets `executed: { paragraph, entered }`,
44
+ // entered summed over the feeds. A finding before the program's first paragraph, or in a program
45
+ // no feed covers, is left as it is.
46
+ export function applyExecution(findings, root, feeds) {
47
+ const good = feeds.filter((f) => !f.problem);
48
+ const byExecution = { entered: 0, 'never-entered': 0 };
49
+ if (!good.length) return { byExecution: null };
50
+ const sources = new Map();
51
+ const programs = (path) => {
52
+ if (!sources.has(path)) {
53
+ let text = null;
54
+ try { text = readFileSync(join(root, path), 'utf8'); } catch { /* not a file the scan can read again */ }
55
+ sources.set(path, text === null ? [] : programsOf(text));
56
+ }
57
+ return sources.get(path);
58
+ };
59
+ for (const f of findings) {
60
+ if (!f.path || !Number.isInteger(f.line)) continue;
61
+ const owner = programs(f.path).filter((p) => p.line <= f.line && good.some((g) => g.programs.has(p.id))).at(-1);
62
+ if (!owner) continue;
63
+ let paragraph = null;
64
+ let entered = 0;
65
+ for (const g of good) {
66
+ const detail = g.programs.get(owner.id);
67
+ const at = detail?.filter((d) => d.line >= owner.line && d.line <= f.line).at(-1);
68
+ if (!at) continue;
69
+ paragraph = at.name;
70
+ entered += at.entered;
71
+ }
72
+ if (paragraph === null) continue;
73
+ f.executed = { paragraph, entered };
74
+ byExecution[entered > 0 ? 'entered' : 'never-entered']++;
75
+ }
76
+ return { byExecution };
77
+ }
package/lib/explain.mjs CHANGED
@@ -7,7 +7,7 @@ import { EVIDENCE } from './kernel/findings.mjs';
7
7
  import { REGISTRY } from './kernel/registry.mjs';
8
8
  import { WHO_ACTS } from './tui/model.mjs';
9
9
  import { verificationPlan } from './verify.mjs';
10
- import { cobolCard, statementCard } from './card.mjs';
10
+ import { cobolCard, statementCard } from './cards.mjs';
11
11
 
12
12
  const HIDDEN_RULES = new Set(Object.keys(REGISTRY.find((s) => s.name === 'hidden').rules));
13
13
 
package/lib/ftp.mjs CHANGED
@@ -1,5 +1,5 @@
1
1
  // SPDX-License-Identifier: AGPL-3.0-or-later
2
- import { statementCard } from './card.mjs';
2
+ import { statementCard } from './cards.mjs';
3
3
 
4
4
  // IBM's batch FTP client takes the host and its options on PARM, or the host as the first line of
5
5
  // its input, reads subcommands from INPUT, and logs on from NETRC or from its input. CardDemo's job