@portll/cobolwork 0.0.1 → 0.2.75

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/LICENSE +661 -0
  2. package/LICENSING.md +93 -0
  3. package/NOTICE +9 -0
  4. package/README.md +323 -3
  5. package/THIRD-PARTY-NOTICES.md +118 -0
  6. package/bin/cobolwork.mjs +354 -0
  7. package/lib/advisories.mjs +133 -0
  8. package/lib/baseline.mjs +154 -0
  9. package/lib/bms.mjs +453 -0
  10. package/lib/build.mjs +402 -0
  11. package/lib/capabilities.mjs +79 -0
  12. package/lib/cics-commands.mjs +281 -0
  13. package/lib/compliance.mjs +81 -0
  14. package/lib/consequence.mjs +139 -0
  15. package/lib/control.mjs +1515 -0
  16. package/lib/csd.mjs +77 -0
  17. package/lib/dataflow.mjs +1506 -0
  18. package/lib/diff.mjs +344 -0
  19. package/lib/explain.mjs +145 -0
  20. package/lib/gate.mjs +383 -0
  21. package/lib/index.mjs +6 -0
  22. package/lib/inventory.mjs +79 -0
  23. package/lib/jcl.mjs +478 -0
  24. package/lib/kernel/findings.mjs +94 -0
  25. package/lib/kernel/identity.mjs +216 -0
  26. package/lib/kernel/memory.mjs +217 -0
  27. package/lib/kernel/printable.mjs +6 -0
  28. package/lib/kernel/registry.mjs +79 -0
  29. package/lib/kernel/ruleset.mjs +72 -0
  30. package/lib/kernel/source-tree.mjs +159 -0
  31. package/lib/kev.mjs +27 -0
  32. package/lib/options.mjs +512 -0
  33. package/lib/packs.mjs +148 -0
  34. package/lib/parser.mjs +2055 -0
  35. package/lib/policy.mjs +163 -0
  36. package/lib/precompile-cics.mjs +169 -0
  37. package/lib/precompile.mjs +544 -0
  38. package/lib/reach.mjs +122 -0
  39. package/lib/revision.json +1 -0
  40. package/lib/revision.mjs +89 -0
  41. package/lib/sarif.mjs +222 -0
  42. package/lib/scan.mjs +272 -0
  43. package/lib/sets/build.mjs +234 -0
  44. package/lib/sets/cics.mjs +306 -0
  45. package/lib/sets/compile.mjs +187 -0
  46. package/lib/sets/copybook.mjs +174 -0
  47. package/lib/sets/flow.mjs +487 -0
  48. package/lib/sets/hidden.mjs +216 -0
  49. package/lib/sets/jcl.mjs +440 -0
  50. package/lib/sets/log.mjs +406 -0
  51. package/lib/sets/opaque.mjs +102 -0
  52. package/lib/sets/priv.mjs +322 -0
  53. package/lib/sets/recon.mjs +267 -0
  54. package/lib/sets/vendor.mjs +117 -0
  55. package/lib/sets/web.mjs +327 -0
  56. package/lib/site.mjs +164 -0
  57. package/lib/sources.mjs +156 -0
  58. package/lib/tui/app.mjs +325 -0
  59. package/lib/tui/keys.mjs +39 -0
  60. package/lib/tui/model.mjs +96 -0
  61. package/lib/tui/run.mjs +38 -0
  62. package/lib/tui/screen.mjs +59 -0
  63. package/lib/tui/terminal.mjs +46 -0
  64. package/lib/utilities.mjs +296 -0
  65. package/lib/version.mjs +15 -0
  66. package/lib/words.mjs +318 -0
  67. package/package.json +45 -6
  68. package/rules/advisories.json +264 -0
  69. package/rules/compliance-dora.json +2151 -0
  70. package/rules/compliance-ffiec.json +2134 -0
  71. package/rules/compliance-nist80053.json +2134 -0
  72. package/rules/gitleaks-mainframe.toml +57 -0
  73. package/rules/kev-ids.json +1729 -0
  74. package/rules/packs/broadcom.json +124 -0
  75. package/rules/packs/connectdirect.json +116 -0
  76. package/rules/packs/controlm.json +114 -0
  77. package/rules/system-layouts.json +28 -0
  78. package/schema/cobolwork-coverage.schema.json +65 -0
  79. package/schema/cobolwork.policy.schema.json +54 -0
@@ -0,0 +1 @@
1
+ {"commit":"b74e3baeadbf548f73f45b0de0212b0579793d3c","tag":"v0.2.75"}
@@ -0,0 +1,89 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // Which commit a report was made by, and which commit it read: summary.toolRevision and
3
+ // summary.revision, each { commit, dirty }. A scanned repository is untrusted, and its .git/config
4
+ // can name commands that porcelain runs - `git status` runs clean filters and fsmonitor - so only
5
+ // plumbing that lists objects and files runs here, and whether the tree differs from HEAD is decided
6
+ // by hashing its files in process.
7
+ import { createHash } from 'node:crypto';
8
+ import { spawnSync } from 'node:child_process';
9
+ import { existsSync, lstatSync, readFileSync, readlinkSync, realpathSync } from 'node:fs';
10
+ import { join } from 'node:path';
11
+ import { fileURLToPath } from 'node:url';
12
+
13
+ const GIT_ENV = { ...process.env, GIT_OPTIONAL_LOCKS: '0', GIT_TERMINAL_PROMPT: '0' };
14
+ const git = (dir, args) => spawnSync('git', ['-c', 'core.fsmonitor=false', '-C', dir, ...args], { env: GIT_ENV, maxBuffer: 256 << 20 });
15
+ // Past this many tracked files the comparison is not made, and dirty is null rather than a guess.
16
+ const MAX_TRACKED = 50000;
17
+
18
+ const nulSplit = (buf) => buf.toString('utf8').split('\0').filter(Boolean);
19
+
20
+ // The object id git would give these bytes as a blob.
21
+ const blobId = (bytes, format) => createHash(format).update(`blob ${bytes.length}\0`).update(bytes).digest('hex');
22
+
23
+ // { commit, dirty } for the repository `dir` is in, over `paths` under it when given, or null
24
+ // outside a repository. dirty is true when a tracked file under `dir` differs from HEAD or is gone,
25
+ // or a file git does not ignore is untracked; null when there were too many files to compare. A
26
+ // checkout that converts line endings reads as dirty, which errs the safe way.
27
+ export function revisionOf(dir, { paths = [] } = {}) {
28
+ const head = git(dir, ['rev-parse', '--verify', '-q', 'HEAD']);
29
+ if (head.status !== 0) return null;
30
+ const commit = head.stdout.toString().trim();
31
+ const fmt = git(dir, ['rev-parse', '--show-object-format']);
32
+ const format = fmt.status === 0 && fmt.stdout.toString().trim() === 'sha256' ? 'sha256' : 'sha1';
33
+ const scope = paths.length ? ['--', ...paths] : [];
34
+ const listed = git(dir, ['ls-tree', '-r', '-z', 'HEAD', ...scope]);
35
+ if (listed.status !== 0) return { commit, dirty: null };
36
+ const entries = nulSplit(listed.stdout);
37
+ if (entries.length > MAX_TRACKED) return { commit, dirty: null };
38
+ for (const entry of entries) {
39
+ const tab = entry.indexOf('\t');
40
+ const [mode, type, id] = entry.slice(0, tab).split(' ');
41
+ const rel = entry.slice(tab + 1);
42
+ if (type !== 'blob') continue;
43
+ const file = join(dir, rel);
44
+ let st;
45
+ try { st = lstatSync(file); } catch { return { commit, dirty: true }; }
46
+ const bytes = mode === '120000'
47
+ ? (st.isSymbolicLink() ? Buffer.from(readlinkSync(file)) : null)
48
+ : (st.isFile() ? readFileSync(file) : null);
49
+ if (!bytes || blobId(bytes, format) !== id) return { commit, dirty: true };
50
+ }
51
+ const untracked = git(dir, ['ls-files', '-z', '--others', '--exclude-standard', ...scope]);
52
+ if (untracked.status !== 0) return { commit, dirty: null };
53
+ return { commit, dirty: nulSplit(untracked.stdout).length > 0 };
54
+ }
55
+
56
+ // { commit, dirty: false } for a named revision, which is a commit and nothing on disk; null if it
57
+ // does not resolve.
58
+ export function commitAt(dir, ref) {
59
+ if (!ref || String(ref).startsWith('-')) return null;
60
+ const r = git(dir, ['rev-parse', '--verify', '-q', `${ref}^{commit}`]);
61
+ return r.status === 0 ? { commit: r.stdout.toString().trim(), dirty: false } : null;
62
+ }
63
+
64
+ const PACKAGE_ROOT = fileURLToPath(new URL('..', import.meta.url));
65
+ // What decides a report: the code, the rules it loads and the schemas it names.
66
+ export const SHIPPED = ['bin', 'lib', 'rules', 'schema', 'package.json'];
67
+ export const PACKED_REVISION = join(PACKAGE_ROOT, 'lib', 'revision.json');
68
+
69
+ // Whether this cobolwork is its own checkout, not a package installed somewhere inside another
70
+ // repository, whose HEAD would say nothing about this code.
71
+ function isOwnCheckout() {
72
+ const top = git(PACKAGE_ROOT, ['rev-parse', '--show-toplevel']);
73
+ if (top.status !== 0) return false;
74
+ try { return realpathSync(top.stdout.toString().trim()) === realpathSync(PACKAGE_ROOT); } catch { return false; }
75
+ }
76
+
77
+ // The revision this cobolwork runs from: read from the checkout it runs in, over the files that
78
+ // decide what it reports, or stamped into a packed release by diag/stamp-revision.mjs.
79
+ export function toolRevision() {
80
+ if (isOwnCheckout()) {
81
+ const r = revisionOf(PACKAGE_ROOT, { paths: SHIPPED });
82
+ return r ? { ...r, from: 'checkout' } : null;
83
+ }
84
+ if (!existsSync(PACKED_REVISION)) return null;
85
+ try {
86
+ const r = JSON.parse(readFileSync(PACKED_REVISION, 'utf8'));
87
+ return { commit: r.commit, dirty: false, ...(r.tag ? { tag: r.tag } : {}), from: 'release' };
88
+ } catch { return null; }
89
+ }
package/lib/sarif.mjs ADDED
@@ -0,0 +1,222 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ // The report as SARIF 2.1.0.
3
+ //
4
+ // SARIF already models nearly everything this project invented separately, and this file used to
5
+ // translate rather than project. Three corrections:
6
+ //
7
+ // a rule set is a toolComponent. SARIF has first-class support for a tool made of several
8
+ // components - `tool.driver` plus `tool.extensions[]` - each with its own rules and its own
9
+ // notifications. That is exactly what a cobolwork rule set is, and modelling it means a consumer
10
+ // can tell which analysis produced a result without parsing its id.
11
+ //
12
+ // files nobody read are notifications. `invocations[].toolExecutionNotifications` is the
13
+ // standard place for "the tool could not read these", and it is where the sentence lib/memory.mjs
14
+ // composes belongs. It used to go into a property bag no SARIF consumer reads.
15
+ //
16
+ // a set that could not run is a configuration notification. `toolConfigurationNotifications`
17
+ // says the tool was not told enough, which is a different claim from having read less than it
18
+ // meant to, and the two call for different actions from different people.
19
+ //
20
+ // What SARIF cannot say is coverage as a ratio with a denominator: "read 8,000 of 12,000, stopped
21
+ // on a memory reserve". That quadruple travels in a namespaced property bag. It does NOT travel
22
+ // there alone: `coverageIncomplete` is this project's distinguishing promise - a finding count over
23
+ // source nobody read is not a clean result - so the verdict stays where nobody can miss it, in
24
+ // `executionSuccessful` and as a flat property, and the property bag carries the detail.
25
+ import { REGISTRY, ALL_RULES } from './kernel/registry.mjs';
26
+ import { stoppedBecause } from './kernel/memory.mjs';
27
+ import { FINGERPRINT_VERSION } from './kernel/identity.mjs';
28
+ import { DIFF_RULES } from './diff.mjs';
29
+
30
+ const LEVEL = { crit: 'error', high: 'error', med: 'warning', low: 'note', info: 'note' };
31
+
32
+ // Locations are URI references: each segment is encoded, lone surrogates from Windows names replaced first.
33
+ const LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/g;
34
+ const uriOf = (p) => String(p).split('/').map((seg) => encodeURIComponent(seg.replace(LONE_SURROGATE, '\uFFFD'))).join('/');
35
+ // Square brackets in a SARIF message are link syntax, and a finding's text can quote the tree.
36
+ const textOf = (s) => (typeof s === 'string' ? s.replace(/[[\]]/g, '\\$&') : s);
37
+ // SARIF's three levels fold critical into high and info into low, so the severity also travels as
38
+ // the number GitHub code scanning ranks by. Info is coverage and context, which rank as nothing.
39
+ const SECURITY_SEVERITY = { crit: '9.5', high: '8.0', med: '5.5', low: '3.0' };
40
+ const DECLARED = new Map([...Object.entries(ALL_RULES), ...Object.entries(DIFF_RULES)].map(([id, r]) => [id, r.sev]));
41
+
42
+ // Which rule set declares each rule. Rules with no set - diff's, and the credential pack's - stay
43
+ // on the driver, because they are not produced by a rule set.
44
+ const SET_OF = new Map();
45
+ for (const s of REGISTRY) for (const id of Object.keys(s.rules)) SET_OF.set(id, s.name);
46
+
47
+ // SARIF's level cannot say what kind of claim a result is, so evidence rides in the property bags.
48
+ const descriptorFor = (report) => (id) => {
49
+ const properties = {
50
+ ...(report.ruleCwe?.[id] ? { cwe: report.ruleCwe[id] } : {}),
51
+ ...(report.ruleEvidence?.[id] ? { evidence: report.ruleEvidence[id] } : {}),
52
+ ...(SECURITY_SEVERITY[DECLARED.get(id)] ? { 'security-severity': SECURITY_SEVERITY[DECLARED.get(id)] } : {}),
53
+ };
54
+ return {
55
+ id,
56
+ shortDescription: { text: report.ruleText?.[id] || id },
57
+ // What the construct lets someone do, and the standard fix - the same for every result of the
58
+ // rule. `help` is where GitHub code scanning shows a recommendation.
59
+ ...(report.ruleImpact?.[id] ? { fullDescription: { text: textOf(report.ruleImpact[id]) } } : {}),
60
+ ...(report.ruleRemedy?.[id] ? { help: { text: textOf(report.ruleRemedy[id]) } } : {}),
61
+ ...(Object.keys(properties).length ? { properties } : {}),
62
+ };
63
+ };
64
+
65
+ function notifications(report) {
66
+ const bySet = report.summary?.bySet || {};
67
+ const execution = [];
68
+ const configuration = [];
69
+
70
+ for (const [set, s] of Object.entries(bySet)) {
71
+ const shortfall = (s.filesNotRead || 0) + (s.filesUnreadable || 0) + (s.filesUnparsed || 0);
72
+ if (!shortfall) continue;
73
+ execution.push({
74
+ descriptor: { id: 'cobolwork/files-not-read' },
75
+ level: 'warning',
76
+ message: {
77
+ text: s.notRead
78
+ || `${set}: ${shortfall} file(s) did not contribute to this run`
79
+ + (s.stoppedBy ? `, because ${stoppedBecause(s.stoppedBy)}` : ''),
80
+ },
81
+ properties: {
82
+ set,
83
+ ...(s.filesNotRead ? { filesNotRead: s.filesNotRead } : {}),
84
+ ...(s.filesUnreadable ? { filesUnreadable: s.filesUnreadable } : {}),
85
+ ...(s.filesUnparsed ? { filesUnparsed: s.filesUnparsed } : {}),
86
+ ...(s.stoppedBy ? { stoppedBy: s.stoppedBy } : {}),
87
+ ...(s.unreadable?.length ? { unreadable: s.unreadable } : {}),
88
+ ...(s.unparsed?.length ? { unparsed: s.unparsed } : {}),
89
+ },
90
+ });
91
+ }
92
+
93
+ // A set that could not run at all, as distinct from one that ran and read less than the tree.
94
+ const named = new Set(execution.map((n) => n.properties.set));
95
+ for (const { set, kind, why } of report.summary?.setsIncomplete || []) {
96
+ if (named.has(set)) continue;
97
+ if (kind === 'coverage') {
98
+ execution.push({
99
+ descriptor: { id: 'cobolwork/read-in-part' },
100
+ level: 'warning',
101
+ message: { text: textOf(`${set}: ${why}`) },
102
+ properties: { set },
103
+ });
104
+ continue;
105
+ }
106
+ configuration.push({
107
+ descriptor: { id: 'cobolwork/rule-set-did-not-run' },
108
+ level: 'warning',
109
+ message: { text: textOf(`${set}: ${why}`) },
110
+ properties: { set },
111
+ });
112
+ }
113
+ // A diff whose two sides were scanned under different configuration.
114
+ for (const { file, change } of report.summary?.configurationChanged || []) {
115
+ configuration.push({
116
+ descriptor: { id: 'cobolwork/configuration-changed' },
117
+ level: 'warning',
118
+ message: { text: textOf(`${file} ${change}`) },
119
+ properties: { file },
120
+ });
121
+ }
122
+ return { execution, configuration };
123
+ }
124
+
125
+ export function toSarif(report, { toolVersion = '0.1.0' } = {}) {
126
+ const describe = descriptorFor(report);
127
+ // A suppressed finding is still a result, marked as one a person judged, so its rule is described.
128
+ const suppressed = report.suppressed || [];
129
+ const present = [...new Set([...report.findings, ...suppressed].map((f) => f.rule))].sort();
130
+
131
+ // One component per rule set that produced anything, in registry order so two runs of the same
132
+ // tree compare.
133
+ const sets = REGISTRY.map((s) => s.name).filter((name) => present.some((id) => SET_OF.get(id) === name));
134
+ const extensions = sets.map((name) => ({
135
+ name: `cobolwork-${name}`,
136
+ version: toolVersion,
137
+ rules: present.filter((id) => SET_OF.get(id) === name).map(describe),
138
+ }));
139
+ const indexOfSet = new Map(sets.map((name, i) => [name, i]));
140
+
141
+ // Anything no rule set declares belongs to the driver: diff's rules, and the credential pack's.
142
+ const driverRules = present.filter((id) => !SET_OF.has(id)).map(describe);
143
+
144
+ const { execution, configuration } = notifications(report);
145
+ const s = report.summary || {};
146
+
147
+ return {
148
+ $schema: 'https://json.schemastore.org/sarif-2.1.0.json',
149
+ version: '2.1.0',
150
+ runs: [{
151
+ tool: {
152
+ driver: {
153
+ name: 'cobolwork',
154
+ version: toolVersion,
155
+ informationUri: 'https://github.com/Portll/cobolwork',
156
+ rules: driverRules,
157
+ },
158
+ ...(extensions.length ? { extensions } : {}),
159
+ },
160
+ invocations: [{
161
+ // The verdict, where a consumer that reads nothing else will still see it.
162
+ executionSuccessful: s.nosrc !== true,
163
+ exitCode: 0,
164
+ ...(execution.length ? { toolExecutionNotifications: execution } : {}),
165
+ ...(configuration.length ? { toolConfigurationNotifications: configuration } : {}),
166
+ properties: {
167
+ ...(s.filesScanned === undefined ? {} : { filesScanned: s.filesScanned }),
168
+ ...(s.coverageIncomplete === undefined ? {} : { coverageIncomplete: s.coverageIncomplete }),
169
+ ...(s.copiesMissing ? { copiesMissing: s.copiesMissing } : {}),
170
+ ...(s.filesUnreadable ? { filesUnreadable: s.filesUnreadable } : {}),
171
+ ...(s.advisoryFeeds ? { 'cobolwork/advisoryFeeds': s.advisoryFeeds } : {}),
172
+ ...(s.advisoryCoverage ? { 'cobolwork/advisoryCoverage': s.advisoryCoverage } : {}),
173
+ ...(s.toolRevision !== undefined ? { 'cobolwork/toolRevision': s.toolRevision } : {}),
174
+ ...(s.revision !== undefined ? { 'cobolwork/revision': s.revision } : {}),
175
+ // The part SARIF has no vocabulary for. Namespaced so it cannot be mistaken for
176
+ // standard fields, and documented in schema/cobolwork-coverage.schema.json.
177
+ 'cobolwork/coverage': {
178
+ coverageIncomplete: s.coverageIncomplete === true,
179
+ filesScanned: s.filesScanned ?? null,
180
+ ...(s.bySet ? { bySet: s.bySet } : {}),
181
+ ...(s.setsIncomplete?.length ? { setsIncomplete: s.setsIncomplete } : {}),
182
+ },
183
+ },
184
+ }],
185
+ results: [...report.findings, ...suppressed].map((f) => {
186
+ const set = SET_OF.get(f.rule);
187
+ const component = set !== undefined ? indexOfSet.get(set) : undefined;
188
+ const within = component === undefined ? null : extensions[component].rules.findIndex((r) => r.id === f.rule);
189
+ return {
190
+ ruleId: f.rule,
191
+ // Which component declares this rule. A consumer that resolves ruleId against the driver
192
+ // still works; one that follows the reference learns which analysis produced the result.
193
+ ...(within != null && within >= 0
194
+ ? { rule: { id: f.rule, index: within, toolComponent: { index: component } } }
195
+ : {}),
196
+ level: LEVEL[f.sev] || 'warning',
197
+ message: { text: textOf(f.detail) },
198
+ // The whole identity, computed by the tool, so it is both the fingerprint and the one
199
+ // partial fingerprint a result-management system needs to track it.
200
+ ...(f.fingerprint ? {
201
+ fingerprints: { [FINGERPRINT_VERSION]: f.fingerprint },
202
+ partialFingerprints: { [FINGERPRINT_VERSION]: f.fingerprint },
203
+ } : {}),
204
+ // Per result: an info pack rule is context under a rule id that is a construct, and a
205
+ // checked path is a step below what its rule declares.
206
+ ...(f.evidence || f.sev ? { properties: {
207
+ ...(f.evidence ? { evidence: f.evidence } : {}),
208
+ ...(f.sev ? { severity: f.sev } : {}),
209
+ ...(SECURITY_SEVERITY[f.sev] ? { 'security-severity': SECURITY_SEVERITY[f.sev] } : {}),
210
+ ...(f.guard ? { guard: f.guard, guardedFrom: f.guardedFrom } : {}),
211
+ // Who can drive the finding and what it runs as, when the estate declared the facts.
212
+ ...(f.reach ? { reach: f.reach } : {}),
213
+ ...(f.effect ? { effect: f.effect } : {}),
214
+ } } : {}),
215
+ locations: [{ physicalLocation: { artifactLocation: { uri: uriOf(f.path), uriBaseId: '%SRCROOT%' }, region: { startLine: f.line || 1 } } }],
216
+ ...(f.related ? { relatedLocations: f.related.map((r) => ({ physicalLocation: { artifactLocation: { uri: uriOf(r.path), uriBaseId: '%SRCROOT%' }, region: { startLine: r.line || 1 } }, message: { text: textOf(r.detail) } })) } : {}),
217
+ ...(f.suppressed ? { suppressions: [{ kind: 'external', status: 'accepted', justification: `${f.suppressed.action} by ${f.suppressed.who} until ${f.suppressed.expires}: ${f.suppressed.reason}` }] } : {}),
218
+ };
219
+ }),
220
+ }],
221
+ };
222
+ }
package/lib/scan.mjs ADDED
@@ -0,0 +1,272 @@
1
+ // SPDX-License-Identifier: AGPL-3.0-or-later
2
+ import { join } from 'node:path';
3
+ import { sortFindings, tally, evidenceMap, EVIDENCE } from './kernel/findings.mjs';
4
+ import { directoryTree } from './kernel/source-tree.mjs';
5
+ import { REGISTRY, RULE_SETS, ALL_RULES, reportKey } from './kernel/registry.mjs';
6
+ import { inventory } from './inventory.mjs';
7
+ import { stoppedBecause } from './kernel/memory.mjs';
8
+ import { complianceMap, FRAMEWORKS_LOADED, FRAMEWORKS_UNLOADED } from './compliance.mjs';
9
+ import { stampFingerprints, FINGERPRINT_VERSION } from './kernel/identity.mjs';
10
+ import { report } from './kernel/ruleset.mjs';
11
+ import { printable } from './kernel/printable.mjs';
12
+ import { rollup } from './sets/flow.mjs';
13
+ import { loadSite } from './site.mjs';
14
+ import { reachFeedPaths, loadReachFeed, reachResolver, stampReach, stampEffect } from './reach.mjs';
15
+
16
+ import { FLOW_MODEL, SCHEMA_VERSION, TOOL_VERSION } from './version.mjs';
17
+
18
+ export { SCHEMA_VERSION };
19
+
20
+ export { ALL_RULES, RULE_SETS };
21
+
22
+ // One report for every rule set, in the shape a consumer can read without knowing which set a
23
+ // finding came from. filesScanned and coverageIncomplete travel with it: a scan of a tree whose
24
+ // copybooks are missing has read less than it appears to have read, and must say so.
25
+ // With several repositories, each is scanned on its own and the reports are merged: program and
26
+ // copybook names are only unique within a repository, so one graph over all of them would link a
27
+ // CALL in one to a same-named program in another and report copybooks in different repositories as
28
+ // shadowing each other.
29
+ // The repository prefix is joined the same way every other report path is built: with a forward
30
+ // slash. path.join would use the platform separator here and rewrite the separators already in the
31
+ // path, so one file would print two ways depending on whether --repos was passed.
32
+ const underRepo = (repo, p) => (repo ? `${repo}/${p}` : p);
33
+
34
+ // Frameworks that did not load; a reason can quote the unparsed file.
35
+ const COMPLIANCE_UNLOADED = FRAMEWORKS_UNLOADED.map((f) => ({ id: f.id, why: printable(f.why) }));
36
+
37
+ function scanEach(root, opts) {
38
+ const reports = opts.repos.map(repo => ({ repo, r: scanAll(join(root, repo), { ...opts, repos: null, repoName: repo }) }));
39
+ const findings = [];
40
+ const checked = [];
41
+ const prefixed = (repo, f) => ({ ...f, path: underRepo(repo, f.path), ...(f.related ? { related: f.related.map(x => ({ ...x, path: underRepo(repo, x.path) })) } : {}) });
42
+ for (const { repo, r } of reports) {
43
+ for (const f of r.findings) findings.push(prefixed(repo, f));
44
+ for (const f of r.checked || []) checked.push(prefixed(repo, f));
45
+ }
46
+ sortFindings(findings);
47
+ const shared = reports.reduce((n, { r }) => n + (r.summary.identity?.shared || 0), 0);
48
+ const sum = (k) => reports.reduce((n, { r }) => n + (r.summary[k] || 0), 0);
49
+ const byRule = tally(findings);
50
+ const bySeverity = {};
51
+ for (const f of findings) bySeverity[f.sev] = (bySeverity[f.sev] || 0) + 1;
52
+ const byEvidence = {};
53
+ for (const f of findings) byEvidence[f.evidence] = (byEvidence[f.evidence] || 0) + 1;
54
+ const perRepo = Object.fromEntries(reports.map(({ repo, r }) => [repo, { findings: r.summary.findings, filesScanned: r.summary.filesScanned, coverageIncomplete: r.summary.coverageIncomplete, nosrc: r.summary.nosrc }]));
55
+ const filesScanned = sum('filesScanned');
56
+ return {
57
+ tool: 'cobolwork',
58
+ schemaVersion: SCHEMA_VERSION,
59
+ summary: {
60
+ findings: findings.length, byRule, bySeverity, byEvidence, repos: reports.length, perRepo,
61
+ filesScanned, programs: sum('programs'), programFiles: sum('programFiles'), copybookFiles: sum('copybookFiles'),
62
+ jclFiles: sum('jclFiles'), copiesMissing: sum('copiesMissing'), filesUnreadable: sum('filesUnreadable'),
63
+ filesEbcdic: sum('filesEbcdic'), filesOverBudget: sum('filesOverBudget'), crossProgram: sum('crossProgram'),
64
+ ruleSets: opts.only || RULE_SETS, nosrc: filesScanned === 0,
65
+ coverageIncomplete: reports.some(({ r }) => r.summary.coverageIncomplete),
66
+ identity: { version: FINGERPRINT_VERSION, shared },
67
+ ...(reports[0]?.r.summary.advisoryFeeds ? { advisoryFeeds: reports[0].r.summary.advisoryFeeds } : {}),
68
+ ...(reports[0]?.r.summary.advisoryCoverage ? { advisoryCoverage: reports[0].r.summary.advisoryCoverage } : {}),
69
+ // The schema has no repository field for a set, so the repository leads the reason.
70
+ setsIncomplete: reports.flatMap(({ repo, r }) => (r.summary.setsIncomplete || []).map((x) => ({ ...x, why: `${repo}: ${x.why}` }))),
71
+ },
72
+ findings,
73
+ ...(checked.length ? { checked } : {}),
74
+ ruleText: Object.fromEntries(Object.entries(ALL_RULES).map(([k, v]) => [k, v.text])),
75
+ ruleCwe: Object.fromEntries(Object.entries(ALL_RULES).filter(([, v]) => v.cwe).map(([k, v]) => [k, v.cwe])),
76
+ ruleEvidence: evidenceMap(ALL_RULES),
77
+ ruleImpact: Object.fromEntries(Object.entries(ALL_RULES).filter(([, v]) => v.impact).map(([k, v]) => [k, v.impact])),
78
+ ruleRemedy: Object.fromEntries(Object.entries(ALL_RULES).filter(([, v]) => v.remedy).map(([k, v]) => [k, v.remedy])),
79
+ evidence: EVIDENCE,
80
+ // Which clause each rule bears on, and what the mapping does not claim. A reader who takes
81
+ // this to an auditor needs both halves.
82
+ ruleCompliance: complianceMap(Object.keys(ALL_RULES)),
83
+ compliance: FRAMEWORKS_LOADED,
84
+ ...(COMPLIANCE_UNLOADED.length ? { complianceUnloaded: COMPLIANCE_UNLOADED } : {}),
85
+ };
86
+ }
87
+
88
+ // A set that throws is reported as not having run, and the other sets' findings stand.
89
+ function runSet(set, root, o) {
90
+ try { return set.scan(root, o); } catch (e) {
91
+ return report(set.name, { rules: set.rules, findings: [], stats: {
92
+ coverageIncomplete: true,
93
+ notRead: `the rule set stopped on an error and reported nothing: ${printable(e?.message ?? e)}`,
94
+ } });
95
+ }
96
+ }
97
+
98
+ export function scanAll(root, opts = {}) {
99
+ if (opts.repos && opts.repos.length > 1) return scanEach(root, opts);
100
+ const only = opts.only || RULE_SETS;
101
+ // One walk of the tree for the whole scan. Every rule set used to call buildFileIndex(root) for
102
+ // itself, so a scan read every directory entry and resolved every symlink nine times over, and
103
+ // added a tenth walk each time a rule set was added.
104
+ const tree = opts.tree || directoryTree(root, opts);
105
+ const o = { ...opts, tree };
106
+ const inv = inventory(root, o);
107
+ // One loop over the registry, in its order. A set that is not registered cannot run, and
108
+ // cannot report a silent zero on its own behalf either.
109
+ const parts = REGISTRY.filter((set) => only.includes(set.name)).map((set) => runSet(set, root, o));
110
+
111
+ const flowPart = parts.find(p => p.tool === 'cobolwork-flow');
112
+ const overBudget = flowPart ? flowPart.summary.filesOverBudget || 0 : 0;
113
+ // Each rule set reads a different slice of the tree: the hidden-content rules read JCL and
114
+ // copybooks that no program-based set does. The count that decides whether anything was examined
115
+ // is the widest of them, so a JCL-only tree is scanned rather than reported as holding nothing.
116
+ const bySet = { inventory: { filesScanned: inv.summary.filesScanned, filesUnreadable: inv.summary.filesUnreadable } };
117
+ // A rule set's own tool name is what keys its counts. When one was renamed and this map was not,
118
+ // that set's counts landed under the key `undefined` in every report. Refusing a name with no key
119
+ // means the next rename fails a test instead of shipping.
120
+ for (const p of parts) {
121
+ const name = reportKey(p.tool);
122
+ const s = p.summary;
123
+ // What each set read, and what it did not. The projection used to copy three keys, so the
124
+ // sentence lib/memory.mjs composes when a scan stops early - which files, and why - was written
125
+ // and then discarded here. A report that says coverage is incomplete without saying where is a
126
+ // report that cannot be acted on.
127
+ bySet[name] = {
128
+ filesScanned: s.filesScanned || 0,
129
+ filesUnreadable: s.filesUnreadable || 0,
130
+ ...(s.filesBinary ? { filesBinary: s.filesBinary } : {}),
131
+ ...(s.filesNotRead ? { filesNotRead: s.filesNotRead } : {}),
132
+ ...(s.filesUnparsed ? { filesUnparsed: s.filesUnparsed } : {}),
133
+ ...(s.stoppedBy ? { stoppedBy: s.stoppedBy } : {}),
134
+ ...(s.unreadable?.length ? { unreadable: s.unreadable } : {}),
135
+ ...(s.unparsed?.length ? { unparsed: s.unparsed } : {}),
136
+ ...(s.notRead ? { notRead: s.notRead } : {}),
137
+ };
138
+ }
139
+ const filesRead = Math.max(...Object.values(bySet).map(x => x.filesScanned));
140
+ const unreadInSomeSet = Object.values(bySet).some(x => x.filesUnreadable > 0);
141
+ // Severity and CWE arrive already on the finding, stamped by the set whose table declares the
142
+ // rule. This used to impute both here, defaulting an unrecognised rule id to 'med', which meant
143
+ // a set called on its own returned findings with no severity at all and a typo became a medium
144
+ // finding nobody had written.
145
+ const findings = [];
146
+ // A finding any set places in a program is reached by whatever starts that program, which only
147
+ // the flow set knows.
148
+ const startedBy = parts.find((p) => p.tool === 'cobolwork-flow')?.startedBy || {};
149
+ for (const p of parts) for (const f of p.findings) findings.push(!f.startedBy && f.program && startedBy[f.program] ? { ...f, startedBy: startedBy[f.program] } : { ...f });
150
+ sortFindings(findings);
151
+ const identity = stampFingerprints(findings, { root, repo: opts.repoName || '' });
152
+ // Routes a check stops keep their identity too, so a finding a fix clears can be found again here.
153
+ const checked = parts.find((p) => p.tool === 'cobolwork-flow')?.checked || [];
154
+ stampFingerprints(checked, { root, repo: opts.repoName || '' });
155
+
156
+ // Who can reach a finding: whether the transactions and jobs that start it are open to any user
157
+ // or restricted, from the estate's own facts - a brought COBOLWORK_REACH extract, or the
158
+ // open/restricted keys in cobolwork.site.json - never from the source. Undeclared until it says.
159
+ const site = loadSite(root, opts.site || null);
160
+ const reachFeeds = reachFeedPaths(opts).map((p) => loadReachFeed(p, { root }));
161
+ const reach = reachResolver({ feeds: reachFeeds, site });
162
+ const withEntries = findings.some((f) => (f.startedBy || []).length);
163
+ const byReach = reach.declared ? stampReach(findings, reach) : null;
164
+ const byEffect = reach.effectDeclared ? stampEffect(findings, reach) : null;
165
+ // A rule that escalates only under elevated authority takes its higher severity where the estate
166
+ // says the entry that reaches it runs privileged.
167
+ for (const f of findings) {
168
+ const raise = ALL_RULES[f.rule]?.whenPrivileged;
169
+ if (raise && f.effect === 'privileged') f.sev = raise;
170
+ }
171
+
172
+ const byRule = tally(findings);
173
+ const bySeverity = {};
174
+ for (const f of findings) bySeverity[f.sev] = (bySeverity[f.sev] || 0) + 1;
175
+ const byEvidence = {};
176
+ for (const f of findings) byEvidence[f.evidence] = (byEvidence[f.evidence] || 0) + 1;
177
+
178
+ const report = {
179
+ tool: 'cobolwork',
180
+ schemaVersion: SCHEMA_VERSION,
181
+ summary: {
182
+ findings: findings.length,
183
+ byRule,
184
+ bySeverity,
185
+ byEvidence,
186
+ filesScanned: filesRead,
187
+ bySet,
188
+ programs: inv.summary.programs,
189
+ programFiles: inv.summary.programFiles,
190
+ copybookFiles: inv.summary.copybookFiles,
191
+ jclFiles: inv.summary.jclFiles,
192
+ copiesMissing: inv.summary.copiesMissing,
193
+ filesUnreadable: inv.summary.filesUnreadable,
194
+ cicsPrograms: parts.find(p => p.tool === 'cobolwork-cics')?.summary.cicsPrograms ?? null,
195
+ // A customer's own advisory extract changes what a clean advisory result means, so a scan
196
+ // that used one says which.
197
+ ...(parts.find(p => p.tool === 'cobolwork-build')?.summary.advisoryFeeds ? { advisoryFeeds: parts.find(p => p.tool === 'cobolwork-build').summary.advisoryFeeds } : {}),
198
+ // Which products the advisory rules searched, so a scan with no advisory finding says which
199
+ // products that silence covers.
200
+ ...(parts.find(p => p.tool === 'cobolwork-build')?.summary.advisoryCoverage ? { advisoryCoverage: parts.find(p => p.tool === 'cobolwork-build').summary.advisoryCoverage } : {}),
201
+ crossProgram: parts.find(p => p.tool === 'cobolwork-flow')?.summary.crossProgram ?? 0,
202
+ ...(parts.find(p => p.tool === 'cobolwork-flow')?.summary.entryPoints?.roots ? rollup(findings) : {}),
203
+ // Reach annotates, it does not re-rank: severity is urgency, reach is who can drive it. When
204
+ // no fact is declared, the report says so rather than reading as a clean bill.
205
+ ...(byReach ? { byReach } : {}),
206
+ ...(byEffect ? { byEffect } : {}),
207
+ ...(reachFeeds.some(f => !f.problem) ? { reachFeeds: reachFeeds.filter(f => !f.problem).map(f => ({ file: f.file, extract: f.extract, retrieved: f.retrieved })) } : {}),
208
+ ...(reachFeeds.some(f => f.problem) ? { reachFeedProblems: reachFeeds.filter(f => f.problem).map(f => `${f.file}: ${f.problem}`) } : {}),
209
+ ...(!reach.declared && withEntries ? { reachNote: 'no reachability facts declared, so no finding was judged reachable: name open/restricted transactions and jobs in cobolwork.site.json, or bring a COBOLWORK_REACH extract' } : {}),
210
+ ...(!reach.effectDeclared && withEntries ? { effectNote: 'no authority facts declared, so no finding was judged to run privileged: name privilegedTransactions/privilegedJobs in cobolwork.site.json, or bring them in the COBOLWORK_REACH extract' } : {}),
211
+ ruleSets: only,
212
+ // Which flow model produced these rows, and which cobolwork. A finding that appears or
213
+ // disappears between two reports is attributable to the model rather than to the code scanned.
214
+ flowModel: FLOW_MODEL,
215
+ toolVersion: TOOL_VERSION,
216
+ // How a finding is known across runs, and how many shared an identity with another.
217
+ identity,
218
+ filesEbcdic: inv.summary.filesEbcdic,
219
+ nosrc: filesRead === 0,
220
+ // Files past the data-flow memory budget were not read by the flow rules. Named, and counted
221
+ // as incomplete coverage, so a partial scan of a very large repository never reads as whole.
222
+ filesOverBudget: overBudget,
223
+ coverageIncomplete: inv.summary.coverageIncomplete || overBudget > 0 || unreadInSomeSet ||
224
+ parts.some(p => p.summary.coverageIncomplete),
225
+ // A rule set that could not run for want of configuration, as distinct from content that
226
+ // could not be read. Named per set, because "recon did not run" and "a copybook is missing"
227
+ // call for different actions from different people.
228
+ // Keyed off both flags. setIncomplete means a set could not run for want of configuration;
229
+ // coverageIncomplete means it ran and did not reach everything. Keying off the first alone
230
+ // meant a set that stopped on the memory reserve set the report's coverage flag and then
231
+ // appeared nowhere in the list of which sets fell short. `kind` says which of the two it is.
232
+ setsIncomplete: parts.filter(p => p.summary.setIncomplete || p.summary.coverageIncomplete)
233
+ .map(p => {
234
+ const s = p.summary;
235
+ const kind = s.coverageIncomplete ? 'coverage' : 'configuration';
236
+ const coverageWhy = s.notRead || s.readInPart
237
+ || (s.filesNotRead ? `${s.filesNotRead} file(s) were not read` : null)
238
+ || (s.filesUnreadable ? `${s.filesUnreadable} file(s) could not be read` : null)
239
+ || (s.programsUnread?.length ? `${s.programsUnread.length} program(s) could not be read, so a step running one reads as unresolved` : null)
240
+ // flow can stop after reading every file, with nothing unread to count
241
+ || (s.stoppedBy ? `${s.filesScanned || 0} file(s) were read, and the analysis stopped because ${stoppedBecause(s.stoppedBy)}` : null);
242
+ const configurationWhy = s.notLooked?.[0] || s.packsRefused?.[0]?.why || s.packCaveats?.[0] || s.problems?.[0];
243
+ return {
244
+ set: reportKey(p.tool),
245
+ kind,
246
+ why: (kind === 'coverage' ? coverageWhy || configurationWhy : configurationWhy || coverageWhy)
247
+ || 'the rule set did not fully run',
248
+ };
249
+ }),
250
+ },
251
+ inventory: {
252
+ formats: inv.summary.formats, missingCopybooks: inv.missingCopybooks, unreadable: inv.unreadable,
253
+ refusedCopies: inv.refusedCopies, ebcdic: inv.ebcdic, symlinks: inv.summary.symlinks, dirsUnreadable: inv.summary.dirsUnreadable,
254
+ },
255
+ findings,
256
+ ...(checked.length ? { checked } : {}),
257
+ ruleText: Object.fromEntries(Object.entries(ALL_RULES).map(([k, v]) => [k, v.text])),
258
+ ruleCwe: Object.fromEntries(Object.entries(ALL_RULES).filter(([, v]) => v.cwe).map(([k, v]) => [k, v.cwe])),
259
+ ruleEvidence: evidenceMap(ALL_RULES),
260
+ ruleImpact: Object.fromEntries(Object.entries(ALL_RULES).filter(([, v]) => v.impact).map(([k, v]) => [k, v.impact])),
261
+ ruleRemedy: Object.fromEntries(Object.entries(ALL_RULES).filter(([, v]) => v.remedy).map(([k, v]) => [k, v.remedy])),
262
+ evidence: EVIDENCE,
263
+ // Which clause each rule bears on, and what the mapping does not claim. A reader who takes
264
+ // this to an auditor needs both halves.
265
+ ruleCompliance: complianceMap(Object.keys(ALL_RULES)),
266
+ compliance: FRAMEWORKS_LOADED,
267
+ ...(COMPLIANCE_UNLOADED.length ? { complianceUnloaded: COMPLIANCE_UNLOADED } : {}),
268
+ };
269
+ const listed = flowPart?.listed;
270
+ if (listed) Object.defineProperty(report, 'listed', { value: listed, enumerable: false });
271
+ return report;
272
+ }