sparkforensics-mcp 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +13 -13
- package/vendor-core/cli/budgets.js +23 -9
- package/vendor-core/cli/collect-run.js +11 -4
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/comparison-verdict.js +22 -22
- package/vendor-core/core-source-hash.txt +1 -1
- package/vendor-core/detectors.js +202 -68
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +1 -1
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +160 -47
- package/vendor-core/event-schemas.js +2 -0
- package/vendor-core/evidence-report.js +18 -10
- package/vendor-core/finding-generic-recommendation.js +20 -1
- package/vendor-core/finding-names.js +7 -0
- package/vendor-core/finding-presentation.js +61 -26
- package/vendor-core/finding-tag-help.js +1 -1
- package/vendor-core/finding-types.js +12 -0
- package/vendor-core/format-utils.js +4 -3
- package/vendor-core/impact-estimator.js +20 -2
- package/vendor-core/impact-format.js +14 -13
- package/vendor-core/impact-model.js +27 -5
- package/vendor-core/ingest.js +4 -2
- package/vendor-core/list-runs.js +5 -2
- package/vendor-core/mcp-tools.js +1 -1
- package/vendor-core/model-assembler.js +23 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +16 -9
- package/vendor-core/redact.js +51 -10
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +43 -24
- package/vendor-core/run-interpretation.js +2 -1
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +3 -4
- package/vendor-core/scorecard-estimates.js +1 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +4 -0
- package/vendor-core/types.js +49 -1
- package/vendor-core/wasted-core-hours.js +10 -7
- package/vendor-core/write-targets.js +312 -0
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
|
|
2
|
+
|
|
3
|
+
/** The plan nodes a stage actually ran: its SQL execution's resolved plan tree, filtered to the
|
|
4
|
+
* nodes attributed to it (`node.stageIds`). Empty when the stage has no SQL execution, the
|
|
5
|
+
* execution has no resolved plan, or no node is attributed to it. The one stage-to-plan mapping:
|
|
6
|
+
* stageIdentity's fingerprint and the Python-stage check both read it. */
|
|
7
|
+
export function planNodesOfStage(stage , sql ) {
|
|
8
|
+
const execId = stage.sqlExecutionId;
|
|
9
|
+
if (execId == null) return [];
|
|
10
|
+
const root = sql.get(execId)?.planTree ?? null;
|
|
11
|
+
if (!root) return [];
|
|
12
|
+
const nodes = [];
|
|
13
|
+
(function collect(node ) {
|
|
14
|
+
if (node.stageIds?.includes(stage.id)) nodes.push(node);
|
|
15
|
+
for (const child of node.children ?? []) collect(child);
|
|
16
|
+
})(root);
|
|
17
|
+
return nodes;
|
|
18
|
+
}
|
|
@@ -56,6 +56,7 @@ const TASK_FIELD_PROPS = TASK_FIELD_DESCRIPTORS.map((d) => d.prop);
|
|
|
56
56
|
|
|
57
57
|
|
|
58
58
|
|
|
59
|
+
|
|
59
60
|
|
|
60
61
|
|
|
61
62
|
export function finalizeStage(
|
|
@@ -124,9 +125,11 @@ export function finalizeStage(
|
|
|
124
125
|
acc.executorCpuTime += t.executorCpuTime;
|
|
125
126
|
acc.inputBytes += t.inputBytes;
|
|
126
127
|
acc.outputBytes += t.outputBytes;
|
|
128
|
+
if (t.outputRecords != null) acc.outputRecords = (acc.outputRecords ?? 0) + t.outputRecords;
|
|
127
129
|
for (let i = 0; i < TASK_FIELD_PROPS.length; i++) buf.push(t[TASK_FIELD_PROPS[i]]);
|
|
128
130
|
}
|
|
129
131
|
stage.taskCount = taskCount;
|
|
132
|
+
stage.peakExecutionMemoryMax = peakExecutionMemoryMax;
|
|
130
133
|
stage.failedTasks = failedTasks;
|
|
131
134
|
stage.speculativeTasks = speculativeTasks;
|
|
132
135
|
stage.taskAttempts = null; // no longer needed after finalize, freeing memory
|
|
@@ -201,6 +204,7 @@ export function finalizeStage(
|
|
|
201
204
|
delete data.failureDetails; // internal-only intern table, summarized by failureGroups
|
|
202
205
|
delete data.speculativeWinners; // internal-only late-TaskEnd pairing state, kept worker-side
|
|
203
206
|
delete data.lateSpeculationWaste;
|
|
207
|
+
delete data.stageAttemptId;
|
|
204
208
|
|
|
205
209
|
return { type: 'stage', data };
|
|
206
210
|
}
|
package/vendor-core/types.js
CHANGED
|
@@ -57,6 +57,24 @@
|
|
|
57
57
|
|
|
58
58
|
|
|
59
59
|
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
|
|
60
78
|
|
|
61
79
|
|
|
62
80
|
|
|
@@ -70,6 +88,20 @@
|
|
|
70
88
|
|
|
71
89
|
|
|
72
90
|
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
|
|
73
105
|
|
|
74
106
|
|
|
75
107
|
|
|
@@ -263,6 +295,10 @@
|
|
|
263
295
|
|
|
264
296
|
|
|
265
297
|
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
|
|
266
302
|
|
|
267
303
|
|
|
268
304
|
|
|
@@ -272,6 +308,9 @@
|
|
|
272
308
|
|
|
273
309
|
|
|
274
310
|
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
|
|
275
314
|
|
|
276
315
|
|
|
277
316
|
|
|
@@ -288,11 +327,20 @@
|
|
|
288
327
|
|
|
289
328
|
|
|
290
329
|
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
|
|
291
339
|
|
|
292
340
|
|
|
293
341
|
// `Finding` is a union discriminated on `type`, one member per emitted finding type: see
|
|
294
342
|
// finding-types.ts for each detector's shape and which of its fields are public evidence.
|
|
295
|
-
|
|
343
|
+
|
|
296
344
|
|
|
297
345
|
|
|
298
346
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
// the memoryUtilization detector's idle-cores math uses: capacity core-time vs.
|
|
3
3
|
// core-time that actually ran tasks. Feeds a report widget only, no detector.
|
|
4
4
|
|
|
5
|
-
import {
|
|
5
|
+
import { computePeakConcurrentCores } from './core-count.js';
|
|
6
6
|
import { MS_PER_CORE_HOUR } from './format-utils.js';
|
|
7
7
|
|
|
8
8
|
const TOP_N = 5;
|
|
@@ -30,21 +30,24 @@ const EMPTY = {
|
|
|
30
30
|
|
|
31
31
|
// `app`/`runAggregates` are nullable because real callers pass null (app is
|
|
32
32
|
// SparkAppInfo | null before parsing completes), which the guard below already
|
|
33
|
-
// tolerates. `resources` is on app's shape so the
|
|
34
|
-
// type-checks; real app objects always carry it.
|
|
33
|
+
// tolerates. `resources` is on app's shape so the computePeakConcurrentCores
|
|
34
|
+
// pass-through type-checks; real app objects always carry it. An omitted
|
|
35
|
+
// `executorsRemoved` means no executor left, where peak cores equal the sum.
|
|
35
36
|
export function computeWastedCoreHours(
|
|
36
37
|
app ,
|
|
37
|
-
executorsAdded
|
|
38
|
+
executorsAdded = [],
|
|
38
39
|
runAggregates ,
|
|
40
|
+
executorsRemoved = [],
|
|
39
41
|
) {
|
|
40
42
|
// Nullish (not falsy) guard on times: a literal startTime:0 is valid.
|
|
41
43
|
if (!runAggregates || app?.startTime == null || app?.endTime == null) return EMPTY;
|
|
42
44
|
const appDurationMs = app.endTime - app.startTime;
|
|
43
45
|
if (appDurationMs <= 0) return EMPTY;
|
|
44
46
|
|
|
45
|
-
// Total cores:
|
|
46
|
-
// configured cores
|
|
47
|
-
|
|
47
|
+
// Total cores: peak concurrent Executor-Added Total Cores, else peak executors ×
|
|
48
|
+
// configured cores. Same capacity as the memoryUtilization detector, so an
|
|
49
|
+
// executor replaced mid-run is not counted twice.
|
|
50
|
+
const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
|
|
48
51
|
if (totalCores <= 0) return EMPTY;
|
|
49
52
|
|
|
50
53
|
const totalCoreHours = (totalCores * appDurationMs) / MS_PER_CORE_HOUR;
|
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
// Write targets: every SQL write command in a run's plans, with the path or table it writes to.
|
|
2
|
+
// Read by automation that halts a session when a target leaves its sandbox, so every rule here
|
|
3
|
+
// fails toward "target unknown" (null), never toward a guessed or partial target:
|
|
4
|
+
// - a write node whose target cannot be parsed is still reported, with target null and its raw
|
|
5
|
+
// simpleString;
|
|
6
|
+
// - a target Spark cut short (no delimiter after it, or a "..." marker in it) is unparseable;
|
|
7
|
+
// - a write-like node outside the known list is reported as an unrecognized write;
|
|
8
|
+
// - an execution whose plan is not in the model (or whose start event could not be read) is
|
|
9
|
+
// listed, never skipped, and the count of unreadable log lines is reported.
|
|
10
|
+
// Targets are verbatim from the simpleString: nothing is resolved (relative paths, ${var}
|
|
11
|
+
// placeholders, catalog-relative table names all pass through as the log states them).
|
|
12
|
+
import { walkPlanTree } from './plan-tree-walk.js';
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
// Write-like detection: a known command, any CamelCase word of the operator name in this set, or a
|
|
54
|
+
// TableAsSelect name.
|
|
55
|
+
const WRITE_WORDS = new Set([
|
|
56
|
+
'Write', 'Insert', 'Save', 'Overwrite', 'Append', 'Merge', 'Update', 'Delete', 'Truncate', 'Replace', 'Drop',
|
|
57
|
+
'Load', 'Vacuum', 'Convert', 'Clone', 'Restore', 'Optimize', 'Alter', 'Reorg', 'Rename', 'Call',
|
|
58
|
+
]);
|
|
59
|
+
// Names that contain no write word of their own: ADD/DROP/RENAME PARTITION (which can name a LOCATION).
|
|
60
|
+
const WRITE_SUBSTRINGS = ['TableAsSelect', 'AddPartition', 'DropPartition', 'RenamePartition', 'RecoverPartitions'];
|
|
61
|
+
// Plan nodes whose name matches a write word but which write no user data: WriteFiles is the child
|
|
62
|
+
// of a write command (the command is the write), the rest are query or streaming-state operators.
|
|
63
|
+
// Any join (SortMergeJoin) is excluded by name in isWriteLike.
|
|
64
|
+
const NOT_WRITES = new Set([
|
|
65
|
+
'WriteFiles', 'AppendColumns', 'AppendColumnsWithObject', 'MergeRows', 'StateStoreSave',
|
|
66
|
+
'StateStoreRestore', 'SessionWindowStateStoreSave', 'SessionWindowStateStoreRestore',
|
|
67
|
+
'UpdateEventTimeWatermarkColumn',
|
|
68
|
+
]);
|
|
69
|
+
|
|
70
|
+
const IDENT_PART = String.raw`(?:\`(?:[^\`]|\`\`)*\`|[A-Za-z0-9_$]+)`;
|
|
71
|
+
const IDENTIFIER = new RegExp(`^${IDENT_PART}(?:\\.${IDENT_PART})*$`);
|
|
72
|
+
const IDENT_PARTS = new RegExp(IDENT_PART, 'g');
|
|
73
|
+
const TRUNCATION_MARKER = /\.\.\.(?:\s*\d+ more fields)?/;
|
|
74
|
+
|
|
75
|
+
function commandOf(nodeName ) {
|
|
76
|
+
return nodeName.replace(/^Execute\s+/, '').trim().replace(/Exec$/, '');
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Only the operator name (the first word) is classified: a scan's node name goes on to print its
|
|
80
|
+
// relation and table, whose identifiers can hold a write word (Scan JDBCRelation(dbo.PriceUpdate)).
|
|
81
|
+
function isWriteLike(command ) {
|
|
82
|
+
if (isKnownWrite(command)) return true;
|
|
83
|
+
const operator = command.split(/\s/, 1)[0];
|
|
84
|
+
if (NOT_WRITES.has(operator) || operator.includes('Join')) return false;
|
|
85
|
+
if (WRITE_SUBSTRINGS.some((part) => operator.includes(part))) return true;
|
|
86
|
+
return (operator.match(/[A-Z][a-z0-9]*/g) ?? []).some((word) => WRITE_WORDS.has(word));
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
// Splits at top-level commas, ignoring commas inside (), [], {} and backticks. `terminated` is
|
|
92
|
+
// true when a comma follows the arg: the last arg of a cut-off string is never terminated.
|
|
93
|
+
function splitArgs(args ) {
|
|
94
|
+
const out = [];
|
|
95
|
+
let depth = 0;
|
|
96
|
+
let quoted = false;
|
|
97
|
+
let start = 0;
|
|
98
|
+
for (let i = 0; i < args.length; i++) {
|
|
99
|
+
const c = args[i];
|
|
100
|
+
if (c === '`') quoted = !quoted;
|
|
101
|
+
else if (quoted) continue;
|
|
102
|
+
else if (c === '(' || c === '[' || c === '{') depth++;
|
|
103
|
+
else if (c === ')' || c === ']' || c === '}') depth--;
|
|
104
|
+
else if (c === ',' && depth === 0) {
|
|
105
|
+
out.push({ text: args.slice(start, i).trim(), terminated: true });
|
|
106
|
+
start = i + 1;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
out.push({ text: args.slice(start).trim(), terminated: false });
|
|
110
|
+
return out;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// Text after the command name in the simpleString. Empty when the simpleString does not start
|
|
114
|
+
// with the command (so nothing after it can be trusted).
|
|
115
|
+
function argsOf(detail , command ) {
|
|
116
|
+
const m = /^(?:Execute\s+)?(\S+)\s*([\s\S]*)$/.exec(detail.trim());
|
|
117
|
+
if (!m || commandOf(m[1]) !== command) return '';
|
|
118
|
+
return m[2];
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function isCut(text ) {
|
|
122
|
+
return text === '' || TRUNCATION_MARKER.test(text);
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// `db`.`t` or db.t -> db.t. A part that itself contains a dot or backtick keeps its quoting.
|
|
126
|
+
function tableName(ident ) {
|
|
127
|
+
if (!IDENTIFIER.test(ident)) return null;
|
|
128
|
+
const parts = (ident.match(IDENT_PARTS) ?? []).map((p) => {
|
|
129
|
+
if (!p.startsWith('`')) return p;
|
|
130
|
+
const inner = p.slice(1, -1).replace(/``/g, '`');
|
|
131
|
+
return /[.`]/.test(inner) ? p : inner;
|
|
132
|
+
});
|
|
133
|
+
return parts.join('.');
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// First arg of a path-writing command. The next arg must be the boolean that follows it in
|
|
137
|
+
// InsertIntoHadoopFsRelationCommand, or the static-partition map (INSERT OVERWRITE ... PARTITION
|
|
138
|
+
// (dt='x')) that precedes that boolean, which Spark prints as `[dt=x]` (or `Map(dt -> x)`); this also rejects a path Spark printed with ", " inside.
|
|
139
|
+
function parseHadoopFsPath(args ) {
|
|
140
|
+
const [path, second, third] = args;
|
|
141
|
+
const flag = second && /^(Map\(.*\)|\[.*\])$/s.test(second.text) && second.terminated ? third : second;
|
|
142
|
+
if (!path?.terminated || isCut(path.text) || !flag || !/^(true|false)$/.test(flag.text)) return null;
|
|
143
|
+
return { kind: 'path', target: path.text };
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
// A quoted or plain identifier, or the `[Database: d, TableName: t, ...]` form Hive commands print.
|
|
147
|
+
function parseTableArg(args ) {
|
|
148
|
+
const first = args[0];
|
|
149
|
+
if (!first) return null;
|
|
150
|
+
const hive = /^\[Database: ([^,\]]+), TableName: ([^,\]]+)[,\]]/.exec(first.text);
|
|
151
|
+
if (hive && /\]$/.test(first.text)) {
|
|
152
|
+
const name = tableName(`${hive[1].trim()}.${hive[2].trim()}`);
|
|
153
|
+
return name && !isCut(name) ? { kind: 'table', target: name } : null;
|
|
154
|
+
}
|
|
155
|
+
if (!first.terminated || isCut(first.text)) return null;
|
|
156
|
+
const name = tableName(first.text);
|
|
157
|
+
return name ? { kind: 'table', target: name } : null;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// Options of `Map(k -> v, k2 -> v2)`, or null when the Map is missing or not closed.
|
|
161
|
+
function parseOptionMap(text ) {
|
|
162
|
+
const m = /^Map\(([\s\S]*)\)$/.exec(text);
|
|
163
|
+
if (!m) return null;
|
|
164
|
+
const body = m[1];
|
|
165
|
+
const keys = [...body.matchAll(/(?:^|, )([A-Za-z0-9_.\-]+) -> /g)];
|
|
166
|
+
const options = new Map ();
|
|
167
|
+
keys.forEach((key, i) => {
|
|
168
|
+
const valueStart = key.index + key[0].length;
|
|
169
|
+
const valueEnd = i + 1 < keys.length ? keys[i + 1].index : body.length;
|
|
170
|
+
options.set(key[1].toLowerCase(), body.slice(valueStart, valueEnd));
|
|
171
|
+
});
|
|
172
|
+
return options;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// SaveIntoDataSourceCommand: `<provider>, Map(options), <mode>`. The target is the `path` option or
|
|
176
|
+
// the JDBC `dbtable`/`table` option; a redacted or missing value, or both options, is unparseable.
|
|
177
|
+
function parseSaveIntoDataSource(args ) {
|
|
178
|
+
const [, options, mode] = args;
|
|
179
|
+
if (!options?.terminated || !mode || mode.text === '') return null;
|
|
180
|
+
const map = parseOptionMap(options.text);
|
|
181
|
+
if (!map) return null;
|
|
182
|
+
const path = map.get('path');
|
|
183
|
+
const table = map.get('dbtable') ?? map.get('table');
|
|
184
|
+
if ((path === undefined) === (table === undefined)) return null;
|
|
185
|
+
const value = (path ?? table) ;
|
|
186
|
+
if (isCut(value) || /\*{5,}\(redacted\)/.test(value)) return null;
|
|
187
|
+
return path !== undefined ? { kind: 'path', target: value } : { kind: 'jdbcTable', target: value };
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
// The one target every match of `pattern` names, or null when there is none, when they name
|
|
191
|
+
// different targets, or when one of them is cut: a string that also prints another plan (a
|
|
192
|
+
// subquery) cannot tell which match is written. `pattern` captures the target, then its closing
|
|
193
|
+
// delimiter; a match without the delimiter was cut before it.
|
|
194
|
+
function soleMatch(detail , pattern , normalize ) {
|
|
195
|
+
const targets = new Set ();
|
|
196
|
+
for (const m of detail.matchAll(pattern)) targets.add(m[2] === undefined || isCut(m[1]) ? null : normalize(m[1]));
|
|
197
|
+
const [target] = targets;
|
|
198
|
+
return targets.size === 1 ? target : null;
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// A V2 name states its catalog only as a leading part (catalog.namespace.table). With two parts or
|
|
202
|
+
// fewer the catalog is implicit, so a database check on the name alone could match another catalog.
|
|
203
|
+
function v2Table(name ) {
|
|
204
|
+
if (!name) return null;
|
|
205
|
+
const parts = name.match(IDENT_PARTS)?.length ?? 0;
|
|
206
|
+
return { kind: parts >= 3 ? 'table' : 'unqualifiedTable', target: name };
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// DataSource V2 writes print the connector's Write object; Iceberg's is IcebergWrite(table=t, ...).
|
|
210
|
+
function parseV2Write(detail ) {
|
|
211
|
+
return v2Table(soleMatch(detail, /\b\w*Write\(table=([^,()\s]*)([,)])?/g, tableName));
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// V2 CTAS/RTAS: `<catalog>@<hash>, <identifier>, <query plan>, ...`. The catalog shape is checked
|
|
215
|
+
// so a plan printed in a different layout is not misread.
|
|
216
|
+
function parseV2TableAsSelect(args ) {
|
|
217
|
+
const [catalog, ident] = args;
|
|
218
|
+
if (!catalog?.terminated || !/^[\w.$]+@[0-9a-f]+$/.test(catalog.text)) return null;
|
|
219
|
+
if (!ident?.terminated || isCut(ident.text)) return null;
|
|
220
|
+
return v2Table(tableName(ident.text));
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// Delta prints a path-based table as delta.`<path>`. Only the command's own first argument is the
|
|
224
|
+
// written table: later arguments print other plans (a source relation, a condition's subquery).
|
|
225
|
+
function parseDeltaPath(args ) {
|
|
226
|
+
const [first] = args;
|
|
227
|
+
const m = first?.terminated ? /^delta\.`([^`]*)`$/.exec(first.text) : null;
|
|
228
|
+
return m && !isCut(m[1]) ? { kind: 'path', target: m[1] } : null;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
const V2_WRITES = new Set([
|
|
232
|
+
'AppendData', 'OverwriteByExpression', 'OverwritePartitionsDynamic', 'ReplaceData', 'WriteDelta',
|
|
233
|
+
'WriteToDataSourceV2',
|
|
234
|
+
// V1-fallback writers (Delta tables): their node names keep the V1 suffix, not ending in Exec.
|
|
235
|
+
'AppendDataExecV1', 'OverwriteByExpressionExecV1',
|
|
236
|
+
]);
|
|
237
|
+
const V2_TABLE_AS_SELECT = new Set([
|
|
238
|
+
'CreateTableAsSelect', 'AtomicCreateTableAsSelect', 'ReplaceTableAsSelect', 'AtomicReplaceTableAsSelect',
|
|
239
|
+
]);
|
|
240
|
+
const DELTA_COMMANDS = new Set([
|
|
241
|
+
'WriteIntoDelta', 'WriteIntoDeltaCommand', 'MergeIntoCommand', 'UpdateCommand', 'DeleteCommand',
|
|
242
|
+
'CreateDeltaTableCommand', 'OptimizeTableCommand', 'RestoreTableCommand', 'DeltaReorgTableCommand',
|
|
243
|
+
]);
|
|
244
|
+
|
|
245
|
+
const ARG_PARSERS = {
|
|
246
|
+
InsertIntoHadoopFsRelationCommand: parseHadoopFsPath,
|
|
247
|
+
InsertIntoHiveTable: parseTableArg,
|
|
248
|
+
CreateHiveTableAsSelectCommand: parseTableArg,
|
|
249
|
+
OptimizedCreateHiveTableAsSelectCommand: parseTableArg,
|
|
250
|
+
CreateDataSourceTableAsSelectCommand: parseTableArg,
|
|
251
|
+
SaveIntoDataSourceCommand: parseSaveIntoDataSource,
|
|
252
|
+
};
|
|
253
|
+
|
|
254
|
+
function isKnownWrite(command ) {
|
|
255
|
+
return command in ARG_PARSERS || V2_WRITES.has(command) || V2_TABLE_AS_SELECT.has(command) || DELTA_COMMANDS.has(command);
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
// [recognized, parsed target or null] for one write-like node.
|
|
259
|
+
function parseWrite(command , detail ) {
|
|
260
|
+
const argParser = ARG_PARSERS[command];
|
|
261
|
+
if (argParser) return [true, argParser(splitArgs(argsOf(detail, command)))];
|
|
262
|
+
if (V2_WRITES.has(command)) return [true, parseV2Write(detail)];
|
|
263
|
+
if (V2_TABLE_AS_SELECT.has(command)) return [true, parseV2TableAsSelect(splitArgs(argsOf(detail, command)))];
|
|
264
|
+
// A MERGE prints its source relation too, so a delta. path in it may be the source, not the target.
|
|
265
|
+
if (command === 'MergeIntoCommand') return [true, null];
|
|
266
|
+
if (DELTA_COMMANDS.has(command)) return [true, parseDeltaPath(splitArgs(argsOf(detail, command)))];
|
|
267
|
+
return [false, null];
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
function outputRowsOf(node ) {
|
|
271
|
+
const metric = node.metrics?.find((m) => m.name === 'number of output rows');
|
|
272
|
+
return metric !== undefined && Number.isFinite(metric.value) ? metric.value : null;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
function collectWrites(executionId , root , out ) {
|
|
276
|
+
walkPlanTree(root, (node) => {
|
|
277
|
+
const command = commandOf(node.name);
|
|
278
|
+
if (!isWriteLike(command)) return;
|
|
279
|
+
const detail = node.detail ?? '';
|
|
280
|
+
const [recognized, parsed] = parseWrite(command, detail);
|
|
281
|
+
out.push({
|
|
282
|
+
sqlExecutionId: executionId,
|
|
283
|
+
nodeId: node.id ?? null,
|
|
284
|
+
command,
|
|
285
|
+
recognized,
|
|
286
|
+
kind: parsed?.kind ?? null,
|
|
287
|
+
target: parsed?.target ?? null,
|
|
288
|
+
outputRows: outputRowsOf(node),
|
|
289
|
+
raw: detail !== '' ? detail : node.name,
|
|
290
|
+
});
|
|
291
|
+
}, { dedupe: true });
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
export function extractWriteTargets(sql , input = {}) {
|
|
301
|
+
const writes = [];
|
|
302
|
+
const withoutPlan = new Map ();
|
|
303
|
+
for (const id of input.unreadableSqlExecutions ?? []) if (!sql.get(id)?.planTree) withoutPlan.set(id, 'unreadableStart');
|
|
304
|
+
for (const id of [...sql.keys()].sort((a, b) => a - b)) {
|
|
305
|
+
const planTree = sql.get(id) .planTree;
|
|
306
|
+
if (!planTree) withoutPlan.set(id, withoutPlan.get(id) ?? 'noPlan');
|
|
307
|
+
else collectWrites(id, planTree, writes);
|
|
308
|
+
}
|
|
309
|
+
const executionsWithoutPlan = [...withoutPlan].sort((a, b) => a[0] - b[0])
|
|
310
|
+
.map(([sqlExecutionId, reason]) => ({ sqlExecutionId, reason }));
|
|
311
|
+
return { writes, executionsWithoutPlan, skippedLines: input.skippedLines ?? null };
|
|
312
|
+
}
|