@intflow/sentinelctl 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +23 -19
- package/package.json +1 -1
- package/src/commands.mjs +201 -36
- package/src/main.mjs +70 -44
package/README.md
CHANGED
|
@@ -46,39 +46,43 @@ Mistyped commands and options get a suggestion (`Did you mean --range?`), and ev
|
|
|
46
46
|
## Fleet overview and metrics
|
|
47
47
|
|
|
48
48
|
```bash
|
|
49
|
-
sentinelctl fleet summary # status, attention, busiest devices, log levels,
|
|
50
|
-
sentinelctl fleet summary -r 1h -o json # about
|
|
49
|
+
sentinelctl fleet summary # status, offline split, attention, busiest devices, trends, log levels, error patterns
|
|
50
|
+
sentinelctl fleet summary -r 1h -o json # about 10KB
|
|
51
51
|
sentinelctl metrics top memory -r 7d # devices ranked by 7-day average memory + fleet distribution
|
|
52
52
|
sentinelctl metrics top disk --agg last --above 85 --top 50
|
|
53
|
+
sentinelctl metrics top diskDaysLeft --below 14 # disks that fill up within two weeks at the current growth
|
|
53
54
|
sentinelctl metrics top cpu --agg p95 -r 24h --compare --sort change
|
|
54
|
-
sentinelctl metrics top reboots -r 7d --above 0
|
|
55
|
+
sentinelctl metrics top reboots -r 7d --above 0 # many devices reboot daily on schedule; compare with p50
|
|
55
56
|
sentinelctl metrics top networkTx -r 24h --label fleet=edge -o json
|
|
56
57
|
```
|
|
57
58
|
|
|
58
|
-
`metrics top` runs one query over every device: `--agg avg|max|min|p95|last` per device over `-r 1h|6h|24h|7d|30d` (or `--from/--to`, up to 30 days), ranked, with min/p50/p90/p95/max across devices. Metrics: cpu, memory, disk, swap, load, temperature, processes, tcpConnections, networkRx, networkTx, reboots. `--compare` adds the previous window of the same length. Device filters work like `devices list`; retired devices are excluded by default.
|
|
59
|
+
`metrics top` runs one query over every device: `--agg avg|max|min|p95|last` per device over `-r 1h|6h|24h|7d|30d` (or `--from/--to`, up to 30 days), ranked, with min/p50/p90/p95/max across devices. Metrics: cpu, memory, disk, swap, load, temperature, processes, tcpConnections, networkRx, networkTx, reboots, diskDaysLeft. `--compare` adds the previous window of the same length. Device filters work like `devices list`; retired devices are excluded by default.
|
|
59
60
|
|
|
60
|
-
## Logs
|
|
61
|
+
## Logs: count first, read last
|
|
61
62
|
|
|
62
63
|
```bash
|
|
63
|
-
sentinelctl logs
|
|
64
|
-
sentinelctl logs
|
|
65
|
-
sentinelctl logs
|
|
66
|
-
sentinelctl logs search -
|
|
67
|
-
sentinelctl logs search -q boom -o ndjson | jq .message
|
|
64
|
+
sentinelctl logs stats -l error -r 24h --by message --compare # which patterns, what changed
|
|
65
|
+
sentinelctl logs groups -l error -r 24h # every device x source x pattern, with count, first/last and latest line
|
|
66
|
+
sentinelctl logs stats -p "<pattern>" --by agent --timeline -r 7d # where and since when
|
|
67
|
+
sentinelctl logs search -p "<pattern>" -a <agent-id> -r 24h -n 50 # a few raw lines
|
|
68
68
|
sentinelctl logs context -a <agent-id> -s app.service -t 2026-09-30T01:02:03.456Z
|
|
69
|
-
sentinelctl logs
|
|
70
|
-
sentinelctl logs
|
|
71
|
-
sentinelctl logs stats -l error -r 24h --by agent --top 10
|
|
72
|
-
sentinelctl logs stats -a <agent-id> -r 7d --by message # message patterns, numbers collapsed to <N>
|
|
73
|
-
sentinelctl logs stats -l error -r 24h --by message --compare --sort change # what is new or growing
|
|
69
|
+
sentinelctl logs facets -l error -r 24h # top devices, sources, severities at a glance
|
|
70
|
+
sentinelctl logs histogram -a <agent-id> -r 30d
|
|
74
71
|
sentinelctl logs sources -r 24h
|
|
72
|
+
sentinelctl logs stats --service payment-api --env prod -l error -r 7d --by message # one application across devices (needs service/env labels)
|
|
73
|
+
sentinelctl logs stats -l error -r 24h --by service
|
|
74
|
+
sentinelctl logs search -l error --since <nextSince> -o ndjson # poll only new logs
|
|
75
75
|
```
|
|
76
76
|
|
|
77
|
-
`logs
|
|
77
|
+
- `logs search` returns at most 10,000 lines (250 per request, `--pages N` up to 40 or `--all`). On a busy fleet that covers minutes, so never download logs to count or summarize them: `logs stats` and `logs groups` count every log in the window on the server.
|
|
78
|
+
- `-p/--pattern` takes a pattern exactly as `logs stats --by message` or `logs groups` print it (numbers as `<N>`) and matches only that pattern. `-q` matches words case-insensitively; add `--case-sensitive` for a much faster exact-case match.
|
|
79
|
+
- `--dedupe` folds lines repeated within the same second; `--collapse` on `logs search` is the same as `logs groups`.
|
|
80
|
+
- Windows: `15m`, `1h`, `6h`, `24h`, `7d`, `30d` or `--from/--to` up to the retention period (31 days by default). Fleet-wide aggregations over 7 days need `-a`, `-s`, `-p` or `--case-sensitive`; the server answers `log_query_too_broad` otherwise.
|
|
81
|
+
- Aggregations over a day are computed per day and finished days are cached, so repeating a long query is cheap; the CLI continues automatically when the server needs several requests.
|
|
82
|
+
- Levels: `emergency`, `critical` and `error` include more severe levels; `warning`, `info` (with notice) and `debug` match that level only.
|
|
83
|
+
- `--fields` (raw fields) is admin-only and audited. `--follow` polls at most every 5 seconds and is meant for humans.
|
|
78
84
|
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
Each user can run 2 log queries and 2 metric aggregations at a time, web and CLI combined; a third concurrent query gets 429 and GET requests retry automatically with backoff.
|
|
85
|
+
Each user can run 2 log queries and 2 metric aggregations at a time (web and CLI combined) and spend 60 seconds of log query time per minute; GET requests that get 429 retry automatically with backoff.
|
|
82
86
|
|
|
83
87
|
## Devices and metadata
|
|
84
88
|
|
package/package.json
CHANGED
package/src/commands.mjs
CHANGED
|
@@ -4,7 +4,7 @@ import { ageText, localTime, percentText, renderKeyValues, renderTable, singleLi
|
|
|
4
4
|
|
|
5
5
|
export const roleLabels = { viewer: 'viewer (read-only)', editor: 'editor', admin: 'admin' };
|
|
6
6
|
export const outputChoices = ['table', 'json', 'ndjson'];
|
|
7
|
-
const logRanges = ['15m', '1h', '6h', '24h', '7d'];
|
|
7
|
+
const logRanges = ['15m', '1h', '6h', '24h', '7d', '30d'];
|
|
8
8
|
const deviceRanges = ['30m', '1h', '6h', '24h', '7d', '30d'];
|
|
9
9
|
const logLevels = ['emergency', 'critical', 'error', 'warning', 'info', 'debug'];
|
|
10
10
|
const deviceStatuses = ['all', 'online', 'delayed', 'offline', 'unknown', 'maintenance', 'retired'];
|
|
@@ -13,15 +13,29 @@ const maxPages = 40;
|
|
|
13
13
|
const deviceSorts = ['attention', 'status', 'displayName', 'cpu', 'memory', 'disk', 'load', 'temperature', 'ageSeconds', 'role', 'site', 'group', 'incidentCount'];
|
|
14
14
|
|
|
15
15
|
const logFilterOptions = {
|
|
16
|
-
query: { alias: 'q', type: 'string', placeholder: 'text', description: '
|
|
16
|
+
query: { alias: 'q', type: 'string', placeholder: 'text', description: 'Words or phrase in the message, case-insensitive by default. Positional words are appended.' },
|
|
17
|
+
pattern: { alias: 'p', type: 'string', placeholder: 'pattern', description: 'Exact message pattern with numbers as <N>, copied from `logs stats --by message` or `logs groups`. Fast and precise.' },
|
|
18
|
+
'case-sensitive': { type: 'boolean', description: 'Match -q case-sensitively. Much faster on long windows, but misses other casings.' },
|
|
17
19
|
agent: { alias: 'a', type: 'string', placeholder: 'id', description: 'Exact agent ID (find it with `devices lookup`).' },
|
|
18
|
-
level: { alias: 'l', type: 'string', choices: logLevels, description: '
|
|
20
|
+
level: { alias: 'l', type: 'string', choices: logLevels, description: 'emergency/critical/error include more severe levels; warning, info (with notice) and debug match that level only.' },
|
|
19
21
|
source: { alias: 's', type: 'string', placeholder: 'source', description: 'Exact source: systemd unit, container name, syslog identifier (see `logs sources`).' },
|
|
22
|
+
service: { type: 'string', placeholder: 'name', description: 'Logical service label (`service` field) across all devices, e.g. payment-api. Logs without the label never match.' },
|
|
23
|
+
env: { type: 'string', placeholder: 'name', description: 'Environment label (`env` field), e.g. prod.' },
|
|
20
24
|
range: { alias: 'r', type: 'string', choices: logRanges, description: 'Relative window. Default 15m when --from/--to are absent.' },
|
|
21
|
-
from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start, e.g. 2026-09-30T00:00:00+09:00. Requires --to;
|
|
25
|
+
from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start, e.g. 2026-09-30T00:00:00+09:00. Requires --to; up to the retention period (31 days by default).' },
|
|
22
26
|
to: { type: 'string', placeholder: 'ISO', description: 'Absolute window end.' },
|
|
23
27
|
};
|
|
24
28
|
|
|
29
|
+
async function getComplete(context, path, query) {
|
|
30
|
+
let body = await context.client.get(path, query);
|
|
31
|
+
for (let round = 0; body.complete === false && round < 20; round += 1) {
|
|
32
|
+
context.note(`Aggregating step ${body.coverage?.done ?? 0}/${body.coverage?.chunks ?? '?'}, continuing…`);
|
|
33
|
+
body = await context.client.get(path, query);
|
|
34
|
+
}
|
|
35
|
+
if (body.complete === false) context.hint(`Partial result: ${body.coverage?.done}/${body.coverage?.chunks} steps aggregated. Run the same command again to continue (finished days are cached).`);
|
|
36
|
+
return body;
|
|
37
|
+
}
|
|
38
|
+
|
|
25
39
|
function logCriteria(options, positionals) {
|
|
26
40
|
const text = [options.query, ...positionals].filter(Boolean).join(' ');
|
|
27
41
|
if ((options.from && !options.to) || (!options.from && options.to)) {
|
|
@@ -33,8 +47,10 @@ function logCriteria(options, positionals) {
|
|
|
33
47
|
throw new UsageError(`Cannot parse --${name}: ${value}`, { hint: 'Use ISO 8601, e.g. 2026-09-30T09:00:00+09:00.' });
|
|
34
48
|
}
|
|
35
49
|
}
|
|
50
|
+
if (options['case-sensitive'] && !text) throw new UsageError('--case-sensitive needs search text (-q).');
|
|
36
51
|
return {
|
|
37
|
-
q: text || undefined, agentId: options.agent, level: options.level, source: options.source,
|
|
52
|
+
q: text || undefined, agentId: options.agent, level: options.level, source: options.source, service: options.service, env: options.env,
|
|
53
|
+
pattern: options.pattern, match: options['case-sensitive'] ? 'exact' : undefined,
|
|
38
54
|
range: options.from ? undefined : (options.range ?? '15m'),
|
|
39
55
|
from: options.from ? new Date(options.from).toISOString() : undefined,
|
|
40
56
|
to: options.to ? new Date(options.to).toISOString() : undefined,
|
|
@@ -94,14 +110,43 @@ function logKey(log) {
|
|
|
94
110
|
function emptyLogHint(criteria) {
|
|
95
111
|
const tips = [];
|
|
96
112
|
if (criteria.range && criteria.range !== '7d') tips.push('widen the window (-r 24h or -r 7d)');
|
|
97
|
-
if (criteria.q) tips.push('shorten or drop the search text');
|
|
113
|
+
if (criteria.q) tips.push(criteria.match === 'exact' ? 'drop --case-sensitive' : 'shorten or drop the search text');
|
|
114
|
+
if (criteria.pattern) tips.push('check the pattern with `sentinelctl logs stats --by message`');
|
|
98
115
|
if (criteria.source) tips.push('check source names with `sentinelctl logs sources -r 24h`');
|
|
99
116
|
if (criteria.agentId) tips.push('check the agent ID with `sentinelctl devices lookup <name>`');
|
|
100
117
|
if (criteria.level) tips.push('drop -l');
|
|
101
118
|
return tips.length ? `No logs matched. Try: ${tips.join('; ')}.` : 'No logs matched.';
|
|
102
119
|
}
|
|
103
120
|
|
|
121
|
+
function timestampNanos(value) {
|
|
122
|
+
const match = /^(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})(?:\.(\d{1,9}))?(Z|[+-]\d{2}:\d{2})$/.exec(String(value));
|
|
123
|
+
if (!match) return BigInt(Date.parse(value)) * 1_000_000n;
|
|
124
|
+
return BigInt(Date.parse(`${match[1]}${match[3]}`)) * 1_000_000n + BigInt((match[2] ?? '').padEnd(9, '0'));
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function dedupeLogs(logs) {
|
|
128
|
+
const result = [];
|
|
129
|
+
for (const log of logs) {
|
|
130
|
+
const previous = result.at(-1);
|
|
131
|
+
const sameSecond = previous && String(previous.timestamp).slice(0, 19) === String(log.timestamp).slice(0, 19);
|
|
132
|
+
const strip = (message) => String(message).replace(/^\[[^\]]+\]\s*/, '');
|
|
133
|
+
if (sameSecond && previous.agentId === log.agentId && previous.source === log.source && strip(previous.message) === strip(log.message)) {
|
|
134
|
+
previous.repeat = (previous.repeat ?? 1) + 1;
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
result.push({ ...log });
|
|
138
|
+
}
|
|
139
|
+
return result;
|
|
140
|
+
}
|
|
141
|
+
|
|
104
142
|
async function logsSearch(context, { options, positionals }) {
|
|
143
|
+
if (options.collapse) return logsGroups(context, { options, positionals });
|
|
144
|
+
if (options.since) {
|
|
145
|
+
if (options.from || options.to || options.range) throw new UsageError('--since cannot be combined with --range/--from/--to.');
|
|
146
|
+
if (!Number.isFinite(Date.parse(options.since))) throw new UsageError(`Cannot parse --since: ${options.since}`, { hint: 'Pass the nextSince value printed by the previous run.' });
|
|
147
|
+
options.from = options.since;
|
|
148
|
+
options.to = new Date().toISOString();
|
|
149
|
+
}
|
|
105
150
|
const criteria = logCriteria(options, positionals);
|
|
106
151
|
const paging = options.all || options.pages > 1;
|
|
107
152
|
const limit = options.fields ? 50 : (options.limit ?? (paging ? 250 : 100));
|
|
@@ -120,8 +165,17 @@ async function logsSearch(context, { options, positionals }) {
|
|
|
120
165
|
if (!last.hasMore || !last.nextCursor) break;
|
|
121
166
|
cursor = last.nextCursor;
|
|
122
167
|
}
|
|
123
|
-
|
|
124
|
-
|
|
168
|
+
if (options.since) {
|
|
169
|
+
const since = timestampNanos(options.since);
|
|
170
|
+
for (let index = logs.length - 1; index >= 0; index -= 1) if (timestampNanos(logs[index].timestamp) <= since) logs.splice(index, 1);
|
|
171
|
+
}
|
|
172
|
+
const shown = options.dedupe ? dedupeLogs(logs) : logs;
|
|
173
|
+
const nextSince = logs[0]?.timestamp ?? options.since ?? null;
|
|
174
|
+
emit(context, context.global.output === 'json'
|
|
175
|
+
? { criteria: last.criteria, window: last.window, count: logs.length, ...(options.dedupe ? { rows: shown.length } : {}), hasMore: last.hasMore, ...(options.since ? { nextSince } : {}), logs: shown }
|
|
176
|
+
: shown,
|
|
177
|
+
() => renderTable(shown, options.dedupe ? [...logColumns.slice(0, 4), { header: 'x', value: (log) => (log.repeat ? `×${log.repeat}` : ''), max: 5 }, logColumns[4]] : logColumns, { width: context.width, wide: options.wide }));
|
|
178
|
+
if (options.since) context.hint(`Next time: --since ${nextSince}`);
|
|
125
179
|
if (!logs.length) context.hint(emptyLogHint(criteria));
|
|
126
180
|
else context.note(`${logs.length} logs · ${localTime(last.window.start)} → ${localTime(last.window.end)}`);
|
|
127
181
|
if (last.hasMore && options.all) context.hint(`Stopped at the ${maxPages * 250}-log cap. Narrow the filters, or use \`sentinelctl logs stats\` for counts.`);
|
|
@@ -157,17 +211,39 @@ async function followLogs(context, query, options) {
|
|
|
157
211
|
}
|
|
158
212
|
|
|
159
213
|
|
|
160
|
-
const statsDimensions = ['source', 'agent', 'level', 'message'];
|
|
214
|
+
const statsDimensions = ['source', 'agent', 'level', 'message', 'service'];
|
|
215
|
+
|
|
216
|
+
const sparkBlocks = ['▁', '▂', '▃', '▄', '▅', '▆', '▇', '█'];
|
|
217
|
+
|
|
218
|
+
function sparkline(points, window, step) {
|
|
219
|
+
if (!points?.length || !step) return '';
|
|
220
|
+
const start = Math.floor(Date.parse(window.start) / 1000 / step) * step;
|
|
221
|
+
const end = Date.parse(window.end) / 1000;
|
|
222
|
+
const counts = new Map(points);
|
|
223
|
+
const values = [];
|
|
224
|
+
for (let at = start; at < end; at += step) values.push(counts.get(at) ?? 0);
|
|
225
|
+
const peak = Math.max(1, ...values);
|
|
226
|
+
return values.map((value) => (value ? sparkBlocks[Math.min(7, Math.floor((value / peak) * 7.999))] : ' ')).join('');
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
function quoteArgument(text) {
|
|
230
|
+
return `"${String(text).replaceAll('\\', '\\\\').replaceAll('"', '\\"')}"`;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
function narrowingHint(criteria) {
|
|
234
|
+
const window = criteria.range ? `-r ${criteria.range}` : `--from ${criteria.from} --to ${criteria.to}`;
|
|
235
|
+
return { window, level: criteria.level ? ` -l ${criteria.level}` : '' };
|
|
236
|
+
}
|
|
161
237
|
|
|
162
238
|
async function logsStats(context, { options, positionals }) {
|
|
163
239
|
const criteria = logCriteria(options, positionals);
|
|
164
240
|
if (options.sort === 'change' && !options.compare) {
|
|
165
241
|
throw new UsageError('--sort change needs --compare.', { hint: 'Example: sentinelctl logs stats -l error -r 24h --by message --compare --sort change' });
|
|
166
242
|
}
|
|
167
|
-
const body = await context
|
|
168
|
-
...criteria, by: options.by, top: options.top, compare: options.compare ? 1 : undefined, sort: options.sort,
|
|
243
|
+
const body = await getComplete(context, '/api/logs/stats', {
|
|
244
|
+
...criteria, by: options.by, top: options.top, compare: options.compare ? 1 : undefined, sort: options.sort, timeline: options.timeline ? 1 : undefined,
|
|
169
245
|
});
|
|
170
|
-
const label = { source: 'SOURCE', agent: 'DEVICE', level: 'LEVEL', message: 'MESSAGE PATTERN' }[body.criteria.by];
|
|
246
|
+
const label = { source: 'SOURCE', agent: 'DEVICE', level: 'LEVEL', message: 'MESSAGE PATTERN', service: 'SERVICE' }[body.criteria.by];
|
|
171
247
|
const signed = (value) => (value === null || value === undefined ? '?' : value > 0 ? `+${value}` : String(value));
|
|
172
248
|
emit(context, context.global.output === 'ndjson' ? body.groups : body, () => renderTable(body.groups, [
|
|
173
249
|
{ header: 'COUNT', value: (group) => String(group.count) },
|
|
@@ -176,10 +252,13 @@ async function logsStats(context, { options, positionals }) {
|
|
|
176
252
|
{ header: 'CHANGE', value: (group) => (group.isNew ? 'new' : signed(group.change)) },
|
|
177
253
|
] : [{ header: 'SHARE', value: (group) => (body.total ? `${((group.count / body.total) * 100).toFixed(1)}%` : '-') }]),
|
|
178
254
|
{ header: 'DEVICES', value: (group) => String(group.devices) },
|
|
255
|
+
...(body.timelineStepSeconds ? [{ header: 'TREND', value: (group) => sparkline(group.timeline, body.window, body.timelineStepSeconds), max: 48 }] : []),
|
|
256
|
+
...(body.timelineStepSeconds ? [{ header: 'LAST', value: (group) => (group.last ? localTime(group.last).slice(5, 16) : '-'), max: 11 }] : []),
|
|
179
257
|
...(body.criteria.by === 'agent' ? [{ header: 'AGENT ID', value: (group) => group.key, max: 36 }] : []),
|
|
180
258
|
{ header: label, value: (group) => (body.criteria.by === 'agent' ? group.displayName ?? group.host ?? group.key : group.key) },
|
|
181
259
|
], { width: context.width, wide: options.wide }));
|
|
182
260
|
context.note(`${body.total} logs from ${body.devices} devices · ${localTime(body.window.start)} → ${localTime(body.window.end)}`);
|
|
261
|
+
if (body.timelineStepSeconds) context.note(`TREND: one column per ${Math.round(body.timelineStepSeconds / 60)} min; first/last (bucket precision) are in -o json.`);
|
|
183
262
|
if (body.previous) {
|
|
184
263
|
context.note(`Previous window ${localTime(body.previous.window.start)} → ${localTime(body.previous.window.end)}: ${body.previous.total} logs (${signed(body.total - body.previous.total)}) · ${body.newGroupCount} new groups`);
|
|
185
264
|
if (body.falling?.length && body.criteria.sort !== 'change') {
|
|
@@ -190,13 +269,45 @@ async function logsStats(context, { options, positionals }) {
|
|
|
190
269
|
if (body.truncated) context.hint(`Showing the top ${body.groups.length} of ${body.groupCount} groups. Raise --top (max 100) or add filters.`);
|
|
191
270
|
if (body.approximate) context.hint('Message patterns are approximate on this server (collapse_nums is unavailable); narrow the window or filters for exact counts.');
|
|
192
271
|
if (!body.total) context.hint(emptyLogHint(criteria));
|
|
193
|
-
const sample = body.groups.find((group) => group.
|
|
272
|
+
const sample = body.groups.find((group) => group.count > 0);
|
|
194
273
|
if (body.criteria.by === 'message' && sample) {
|
|
195
|
-
|
|
274
|
+
const { window, level } = narrowingHint(criteria);
|
|
275
|
+
context.hint(`Next: who and where → sentinelctl logs groups -p ${quoteArgument(sample.key)} ${window}${level}; raw lines → sentinelctl logs search -p ${quoteArgument(sample.key)} ${window} -n 50`);
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
const groupColumns = [
|
|
280
|
+
{ header: 'COUNT', value: (group) => String(group.count) },
|
|
281
|
+
{ header: 'LAST', value: (group) => (group.last ? localTime(group.last).slice(5, 16) : '-'), max: 11 },
|
|
282
|
+
{ header: 'DEVICE', value: (group) => group.displayName ?? group.agentId, max: 26 },
|
|
283
|
+
{ header: 'SOURCE', value: (group) => group.source, max: 22 },
|
|
284
|
+
{ header: 'PATTERN', value: (group) => group.pattern },
|
|
285
|
+
];
|
|
286
|
+
|
|
287
|
+
async function logsGroups(context, { options, positionals }) {
|
|
288
|
+
const criteria = logCriteria(options, positionals);
|
|
289
|
+
const body = await getComplete(context, '/api/logs/groups', { ...criteria, top: options.top ?? options.limit });
|
|
290
|
+
emit(context, context.global.output === 'ndjson' ? body.groups : body, () => renderTable(body.groups, groupColumns, { width: context.width, wide: options.wide }));
|
|
291
|
+
context.note(`${body.total} logs in ${body.groupCount} groups (${body.patternCount} patterns, ${body.devices} devices) · ${localTime(body.window.start)} → ${localTime(body.window.end)}`);
|
|
292
|
+
if (body.truncated) context.hint(`Showing ${body.groups.length} of ${body.groupCount} groups, largest first. Raise --top (max 500) or add filters.`);
|
|
293
|
+
if (body.approximate) context.hint('Patterns are approximate on this server (collapse_nums is unavailable).');
|
|
294
|
+
if (!body.total) context.hint(emptyLogHint(criteria));
|
|
295
|
+
const first = body.groups[0];
|
|
296
|
+
if (first) {
|
|
297
|
+
const { window } = narrowingHint(criteria);
|
|
298
|
+
context.hint(`Each group keeps its latest line in -o json (latest.message). Raw lines of one group: sentinelctl logs search -a ${first.agentId} -p ${quoteArgument(first.pattern)} ${window} -n 50`);
|
|
196
299
|
}
|
|
197
300
|
}
|
|
198
301
|
|
|
199
|
-
|
|
302
|
+
async function logsFacets(context, { options, positionals }) {
|
|
303
|
+
const criteria = logCriteria({ ...options, range: options.range ?? (options.from ? undefined : '1h') }, positionals);
|
|
304
|
+
const body = await context.client.get('/api/logs/facets', { ...criteria, top: options.top });
|
|
305
|
+
emit(context, body, () => Object.entries(body.facets).map(([field, values]) => `${field}:\n${values
|
|
306
|
+
.map((item) => ` ${String(item.count).padStart(9)} ${item.displayName && item.displayName !== item.value ? `${item.displayName} (${item.value})` : item.value}`).join('\n')}`).join('\n'));
|
|
307
|
+
context.note(`${localTime(body.window.start)} → ${localTime(body.window.end)}`);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
const metricNames = ['cpu', 'memory', 'disk', 'swap', 'load', 'temperature', 'processes', 'tcpConnections', 'networkRx', 'networkTx', 'reboots', 'diskDaysLeft'];
|
|
200
311
|
const metricWindows = ['1h', '6h', '24h', '7d', '30d'];
|
|
201
312
|
const deviceSeries = ['cpu', 'memory', 'disk', 'load', 'temperature', 'processes', 'tcpConnections', 'networkRx', 'networkTx', 'swap', 'uptime'];
|
|
202
313
|
|
|
@@ -247,7 +358,7 @@ async function metricsTop(context, { options, positionals }) {
|
|
|
247
358
|
}
|
|
248
359
|
|
|
249
360
|
async function fleetSummary(context, { options }) {
|
|
250
|
-
const body = await context.client.get('/api/fleet/summary', { range: options.range, logs: options.logs === false ? 0 : undefined });
|
|
361
|
+
const body = await context.client.get('/api/fleet/summary', { range: options.range, logs: options.logs === false ? 0 : undefined, trends: options.trends === false ? 0 : undefined });
|
|
251
362
|
emit(context, body, () => {
|
|
252
363
|
const counts = (record) => Object.entries(record ?? {}).sort((left, right) => right[1] - left[1]).map(([key, value]) => `${key} ${value}`).join(', ');
|
|
253
364
|
const names = (devices) => devices.map((device) => device.displayName).join(', ');
|
|
@@ -264,9 +375,23 @@ async function fleetSummary(context, { options }) {
|
|
|
264
375
|
{ header: 'EXAMPLES', value: (entry) => names(entry.devices) },
|
|
265
376
|
], { width: context.width })
|
|
266
377
|
: ' none');
|
|
378
|
+
const offline = body.devices.offline;
|
|
379
|
+
if (offline) {
|
|
380
|
+
sections.push('', `Offline: ${offline.within24h.length} within 24h${offline.within24h.length ? ` (${offline.within24h.map((device) => `${device.displayName} ${device.offlineHours}h`).join(', ')})` : ''} · ${offline.otherCount} for 1-7 days · ${offline.over7dCount} over 7 days`);
|
|
381
|
+
}
|
|
267
382
|
sections.push('', 'Highest now (online, active):', renderKeyValues(Object.entries(body.resources).map(([field, devices]) => [
|
|
268
383
|
field, devices.map((device) => `${device.displayName} ${field === 'temperature' ? `${device.value}°C` : `${device.value}%`}`).join(', ') || '-',
|
|
269
384
|
])));
|
|
385
|
+
if (body.trends?.error) sections.push('', `Trends: ${body.trends.error}`);
|
|
386
|
+
else if (body.trends) {
|
|
387
|
+
const list = (devices, format) => devices.map((device) => `${device.displayName} ${format(device)}`).join(', ') || 'none';
|
|
388
|
+
sections.push('', 'Trends:', renderKeyValues([
|
|
389
|
+
['disk full <14d', list(body.trends.diskFullWithin14Days, (device) => `${device.value}d`)],
|
|
390
|
+
['temperature +15°C', list(body.trends.temperatureRise, (device) => `${device.previous}→${device.value}°C`)],
|
|
391
|
+
[`reboots >${body.trends.rebootOutliers.threshold}/7d`, `${list(body.trends.rebootOutliers.devices, (device) => `${device.value}`)} (typical ${body.trends.rebootOutliers.typicalPerWeek}/7d)`],
|
|
392
|
+
]));
|
|
393
|
+
}
|
|
394
|
+
if (body.logs?.complete === false) sections.push('', 'Logs: partial (some days still aggregating); run again to complete.');
|
|
270
395
|
if (body.logs?.error) sections.push('', `Logs: ${body.logs.error}`);
|
|
271
396
|
else if (body.logs) {
|
|
272
397
|
const errors = body.logs.errors;
|
|
@@ -287,7 +412,7 @@ async function fleetSummary(context, { options }) {
|
|
|
287
412
|
}
|
|
288
413
|
return sections.join('\n');
|
|
289
414
|
});
|
|
290
|
-
context.hint('Drill down:
|
|
415
|
+
context.hint('Drill down: logs groups -l error -r 24h · logs stats -l error --by message --compare --timeline · metrics top diskDaysLeft · devices overview <id>');
|
|
291
416
|
}
|
|
292
417
|
|
|
293
418
|
async function logsContext(context, { options }) {
|
|
@@ -305,7 +430,7 @@ async function logsContext(context, { options }) {
|
|
|
305
430
|
}
|
|
306
431
|
|
|
307
432
|
async function logsHistogram(context, { options, positionals }) {
|
|
308
|
-
const body = await context
|
|
433
|
+
const body = await getComplete(context, '/api/logs/histogram', logCriteria(options, positionals));
|
|
309
434
|
emit(context, context.global.output === 'ndjson' ? body.buckets : body, () => {
|
|
310
435
|
const peak = Math.max(1, ...body.buckets.map((bucket) => bucket.total));
|
|
311
436
|
return renderTable(body.buckets, [
|
|
@@ -324,7 +449,7 @@ async function logsSources(context, { options }) {
|
|
|
324
449
|
{ header: 'SOURCE', value: (source) => source.value },
|
|
325
450
|
{ header: 'COUNT', value: (source) => String(source.count) },
|
|
326
451
|
], { width: context.width }));
|
|
327
|
-
context.note(`Top sources
|
|
452
|
+
context.note(`Top ${body.sources.length} sources by exact count over ${body.range} (${body.total} logs).`);
|
|
328
453
|
}
|
|
329
454
|
|
|
330
455
|
const deviceColumns = [
|
|
@@ -696,8 +821,8 @@ async function rawApi(context, { options, positionals }) {
|
|
|
696
821
|
const commands = [
|
|
697
822
|
{
|
|
698
823
|
group: 'logs', name: 'search', run: logsSearch, role: 'viewer', mutates: false,
|
|
699
|
-
summary: '
|
|
700
|
-
description: 'Newest first. Default 15m window and 100 logs. When fetching several pages, 250 logs are fetched per request, up to 10,000 in total.
|
|
824
|
+
summary: 'Raw log lines by window, device, severity, source, text or exact pattern; --collapse for groups',
|
|
825
|
+
description: 'Newest first. Default 15m window and 100 logs. When fetching several pages, 250 logs are fetched per request, up to 10,000 in total, which on busy fleets covers only minutes. Use it for a handful of example lines after narrowing with `logs stats`/`logs groups` and -p. --collapse returns every matching group instead of lines (same as `logs groups`).',
|
|
701
826
|
args: [{ name: 'text', optional: true, description: 'Same as -q (words are joined with spaces).' }],
|
|
702
827
|
options: {
|
|
703
828
|
...logFilterOptions,
|
|
@@ -705,6 +830,10 @@ const commands = [
|
|
|
705
830
|
pages: { type: 'integer', placeholder: 'N', description: `Follow the cursor for up to N pages (1-${maxPages}).` },
|
|
706
831
|
all: { type: 'boolean', description: 'Fetch every page of the window, stopping at 10,000 logs.' },
|
|
707
832
|
fields: { type: 'boolean', description: 'Include raw fields (admin only, audited, 50 logs per page).' },
|
|
833
|
+
collapse: { type: 'boolean', description: 'Group by device, source and pattern over the whole window with counts, first/last time and the latest line (see logs groups).' },
|
|
834
|
+
top: { type: 'integer', placeholder: 'N', description: 'With --collapse: groups to return (1-500, default 50).' },
|
|
835
|
+
dedupe: { type: 'boolean', description: 'Fold lines repeated within the same second on the same device and source into one with a repeat count.' },
|
|
836
|
+
since: { type: 'string', placeholder: 'ISO', description: 'Only logs after this timestamp (exclusive) up to now; pass the printed nextSince to poll without re-reading.' },
|
|
708
837
|
follow: { alias: 'f', type: 'boolean', description: 'Keep printing new logs (for humans; never ends, agents should not use it).' },
|
|
709
838
|
interval: { type: 'integer', default: 5, placeholder: 'seconds', description: 'Polling interval for --follow (min 5).' },
|
|
710
839
|
wide: { alias: 'w', type: 'boolean', description: 'Do not truncate messages in table output.' },
|
|
@@ -713,22 +842,26 @@ const commands = [
|
|
|
713
842
|
roleNote: '--fields requires admin',
|
|
714
843
|
examples: [
|
|
715
844
|
'sentinelctl logs search -l error -r 1h',
|
|
845
|
+
'sentinelctl logs search -p "tegra-xusb <N>.usb: not all ports suspended: -<N>" -r 24h -n 50 -o json',
|
|
846
|
+
'sentinelctl logs search -l error -r 24h --collapse -o json',
|
|
847
|
+
'sentinelctl logs search -l error --since 2026-09-30T01:02:03.456789Z -o ndjson',
|
|
716
848
|
'sentinelctl logs search "connection reset" -r 24h -o ndjson',
|
|
717
849
|
'sentinelctl logs search -a <agent-id> -s ssh.service -r 6h --all -o json',
|
|
718
850
|
'sentinelctl logs search --from 2026-09-30T09:00:00+09:00 --to 2026-09-30T10:00:00+09:00 -l warning',
|
|
719
851
|
],
|
|
720
|
-
output: 'logs[]: timestamp (UTC ISO), severity, priority, agentId, host, displayName, source, message (+fields). -o json wraps them as {criteria, window, count, hasMore, logs}.',
|
|
852
|
+
output: 'logs[]: timestamp (UTC ISO), severity, priority, agentId, host, displayName, source, message (+fields, +repeat with --dedupe). -o json wraps them as {criteria, window, count, hasMore, nextSince?, logs}.',
|
|
721
853
|
},
|
|
722
854
|
{
|
|
723
855
|
group: 'logs', name: 'stats', run: logsStats, role: 'viewer', mutates: false,
|
|
724
|
-
summary: 'Count and rank logs by source, device, severity or message pattern
|
|
725
|
-
description: 'Aggregates in VictoriaLogs instead of downloading logs, so counts are exact over the whole window. message groups by pattern with numbers collapsed to <N
|
|
856
|
+
summary: 'Count and rank logs by source, device, severity or message pattern; compare windows; trends',
|
|
857
|
+
description: 'Aggregates in VictoriaLogs instead of downloading logs, so counts are exact over the whole window. message groups by pattern with numbers collapsed to <N>; pass a pattern to -p for exact follow-ups. Windows over one day are computed per day and finished days are cached, so repeating a long query is cheap; long requests continue automatically.',
|
|
726
858
|
args: [{ name: 'text', optional: true, description: 'Same as -q.' }],
|
|
727
859
|
options: {
|
|
728
860
|
...logFilterOptions,
|
|
729
|
-
by: { alias: 'b', type: 'string', choices: statsDimensions, default: 'source', description: 'Group by source, agent (device), level or
|
|
861
|
+
by: { alias: 'b', type: 'string', choices: statsDimensions, default: 'source', description: 'Group by source, agent (device), level, message pattern or service label.' },
|
|
730
862
|
top: { type: 'integer', default: 20, placeholder: 'N', description: 'Groups to return (1-100).' },
|
|
731
863
|
compare: { type: 'boolean', description: 'Also count the previous window of the same length: previousCount, change and isNew per group.' },
|
|
864
|
+
timeline: { type: 'boolean', description: 'Add per-group counts over time (about 48 buckets): when a pattern started, stopped or spiked.' },
|
|
732
865
|
sort: { type: 'string', choices: ['count', 'change'], description: 'Rank by count (default) or by increase over the previous window (needs --compare).' },
|
|
733
866
|
wide: { alias: 'w', type: 'boolean', description: 'Do not truncate long keys in table output.' },
|
|
734
867
|
},
|
|
@@ -736,10 +869,40 @@ const commands = [
|
|
|
736
869
|
'sentinelctl logs stats -l error -r 1h --by source',
|
|
737
870
|
'sentinelctl logs stats -l error -r 24h --by agent --top 10 -o json',
|
|
738
871
|
'sentinelctl logs stats -l error -r 24h --by message --compare --sort change',
|
|
872
|
+
'sentinelctl logs stats -l error -r 7d --by message --timeline --top 10',
|
|
739
873
|
'sentinelctl logs stats -a <agent-id> -r 7d --by message',
|
|
740
874
|
'sentinelctl logs stats "connection refused" -r 24h --by agent',
|
|
741
875
|
],
|
|
742
|
-
output: '{criteria, window, total, devices, groupCount, truncated, approximate, groups[]: {key, count, devices (distinct agents),
|
|
876
|
+
output: '{criteria, window, total, devices, groupCount, truncated, approximate, complete, coverage{chunks, done}, groups[]: {key, count, devices (distinct agents), first?, last? (with --timeline, bucket precision), host?, displayName?, searchText?, previousCount?, change?, isNew?, timeline?: [[epochSeconds, count]]}, timelineStepSeconds?, previous?: {window, total, devices, exhaustive}, newGroupCount?, rising?[≤5], falling?[≤5] (largest increases and decreases across all groups, not only the returned top)}.',
|
|
877
|
+
},
|
|
878
|
+
{
|
|
879
|
+
group: 'logs', name: 'groups', run: logsGroups, role: 'viewer', mutates: false,
|
|
880
|
+
summary: 'Every distinct (device, source, message pattern) in the window with counts, first/last time and the latest line',
|
|
881
|
+
description: 'The lossless way to see what happened: each group is counted over the whole window and keeps its latest raw line, so nothing is skipped the way a 10,000-line download skips. Typically 50-100x smaller than the raw logs. Windows over one day are computed per day and cached.',
|
|
882
|
+
args: [{ name: 'text', optional: true, description: 'Same as -q.' }],
|
|
883
|
+
options: {
|
|
884
|
+
...logFilterOptions,
|
|
885
|
+
top: { type: 'integer', default: 50, placeholder: 'N', description: 'Groups to return, largest first (1-500).' },
|
|
886
|
+
wide: { alias: 'w', type: 'boolean', description: 'Do not truncate patterns in table output.' },
|
|
887
|
+
},
|
|
888
|
+
examples: [
|
|
889
|
+
'sentinelctl logs groups -l error -r 1h',
|
|
890
|
+
'sentinelctl logs groups -a <agent-id> -r 7d -o json',
|
|
891
|
+
'sentinelctl logs groups -p "Failed to start <N>.service" -r 24h',
|
|
892
|
+
],
|
|
893
|
+
output: '{criteria, window, complete, coverage, total, devices, groupCount, patternCount, truncated, approximate, groups[]: {agentId, displayName, source, pattern, count, first, last, latest{timestamp, severity, message}}}.',
|
|
894
|
+
},
|
|
895
|
+
{
|
|
896
|
+
group: 'logs', name: 'facets', run: logsFacets, role: 'viewer', mutates: false,
|
|
897
|
+
summary: 'Top values of device, host, severity, source, agent version and profile in one query',
|
|
898
|
+
description: 'Orientation for an unfamiliar window: which devices, sources and severities dominate. Fleet-wide up to 24h; up to 7 days with -a, -s or -p.',
|
|
899
|
+
options: {
|
|
900
|
+
...logFilterOptions,
|
|
901
|
+
range: { alias: 'r', type: 'string', choices: logRanges, description: 'Window (default 1h).' },
|
|
902
|
+
top: { type: 'integer', default: 10, placeholder: 'N', description: 'Values per field (1-50).' },
|
|
903
|
+
},
|
|
904
|
+
examples: ['sentinelctl logs facets -l error -r 24h', 'sentinelctl logs facets -a <agent-id> -r 7d -o json'],
|
|
905
|
+
output: '{criteria, window, facets{agent_id[{value, count, displayName}], host[], severity[], _SYSTEMD_UNIT[], …}}.',
|
|
743
906
|
},
|
|
744
907
|
{
|
|
745
908
|
group: 'logs', name: 'context', run: logsContext, role: 'viewer', mutates: false,
|
|
@@ -761,15 +924,15 @@ const commands = [
|
|
|
761
924
|
summary: 'Log volume and error counts over time',
|
|
762
925
|
args: [{ name: 'text', optional: true, description: 'Same as -q.' }],
|
|
763
926
|
options: logFilterOptions,
|
|
764
|
-
examples: ['sentinelctl logs histogram -r 24h', 'sentinelctl logs histogram -a <agent-id> -r
|
|
765
|
-
output: '{start, end, stepSeconds, total, buckets[]: {at (epoch seconds), total, errors}}.
|
|
927
|
+
examples: ['sentinelctl logs histogram -r 24h', 'sentinelctl logs histogram -a <agent-id> -r 30d -o json'],
|
|
928
|
+
output: '{start, end, complete, coverage, stepSeconds, total, buckets[]: {at (epoch seconds), total, errors}}. Finished days are cached.',
|
|
766
929
|
},
|
|
767
930
|
{
|
|
768
931
|
group: 'logs', name: 'sources', run: logsSources, role: 'viewer', mutates: false,
|
|
769
|
-
summary: 'Log sources (systemd units, containers)
|
|
932
|
+
summary: 'Log sources (systemd units, containers) ranked by exact count',
|
|
770
933
|
options: { range: { alias: 'r', type: 'string', choices: logRanges, default: '24h', description: 'Sample window.' } },
|
|
771
934
|
examples: ['sentinelctl logs sources -r 1h'],
|
|
772
|
-
output: '{range,
|
|
935
|
+
output: '{range, total, complete, sources[]: {value, count}}: top 50 by exact count.',
|
|
773
936
|
},
|
|
774
937
|
{
|
|
775
938
|
group: 'devices', name: 'list', run: devicesList, role: 'viewer', mutates: false,
|
|
@@ -828,7 +991,7 @@ const commands = [
|
|
|
828
991
|
{
|
|
829
992
|
group: 'metrics', name: 'top', run: metricsTop, role: 'viewer', mutates: false,
|
|
830
993
|
summary: 'Rank devices by a metric aggregated over a window, with the fleet-wide distribution',
|
|
831
|
-
description: 'One server-side query over every device: avg/max/min/p95/last of the metric per device, ranked, plus min/p50/p90/p95/max across devices. --compare adds the previous window of the same length. reboots counts uptime resets. Device filters work like devices list (retired devices are excluded by default). Cached for 60s.',
|
|
994
|
+
description: 'One server-side query over every device: avg/max/min/p95/last of the metric per device, ranked, plus min/p50/p90/p95/max across devices. --compare adds the previous window of the same length. reboots counts uptime resets (many devices reboot daily on schedule; compare with the fleet p50). diskDaysLeft forecasts days until the root disk is full from the growth over the window (only growing disks; soonest first). Device filters work like devices list (retired devices are excluded by default). Cached for 60s.',
|
|
832
995
|
args: [{ name: 'metric', description: metricNames.join(', ') }],
|
|
833
996
|
options: {
|
|
834
997
|
agg: { type: 'string', choices: ['avg', 'max', 'min', 'p95', 'last'], default: 'avg', description: 'Per-device aggregation over the window (ignored for reboots).' },
|
|
@@ -836,7 +999,7 @@ const commands = [
|
|
|
836
999
|
from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start (with --to, at most 30 days).' },
|
|
837
1000
|
to: { type: 'string', placeholder: 'ISO', description: 'Absolute window end.' },
|
|
838
1001
|
top: { type: 'integer', default: 10, placeholder: 'N', description: 'Devices to return (1-100).' },
|
|
839
|
-
order: { type: 'string', choices: ['desc', 'asc'],
|
|
1002
|
+
order: { type: 'string', choices: ['desc', 'asc'], description: 'desc = highest first, asc = lowest first (default desc; asc for diskDaysLeft).' },
|
|
840
1003
|
compare: { type: 'boolean', description: 'Also aggregate the previous window: previous and change per device.' },
|
|
841
1004
|
sort: { type: 'string', choices: ['value', 'change'], description: 'Rank by value (default) or by change (needs --compare).' },
|
|
842
1005
|
above: { type: 'number', placeholder: 'N', description: 'Only devices whose value is above N (e.g. --above 90).' },
|
|
@@ -856,20 +1019,22 @@ const commands = [
|
|
|
856
1019
|
'sentinelctl metrics top disk --agg last --above 85 --top 50',
|
|
857
1020
|
'sentinelctl metrics top cpu --agg p95 -r 24h --compare --sort change',
|
|
858
1021
|
'sentinelctl metrics top reboots -r 7d --above 0',
|
|
1022
|
+
'sentinelctl metrics top diskDaysLeft -r 24h --below 14',
|
|
859
1023
|
'sentinelctl metrics top networkTx -r 24h --label fleet=edge -o json',
|
|
860
1024
|
],
|
|
861
1025
|
output: '{metric, aggregation, unit, window{start, end, seconds, range}, criteria, scope{devices, withData, withoutData, matched}, distribution{count, min, avg, p50, p90, p95, max}, previousDistribution?, devices[]: {agentId, displayName, host, status, value, previous?, change?}}.',
|
|
862
1026
|
},
|
|
863
1027
|
{
|
|
864
1028
|
group: 'fleet', name: 'summary', run: fleetSummary, role: 'viewer', mutates: false,
|
|
865
|
-
summary: 'One-call fleet briefing: status
|
|
866
|
-
description: 'Start here to answer "how is the infrastructure?". Combines the device inventory, attention reasons with example devices, the highest current CPU/memory/disk/temperature, and log statistics for the window: totals by level, top and
|
|
1029
|
+
summary: 'One-call fleet briefing: status, attention, busiest devices, metric trends, log levels and error patterns',
|
|
1030
|
+
description: 'Start here to answer "how is the infrastructure?". Combines the device inventory, attention reasons with example devices, the highest current CPU/memory/disk/temperature, offline devices split by how long, metric trends (disks full within 14 days, temperature rises of 15°C or more, reboot counts well above the fleet norm) and log statistics for the window: totals by level, top, rising, falling and new error patterns (compared with the previous window) and the devices with the most errors. About 10KB as JSON.',
|
|
867
1031
|
options: {
|
|
868
1032
|
range: { alias: 'r', type: 'string', choices: logRanges, default: '24h', description: 'Log window.' },
|
|
869
1033
|
logs: { type: 'boolean', description: 'Include log statistics (default on; --no-logs skips them).' },
|
|
1034
|
+
trends: { type: 'boolean', description: 'Include metric trends: disks full within 14 days, temperature rises, unusual reboot counts (default on; --no-trends skips them).' },
|
|
870
1035
|
},
|
|
871
1036
|
examples: ['sentinelctl fleet summary', 'sentinelctl fleet summary -r 1h -o json', 'sentinelctl fleet summary --no-logs'],
|
|
872
|
-
output: '{range, stale, devices{total, status, lifecycle, metrics, logs}, attention[]: {reason, count, devices[≤5]}, resources{cpu, memory, disk, temperature: [{agentId, displayName, value}]}, logs{window, total, devices, levels, errors{total, previousTotal, devices, newPatterns}, topErrorPatterns[], risingErrorPatterns[], fallingErrorPatterns[] (shrunk or gone since the previous window), topErrorDevices[]}}.',
|
|
1037
|
+
output: '{range, stale, devices{total, status, lifecycle, metrics, logs, offline{within24h[], over7dCount, otherCount}}, attention[]: {reason, count, devices[≤5]}, resources{cpu, memory, disk, temperature: [{agentId, displayName, value}]}, trends{diskFullWithin14Days[], temperatureRise[], rebootOutliers{typicalPerWeek, threshold, devices[]}}, logs{complete, window, total, devices, levels, errors{total, previousTotal, devices, newPatterns}, topErrorPatterns[], risingErrorPatterns[], fallingErrorPatterns[] (shrunk or gone since the previous window), topErrorDevices[]}}.',
|
|
873
1038
|
},
|
|
874
1039
|
{
|
|
875
1040
|
group: 'metadata', name: 'get', run: metadataGet, role: 'viewer', mutates: false,
|
|
@@ -1007,7 +1172,7 @@ const commands = [
|
|
|
1007
1172
|
export const catalog = commands;
|
|
1008
1173
|
export const groups = {
|
|
1009
1174
|
fleet: 'One-call briefing of the whole fleet',
|
|
1010
|
-
logs: '
|
|
1175
|
+
logs: 'Count, group and compare logs on the server; exact pattern search; context; volume; sources',
|
|
1011
1176
|
metrics: 'Rank devices by an aggregated metric across the fleet',
|
|
1012
1177
|
devices: 'Device list, details, lookup, reception overview',
|
|
1013
1178
|
metadata: 'Device metadata: read, edit (editor), import/export (admin)',
|
package/src/main.mjs
CHANGED
|
@@ -121,50 +121,67 @@ function overviewHelp() {
|
|
|
121
121
|
|
|
122
122
|
const agentGuide = `Using sentinelctl from AI agents and scripts
|
|
123
123
|
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
124
|
+
SETUP
|
|
125
|
+
- A human logs in once (\`sentinelctl login\`, approve in the browser; 30 days). Later commands are
|
|
126
|
+
non-interactive. Elsewhere pass SENTINEL_TOKEN and SENTINEL_URL.
|
|
127
|
+
- Use -o json or -o ndjson. stdout is data only; notes, hints and errors go to stderr (--quiet drops
|
|
128
|
+
notes and hints). A json/ndjson error is one stderr line {"error":{status,code,message,hint,exitCode}}.
|
|
129
|
+
- Exit codes: 0 ok · 1 server/network (retry later) · 2 usage (fix the arguments) · 3 login/role (do
|
|
130
|
+
not retry). \`sentinelctl commands <group> --json\` describes options and output fields.
|
|
131
|
+
|
|
132
|
+
HOW TO INVESTIGATE: broad to narrow, count before you read
|
|
133
|
+
1. Overview fleet summary -o json
|
|
134
|
+
status, attention reasons, busiest devices, trends (disk full soon, temperature
|
|
135
|
+
rises, unusual reboots), error levels, top/rising/falling/new error patterns.
|
|
136
|
+
2. What and where logs stats -l error -r 24h --by message --compare -o json (patterns, change)
|
|
137
|
+
logs groups -l error -r 24h -o json (every device x source x pattern with
|
|
138
|
+
count, first/last time, latest line)
|
|
139
|
+
3. When logs stats -p "<pattern>" --by agent --timeline -r 7d -o json
|
|
140
|
+
logs histogram -a <id> -r 7d -o json
|
|
141
|
+
4. Read a few lines logs search -p "<pattern>" -a <id> -r 24h -n 50 -o json
|
|
142
|
+
logs context -a <id> -s <source> -t <timestamp> (lines around one log)
|
|
143
|
+
5. Device state devices overview <id> -r 7d -o json · devices show <id> --series cpu,memory -o json
|
|
144
|
+
6. Fleet metrics metrics top memory -r 7d · metrics top disk --agg last --above 85
|
|
145
|
+
metrics top diskDaysLeft --below 14 · metrics top cpu --agg p95 --compare
|
|
146
|
+
metrics top reboots -r 7d (most devices reboot daily; compare with p50)
|
|
147
|
+
|
|
148
|
+
LOG SEARCH RULES (complete answers at low cost)
|
|
149
|
+
- Never download logs to count, rank or summarize them. logs search returns at most 10,000 lines,
|
|
150
|
+
which on this fleet covers minutes, not hours: the rest of the window is silently missing.
|
|
151
|
+
logs stats and logs groups count every log in the window on the server.
|
|
152
|
+
- Prefer logs groups over logs search: errors in one hour are typically ~10,000 lines but ~150
|
|
153
|
+
groups, each with its exact count and latest line.
|
|
154
|
+
- Copy patterns exactly. Patterns from logs stats --by message / logs groups (numbers shown as <N>)
|
|
155
|
+
go to -p/--pattern, which matches that pattern exactly and quickly. -q matches words
|
|
156
|
+
case-insensitively and can also hit unrelated messages.
|
|
157
|
+
- Narrow before widening: device (-a), source (-s), service (--service), severity (-l), pattern (-p).
|
|
158
|
+
Start with 1h-24h; widen to 7d or 30d only when you need history.
|
|
159
|
+
- Fleet-wide windows over 7 days need a narrowing filter (-a, -s, --service, -p or -q with
|
|
160
|
+
--case-sensitive); otherwise the server answers 422 log_query_too_broad.
|
|
161
|
+
Single-device, single-service and single-pattern queries can use the whole retention period
|
|
162
|
+
(31 days by default).
|
|
163
|
+
- Windows over one day are aggregated per day and finished days are cached: repeating a long query
|
|
164
|
+
is cheap, and the CLI continues automatically if the server needs several requests. A partial
|
|
165
|
+
result says so in complete/coverage.
|
|
166
|
+
- To watch for new logs, poll with --since <nextSince from the previous run>, not --follow.
|
|
167
|
+
- --dedupe folds lines repeated within the same second (the agent writes some messages twice).
|
|
168
|
+
- Levels: emergency/critical/error include more severe levels; warning, info and debug are exact.
|
|
169
|
+
- Applications are found by source (-s: systemd unit, container) on one device, or by the service
|
|
170
|
+
label (--service, --env) across devices when the collector sets it; logs stats --by service ranks them.
|
|
171
|
+
|
|
172
|
+
LIMITS (shared with the web console, per user)
|
|
173
|
+
- 2 log queries and 2 metric aggregations at a time; 60 seconds of log query time per minute.
|
|
174
|
+
429 responses are retried automatically with backoff; run commands one after another.
|
|
175
|
+
- Retrying a 422 or 400 without changing the arguments will fail again; read the hint.
|
|
176
|
+
|
|
177
|
+
KEEP OUTPUT SMALL
|
|
178
|
+
- --select a,b.c keeps only those fields (devices list --all -o json: ~700KB -> ~20KB).
|
|
179
|
+
- devices show --series none drops time series; logs groups/stats return compact rows.
|
|
180
|
+
|
|
181
|
+
CHANGES
|
|
182
|
+
- metadata set/bulk/rollback/import and admin commands change data: run with --dry-run first when
|
|
183
|
+
available, check the result, then apply. Viewer tokens cannot change anything.
|
|
184
|
+
- Server error messages are Korean (shared with the web console); hints are English.`;
|
|
168
185
|
|
|
169
186
|
function commandsCatalog(selected = null) {
|
|
170
187
|
return {
|
|
@@ -282,6 +299,15 @@ function apiHint(error, context, command) {
|
|
|
282
299
|
? 'Check the agent ID with `sentinelctl devices lookup <part of the name>`.' : 'Not found; check the ID.';
|
|
283
300
|
}
|
|
284
301
|
if (error.status === 409) return 'Someone saved a change first. Run the command again to apply on top of the latest revision.';
|
|
302
|
+
if (error.code === 'log_query_too_broad') {
|
|
303
|
+
return 'Narrow it: add -a <agent-id>, -s <source>, --service <name>, -p "<pattern>" (from `logs stats --by message`) or --case-sensitive with -q, or use a window of 7 days or less.';
|
|
304
|
+
}
|
|
305
|
+
if (error.code === 'log_budget_exceeded') {
|
|
306
|
+
return `Your log query time for this minute is used up${error.retryAfterSeconds ? `; wait ${error.retryAfterSeconds}s` : ''}. Prefer logs stats/groups with filters over repeated wide searches.`;
|
|
307
|
+
}
|
|
308
|
+
if (error.status === 504) {
|
|
309
|
+
return 'The log/metric backend timed out. Narrow the window or add -a/-s/-l, use -p "<pattern>" instead of -q, or add --case-sensitive. Long aggregations resume from cache if you rerun.';
|
|
310
|
+
}
|
|
285
311
|
if (error.status === 429) return `Too many requests${error.retryAfterSeconds ? `; wait ${error.retryAfterSeconds}s` : ''} and retry. Narrower filters reduce load.`;
|
|
286
312
|
if ([400, 413, 415, 422].includes(error.status)) {
|
|
287
313
|
return command ? `Check the input: sentinelctl ${commandPath(command)} --help` : 'Check the input.';
|