@intflow/sentinelctl 0.3.2 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -38,29 +38,51 @@ sentinelctl --help # overview
38
38
  sentinelctl logs search --help # one command: options, examples, required role, output fields
39
39
  sentinelctl help agents # guide for AI agents and scripts
40
40
  sentinelctl commands --json # machine-readable catalog of every command
41
+ sentinelctl commands logs --json # only one group (or "logs stats" for one command)
41
42
  ```
42
43
 
43
44
  Mistyped commands and options get a suggestion (`Did you mean --range?`), and every error comes with a hint for the next step.
44
45
 
45
- ## Logs
46
+ ## Fleet overview and metrics
46
47
 
47
48
  ```bash
48
- sentinelctl logs search timeout -r 1h
49
- sentinelctl logs search -a <agent-id> -l error -r 24h --all
50
- sentinelctl logs search -s sshd.service --from 2026-09-30T00:00:00Z --to 2026-09-30T06:00:00Z
51
- sentinelctl logs search -l error -f # follow new errors (Ctrl+C to stop)
52
- sentinelctl logs search -q boom -o ndjson | jq .message
49
+ sentinelctl fleet summary # status, offline split, attention, busiest devices, trends, log levels, error patterns
50
+ sentinelctl fleet summary -r 1h -o json # about 10KB
51
+ sentinelctl metrics top memory -r 7d # devices ranked by 7-day average memory + fleet distribution
52
+ sentinelctl metrics top disk --agg last --above 85 --top 50
53
+ sentinelctl metrics top diskDaysLeft --below 14 # disks that fill up within two weeks at the current growth
54
+ sentinelctl metrics top cpu --agg p95 -r 24h --compare --sort change
55
+ sentinelctl metrics top reboots -r 7d --above 0 # many devices reboot daily on schedule; compare with p50
56
+ sentinelctl metrics top networkTx -r 24h --label fleet=edge -o json
57
+ ```
58
+
59
+ `metrics top` runs one query over every device: `--agg avg|max|min|p95|last` per device over `-r 1h|6h|24h|7d|30d` (or `--from/--to`, up to 30 days), ranked, with min/p50/p90/p95/max across devices. Metrics: cpu, memory, disk, swap, load, temperature, processes, tcpConnections, networkRx, networkTx, reboots, diskDaysLeft. `--compare` adds the previous window of the same length. Device filters work like `devices list`; retired devices are excluded by default.
60
+
61
+ ## Logs: count first, read last
62
+
63
+ ```bash
64
+ sentinelctl logs stats -l error -r 24h --by message --compare # which patterns, what changed
65
+ sentinelctl logs groups -l error -r 24h # every device x source x pattern, with count, first/last and latest line
66
+ sentinelctl logs stats -p "<pattern>" --by agent --timeline -r 7d # where and since when
67
+ sentinelctl logs search -p "<pattern>" -a <agent-id> -r 24h -n 50 # a few raw lines
53
68
  sentinelctl logs context -a <agent-id> -s app.service -t 2026-09-30T01:02:03.456Z
54
- sentinelctl logs histogram -l error -r 24h
55
- sentinelctl logs stats -l error -r 1h --by source # exact counts, one server-side query
56
- sentinelctl logs stats -l error -r 24h --by agent --top 10
57
- sentinelctl logs stats -a <agent-id> -r 7d --by message # message patterns, numbers collapsed to <N>
69
+ sentinelctl logs facets -l error -r 24h # top devices, sources, severities at a glance
70
+ sentinelctl logs histogram -a <agent-id> -r 30d
58
71
  sentinelctl logs sources -r 24h
72
+ sentinelctl logs stats --service payment-api --env prod -l error -r 7d --by message # one application across devices (needs service/env labels)
73
+ sentinelctl logs stats -l error -r 24h --by service
74
+ sentinelctl logs search -l error --since <nextSince> -o ndjson # poll only new logs
59
75
  ```
60
76
 
61
- Windows are `15m`, `1h`, `6h`, `24h`, `7d`, or `--from`/`--to` up to 30 days. One page holds 50, 100 or 250 logs; when you ask for several pages (`--pages N`, at most 40, or `--all`) the CLI fetches 250 per request and stops at 10,000 logs. To answer "how many" or "which devices the most", use `logs stats`: it aggregates on the server and returns exact counts instead of downloading logs. `--fields` (raw fields) is admin-only and audited. `--follow` polls at most every 5 seconds.
77
+ - `logs search` returns at most 10,000 lines (250 per request, `--pages N` up to 40 or `--all`). On a busy fleet that covers minutes, so never download logs to count or summarize them: `logs stats` and `logs groups` count every log in the window on the server.
78
+ - `-p/--pattern` takes a pattern exactly as `logs stats --by message` or `logs groups` print it (numbers as `<N>`) and matches only that pattern. `-q` matches words case-insensitively; add `--case-sensitive` for a much faster exact-case match.
79
+ - `--dedupe` folds lines repeated within the same second; `--collapse` on `logs search` is the same as `logs groups`.
80
+ - Windows: `15m`, `1h`, `6h`, `24h`, `7d`, `30d` or `--from/--to` up to the retention period (31 days by default). Fleet-wide aggregations over 7 days need `-a`, `-s`, `-p` or `--case-sensitive`; the server answers `log_query_too_broad` otherwise.
81
+ - Aggregations over a day are computed per day and finished days are cached, so repeating a long query is cheap; the CLI continues automatically when the server needs several requests.
82
+ - Levels: `emergency`, `critical` and `error` include more severe levels; `warning`, `info` (with notice) and `debug` match that level only.
83
+ - `--fields` (raw fields) is admin-only and audited. `--follow` polls at most every 5 seconds and is meant for humans.
62
84
 
63
- Each user can run 2 log queries at a time, web and CLI combined; a third concurrent query gets 429 and GET requests retry automatically.
85
+ Each user can run 2 log queries and 2 metric aggregations at a time (web and CLI combined) and spend 60 seconds of log query time per minute; GET requests that get 429 retry automatically with backoff.
64
86
 
65
87
  ## Devices and metadata
66
88
 
@@ -101,6 +123,7 @@ Any other endpoint: `sentinelctl api "/api/devices?status=offline"`.
101
123
 
102
124
  ## Output and exit codes
103
125
 
126
+ - `--select agentId,displayName,status,metadata.role` keeps only those fields of each item in json/ndjson output (plus top-level scalars), e.g. `devices list --all -o json` drops from ~700KB to ~20KB. `devices show --series none` (or `--series cpu,memory`) drops time series.
104
127
  - `-o table` (default), `-o json` (`--json`), `-o ndjson`. Data goes to stdout; summaries, hints and errors go to stderr (`--quiet` drops summaries and hints). With json/ndjson, an error is one stderr line: `{"error":{"status","code","message","hint","exitCode"}}`.
105
128
  - Exit codes: `0` success, `1` server or network error, `2` usage error, `3` not logged in or insufficient role.
106
129
  - Server error messages are Korean because the web console shares them; hints are English.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@intflow/sentinelctl",
3
- "version": "0.3.2",
3
+ "version": "0.5.0",
4
4
  "description": "Sentinel Fleet Console CLI: search device logs, inspect status and metrics, edit metadata and manage roles",
5
5
  "type": "module",
6
6
  "bin": {
package/src/args.mjs CHANGED
@@ -82,6 +82,10 @@ export function parseArguments(argv, definitions) {
82
82
  parsed = Number(value);
83
83
  if (!Number.isInteger(parsed)) throw new UsageError(`--${name} must be an integer: ${value}`);
84
84
  }
85
+ if (spec.type === 'number') {
86
+ parsed = Number(value);
87
+ if (value.trim() === '' || !Number.isFinite(parsed)) throw new UsageError(`--${name} must be a number: ${value}`);
88
+ }
85
89
  if (spec.choices && !spec.choices.includes(parsed)) {
86
90
  const suggestion = closest(String(parsed), spec.choices.map(String));
87
91
  throw new UsageError(`--${name} must be one of ${spec.choices.join(', ')}: ${value}`,
package/src/commands.mjs CHANGED
@@ -1,10 +1,10 @@
1
1
  import { readFile, writeFile } from 'node:fs/promises';
2
- import { keyValue, UsageError } from './args.mjs';
2
+ import { closest, keyValue, UsageError } from './args.mjs';
3
3
  import { ageText, localTime, percentText, renderKeyValues, renderTable, singleLine } from './format.mjs';
4
4
 
5
5
  export const roleLabels = { viewer: 'viewer (read-only)', editor: 'editor', admin: 'admin' };
6
6
  export const outputChoices = ['table', 'json', 'ndjson'];
7
- const logRanges = ['15m', '1h', '6h', '24h', '7d'];
7
+ const logRanges = ['15m', '1h', '6h', '24h', '7d', '30d'];
8
8
  const deviceRanges = ['30m', '1h', '6h', '24h', '7d', '30d'];
9
9
  const logLevels = ['emergency', 'critical', 'error', 'warning', 'info', 'debug'];
10
10
  const deviceStatuses = ['all', 'online', 'delayed', 'offline', 'unknown', 'maintenance', 'retired'];
@@ -13,15 +13,29 @@ const maxPages = 40;
13
13
  const deviceSorts = ['attention', 'status', 'displayName', 'cpu', 'memory', 'disk', 'load', 'temperature', 'ageSeconds', 'role', 'site', 'group', 'incidentCount'];
14
14
 
15
15
  const logFilterOptions = {
16
- query: { alias: 'q', type: 'string', placeholder: 'text', description: 'Case-insensitive text in the message. Positional words are appended.' },
16
+ query: { alias: 'q', type: 'string', placeholder: 'text', description: 'Words or phrase in the message, case-insensitive by default. Positional words are appended.' },
17
+ pattern: { alias: 'p', type: 'string', placeholder: 'pattern', description: 'Exact message pattern with numbers as <N>, copied from `logs stats --by message` or `logs groups`. Fast and precise.' },
18
+ 'case-sensitive': { type: 'boolean', description: 'Match -q case-sensitively. Much faster on long windows, but misses other casings.' },
17
19
  agent: { alias: 'a', type: 'string', placeholder: 'id', description: 'Exact agent ID (find it with `devices lookup`).' },
18
- level: { alias: 'l', type: 'string', choices: logLevels, description: 'This severity and above (error includes emergency/alert/critical).' },
20
+ level: { alias: 'l', type: 'string', choices: logLevels, description: 'emergency/critical/error include more severe levels; warning, info (with notice) and debug match that level only.' },
19
21
  source: { alias: 's', type: 'string', placeholder: 'source', description: 'Exact source: systemd unit, container name, syslog identifier (see `logs sources`).' },
22
+ service: { type: 'string', placeholder: 'name', description: 'Logical service label (`service` field) across all devices, e.g. payment-api. Logs without the label never match.' },
23
+ env: { type: 'string', placeholder: 'name', description: 'Environment label (`env` field), e.g. prod.' },
20
24
  range: { alias: 'r', type: 'string', choices: logRanges, description: 'Relative window. Default 15m when --from/--to are absent.' },
21
- from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start, e.g. 2026-09-30T00:00:00+09:00. Requires --to; at most 30 days.' },
25
+ from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start, e.g. 2026-09-30T00:00:00+09:00. Requires --to; up to the retention period (31 days by default).' },
22
26
  to: { type: 'string', placeholder: 'ISO', description: 'Absolute window end.' },
23
27
  };
24
28
 
29
+ async function getComplete(context, path, query) {
30
+ let body = await context.client.get(path, query);
31
+ for (let round = 0; body.complete === false && round < 20; round += 1) {
32
+ context.note(`Aggregating step ${body.coverage?.done ?? 0}/${body.coverage?.chunks ?? '?'}, continuing…`);
33
+ body = await context.client.get(path, query);
34
+ }
35
+ if (body.complete === false) context.hint(`Partial result: ${body.coverage?.done}/${body.coverage?.chunks} steps aggregated. Run the same command again to continue (finished days are cached).`);
36
+ return body;
37
+ }
38
+
25
39
  function logCriteria(options, positionals) {
26
40
  const text = [options.query, ...positionals].filter(Boolean).join(' ');
27
41
  if ((options.from && !options.to) || (!options.from && options.to)) {
@@ -33,18 +47,51 @@ function logCriteria(options, positionals) {
33
47
  throw new UsageError(`Cannot parse --${name}: ${value}`, { hint: 'Use ISO 8601, e.g. 2026-09-30T09:00:00+09:00.' });
34
48
  }
35
49
  }
50
+ if (options['case-sensitive'] && !text) throw new UsageError('--case-sensitive needs search text (-q).');
36
51
  return {
37
- q: text || undefined, agentId: options.agent, level: options.level, source: options.source,
52
+ q: text || undefined, agentId: options.agent, level: options.level, source: options.source, service: options.service, env: options.env,
53
+ pattern: options.pattern, match: options['case-sensitive'] ? 'exact' : undefined,
38
54
  range: options.from ? undefined : (options.range ?? '15m'),
39
55
  from: options.from ? new Date(options.from).toISOString() : undefined,
40
56
  to: options.to ? new Date(options.to).toISOString() : undefined,
41
57
  };
42
58
  }
43
59
 
60
+ const listKeys = ['devices', 'logs', 'groups', 'buckets', 'sources', 'sessions', 'users', 'history', 'results', 'attention'];
61
+
62
+ function pick(item, paths) {
63
+ if (!item || typeof item !== 'object') return item;
64
+ const picked = {};
65
+ for (const path of paths) {
66
+ const parts = path.split('.');
67
+ let value = item;
68
+ for (const part of parts) {
69
+ value = value !== null && typeof value === 'object' && Object.hasOwn(value, part) ? value[part] : undefined;
70
+ if (value === undefined) break;
71
+ }
72
+ if (value === undefined) continue;
73
+ let target = picked;
74
+ for (const part of parts.slice(0, -1)) target = (target[part] ??= {});
75
+ target[parts.at(-1)] = value;
76
+ }
77
+ return picked;
78
+ }
79
+
80
+ export function project(value, paths) {
81
+ if (!paths?.length) return value;
82
+ if (Array.isArray(value)) return value.map((item) => pick(item, paths));
83
+ const listKey = listKeys.find((key) => Array.isArray(value?.[key]));
84
+ if (!listKey) return pick(value, paths);
85
+ const meta = Object.fromEntries(Object.entries(value)
86
+ .filter(([key, entry]) => key !== listKey && (key === 'window' || entry === null || typeof entry !== 'object')));
87
+ return { ...meta, [listKey]: value[listKey].map((item) => pick(item, paths)) };
88
+ }
89
+
44
90
  function emit(context, value, table) {
45
91
  const { output } = context.global;
46
- if (output === 'json') context.out(JSON.stringify(value, null, 2));
47
- else if (output === 'ndjson') for (const item of Array.isArray(value) ? value : [value]) context.out(JSON.stringify(item));
92
+ const selected = project(value, context.select);
93
+ if (output === 'json') context.out(JSON.stringify(selected, null, 2));
94
+ else if (output === 'ndjson') for (const item of Array.isArray(selected) ? selected : [selected]) context.out(JSON.stringify(item));
48
95
  else context.out(table());
49
96
  }
50
97
 
@@ -63,14 +110,43 @@ function logKey(log) {
63
110
  function emptyLogHint(criteria) {
64
111
  const tips = [];
65
112
  if (criteria.range && criteria.range !== '7d') tips.push('widen the window (-r 24h or -r 7d)');
66
- if (criteria.q) tips.push('shorten or drop the search text');
113
+ if (criteria.q) tips.push(criteria.match === 'exact' ? 'drop --case-sensitive' : 'shorten or drop the search text');
114
+ if (criteria.pattern) tips.push('check the pattern with `sentinelctl logs stats --by message`');
67
115
  if (criteria.source) tips.push('check source names with `sentinelctl logs sources -r 24h`');
68
116
  if (criteria.agentId) tips.push('check the agent ID with `sentinelctl devices lookup <name>`');
69
117
  if (criteria.level) tips.push('drop -l');
70
118
  return tips.length ? `No logs matched. Try: ${tips.join('; ')}.` : 'No logs matched.';
71
119
  }
72
120
 
121
+ function timestampNanos(value) {
122
+ const match = /^(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2})(?:\.(\d{1,9}))?(Z|[+-]\d{2}:\d{2})$/.exec(String(value));
123
+ if (!match) return BigInt(Date.parse(value)) * 1_000_000n;
124
+ return BigInt(Date.parse(`${match[1]}${match[3]}`)) * 1_000_000n + BigInt((match[2] ?? '').padEnd(9, '0'));
125
+ }
126
+
127
+ function dedupeLogs(logs) {
128
+ const result = [];
129
+ for (const log of logs) {
130
+ const previous = result.at(-1);
131
+ const sameSecond = previous && String(previous.timestamp).slice(0, 19) === String(log.timestamp).slice(0, 19);
132
+ const strip = (message) => String(message).replace(/^\[[^\]]+\]\s*/, '');
133
+ if (sameSecond && previous.agentId === log.agentId && previous.source === log.source && strip(previous.message) === strip(log.message)) {
134
+ previous.repeat = (previous.repeat ?? 1) + 1;
135
+ continue;
136
+ }
137
+ result.push({ ...log });
138
+ }
139
+ return result;
140
+ }
141
+
73
142
  async function logsSearch(context, { options, positionals }) {
143
+ if (options.collapse) return logsGroups(context, { options, positionals });
144
+ if (options.since) {
145
+ if (options.from || options.to || options.range) throw new UsageError('--since cannot be combined with --range/--from/--to.');
146
+ if (!Number.isFinite(Date.parse(options.since))) throw new UsageError(`Cannot parse --since: ${options.since}`, { hint: 'Pass the nextSince value printed by the previous run.' });
147
+ options.from = options.since;
148
+ options.to = new Date().toISOString();
149
+ }
74
150
  const criteria = logCriteria(options, positionals);
75
151
  const paging = options.all || options.pages > 1;
76
152
  const limit = options.fields ? 50 : (options.limit ?? (paging ? 250 : 100));
@@ -89,11 +165,17 @@ async function logsSearch(context, { options, positionals }) {
89
165
  if (!last.hasMore || !last.nextCursor) break;
90
166
  cursor = last.nextCursor;
91
167
  }
92
- if (context.global.output === 'json') {
93
- context.out(JSON.stringify({ criteria: last.criteria, window: last.window, count: logs.length, hasMore: last.hasMore, logs }, null, 2));
94
- } else {
95
- emit(context, logs, () => renderTable(logs, logColumns, { width: context.width, wide: options.wide }));
168
+ if (options.since) {
169
+ const since = timestampNanos(options.since);
170
+ for (let index = logs.length - 1; index >= 0; index -= 1) if (timestampNanos(logs[index].timestamp) <= since) logs.splice(index, 1);
96
171
  }
172
+ const shown = options.dedupe ? dedupeLogs(logs) : logs;
173
+ const nextSince = logs[0]?.timestamp ?? options.since ?? null;
174
+ emit(context, context.global.output === 'json'
175
+ ? { criteria: last.criteria, window: last.window, count: logs.length, ...(options.dedupe ? { rows: shown.length } : {}), hasMore: last.hasMore, ...(options.since ? { nextSince } : {}), logs: shown }
176
+ : shown,
177
+ () => renderTable(shown, options.dedupe ? [...logColumns.slice(0, 4), { header: 'x', value: (log) => (log.repeat ? `×${log.repeat}` : ''), max: 5 }, logColumns[4]] : logColumns, { width: context.width, wide: options.wide }));
178
+ if (options.since) context.hint(`Next time: --since ${nextSince}`);
97
179
  if (!logs.length) context.hint(emptyLogHint(criteria));
98
180
  else context.note(`${logs.length} logs · ${localTime(last.window.start)} → ${localTime(last.window.end)}`);
99
181
  if (last.hasMore && options.all) context.hint(`Stopped at the ${maxPages * 250}-log cap. Narrow the filters, or use \`sentinelctl logs stats\` for counts.`);
@@ -129,23 +211,208 @@ async function followLogs(context, query, options) {
129
211
  }
130
212
 
131
213
 
132
- const statsDimensions = ['source', 'agent', 'level', 'message'];
214
+ const statsDimensions = ['source', 'agent', 'level', 'message', 'service'];
215
+
216
+ const sparkBlocks = ['▁', '▂', '▃', '▄', '▅', '▆', '▇', '█'];
217
+
218
+ function sparkline(points, window, step) {
219
+ if (!points?.length || !step) return '';
220
+ const start = Math.floor(Date.parse(window.start) / 1000 / step) * step;
221
+ const end = Date.parse(window.end) / 1000;
222
+ const counts = new Map(points);
223
+ const values = [];
224
+ for (let at = start; at < end; at += step) values.push(counts.get(at) ?? 0);
225
+ const peak = Math.max(1, ...values);
226
+ return values.map((value) => (value ? sparkBlocks[Math.min(7, Math.floor((value / peak) * 7.999))] : ' ')).join('');
227
+ }
228
+
229
+ function quoteArgument(text) {
230
+ return `"${String(text).replaceAll('\\', '\\\\').replaceAll('"', '\\"')}"`;
231
+ }
232
+
233
+ function narrowingHint(criteria) {
234
+ const window = criteria.range ? `-r ${criteria.range}` : `--from ${criteria.from} --to ${criteria.to}`;
235
+ return { window, level: criteria.level ? ` -l ${criteria.level}` : '' };
236
+ }
133
237
 
134
238
  async function logsStats(context, { options, positionals }) {
135
239
  const criteria = logCriteria(options, positionals);
136
- const body = await context.client.get('/api/logs/stats', { ...criteria, by: options.by, top: options.top });
137
- const label = { source: 'SOURCE', agent: 'DEVICE', level: 'LEVEL', message: 'MESSAGE PATTERN' }[body.criteria.by];
240
+ if (options.sort === 'change' && !options.compare) {
241
+ throw new UsageError('--sort change needs --compare.', { hint: 'Example: sentinelctl logs stats -l error -r 24h --by message --compare --sort change' });
242
+ }
243
+ const body = await getComplete(context, '/api/logs/stats', {
244
+ ...criteria, by: options.by, top: options.top, compare: options.compare ? 1 : undefined, sort: options.sort, timeline: options.timeline ? 1 : undefined,
245
+ });
246
+ const label = { source: 'SOURCE', agent: 'DEVICE', level: 'LEVEL', message: 'MESSAGE PATTERN', service: 'SERVICE' }[body.criteria.by];
247
+ const signed = (value) => (value === null || value === undefined ? '?' : value > 0 ? `+${value}` : String(value));
138
248
  emit(context, context.global.output === 'ndjson' ? body.groups : body, () => renderTable(body.groups, [
139
249
  { header: 'COUNT', value: (group) => String(group.count) },
140
- { header: 'SHARE', value: (group) => (body.total ? `${((group.count / body.total) * 100).toFixed(1)}%` : '-') },
250
+ ...(body.previous ? [
251
+ { header: 'PREV', value: (group) => (group.previousCount === null ? '?' : String(group.previousCount)) },
252
+ { header: 'CHANGE', value: (group) => (group.isNew ? 'new' : signed(group.change)) },
253
+ ] : [{ header: 'SHARE', value: (group) => (body.total ? `${((group.count / body.total) * 100).toFixed(1)}%` : '-') }]),
141
254
  { header: 'DEVICES', value: (group) => String(group.devices) },
255
+ ...(body.timelineStepSeconds ? [{ header: 'TREND', value: (group) => sparkline(group.timeline, body.window, body.timelineStepSeconds), max: 48 }] : []),
256
+ ...(body.timelineStepSeconds ? [{ header: 'LAST', value: (group) => (group.last ? localTime(group.last).slice(5, 16) : '-'), max: 11 }] : []),
142
257
  ...(body.criteria.by === 'agent' ? [{ header: 'AGENT ID', value: (group) => group.key, max: 36 }] : []),
143
258
  { header: label, value: (group) => (body.criteria.by === 'agent' ? group.displayName ?? group.host ?? group.key : group.key) },
144
259
  ], { width: context.width, wide: options.wide }));
145
260
  context.note(`${body.total} logs from ${body.devices} devices · ${localTime(body.window.start)} → ${localTime(body.window.end)}`);
261
+ if (body.timelineStepSeconds) context.note(`TREND: one column per ${Math.round(body.timelineStepSeconds / 60)} min; first/last (bucket precision) are in -o json.`);
262
+ if (body.previous) {
263
+ context.note(`Previous window ${localTime(body.previous.window.start)} → ${localTime(body.previous.window.end)}: ${body.previous.total} logs (${signed(body.total - body.previous.total)}) · ${body.newGroupCount} new groups`);
264
+ if (body.falling?.length && body.criteria.sort !== 'change') {
265
+ context.note(`Largest drops: ${body.falling.map((group) => `${singleLine(group.key).slice(0, 50)} (${group.change})`).join(' | ')}`);
266
+ }
267
+ if (!body.previous.exhaustive) context.hint('The previous window has too many groups to compare exactly; "?" marks groups outside its top list.');
268
+ }
146
269
  if (body.truncated) context.hint(`Showing the top ${body.groups.length} of ${body.groupCount} groups. Raise --top (max 100) or add filters.`);
147
270
  if (body.approximate) context.hint('Message patterns are approximate on this server (collapse_nums is unavailable); narrow the window or filters for exact counts.');
148
271
  if (!body.total) context.hint(emptyLogHint(criteria));
272
+ const sample = body.groups.find((group) => group.count > 0);
273
+ if (body.criteria.by === 'message' && sample) {
274
+ const { window, level } = narrowingHint(criteria);
275
+ context.hint(`Next: who and where → sentinelctl logs groups -p ${quoteArgument(sample.key)} ${window}${level}; raw lines → sentinelctl logs search -p ${quoteArgument(sample.key)} ${window} -n 50`);
276
+ }
277
+ }
278
+
279
+ const groupColumns = [
280
+ { header: 'COUNT', value: (group) => String(group.count) },
281
+ { header: 'LAST', value: (group) => (group.last ? localTime(group.last).slice(5, 16) : '-'), max: 11 },
282
+ { header: 'DEVICE', value: (group) => group.displayName ?? group.agentId, max: 26 },
283
+ { header: 'SOURCE', value: (group) => group.source, max: 22 },
284
+ { header: 'PATTERN', value: (group) => group.pattern },
285
+ ];
286
+
287
+ async function logsGroups(context, { options, positionals }) {
288
+ const criteria = logCriteria(options, positionals);
289
+ const body = await getComplete(context, '/api/logs/groups', { ...criteria, top: options.top ?? options.limit });
290
+ emit(context, context.global.output === 'ndjson' ? body.groups : body, () => renderTable(body.groups, groupColumns, { width: context.width, wide: options.wide }));
291
+ context.note(`${body.total} logs in ${body.groupCount} groups (${body.patternCount} patterns, ${body.devices} devices) · ${localTime(body.window.start)} → ${localTime(body.window.end)}`);
292
+ if (body.truncated) context.hint(`Showing ${body.groups.length} of ${body.groupCount} groups, largest first. Raise --top (max 500) or add filters.`);
293
+ if (body.approximate) context.hint('Patterns are approximate on this server (collapse_nums is unavailable).');
294
+ if (!body.total) context.hint(emptyLogHint(criteria));
295
+ const first = body.groups[0];
296
+ if (first) {
297
+ const { window } = narrowingHint(criteria);
298
+ context.hint(`Each group keeps its latest line in -o json (latest.message). Raw lines of one group: sentinelctl logs search -a ${first.agentId} -p ${quoteArgument(first.pattern)} ${window} -n 50`);
299
+ }
300
+ }
301
+
302
+ async function logsFacets(context, { options, positionals }) {
303
+ const criteria = logCriteria({ ...options, range: options.range ?? (options.from ? undefined : '1h') }, positionals);
304
+ const body = await context.client.get('/api/logs/facets', { ...criteria, top: options.top });
305
+ emit(context, body, () => Object.entries(body.facets).map(([field, values]) => `${field}:\n${values
306
+ .map((item) => ` ${String(item.count).padStart(9)} ${item.displayName && item.displayName !== item.value ? `${item.displayName} (${item.value})` : item.value}`).join('\n')}`).join('\n'));
307
+ context.note(`${localTime(body.window.start)} → ${localTime(body.window.end)}`);
308
+ }
309
+
310
+ const metricNames = ['cpu', 'memory', 'disk', 'swap', 'load', 'temperature', 'processes', 'tcpConnections', 'networkRx', 'networkTx', 'reboots', 'diskDaysLeft'];
311
+ const metricWindows = ['1h', '6h', '24h', '7d', '30d'];
312
+ const deviceSeries = ['cpu', 'memory', 'disk', 'load', 'temperature', 'processes', 'tcpConnections', 'networkRx', 'networkTx', 'swap', 'uptime'];
313
+
314
+ function metricValueText(value, unit) {
315
+ if (value === null || value === undefined) return '-';
316
+ if (unit === '%') return `${value.toFixed(1)}%`;
317
+ if (unit === '°C') return `${value.toFixed(1)}°C`;
318
+ if (unit === 'B/s') return `${bytesText(value)}/s`;
319
+ return String(Math.round(value * 100) / 100);
320
+ }
321
+
322
+ async function metricsTop(context, { options, positionals }) {
323
+ const metric = positionals[0];
324
+ if (!metric) throw new UsageError('A metric name is required.', { hint: `One of: ${metricNames.join(', ')}. Example: sentinelctl metrics top memory -r 7d` });
325
+ if (!metricNames.includes(metric)) {
326
+ const suggestion = closest(metric, metricNames);
327
+ throw new UsageError(`Unknown metric: ${metric}`, { hint: `${suggestion ? `Did you mean ${suggestion}? ` : ''}One of: ${metricNames.join(', ')}` });
328
+ }
329
+ if ((options.from && !options.to) || (!options.from && options.to)) throw new UsageError('--from and --to must be given together.');
330
+ if (options.from && options.range) throw new UsageError('--range cannot be combined with --from/--to.');
331
+ if (options.sort === 'change' && !options.compare) throw new UsageError('--sort change needs --compare.', { hint: `Example: sentinelctl metrics top ${metric} -r 24h --compare --sort change` });
332
+ for (const label of options.label ?? []) keyValue(label, '--label');
333
+ const body = await context.client.get('/api/metrics/summary', {
334
+ metric, agg: options.agg, range: options.from ? undefined : options.range,
335
+ from: options.from ? new Date(options.from).toISOString() : undefined, to: options.to ? new Date(options.to).toISOString() : undefined,
336
+ top: options.top, order: options.order, sort: options.sort, compare: options.compare ? 1 : undefined, above: options.above, below: options.below,
337
+ query: options.query, status: options.status, lifecycle: options.lifecycle, role: options.role, site: options.site, group: options.group,
338
+ maintainer: options.maintainer, fleet: options.fleet, labels: options.label?.length ? JSON.stringify(options.label) : undefined,
339
+ });
340
+ const text = (value) => metricValueText(value, body.unit);
341
+ const signed = (value) => (value === null || value === undefined ? '?' : `${value > 0 ? '+' : ''}${text(value)}`);
342
+ emit(context, context.global.output === 'ndjson' ? body.devices : body, () => renderTable(body.devices, [
343
+ { header: 'AGENT ID', value: (device) => device.agentId, max: 36 },
344
+ { header: 'NAME', value: (device) => device.displayName, max: 32 },
345
+ { header: 'STATUS', value: (device) => device.status, max: 9 },
346
+ { header: body.aggregation.toUpperCase(), value: (device) => text(device.value) },
347
+ ...(body.criteria.compare ? [
348
+ { header: 'PREV', value: (device) => text(device.previous) },
349
+ { header: 'CHANGE', value: (device) => signed(device.change) },
350
+ ] : []),
351
+ ], { width: context.width }));
352
+ const spread = body.distribution;
353
+ context.note(`${body.metric} ${body.aggregation} over ${body.window.range ?? `${localTime(body.window.start)} → ${localTime(body.window.end)}`} · ${body.scope.withData} of ${body.scope.devices} devices have data`
354
+ + (spread.count ? ` · min ${text(spread.min)} · p50 ${text(spread.p50)} · p90 ${text(spread.p90)} · max ${text(spread.max)}` : ''));
355
+ if (body.criteria.above !== null || body.criteria.below !== null) context.note(`${body.scope.matched} devices within the --above/--below bounds.`);
356
+ if (!body.scope.withData) context.hint('No device reported this metric in the window. Widen -r or drop device filters.');
357
+ else if (body.scope.matched > body.devices.length) context.hint(`Showing ${body.devices.length} of ${body.scope.matched}. Raise --top (max 100).`);
358
+ }
359
+
360
+ async function fleetSummary(context, { options }) {
361
+ const body = await context.client.get('/api/fleet/summary', { range: options.range, logs: options.logs === false ? 0 : undefined, trends: options.trends === false ? 0 : undefined });
362
+ emit(context, body, () => {
363
+ const counts = (record) => Object.entries(record ?? {}).sort((left, right) => right[1] - left[1]).map(([key, value]) => `${key} ${value}`).join(', ');
364
+ const names = (devices) => devices.map((device) => device.displayName).join(', ');
365
+ const sections = [renderKeyValues([
366
+ ['Devices', `${body.devices.total} (${counts(body.devices.status)})`],
367
+ ['Metrics', counts(body.devices.metrics)],
368
+ ['Logs', counts(body.devices.logs)],
369
+ ['Data', body.stale ? `stale (${body.staleReason ?? ''})` : 'fresh'],
370
+ ])];
371
+ sections.push('', 'Attention:', body.attention.length
372
+ ? renderTable(body.attention, [
373
+ { header: 'COUNT', value: (entry) => String(entry.count) },
374
+ { header: 'REASON', value: (entry) => entry.reason, max: 24 },
375
+ { header: 'EXAMPLES', value: (entry) => names(entry.devices) },
376
+ ], { width: context.width })
377
+ : ' none');
378
+ const offline = body.devices.offline;
379
+ if (offline) {
380
+ sections.push('', `Offline: ${offline.within24h.length} within 24h${offline.within24h.length ? ` (${offline.within24h.map((device) => `${device.displayName} ${device.offlineHours}h`).join(', ')})` : ''} · ${offline.otherCount} for 1-7 days · ${offline.over7dCount} over 7 days`);
381
+ }
382
+ sections.push('', 'Highest now (online, active):', renderKeyValues(Object.entries(body.resources).map(([field, devices]) => [
383
+ field, devices.map((device) => `${device.displayName} ${field === 'temperature' ? `${device.value}°C` : `${device.value}%`}`).join(', ') || '-',
384
+ ])));
385
+ if (body.trends?.error) sections.push('', `Trends: ${body.trends.error}`);
386
+ else if (body.trends) {
387
+ const list = (devices, format) => devices.map((device) => `${device.displayName} ${format(device)}`).join(', ') || 'none';
388
+ sections.push('', 'Trends:', renderKeyValues([
389
+ ['disk full <14d', list(body.trends.diskFullWithin14Days, (device) => `${device.value}d`)],
390
+ ['temperature +15°C', list(body.trends.temperatureRise, (device) => `${device.previous}→${device.value}°C`)],
391
+ [`reboots >${body.trends.rebootOutliers.threshold}/7d`, `${list(body.trends.rebootOutliers.devices, (device) => `${device.value}`)} (typical ${body.trends.rebootOutliers.typicalPerWeek}/7d)`],
392
+ ]));
393
+ }
394
+ if (body.logs?.complete === false) sections.push('', 'Logs: partial (some days still aggregating); run again to complete.');
395
+ if (body.logs?.error) sections.push('', `Logs: ${body.logs.error}`);
396
+ else if (body.logs) {
397
+ const errors = body.logs.errors;
398
+ sections.push('', `Logs (${body.range}): ${body.logs.total} from ${body.logs.devices} devices · ${counts(body.logs.levels)}`,
399
+ `Errors: ${errors.total}${errors.previousTotal === null ? '' : ` (previous window ${errors.previousTotal})`} from ${errors.devices} devices · ${errors.newPatterns} new patterns`,
400
+ '', 'Top error patterns:', renderTable(body.logs.topErrorPatterns, [
401
+ { header: 'COUNT', value: (group) => String(group.count) },
402
+ { header: 'CHANGE', value: (group) => (group.isNew ? 'new' : group.change === null ? '?' : `${group.change > 0 ? '+' : ''}${group.change}`) },
403
+ { header: 'DEVICES', value: (group) => String(group.devices) },
404
+ { header: 'PATTERN', value: (group) => group.pattern },
405
+ ], { width: context.width }),
406
+ '', `Top error devices: ${body.logs.topErrorDevices.map((device) => `${device.displayName} ${device.count}`).join(', ') || '-'}`);
407
+ const changes = (label, groups) => {
408
+ if (groups?.length) sections.push(`${label}: ${groups.map((group) => `${singleLine(group.pattern).slice(0, 60)} (${group.change > 0 ? '+' : ''}${group.change})`).join(' | ')}`);
409
+ };
410
+ changes('Rising', body.logs.risingErrorPatterns);
411
+ changes('Falling', body.logs.fallingErrorPatterns);
412
+ }
413
+ return sections.join('\n');
414
+ });
415
+ context.hint('Drill down: logs groups -l error -r 24h · logs stats -l error --by message --compare --timeline · metrics top diskDaysLeft · devices overview <id>');
149
416
  }
150
417
 
151
418
  async function logsContext(context, { options }) {
@@ -163,7 +430,7 @@ async function logsContext(context, { options }) {
163
430
  }
164
431
 
165
432
  async function logsHistogram(context, { options, positionals }) {
166
- const body = await context.client.get('/api/logs/histogram', logCriteria(options, positionals));
433
+ const body = await getComplete(context, '/api/logs/histogram', logCriteria(options, positionals));
167
434
  emit(context, context.global.output === 'ndjson' ? body.buckets : body, () => {
168
435
  const peak = Math.max(1, ...body.buckets.map((bucket) => bucket.total));
169
436
  return renderTable(body.buckets, [
@@ -182,7 +449,7 @@ async function logsSources(context, { options }) {
182
449
  { header: 'SOURCE', value: (source) => source.value },
183
450
  { header: 'COUNT', value: (source) => String(source.count) },
184
451
  ], { width: context.width }));
185
- context.note(`Top sources in a sample of the latest ${body.sampleSize} logs.`);
452
+ context.note(`Top ${body.sources.length} sources by exact count over ${body.range} (${body.total} logs).`);
186
453
  }
187
454
 
188
455
  const deviceColumns = [
@@ -217,11 +484,8 @@ async function devicesList(context, { options, positionals }) {
217
484
  }
218
485
  for (const page of pages) devices.push(...results.get(page));
219
486
  }
220
- if (context.global.output === 'json') {
221
- context.out(JSON.stringify({ matched: first.matched, counts: first.counts, stale: first.stale, staleReason: first.staleReason, devices }, null, 2));
222
- } else {
223
- emit(context, devices, () => renderTable(devices, deviceColumns, { width: context.width }));
224
- }
487
+ emit(context, context.global.output === 'json' ? { matched: first.matched, counts: first.counts, stale: first.stale, staleReason: first.staleReason, devices } : devices,
488
+ () => renderTable(devices, deviceColumns, { width: context.width }));
225
489
  context.note(`${devices.length} of ${first.matched} devices${options.all ? '' : ` · page ${first.page}/${first.totalPages}`}`);
226
490
  if (first.stale) context.hint(`Metrics are stale (${first.staleReason ?? 'metrics query failed'}). Check \`sentinelctl health\`.`);
227
491
  if (!devices.length) context.hint('No devices matched. Add --lifecycle all to include maintenance/retired devices, or shorten the search text.');
@@ -235,7 +499,11 @@ function requireAgentId(positionals) {
235
499
 
236
500
  async function devicesShow(context, { options, positionals }) {
237
501
  const agentId = requireAgentId(positionals);
502
+ const keep = options.series === undefined ? null : options.series === 'none' ? [] : options.series.split(',').map((name) => name.trim()).filter(Boolean);
503
+ const unknownSeries = (keep ?? []).filter((name) => !deviceSeries.includes(name));
504
+ if (unknownSeries.length) throw new UsageError(`Unknown series: ${unknownSeries.join(', ')}`, { hint: `Use none or a comma-separated list of: ${deviceSeries.join(', ')}` });
238
505
  const body = await context.client.get(`/api/devices/${encodeURIComponent(agentId)}`, { range: options.range, refresh: options.refresh ? 1 : undefined });
506
+ if (keep) body.series = Object.fromEntries(Object.entries(body.series ?? {}).filter(([name]) => keep.includes(name)));
239
507
  const device = body.device;
240
508
  const current = body.current ?? {};
241
509
  emit(context, body, () => renderKeyValues([
@@ -246,7 +514,7 @@ async function devicesShow(context, { options, positionals }) {
246
514
  ['Maintainer', device.metadata?.maintainer], ['Lifecycle', device.metadata?.lifecycle], ['Notes', device.metadata?.description],
247
515
  ['Data', body.stale ? `stale (${body.staleReason ?? ''})` : 'fresh'],
248
516
  ]));
249
- if (context.global.output === 'table') context.hint(`Time series (${body.range}, ${body.stepSeconds}s step) are in \`series\` with -o json.`);
517
+ if (context.global.output === 'table') context.hint(`Time series (${body.range}, ${body.stepSeconds}s step) are in \`series\` with -o json; --series cpu,memory or --series none keeps the output small.`);
250
518
  }
251
519
 
252
520
  async function devicesLookup(context, { positionals }) {
@@ -547,14 +815,14 @@ async function rawApi(context, { options, positionals }) {
547
815
  }
548
816
  const url = new URL(path, context.client.server);
549
817
  const result = await context.client.request(options.method, url.pathname, { query: Object.fromEntries(url.searchParams), body });
550
- context.out(JSON.stringify(result.body, null, 2));
818
+ context.out(JSON.stringify(project(result.body, context.select), null, 2));
551
819
  }
552
820
 
553
821
  const commands = [
554
822
  {
555
823
  group: 'logs', name: 'search', run: logsSearch, role: 'viewer', mutates: false,
556
- summary: 'Search logs by window, device, severity, source and text; or follow live',
557
- description: 'Newest first. Default 15m window and 100 logs. When fetching several pages, 250 logs are fetched per request, up to 10,000 in total. To count or rank logs, use `logs stats` instead.',
824
+ summary: 'Raw log lines by window, device, severity, source, text or exact pattern; --collapse for groups',
825
+ description: 'Newest first. Default 15m window and 100 logs. When fetching several pages, 250 logs are fetched per request, up to 10,000 in total, which on busy fleets covers only minutes. Use it for a handful of example lines after narrowing with `logs stats`/`logs groups` and -p. --collapse returns every matching group instead of lines (same as `logs groups`).',
558
826
  args: [{ name: 'text', optional: true, description: 'Same as -q (words are joined with spaces).' }],
559
827
  options: {
560
828
  ...logFilterOptions,
@@ -562,6 +830,10 @@ const commands = [
562
830
  pages: { type: 'integer', placeholder: 'N', description: `Follow the cursor for up to N pages (1-${maxPages}).` },
563
831
  all: { type: 'boolean', description: 'Fetch every page of the window, stopping at 10,000 logs.' },
564
832
  fields: { type: 'boolean', description: 'Include raw fields (admin only, audited, 50 logs per page).' },
833
+ collapse: { type: 'boolean', description: 'Group by device, source and pattern over the whole window with counts, first/last time and the latest line (see logs groups).' },
834
+ top: { type: 'integer', placeholder: 'N', description: 'With --collapse: groups to return (1-500, default 50).' },
835
+ dedupe: { type: 'boolean', description: 'Fold lines repeated within the same second on the same device and source into one with a repeat count.' },
836
+ since: { type: 'string', placeholder: 'ISO', description: 'Only logs after this timestamp (exclusive) up to now; pass the printed nextSince to poll without re-reading.' },
565
837
  follow: { alias: 'f', type: 'boolean', description: 'Keep printing new logs (for humans; never ends, agents should not use it).' },
566
838
  interval: { type: 'integer', default: 5, placeholder: 'seconds', description: 'Polling interval for --follow (min 5).' },
567
839
  wide: { alias: 'w', type: 'boolean', description: 'Do not truncate messages in table output.' },
@@ -570,30 +842,67 @@ const commands = [
570
842
  roleNote: '--fields requires admin',
571
843
  examples: [
572
844
  'sentinelctl logs search -l error -r 1h',
845
+ 'sentinelctl logs search -p "tegra-xusb <N>.usb: not all ports suspended: -<N>" -r 24h -n 50 -o json',
846
+ 'sentinelctl logs search -l error -r 24h --collapse -o json',
847
+ 'sentinelctl logs search -l error --since 2026-09-30T01:02:03.456789Z -o ndjson',
573
848
  'sentinelctl logs search "connection reset" -r 24h -o ndjson',
574
849
  'sentinelctl logs search -a <agent-id> -s ssh.service -r 6h --all -o json',
575
850
  'sentinelctl logs search --from 2026-09-30T09:00:00+09:00 --to 2026-09-30T10:00:00+09:00 -l warning',
576
851
  ],
577
- output: 'logs[]: timestamp (UTC ISO), severity, priority, agentId, host, displayName, source, message (+fields). -o json wraps them as {criteria, window, count, hasMore, logs}.',
852
+ output: 'logs[]: timestamp (UTC ISO), severity, priority, agentId, host, displayName, source, message (+fields, +repeat with --dedupe). -o json wraps them as {criteria, window, count, hasMore, nextSince?, logs}.',
578
853
  },
579
854
  {
580
855
  group: 'logs', name: 'stats', run: logsStats, role: 'viewer', mutates: false,
581
- summary: 'Count and rank logs by source, device, severity or message pattern (one server-side query)',
582
- description: 'Aggregates in VictoriaLogs instead of downloading logs, so counts are exact over the whole window. message groups by pattern with numbers collapsed to <N>. Cached for 60s on the server.',
856
+ summary: 'Count and rank logs by source, device, severity or message pattern; compare windows; trends',
857
+ description: 'Aggregates in VictoriaLogs instead of downloading logs, so counts are exact over the whole window. message groups by pattern with numbers collapsed to <N>; pass a pattern to -p for exact follow-ups. Windows over one day are computed per day and finished days are cached, so repeating a long query is cheap; long requests continue automatically.',
583
858
  args: [{ name: 'text', optional: true, description: 'Same as -q.' }],
584
859
  options: {
585
860
  ...logFilterOptions,
586
- by: { alias: 'b', type: 'string', choices: statsDimensions, default: 'source', description: 'Group by source, agent (device), level or message pattern.' },
861
+ by: { alias: 'b', type: 'string', choices: statsDimensions, default: 'source', description: 'Group by source, agent (device), level, message pattern or service label.' },
587
862
  top: { type: 'integer', default: 20, placeholder: 'N', description: 'Groups to return (1-100).' },
863
+ compare: { type: 'boolean', description: 'Also count the previous window of the same length: previousCount, change and isNew per group.' },
864
+ timeline: { type: 'boolean', description: 'Add per-group counts over time (about 48 buckets): when a pattern started, stopped or spiked.' },
865
+ sort: { type: 'string', choices: ['count', 'change'], description: 'Rank by count (default) or by increase over the previous window (needs --compare).' },
588
866
  wide: { alias: 'w', type: 'boolean', description: 'Do not truncate long keys in table output.' },
589
867
  },
590
868
  examples: [
591
869
  'sentinelctl logs stats -l error -r 1h --by source',
592
870
  'sentinelctl logs stats -l error -r 24h --by agent --top 10 -o json',
871
+ 'sentinelctl logs stats -l error -r 24h --by message --compare --sort change',
872
+ 'sentinelctl logs stats -l error -r 7d --by message --timeline --top 10',
593
873
  'sentinelctl logs stats -a <agent-id> -r 7d --by message',
594
874
  'sentinelctl logs stats "connection refused" -r 24h --by agent',
595
875
  ],
596
- output: '{criteria, window, total, devices, groupCount, truncated, approximate, groups[]: {key, count, devices (distinct agents), host?, displayName?}}.',
876
+ output: '{criteria, window, total, devices, groupCount, truncated, approximate, complete, coverage{chunks, done}, groups[]: {key, count, devices (distinct agents), first?, last? (with --timeline, bucket precision), host?, displayName?, searchText?, previousCount?, change?, isNew?, timeline?: [[epochSeconds, count]]}, timelineStepSeconds?, previous?: {window, total, devices, exhaustive}, newGroupCount?, rising?[≤5], falling?[≤5] (largest increases and decreases across all groups, not only the returned top)}.',
877
+ },
878
+ {
879
+ group: 'logs', name: 'groups', run: logsGroups, role: 'viewer', mutates: false,
880
+ summary: 'Every distinct (device, source, message pattern) in the window with counts, first/last time and the latest line',
881
+ description: 'The lossless way to see what happened: each group is counted over the whole window and keeps its latest raw line, so nothing is skipped the way a 10,000-line download skips. Typically 50-100x smaller than the raw logs. Windows over one day are computed per day and cached.',
882
+ args: [{ name: 'text', optional: true, description: 'Same as -q.' }],
883
+ options: {
884
+ ...logFilterOptions,
885
+ top: { type: 'integer', default: 50, placeholder: 'N', description: 'Groups to return, largest first (1-500).' },
886
+ wide: { alias: 'w', type: 'boolean', description: 'Do not truncate patterns in table output.' },
887
+ },
888
+ examples: [
889
+ 'sentinelctl logs groups -l error -r 1h',
890
+ 'sentinelctl logs groups -a <agent-id> -r 7d -o json',
891
+ 'sentinelctl logs groups -p "Failed to start <N>.service" -r 24h',
892
+ ],
893
+ output: '{criteria, window, complete, coverage, total, devices, groupCount, patternCount, truncated, approximate, groups[]: {agentId, displayName, source, pattern, count, first, last, latest{timestamp, severity, message}}}.',
894
+ },
895
+ {
896
+ group: 'logs', name: 'facets', run: logsFacets, role: 'viewer', mutates: false,
897
+ summary: 'Top values of device, host, severity, source, agent version and profile in one query',
898
+ description: 'Orientation for an unfamiliar window: which devices, sources and severities dominate. Fleet-wide up to 24h; up to 7 days with -a, -s or -p.',
899
+ options: {
900
+ ...logFilterOptions,
901
+ range: { alias: 'r', type: 'string', choices: logRanges, description: 'Window (default 1h).' },
902
+ top: { type: 'integer', default: 10, placeholder: 'N', description: 'Values per field (1-50).' },
903
+ },
904
+ examples: ['sentinelctl logs facets -l error -r 24h', 'sentinelctl logs facets -a <agent-id> -r 7d -o json'],
905
+ output: '{criteria, window, facets{agent_id[{value, count, displayName}], host[], severity[], _SYSTEMD_UNIT[], …}}.',
597
906
  },
598
907
  {
599
908
  group: 'logs', name: 'context', run: logsContext, role: 'viewer', mutates: false,
@@ -615,15 +924,15 @@ const commands = [
615
924
  summary: 'Log volume and error counts over time',
616
925
  args: [{ name: 'text', optional: true, description: 'Same as -q.' }],
617
926
  options: logFilterOptions,
618
- examples: ['sentinelctl logs histogram -r 24h', 'sentinelctl logs histogram -a <agent-id> -r 7d -o json'],
619
- output: '{start, end, stepSeconds, total, buckets[]: {at (epoch seconds), total, errors}}. Cached for 60s on the server.',
927
+ examples: ['sentinelctl logs histogram -r 24h', 'sentinelctl logs histogram -a <agent-id> -r 30d -o json'],
928
+ output: '{start, end, complete, coverage, stepSeconds, total, buckets[]: {at (epoch seconds), total, errors}}. Finished days are cached.',
620
929
  },
621
930
  {
622
931
  group: 'logs', name: 'sources', run: logsSources, role: 'viewer', mutates: false,
623
- summary: 'Log sources (systemd units, containers) seen recently',
932
+ summary: 'Log sources (systemd units, containers) ranked by exact count',
624
933
  options: { range: { alias: 'r', type: 'string', choices: logRanges, default: '24h', description: 'Sample window.' } },
625
934
  examples: ['sentinelctl logs sources -r 1h'],
626
- output: '{range, sampleSize, sources[]: {value, count}}: top 50 in a sample of the latest 250 logs.',
935
+ output: '{range, total, complete, sources[]: {value, count}}: top 50 by exact count.',
627
936
  },
628
937
  {
629
938
  group: 'devices', name: 'list', run: devicesList, role: 'viewer', mutates: false,
@@ -657,9 +966,10 @@ const commands = [
657
966
  args: [{ name: 'agent-id' }],
658
967
  options: {
659
968
  range: { alias: 'r', type: 'string', choices: deviceRanges, default: '6h', description: 'Time-series window (about 120 points).' },
969
+ series: { type: 'string', placeholder: 'names|none', description: `Keep only these series in JSON (comma-separated: ${deviceSeries.join(', ')}) or none. Cuts output from ~80KB to ~2KB.` },
660
970
  refresh: { type: 'boolean', description: 'Bypass the server cache.' },
661
971
  },
662
- examples: ['sentinelctl devices show <agent-id> -r 24h -o json'],
972
+ examples: ['sentinelctl devices show <agent-id> -o json --series none', 'sentinelctl devices show <agent-id> -r 24h -o json --series cpu,memory'],
663
973
  output: '{device, current, series{cpu, memory, disk, load, temperature, networkRx, networkTx, …: [[epochSeconds, value], …]}, range, stepSeconds, stale}.',
664
974
  },
665
975
  {
@@ -678,6 +988,54 @@ const commands = [
678
988
  examples: ['sentinelctl devices overview <agent-id> -r 7d -o json'],
679
989
  output: '{summary{metrics, logs}: {state, outages, delays, missingMinutes, longestOutageMinutes, normalRatio}, agentId, device, products, attentionReasons[], history, related{parent, children[], expectedParent}}.',
680
990
  },
991
+ {
992
+ group: 'metrics', name: 'top', run: metricsTop, role: 'viewer', mutates: false,
993
+ summary: 'Rank devices by a metric aggregated over a window, with the fleet-wide distribution',
994
+ description: 'One server-side query over every device: avg/max/min/p95/last of the metric per device, ranked, plus min/p50/p90/p95/max across devices. --compare adds the previous window of the same length. reboots counts uptime resets (many devices reboot daily on schedule; compare with the fleet p50). diskDaysLeft forecasts days until the root disk is full from the growth over the window (only growing disks; soonest first). Device filters work like devices list (retired devices are excluded by default). Cached for 60s.',
995
+ args: [{ name: 'metric', description: metricNames.join(', ') }],
996
+ options: {
997
+ agg: { type: 'string', choices: ['avg', 'max', 'min', 'p95', 'last'], default: 'avg', description: 'Per-device aggregation over the window (ignored for reboots).' },
998
+ range: { alias: 'r', type: 'string', choices: metricWindows, description: 'Window (default 24h).' },
999
+ from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start (with --to, at most 30 days).' },
1000
+ to: { type: 'string', placeholder: 'ISO', description: 'Absolute window end.' },
1001
+ top: { type: 'integer', default: 10, placeholder: 'N', description: 'Devices to return (1-100).' },
1002
+ order: { type: 'string', choices: ['desc', 'asc'], description: 'desc = highest first, asc = lowest first (default desc; asc for diskDaysLeft).' },
1003
+ compare: { type: 'boolean', description: 'Also aggregate the previous window: previous and change per device.' },
1004
+ sort: { type: 'string', choices: ['value', 'change'], description: 'Rank by value (default) or by change (needs --compare).' },
1005
+ above: { type: 'number', placeholder: 'N', description: 'Only devices whose value is above N (e.g. --above 90).' },
1006
+ below: { type: 'number', placeholder: 'N', description: 'Only devices whose value is below N.' },
1007
+ query: { alias: 'q', type: 'string', placeholder: 'text', description: 'Device text filter (same as devices list).' },
1008
+ status: { type: 'string', choices: deviceStatuses, description: 'Reception status.' },
1009
+ lifecycle: { type: 'string', choices: lifecycles, description: 'Lifecycle scope (default operational).' },
1010
+ role: { type: 'string', placeholder: 'role', description: 'Device role.' },
1011
+ site: { type: 'string', placeholder: 'site', description: 'Site.' },
1012
+ group: { type: 'string', placeholder: 'group', description: 'Group.' },
1013
+ maintainer: { type: 'string', placeholder: 'name', description: 'Maintainer.' },
1014
+ fleet: { type: 'string', choices: ['edge', 'infrastructure', 'metrics-only'], description: 'Fleet type.' },
1015
+ label: { type: 'string', multiple: true, placeholder: 'k=v', description: 'Label match (up to 8).' },
1016
+ },
1017
+ examples: [
1018
+ 'sentinelctl metrics top memory -r 7d',
1019
+ 'sentinelctl metrics top disk --agg last --above 85 --top 50',
1020
+ 'sentinelctl metrics top cpu --agg p95 -r 24h --compare --sort change',
1021
+ 'sentinelctl metrics top reboots -r 7d --above 0',
1022
+ 'sentinelctl metrics top diskDaysLeft -r 24h --below 14',
1023
+ 'sentinelctl metrics top networkTx -r 24h --label fleet=edge -o json',
1024
+ ],
1025
+ output: '{metric, aggregation, unit, window{start, end, seconds, range}, criteria, scope{devices, withData, withoutData, matched}, distribution{count, min, avg, p50, p90, p95, max}, previousDistribution?, devices[]: {agentId, displayName, host, status, value, previous?, change?}}.',
1026
+ },
1027
+ {
1028
+ group: 'fleet', name: 'summary', run: fleetSummary, role: 'viewer', mutates: false,
1029
+ summary: 'One-call fleet briefing: status, attention, busiest devices, metric trends, log levels and error patterns',
1030
+ description: 'Start here to answer "how is the infrastructure?". Combines the device inventory, attention reasons with example devices, the highest current CPU/memory/disk/temperature, offline devices split by how long, metric trends (disks full within 14 days, temperature rises of 15°C or more, reboot counts well above the fleet norm) and log statistics for the window: totals by level, top, rising, falling and new error patterns (compared with the previous window) and the devices with the most errors. About 10KB as JSON.',
1031
+ options: {
1032
+ range: { alias: 'r', type: 'string', choices: logRanges, default: '24h', description: 'Log window.' },
1033
+ logs: { type: 'boolean', description: 'Include log statistics (default on; --no-logs skips them).' },
1034
+ trends: { type: 'boolean', description: 'Include metric trends: disks full within 14 days, temperature rises, unusual reboot counts (default on; --no-trends skips them).' },
1035
+ },
1036
+ examples: ['sentinelctl fleet summary', 'sentinelctl fleet summary -r 1h -o json', 'sentinelctl fleet summary --no-logs'],
1037
+ output: '{range, stale, devices{total, status, lifecycle, metrics, logs, offline{within24h[], over7dCount, otherCount}}, attention[]: {reason, count, devices[≤5]}, resources{cpu, memory, disk, temperature: [{agentId, displayName, value}]}, trends{diskFullWithin14Days[], temperatureRise[], rebootOutliers{typicalPerWeek, threshold, devices[]}}, logs{complete, window, total, devices, levels, errors{total, previousTotal, devices, newPatterns}, topErrorPatterns[], risingErrorPatterns[], fallingErrorPatterns[] (shrunk or gone since the previous window), topErrorDevices[]}}.',
1038
+ },
681
1039
  {
682
1040
  group: 'metadata', name: 'get', run: metadataGet, role: 'viewer', mutates: false,
683
1041
  summary: 'Device metadata: name, role, maintainer, labels, lifecycle',
@@ -813,7 +1171,9 @@ const commands = [
813
1171
 
814
1172
  export const catalog = commands;
815
1173
  export const groups = {
816
- logs: 'Search, count and rank logs; context around a log; volume trends; sources',
1174
+ fleet: 'One-call briefing of the whole fleet',
1175
+ logs: 'Count, group and compare logs on the server; exact pattern search; context; volume; sources',
1176
+ metrics: 'Rank devices by an aggregated metric across the fleet',
817
1177
  devices: 'Device list, details, lookup, reception overview',
818
1178
  metadata: 'Device metadata: read, edit (editor), import/export (admin)',
819
1179
  sessions: 'Your sessions and CLI tokens',
package/src/main.mjs CHANGED
@@ -14,6 +14,7 @@ const globalDefinitions = {
14
14
  profile: { type: 'string', placeholder: 'name', description: 'Saved login profile (default "default").' },
15
15
  output: { alias: 'o', type: 'string', choices: outputChoices, description: 'Output format. Use json or ndjson from agents and scripts.' },
16
16
  json: { type: 'boolean', description: 'Same as -o json.' },
17
+ select: { type: 'string', placeholder: 'fields', description: 'json/ndjson: keep only these comma-separated fields (dotted paths like metadata.role) of each item in the main list, plus top-level scalars. Example: --select agentId,displayName,status,memory' },
17
18
  quiet: { type: 'boolean', description: 'Suppress summaries and hints on stderr (errors are still printed).' },
18
19
  help: { alias: 'h', type: 'boolean', description: 'Show help.' },
19
20
  version: { alias: 'V', type: 'boolean', description: 'Show the version.' },
@@ -22,7 +23,7 @@ const specialCommands = {
22
23
  login: { summary: 'Approve in the browser with your intflow.ai Google Workspace account and store a 30-day CLI token', usage: 'sentinelctl login [--no-browser] [--name <client-name>]' },
23
24
  logout: { summary: 'Revoke the CLI token on the server and delete local credentials', usage: 'sentinelctl logout' },
24
25
  whoami: { summary: 'Show the server, user, role and token expiry', usage: 'sentinelctl whoami' },
25
- commands: { summary: 'List every command; with --json, a machine-readable catalog with options, roles and examples', usage: 'sentinelctl commands [--json]' },
26
+ commands: { summary: 'List every command; with --json, a machine-readable catalog with options, roles and examples (optionally for one group or command)', usage: 'sentinelctl commands [<group> [<command>]] [--json]' },
26
27
  help: { summary: 'Help: help <command>, help agents (guide for AI agents)', usage: 'sentinelctl help [<command> | agents]' },
27
28
  };
28
29
 
@@ -95,6 +96,7 @@ function overviewHelp() {
95
96
  '',
96
97
  'Getting started:',
97
98
  ' sentinelctl login approve once in the browser (valid for 30 days)',
99
+ ' sentinelctl fleet summary one-call briefing of the whole fleet',
98
100
  ' sentinelctl devices list --status offline',
99
101
  ' sentinelctl logs search -l error -r 1h',
100
102
  '',
@@ -119,46 +121,76 @@ function overviewHelp() {
119
121
 
120
122
  const agentGuide = `Using sentinelctl from AI agents and scripts
121
123
 
122
- 1. A human logs in once: \`sentinelctl login\` and approve in the browser. The token lasts 30 days;
123
- every later command is non-interactive. The token carries the logged-in user's role, so log in
124
- with a viewer account for agents that should only read. Elsewhere, pass SENTINEL_TOKEN and
125
- SENTINEL_URL as environment variables.
126
- 2. Use -o json (one document) or -o ndjson (one item per line). stdout carries data only; summaries,
127
- hints and errors go to stderr. --quiet drops summaries and hints. With json/ndjson output an error
128
- is one stderr line: {"error":{"status","code","message","hint","exitCode"}}.
129
- 3. Exit codes: 0 success · 1 server/network error · 2 usage error (fix the arguments and retry)
130
- · 3 not logged in or insufficient role (do not retry). GET requests that hit 429 are retried
131
- automatically after Retry-After.
132
- 4. \`sentinelctl commands --json\` returns every command with its options, choices, required role,
133
- whether it changes data, examples and output fields.
134
- 5. Typical investigation
135
- - find a device: devices lookup <part of name> -> agentId
136
- - status: devices list --status offline -o json | devices overview <id> -o json
137
- - metrics: devices show <id> -r 24h -o json (series.cpu etc. are [epochSeconds, value])
138
- - logs: logs search -a <id> -l error -r 24h -o ndjson
139
- then logs context -a <id> -s <source> -t <timestamp> for surrounding lines
140
- - counts/ranking: logs stats -l error -r 24h --by agent|source|message -o json
141
- - trend: logs histogram -a <id> -r 7d -o json
142
- 6. Efficiency
143
- - To answer "how many" or "which devices/sources the most", use logs stats: one server-side
144
- aggregation with exact counts. Do not download thousands of logs to count them.
145
- - Narrow the window and filters (-a, -s, -l, -q) first. Use --pages N (max 40) to take only what you
146
- need; --all stops at 10,000 logs.
147
- - Each user can run 2 log queries at a time (web and CLI combined); run commands one after another.
148
- - Do not use --follow (it never ends). Repeat a -r 15m search instead.
149
- - The server caches identical log searches for 5s, histograms and sources for 60s.
150
- 7. For commands that change data (metadata set/bulk/rollback/import, admin ...), run with --dry-run
151
- first when available, check the result, then apply.
152
- 8. Server error messages are in Korean (shared with the web console); the "hint" field is English.`;
124
+ SETUP
125
+ - A human logs in once (\`sentinelctl login\`, approve in the browser; 30 days). Later commands are
126
+ non-interactive. Elsewhere pass SENTINEL_TOKEN and SENTINEL_URL.
127
+ - Use -o json or -o ndjson. stdout is data only; notes, hints and errors go to stderr (--quiet drops
128
+ notes and hints). A json/ndjson error is one stderr line {"error":{status,code,message,hint,exitCode}}.
129
+ - Exit codes: 0 ok · 1 server/network (retry later) · 2 usage (fix the arguments) · 3 login/role (do
130
+ not retry). \`sentinelctl commands <group> --json\` describes options and output fields.
153
131
 
154
- function commandsCatalog() {
132
+ HOW TO INVESTIGATE: broad to narrow, count before you read
133
+ 1. Overview fleet summary -o json
134
+ status, attention reasons, busiest devices, trends (disk full soon, temperature
135
+ rises, unusual reboots), error levels, top/rising/falling/new error patterns.
136
+ 2. What and where logs stats -l error -r 24h --by message --compare -o json (patterns, change)
137
+ logs groups -l error -r 24h -o json (every device x source x pattern with
138
+ count, first/last time, latest line)
139
+ 3. When logs stats -p "<pattern>" --by agent --timeline -r 7d -o json
140
+ logs histogram -a <id> -r 7d -o json
141
+ 4. Read a few lines logs search -p "<pattern>" -a <id> -r 24h -n 50 -o json
142
+ logs context -a <id> -s <source> -t <timestamp> (lines around one log)
143
+ 5. Device state devices overview <id> -r 7d -o json · devices show <id> --series cpu,memory -o json
144
+ 6. Fleet metrics metrics top memory -r 7d · metrics top disk --agg last --above 85
145
+ metrics top diskDaysLeft --below 14 · metrics top cpu --agg p95 --compare
146
+ metrics top reboots -r 7d (most devices reboot daily; compare with p50)
147
+
148
+ LOG SEARCH RULES (complete answers at low cost)
149
+ - Never download logs to count, rank or summarize them. logs search returns at most 10,000 lines,
150
+ which on this fleet covers minutes, not hours: the rest of the window is silently missing.
151
+ logs stats and logs groups count every log in the window on the server.
152
+ - Prefer logs groups over logs search: errors in one hour are typically ~10,000 lines but ~150
153
+ groups, each with its exact count and latest line.
154
+ - Copy patterns exactly. Patterns from logs stats --by message / logs groups (numbers shown as <N>)
155
+ go to -p/--pattern, which matches that pattern exactly and quickly. -q matches words
156
+ case-insensitively and can also hit unrelated messages.
157
+ - Narrow before widening: device (-a), source (-s), service (--service), severity (-l), pattern (-p).
158
+ Start with 1h-24h; widen to 7d or 30d only when you need history.
159
+ - Fleet-wide windows over 7 days need a narrowing filter (-a, -s, --service, -p or -q with
160
+ --case-sensitive); otherwise the server answers 422 log_query_too_broad.
161
+ Single-device, single-service and single-pattern queries can use the whole retention period
162
+ (31 days by default).
163
+ - Windows over one day are aggregated per day and finished days are cached: repeating a long query
164
+ is cheap, and the CLI continues automatically if the server needs several requests. A partial
165
+ result says so in complete/coverage.
166
+ - To watch for new logs, poll with --since <nextSince from the previous run>, not --follow.
167
+ - --dedupe folds lines repeated within the same second (the agent writes some messages twice).
168
+ - Levels: emergency/critical/error include more severe levels; warning, info and debug are exact.
169
+ - Applications are found by source (-s: systemd unit, container) on one device, or by the service
170
+ label (--service, --env) across devices when the collector sets it; logs stats --by service ranks them.
171
+
172
+ LIMITS (shared with the web console, per user)
173
+ - 2 log queries and 2 metric aggregations at a time; 60 seconds of log query time per minute.
174
+ 429 responses are retried automatically with backoff; run commands one after another.
175
+ - Retrying a 422 or 400 without changing the arguments will fail again; read the hint.
176
+
177
+ KEEP OUTPUT SMALL
178
+ - --select a,b.c keeps only those fields (devices list --all -o json: ~700KB -> ~20KB).
179
+ - devices show --series none drops time series; logs groups/stats return compact rows.
180
+
181
+ CHANGES
182
+ - metadata set/bulk/rollback/import and admin commands change data: run with --dry-run first when
183
+ available, check the result, then apply. Viewer tokens cannot change anything.
184
+ - Server error messages are Korean (shared with the web console); hints are English.`;
185
+
186
+ function commandsCatalog(selected = null) {
155
187
  return {
156
188
  version,
157
189
  globalOptions: globalDefinitions,
158
190
  exitCodes,
159
191
  commands: [
160
- ...Object.entries(specialCommands).map(([name, command]) => ({ command: name, usage: command.usage, summary: command.summary, role: null, mutates: name === 'logout' })),
161
- ...catalog.map((command) => ({
192
+ ...(selected ? [] : Object.entries(specialCommands).map(([name, command]) => ({ command: name, usage: command.usage, summary: command.summary, role: null, mutates: name === 'logout' }))),
193
+ ...(selected ?? catalog).map((command) => ({
162
194
  command: commandPath(command), usage: usageLine(command), summary: command.summary, description: command.description ?? null,
163
195
  role: command.role, roleNote: command.roleNote ?? null, mutates: command.mutates, args: command.args ?? [],
164
196
  options: Object.fromEntries(Object.entries(command.options).map(([name, spec]) => [name, {
@@ -267,6 +299,15 @@ function apiHint(error, context, command) {
267
299
  ? 'Check the agent ID with `sentinelctl devices lookup <part of the name>`.' : 'Not found; check the ID.';
268
300
  }
269
301
  if (error.status === 409) return 'Someone saved a change first. Run the command again to apply on top of the latest revision.';
302
+ if (error.code === 'log_query_too_broad') {
303
+ return 'Narrow it: add -a <agent-id>, -s <source>, --service <name>, -p "<pattern>" (from `logs stats --by message`) or --case-sensitive with -q, or use a window of 7 days or less.';
304
+ }
305
+ if (error.code === 'log_budget_exceeded') {
306
+ return `Your log query time for this minute is used up${error.retryAfterSeconds ? `; wait ${error.retryAfterSeconds}s` : ''}. Prefer logs stats/groups with filters over repeated wide searches.`;
307
+ }
308
+ if (error.status === 504) {
309
+ return 'The log/metric backend timed out. Narrow the window or add -a/-s/-l, use -p "<pattern>" instead of -q, or add --case-sensitive. Long aggregations resume from cache if you rerun.';
310
+ }
270
311
  if (error.status === 429) return `Too many requests${error.retryAfterSeconds ? `; wait ${error.retryAfterSeconds}s` : ''} and retry. Narrower filters reduce load.`;
271
312
  if ([400, 413, 415, 422].includes(error.status)) {
272
313
  return command ? `Check the input: sentinelctl ${commandPath(command)} --help` : 'Check the input.';
@@ -329,10 +370,13 @@ export async function main(argv, {
329
370
  return 0;
330
371
  }
331
372
  if (name === 'commands') {
373
+ const scope = [subcommand, ...commandArguments].filter(Boolean).join(' ');
374
+ const selected = catalog.filter((item) => !scope || commandPath(item) === scope || item.group === scope);
375
+ if (scope && !selected.length) throw unknownCommandError(scope, [...Object.keys(groups), ...catalog.map(commandPath)], 'commands ');
332
376
  if (global.output === 'table') {
333
- out(catalog.map((item) => `${commandPath(item).padEnd(20)} ${item.role.padEnd(7)} ${item.mutates ? 'write' : 'read '} ${item.summary}`).join('\n'));
377
+ out(selected.map((item) => `${commandPath(item).padEnd(20)} ${item.role.padEnd(7)} ${item.mutates ? 'write' : 'read '} ${item.summary}`).join('\n'));
334
378
  } else {
335
- out(JSON.stringify(commandsCatalog(), null, global.output === 'json' ? 2 : 0));
379
+ out(JSON.stringify(commandsCatalog(scope ? selected : null), null, global.output === 'json' ? 2 : 0));
336
380
  }
337
381
  return 0;
338
382
  }
@@ -365,6 +409,7 @@ export async function main(argv, {
365
409
  path: credentialsPath(environment) });
366
410
  context = {
367
411
  global, connection, out, err, fetch, sleep, signal, openBrowser: open, width: stdout.columns ?? 120,
412
+ select: global.select ? global.select.split(',').map((field) => field.trim()).filter(Boolean) : null,
368
413
  hint: (text) => { if (!global.quiet) err(`hint: ${text}`); },
369
414
  note: (text) => { if (!global.quiet && global.output === 'table') err(text); },
370
415
  client: createClient({ server: connection.server, token: connection.token, fetch, sleep }),