@intflow/sentinelctl 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -38,10 +38,25 @@ sentinelctl --help # overview
38
38
  sentinelctl logs search --help # one command: options, examples, required role, output fields
39
39
  sentinelctl help agents # guide for AI agents and scripts
40
40
  sentinelctl commands --json # machine-readable catalog of every command
41
+ sentinelctl commands logs --json # only one group (or "logs stats" for one command)
41
42
  ```
42
43
 
43
44
  Mistyped commands and options get a suggestion (`Did you mean --range?`), and every error comes with a hint for the next step.
44
45
 
46
+ ## Fleet overview and metrics
47
+
48
+ ```bash
49
+ sentinelctl fleet summary # status, attention, busiest devices, log levels, top/rising/new error patterns
50
+ sentinelctl fleet summary -r 1h -o json # about 5KB
51
+ sentinelctl metrics top memory -r 7d # devices ranked by 7-day average memory + fleet distribution
52
+ sentinelctl metrics top disk --agg last --above 85 --top 50
53
+ sentinelctl metrics top cpu --agg p95 -r 24h --compare --sort change
54
+ sentinelctl metrics top reboots -r 7d --above 0
55
+ sentinelctl metrics top networkTx -r 24h --label fleet=edge -o json
56
+ ```
57
+
58
+ `metrics top` runs one query over every device: `--agg avg|max|min|p95|last` per device over `-r 1h|6h|24h|7d|30d` (or `--from/--to`, up to 30 days), ranked, with min/p50/p90/p95/max across devices. Metrics: cpu, memory, disk, swap, load, temperature, processes, tcpConnections, networkRx, networkTx, reboots. `--compare` adds the previous window of the same length. Device filters work like `devices list`; retired devices are excluded by default.
59
+
45
60
  ## Logs
46
61
 
47
62
  ```bash
@@ -55,12 +70,15 @@ sentinelctl logs histogram -l error -r 24h
55
70
  sentinelctl logs stats -l error -r 1h --by source # exact counts, one server-side query
56
71
  sentinelctl logs stats -l error -r 24h --by agent --top 10
57
72
  sentinelctl logs stats -a <agent-id> -r 7d --by message # message patterns, numbers collapsed to <N>
73
+ sentinelctl logs stats -l error -r 24h --by message --compare --sort change # what is new or growing
58
74
  sentinelctl logs sources -r 24h
59
75
  ```
60
76
 
77
+ `logs stats --compare` also counts the previous window of the same length and adds `previousCount`, `change` and `isNew` per group. Message groups carry `searchText`, ready for `logs search -q`.
78
+
61
79
  Windows are `15m`, `1h`, `6h`, `24h`, `7d`, or `--from`/`--to` up to 30 days. One page holds 50, 100 or 250 logs; when you ask for several pages (`--pages N`, at most 40, or `--all`) the CLI fetches 250 per request and stops at 10,000 logs. To answer "how many" or "which devices the most", use `logs stats`: it aggregates on the server and returns exact counts instead of downloading logs. `--fields` (raw fields) is admin-only and audited. `--follow` polls at most every 5 seconds.
62
80
 
63
- Each user can run 2 log queries at a time, web and CLI combined; a third concurrent query gets 429 and GET requests retry automatically.
81
+ Each user can run 2 log queries and 2 metric aggregations at a time, web and CLI combined; a third concurrent query gets 429 and GET requests retry automatically with backoff.
64
82
 
65
83
  ## Devices and metadata
66
84
 
@@ -101,6 +119,7 @@ Any other endpoint: `sentinelctl api "/api/devices?status=offline"`.
101
119
 
102
120
  ## Output and exit codes
103
121
 
122
+ - `--select agentId,displayName,status,metadata.role` keeps only those fields of each item in json/ndjson output (plus top-level scalars), e.g. `devices list --all -o json` drops from ~700KB to ~20KB. `devices show --series none` (or `--series cpu,memory`) drops time series.
104
123
  - `-o table` (default), `-o json` (`--json`), `-o ndjson`. Data goes to stdout; summaries, hints and errors go to stderr (`--quiet` drops summaries and hints). With json/ndjson, an error is one stderr line: `{"error":{"status","code","message","hint","exitCode"}}`.
105
124
  - Exit codes: `0` success, `1` server or network error, `2` usage error, `3` not logged in or insufficient role.
106
125
  - Server error messages are Korean because the web console shares them; hints are English.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@intflow/sentinelctl",
3
- "version": "0.3.1",
3
+ "version": "0.4.0",
4
4
  "description": "Sentinel Fleet Console CLI: search device logs, inspect status and metrics, edit metadata and manage roles",
5
5
  "type": "module",
6
6
  "bin": {
package/src/args.mjs CHANGED
@@ -82,6 +82,10 @@ export function parseArguments(argv, definitions) {
82
82
  parsed = Number(value);
83
83
  if (!Number.isInteger(parsed)) throw new UsageError(`--${name} must be an integer: ${value}`);
84
84
  }
85
+ if (spec.type === 'number') {
86
+ parsed = Number(value);
87
+ if (value.trim() === '' || !Number.isFinite(parsed)) throw new UsageError(`--${name} must be a number: ${value}`);
88
+ }
85
89
  if (spec.choices && !spec.choices.includes(parsed)) {
86
90
  const suggestion = closest(String(parsed), spec.choices.map(String));
87
91
  throw new UsageError(`--${name} must be one of ${spec.choices.join(', ')}: ${value}`,
package/src/client.mjs CHANGED
@@ -11,7 +11,7 @@ export class ApiError extends Error {
11
11
 
12
12
  const userAgent = 'sentinelctl';
13
13
 
14
- export function createClient({ server, token = null, fetch: fetchImplementation = globalThis.fetch, timeoutMs = 60_000, maxRetries = 2, sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)) }) {
14
+ export function createClient({ server, token = null, fetch: fetchImplementation = globalThis.fetch, timeoutMs = 60_000, maxRetries = 5, sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)) }) {
15
15
  async function request(method, path, { query, body, auth = true, allowStatuses = [] } = {}) {
16
16
  const url = new URL(path, server);
17
17
  for (const [key, value] of Object.entries(query ?? {})) {
@@ -45,7 +45,7 @@ export function createClient({ server, token = null, fetch: fetchImplementation
45
45
  }
46
46
  const retryAfterSeconds = Number(response.headers.get('retry-after')) || null;
47
47
  if (response.status === 429 && method === 'GET' && attempt < maxRetries) {
48
- await sleep(Math.min(retryAfterSeconds ?? 1, 30) * 1_000);
48
+ await sleep(Math.min(Math.max(retryAfterSeconds ?? 1, 2 ** attempt), 15) * 1_000);
49
49
  continue;
50
50
  }
51
51
  if (response.ok || allowStatuses.includes(response.status)) return { status: response.status, body: payload };
package/src/commands.mjs CHANGED
@@ -1,5 +1,5 @@
1
1
  import { readFile, writeFile } from 'node:fs/promises';
2
- import { keyValue, UsageError } from './args.mjs';
2
+ import { closest, keyValue, UsageError } from './args.mjs';
3
3
  import { ageText, localTime, percentText, renderKeyValues, renderTable, singleLine } from './format.mjs';
4
4
 
5
5
  export const roleLabels = { viewer: 'viewer (read-only)', editor: 'editor', admin: 'admin' };
@@ -41,10 +41,41 @@ function logCriteria(options, positionals) {
41
41
  };
42
42
  }
43
43
 
44
+ const listKeys = ['devices', 'logs', 'groups', 'buckets', 'sources', 'sessions', 'users', 'history', 'results', 'attention'];
45
+
46
+ function pick(item, paths) {
47
+ if (!item || typeof item !== 'object') return item;
48
+ const picked = {};
49
+ for (const path of paths) {
50
+ const parts = path.split('.');
51
+ let value = item;
52
+ for (const part of parts) {
53
+ value = value !== null && typeof value === 'object' && Object.hasOwn(value, part) ? value[part] : undefined;
54
+ if (value === undefined) break;
55
+ }
56
+ if (value === undefined) continue;
57
+ let target = picked;
58
+ for (const part of parts.slice(0, -1)) target = (target[part] ??= {});
59
+ target[parts.at(-1)] = value;
60
+ }
61
+ return picked;
62
+ }
63
+
64
+ export function project(value, paths) {
65
+ if (!paths?.length) return value;
66
+ if (Array.isArray(value)) return value.map((item) => pick(item, paths));
67
+ const listKey = listKeys.find((key) => Array.isArray(value?.[key]));
68
+ if (!listKey) return pick(value, paths);
69
+ const meta = Object.fromEntries(Object.entries(value)
70
+ .filter(([key, entry]) => key !== listKey && (key === 'window' || entry === null || typeof entry !== 'object')));
71
+ return { ...meta, [listKey]: value[listKey].map((item) => pick(item, paths)) };
72
+ }
73
+
44
74
  function emit(context, value, table) {
45
75
  const { output } = context.global;
46
- if (output === 'json') context.out(JSON.stringify(value, null, 2));
47
- else if (output === 'ndjson') for (const item of Array.isArray(value) ? value : [value]) context.out(JSON.stringify(item));
76
+ const selected = project(value, context.select);
77
+ if (output === 'json') context.out(JSON.stringify(selected, null, 2));
78
+ else if (output === 'ndjson') for (const item of Array.isArray(selected) ? selected : [selected]) context.out(JSON.stringify(item));
48
79
  else context.out(table());
49
80
  }
50
81
 
@@ -89,11 +120,8 @@ async function logsSearch(context, { options, positionals }) {
89
120
  if (!last.hasMore || !last.nextCursor) break;
90
121
  cursor = last.nextCursor;
91
122
  }
92
- if (context.global.output === 'json') {
93
- context.out(JSON.stringify({ criteria: last.criteria, window: last.window, count: logs.length, hasMore: last.hasMore, logs }, null, 2));
94
- } else {
95
- emit(context, logs, () => renderTable(logs, logColumns, { width: context.width, wide: options.wide }));
96
- }
123
+ emit(context, context.global.output === 'json' ? { criteria: last.criteria, window: last.window, count: logs.length, hasMore: last.hasMore, logs } : logs,
124
+ () => renderTable(logs, logColumns, { width: context.width, wide: options.wide }));
97
125
  if (!logs.length) context.hint(emptyLogHint(criteria));
98
126
  else context.note(`${logs.length} logs · ${localTime(last.window.start)} → ${localTime(last.window.end)}`);
99
127
  if (last.hasMore && options.all) context.hint(`Stopped at the ${maxPages * 250}-log cap. Narrow the filters, or use \`sentinelctl logs stats\` for counts.`);
@@ -133,19 +161,133 @@ const statsDimensions = ['source', 'agent', 'level', 'message'];
133
161
 
134
162
  async function logsStats(context, { options, positionals }) {
135
163
  const criteria = logCriteria(options, positionals);
136
- const body = await context.client.get('/api/logs/stats', { ...criteria, by: options.by, top: options.top });
164
+ if (options.sort === 'change' && !options.compare) {
165
+ throw new UsageError('--sort change needs --compare.', { hint: 'Example: sentinelctl logs stats -l error -r 24h --by message --compare --sort change' });
166
+ }
167
+ const body = await context.client.get('/api/logs/stats', {
168
+ ...criteria, by: options.by, top: options.top, compare: options.compare ? 1 : undefined, sort: options.sort,
169
+ });
137
170
  const label = { source: 'SOURCE', agent: 'DEVICE', level: 'LEVEL', message: 'MESSAGE PATTERN' }[body.criteria.by];
171
+ const signed = (value) => (value === null || value === undefined ? '?' : value > 0 ? `+${value}` : String(value));
138
172
  emit(context, context.global.output === 'ndjson' ? body.groups : body, () => renderTable(body.groups, [
139
173
  { header: 'COUNT', value: (group) => String(group.count) },
140
- { header: 'SHARE', value: (group) => (body.total ? `${((group.count / body.total) * 100).toFixed(1)}%` : '-') },
174
+ ...(body.previous ? [
175
+ { header: 'PREV', value: (group) => (group.previousCount === null ? '?' : String(group.previousCount)) },
176
+ { header: 'CHANGE', value: (group) => (group.isNew ? 'new' : signed(group.change)) },
177
+ ] : [{ header: 'SHARE', value: (group) => (body.total ? `${((group.count / body.total) * 100).toFixed(1)}%` : '-') }]),
141
178
  { header: 'DEVICES', value: (group) => String(group.devices) },
142
179
  ...(body.criteria.by === 'agent' ? [{ header: 'AGENT ID', value: (group) => group.key, max: 36 }] : []),
143
180
  { header: label, value: (group) => (body.criteria.by === 'agent' ? group.displayName ?? group.host ?? group.key : group.key) },
144
181
  ], { width: context.width, wide: options.wide }));
145
182
  context.note(`${body.total} logs from ${body.devices} devices · ${localTime(body.window.start)} → ${localTime(body.window.end)}`);
183
+ if (body.previous) {
184
+ context.note(`Previous window ${localTime(body.previous.window.start)} → ${localTime(body.previous.window.end)}: ${body.previous.total} logs (${signed(body.total - body.previous.total)}) · ${body.newGroupCount} new groups`);
185
+ if (body.falling?.length && body.criteria.sort !== 'change') {
186
+ context.note(`Largest drops: ${body.falling.map((group) => `${singleLine(group.key).slice(0, 50)} (${group.change})`).join(' | ')}`);
187
+ }
188
+ if (!body.previous.exhaustive) context.hint('The previous window has too many groups to compare exactly; "?" marks groups outside its top list.');
189
+ }
146
190
  if (body.truncated) context.hint(`Showing the top ${body.groups.length} of ${body.groupCount} groups. Raise --top (max 100) or add filters.`);
147
191
  if (body.approximate) context.hint('Message patterns are approximate on this server (collapse_nums is unavailable); narrow the window or filters for exact counts.');
148
192
  if (!body.total) context.hint(emptyLogHint(criteria));
193
+ const sample = body.groups.find((group) => group.searchText);
194
+ if (body.criteria.by === 'message' && sample) {
195
+ context.hint(`See matching logs: sentinelctl logs search -q "${sample.searchText.replaceAll('"', '\\"')}" -r ${criteria.range ?? '24h'}${criteria.level ? ` -l ${criteria.level}` : ''} (searchText in -o json)`);
196
+ }
197
+ }
198
+
199
+ const metricNames = ['cpu', 'memory', 'disk', 'swap', 'load', 'temperature', 'processes', 'tcpConnections', 'networkRx', 'networkTx', 'reboots'];
200
+ const metricWindows = ['1h', '6h', '24h', '7d', '30d'];
201
+ const deviceSeries = ['cpu', 'memory', 'disk', 'load', 'temperature', 'processes', 'tcpConnections', 'networkRx', 'networkTx', 'swap', 'uptime'];
202
+
203
+ function metricValueText(value, unit) {
204
+ if (value === null || value === undefined) return '-';
205
+ if (unit === '%') return `${value.toFixed(1)}%`;
206
+ if (unit === '°C') return `${value.toFixed(1)}°C`;
207
+ if (unit === 'B/s') return `${bytesText(value)}/s`;
208
+ return String(Math.round(value * 100) / 100);
209
+ }
210
+
211
+ async function metricsTop(context, { options, positionals }) {
212
+ const metric = positionals[0];
213
+ if (!metric) throw new UsageError('A metric name is required.', { hint: `One of: ${metricNames.join(', ')}. Example: sentinelctl metrics top memory -r 7d` });
214
+ if (!metricNames.includes(metric)) {
215
+ const suggestion = closest(metric, metricNames);
216
+ throw new UsageError(`Unknown metric: ${metric}`, { hint: `${suggestion ? `Did you mean ${suggestion}? ` : ''}One of: ${metricNames.join(', ')}` });
217
+ }
218
+ if ((options.from && !options.to) || (!options.from && options.to)) throw new UsageError('--from and --to must be given together.');
219
+ if (options.from && options.range) throw new UsageError('--range cannot be combined with --from/--to.');
220
+ if (options.sort === 'change' && !options.compare) throw new UsageError('--sort change needs --compare.', { hint: `Example: sentinelctl metrics top ${metric} -r 24h --compare --sort change` });
221
+ for (const label of options.label ?? []) keyValue(label, '--label');
222
+ const body = await context.client.get('/api/metrics/summary', {
223
+ metric, agg: options.agg, range: options.from ? undefined : options.range,
224
+ from: options.from ? new Date(options.from).toISOString() : undefined, to: options.to ? new Date(options.to).toISOString() : undefined,
225
+ top: options.top, order: options.order, sort: options.sort, compare: options.compare ? 1 : undefined, above: options.above, below: options.below,
226
+ query: options.query, status: options.status, lifecycle: options.lifecycle, role: options.role, site: options.site, group: options.group,
227
+ maintainer: options.maintainer, fleet: options.fleet, labels: options.label?.length ? JSON.stringify(options.label) : undefined,
228
+ });
229
+ const text = (value) => metricValueText(value, body.unit);
230
+ const signed = (value) => (value === null || value === undefined ? '?' : `${value > 0 ? '+' : ''}${text(value)}`);
231
+ emit(context, context.global.output === 'ndjson' ? body.devices : body, () => renderTable(body.devices, [
232
+ { header: 'AGENT ID', value: (device) => device.agentId, max: 36 },
233
+ { header: 'NAME', value: (device) => device.displayName, max: 32 },
234
+ { header: 'STATUS', value: (device) => device.status, max: 9 },
235
+ { header: body.aggregation.toUpperCase(), value: (device) => text(device.value) },
236
+ ...(body.criteria.compare ? [
237
+ { header: 'PREV', value: (device) => text(device.previous) },
238
+ { header: 'CHANGE', value: (device) => signed(device.change) },
239
+ ] : []),
240
+ ], { width: context.width }));
241
+ const spread = body.distribution;
242
+ context.note(`${body.metric} ${body.aggregation} over ${body.window.range ?? `${localTime(body.window.start)} → ${localTime(body.window.end)}`} · ${body.scope.withData} of ${body.scope.devices} devices have data`
243
+ + (spread.count ? ` · min ${text(spread.min)} · p50 ${text(spread.p50)} · p90 ${text(spread.p90)} · max ${text(spread.max)}` : ''));
244
+ if (body.criteria.above !== null || body.criteria.below !== null) context.note(`${body.scope.matched} devices within the --above/--below bounds.`);
245
+ if (!body.scope.withData) context.hint('No device reported this metric in the window. Widen -r or drop device filters.');
246
+ else if (body.scope.matched > body.devices.length) context.hint(`Showing ${body.devices.length} of ${body.scope.matched}. Raise --top (max 100).`);
247
+ }
248
+
249
+ async function fleetSummary(context, { options }) {
250
+ const body = await context.client.get('/api/fleet/summary', { range: options.range, logs: options.logs === false ? 0 : undefined });
251
+ emit(context, body, () => {
252
+ const counts = (record) => Object.entries(record ?? {}).sort((left, right) => right[1] - left[1]).map(([key, value]) => `${key} ${value}`).join(', ');
253
+ const names = (devices) => devices.map((device) => device.displayName).join(', ');
254
+ const sections = [renderKeyValues([
255
+ ['Devices', `${body.devices.total} (${counts(body.devices.status)})`],
256
+ ['Metrics', counts(body.devices.metrics)],
257
+ ['Logs', counts(body.devices.logs)],
258
+ ['Data', body.stale ? `stale (${body.staleReason ?? ''})` : 'fresh'],
259
+ ])];
260
+ sections.push('', 'Attention:', body.attention.length
261
+ ? renderTable(body.attention, [
262
+ { header: 'COUNT', value: (entry) => String(entry.count) },
263
+ { header: 'REASON', value: (entry) => entry.reason, max: 24 },
264
+ { header: 'EXAMPLES', value: (entry) => names(entry.devices) },
265
+ ], { width: context.width })
266
+ : ' none');
267
+ sections.push('', 'Highest now (online, active):', renderKeyValues(Object.entries(body.resources).map(([field, devices]) => [
268
+ field, devices.map((device) => `${device.displayName} ${field === 'temperature' ? `${device.value}°C` : `${device.value}%`}`).join(', ') || '-',
269
+ ])));
270
+ if (body.logs?.error) sections.push('', `Logs: ${body.logs.error}`);
271
+ else if (body.logs) {
272
+ const errors = body.logs.errors;
273
+ sections.push('', `Logs (${body.range}): ${body.logs.total} from ${body.logs.devices} devices · ${counts(body.logs.levels)}`,
274
+ `Errors: ${errors.total}${errors.previousTotal === null ? '' : ` (previous window ${errors.previousTotal})`} from ${errors.devices} devices · ${errors.newPatterns} new patterns`,
275
+ '', 'Top error patterns:', renderTable(body.logs.topErrorPatterns, [
276
+ { header: 'COUNT', value: (group) => String(group.count) },
277
+ { header: 'CHANGE', value: (group) => (group.isNew ? 'new' : group.change === null ? '?' : `${group.change > 0 ? '+' : ''}${group.change}`) },
278
+ { header: 'DEVICES', value: (group) => String(group.devices) },
279
+ { header: 'PATTERN', value: (group) => group.pattern },
280
+ ], { width: context.width }),
281
+ '', `Top error devices: ${body.logs.topErrorDevices.map((device) => `${device.displayName} ${device.count}`).join(', ') || '-'}`);
282
+ const changes = (label, groups) => {
283
+ if (groups?.length) sections.push(`${label}: ${groups.map((group) => `${singleLine(group.pattern).slice(0, 60)} (${group.change > 0 ? '+' : ''}${group.change})`).join(' | ')}`);
284
+ };
285
+ changes('Rising', body.logs.risingErrorPatterns);
286
+ changes('Falling', body.logs.fallingErrorPatterns);
287
+ }
288
+ return sections.join('\n');
289
+ });
290
+ context.hint('Drill down: devices list --focus attention · logs stats -l error --by message --compare · metrics top memory -r 24h');
149
291
  }
150
292
 
151
293
  async function logsContext(context, { options }) {
@@ -217,11 +359,8 @@ async function devicesList(context, { options, positionals }) {
217
359
  }
218
360
  for (const page of pages) devices.push(...results.get(page));
219
361
  }
220
- if (context.global.output === 'json') {
221
- context.out(JSON.stringify({ matched: first.matched, counts: first.counts, stale: first.stale, staleReason: first.staleReason, devices }, null, 2));
222
- } else {
223
- emit(context, devices, () => renderTable(devices, deviceColumns, { width: context.width }));
224
- }
362
+ emit(context, context.global.output === 'json' ? { matched: first.matched, counts: first.counts, stale: first.stale, staleReason: first.staleReason, devices } : devices,
363
+ () => renderTable(devices, deviceColumns, { width: context.width }));
225
364
  context.note(`${devices.length} of ${first.matched} devices${options.all ? '' : ` · page ${first.page}/${first.totalPages}`}`);
226
365
  if (first.stale) context.hint(`Metrics are stale (${first.staleReason ?? 'metrics query failed'}). Check \`sentinelctl health\`.`);
227
366
  if (!devices.length) context.hint('No devices matched. Add --lifecycle all to include maintenance/retired devices, or shorten the search text.');
@@ -235,17 +374,22 @@ function requireAgentId(positionals) {
235
374
 
236
375
  async function devicesShow(context, { options, positionals }) {
237
376
  const agentId = requireAgentId(positionals);
377
+ const keep = options.series === undefined ? null : options.series === 'none' ? [] : options.series.split(',').map((name) => name.trim()).filter(Boolean);
378
+ const unknownSeries = (keep ?? []).filter((name) => !deviceSeries.includes(name));
379
+ if (unknownSeries.length) throw new UsageError(`Unknown series: ${unknownSeries.join(', ')}`, { hint: `Use none or a comma-separated list of: ${deviceSeries.join(', ')}` });
238
380
  const body = await context.client.get(`/api/devices/${encodeURIComponent(agentId)}`, { range: options.range, refresh: options.refresh ? 1 : undefined });
381
+ if (keep) body.series = Object.fromEntries(Object.entries(body.series ?? {}).filter(([name]) => keep.includes(name)));
239
382
  const device = body.device;
383
+ const current = body.current ?? {};
240
384
  emit(context, body, () => renderKeyValues([
241
385
  ['Agent ID', device.agentId], ['Name', device.displayName], ['Host', device.host],
242
386
  ['Status', device.observedStatus ?? device.status], ['Last seen', ageText(device.ageSeconds)],
243
- ['CPU', percentText(device.cpu)], ['Memory', percentText(device.memory)], ['Disk', percentText(device.disk)],
244
- ['Load', device.load], ['Temperature', device.temperature], ['Role', device.metadata?.role ?? device.role],
387
+ ['CPU', percentText(current.cpu ?? device.cpu)], ['Memory', percentText(current.memory ?? device.memory)], ['Disk', percentText(current.disk ?? device.disk)],
388
+ ['Load', current.load ?? device.load], ['Temperature', current.temperature ?? device.temperature], ['Role', device.metadata?.role ?? device.role],
245
389
  ['Maintainer', device.metadata?.maintainer], ['Lifecycle', device.metadata?.lifecycle], ['Notes', device.metadata?.description],
246
390
  ['Data', body.stale ? `stale (${body.staleReason ?? ''})` : 'fresh'],
247
391
  ]));
248
- if (context.global.output === 'table') context.hint(`Time series (${body.range}, ${body.stepSeconds}s step) are in \`series\` with -o json.`);
392
+ if (context.global.output === 'table') context.hint(`Time series (${body.range}, ${body.stepSeconds}s step) are in \`series\` with -o json; --series cpu,memory or --series none keeps the output small.`);
249
393
  }
250
394
 
251
395
  async function devicesLookup(context, { positionals }) {
@@ -546,7 +690,7 @@ async function rawApi(context, { options, positionals }) {
546
690
  }
547
691
  const url = new URL(path, context.client.server);
548
692
  const result = await context.client.request(options.method, url.pathname, { query: Object.fromEntries(url.searchParams), body });
549
- context.out(JSON.stringify(result.body, null, 2));
693
+ context.out(JSON.stringify(project(result.body, context.select), null, 2));
550
694
  }
551
695
 
552
696
  const commands = [
@@ -584,15 +728,18 @@ const commands = [
584
728
  ...logFilterOptions,
585
729
  by: { alias: 'b', type: 'string', choices: statsDimensions, default: 'source', description: 'Group by source, agent (device), level or message pattern.' },
586
730
  top: { type: 'integer', default: 20, placeholder: 'N', description: 'Groups to return (1-100).' },
731
+ compare: { type: 'boolean', description: 'Also count the previous window of the same length: previousCount, change and isNew per group.' },
732
+ sort: { type: 'string', choices: ['count', 'change'], description: 'Rank by count (default) or by increase over the previous window (needs --compare).' },
587
733
  wide: { alias: 'w', type: 'boolean', description: 'Do not truncate long keys in table output.' },
588
734
  },
589
735
  examples: [
590
736
  'sentinelctl logs stats -l error -r 1h --by source',
591
737
  'sentinelctl logs stats -l error -r 24h --by agent --top 10 -o json',
738
+ 'sentinelctl logs stats -l error -r 24h --by message --compare --sort change',
592
739
  'sentinelctl logs stats -a <agent-id> -r 7d --by message',
593
740
  'sentinelctl logs stats "connection refused" -r 24h --by agent',
594
741
  ],
595
- output: '{criteria, window, total, devices, groupCount, truncated, approximate, groups[]: {key, count, devices (distinct agents), host?, displayName?}}.',
742
+ output: '{criteria, window, total, devices, groupCount, truncated, approximate, groups[]: {key, count, devices (distinct agents), host?, displayName?, searchText? (message: text to pass to logs search -q), previousCount?, change?, isNew?}, previous?: {window, total, devices, exhaustive}, newGroupCount?, rising?[≤5], falling?[≤5] (largest increases and decreases across all groups, not only the returned top)}.',
596
743
  },
597
744
  {
598
745
  group: 'logs', name: 'context', run: logsContext, role: 'viewer', mutates: false,
@@ -656,9 +803,10 @@ const commands = [
656
803
  args: [{ name: 'agent-id' }],
657
804
  options: {
658
805
  range: { alias: 'r', type: 'string', choices: deviceRanges, default: '6h', description: 'Time-series window (about 120 points).' },
806
+ series: { type: 'string', placeholder: 'names|none', description: `Keep only these series in JSON (comma-separated: ${deviceSeries.join(', ')}) or none. Cuts output from ~80KB to ~2KB.` },
659
807
  refresh: { type: 'boolean', description: 'Bypass the server cache.' },
660
808
  },
661
- examples: ['sentinelctl devices show <agent-id> -r 24h -o json'],
809
+ examples: ['sentinelctl devices show <agent-id> -o json --series none', 'sentinelctl devices show <agent-id> -r 24h -o json --series cpu,memory'],
662
810
  output: '{device, current, series{cpu, memory, disk, load, temperature, networkRx, networkTx, …: [[epochSeconds, value], …]}, range, stepSeconds, stale}.',
663
811
  },
664
812
  {
@@ -677,6 +825,52 @@ const commands = [
677
825
  examples: ['sentinelctl devices overview <agent-id> -r 7d -o json'],
678
826
  output: '{summary{metrics, logs}: {state, outages, delays, missingMinutes, longestOutageMinutes, normalRatio}, agentId, device, products, attentionReasons[], history, related{parent, children[], expectedParent}}.',
679
827
  },
828
+ {
829
+ group: 'metrics', name: 'top', run: metricsTop, role: 'viewer', mutates: false,
830
+ summary: 'Rank devices by a metric aggregated over a window, with the fleet-wide distribution',
831
+ description: 'One server-side query over every device: avg/max/min/p95/last of the metric per device, ranked, plus min/p50/p90/p95/max across devices. --compare adds the previous window of the same length. reboots counts uptime resets. Device filters work like devices list (retired devices are excluded by default). Cached for 60s.',
832
+ args: [{ name: 'metric', description: metricNames.join(', ') }],
833
+ options: {
834
+ agg: { type: 'string', choices: ['avg', 'max', 'min', 'p95', 'last'], default: 'avg', description: 'Per-device aggregation over the window (ignored for reboots).' },
835
+ range: { alias: 'r', type: 'string', choices: metricWindows, description: 'Window (default 24h).' },
836
+ from: { type: 'string', placeholder: 'ISO', description: 'Absolute window start (with --to, at most 30 days).' },
837
+ to: { type: 'string', placeholder: 'ISO', description: 'Absolute window end.' },
838
+ top: { type: 'integer', default: 10, placeholder: 'N', description: 'Devices to return (1-100).' },
839
+ order: { type: 'string', choices: ['desc', 'asc'], default: 'desc', description: 'desc = highest first, asc = lowest first.' },
840
+ compare: { type: 'boolean', description: 'Also aggregate the previous window: previous and change per device.' },
841
+ sort: { type: 'string', choices: ['value', 'change'], description: 'Rank by value (default) or by change (needs --compare).' },
842
+ above: { type: 'number', placeholder: 'N', description: 'Only devices whose value is above N (e.g. --above 90).' },
843
+ below: { type: 'number', placeholder: 'N', description: 'Only devices whose value is below N.' },
844
+ query: { alias: 'q', type: 'string', placeholder: 'text', description: 'Device text filter (same as devices list).' },
845
+ status: { type: 'string', choices: deviceStatuses, description: 'Reception status.' },
846
+ lifecycle: { type: 'string', choices: lifecycles, description: 'Lifecycle scope (default operational).' },
847
+ role: { type: 'string', placeholder: 'role', description: 'Device role.' },
848
+ site: { type: 'string', placeholder: 'site', description: 'Site.' },
849
+ group: { type: 'string', placeholder: 'group', description: 'Group.' },
850
+ maintainer: { type: 'string', placeholder: 'name', description: 'Maintainer.' },
851
+ fleet: { type: 'string', choices: ['edge', 'infrastructure', 'metrics-only'], description: 'Fleet type.' },
852
+ label: { type: 'string', multiple: true, placeholder: 'k=v', description: 'Label match (up to 8).' },
853
+ },
854
+ examples: [
855
+ 'sentinelctl metrics top memory -r 7d',
856
+ 'sentinelctl metrics top disk --agg last --above 85 --top 50',
857
+ 'sentinelctl metrics top cpu --agg p95 -r 24h --compare --sort change',
858
+ 'sentinelctl metrics top reboots -r 7d --above 0',
859
+ 'sentinelctl metrics top networkTx -r 24h --label fleet=edge -o json',
860
+ ],
861
+ output: '{metric, aggregation, unit, window{start, end, seconds, range}, criteria, scope{devices, withData, withoutData, matched}, distribution{count, min, avg, p50, p90, p95, max}, previousDistribution?, devices[]: {agentId, displayName, host, status, value, previous?, change?}}.',
862
+ },
863
+ {
864
+ group: 'fleet', name: 'summary', run: fleetSummary, role: 'viewer', mutates: false,
865
+ summary: 'One-call fleet briefing: status counts, attention reasons, busiest devices, log levels and error patterns',
866
+ description: 'Start here to answer "how is the infrastructure?". Combines the device inventory, attention reasons with example devices, the highest current CPU/memory/disk/temperature, and log statistics for the window: totals by level, top and rising error patterns (compared with the previous window) and the devices with the most errors. About 5KB as JSON.',
867
+ options: {
868
+ range: { alias: 'r', type: 'string', choices: logRanges, default: '24h', description: 'Log window.' },
869
+ logs: { type: 'boolean', description: 'Include log statistics (default on; --no-logs skips them).' },
870
+ },
871
+ examples: ['sentinelctl fleet summary', 'sentinelctl fleet summary -r 1h -o json', 'sentinelctl fleet summary --no-logs'],
872
+ output: '{range, stale, devices{total, status, lifecycle, metrics, logs}, attention[]: {reason, count, devices[≤5]}, resources{cpu, memory, disk, temperature: [{agentId, displayName, value}]}, logs{window, total, devices, levels, errors{total, previousTotal, devices, newPatterns}, topErrorPatterns[], risingErrorPatterns[], fallingErrorPatterns[] (shrunk or gone since the previous window), topErrorDevices[]}}.',
873
+ },
680
874
  {
681
875
  group: 'metadata', name: 'get', run: metadataGet, role: 'viewer', mutates: false,
682
876
  summary: 'Device metadata: name, role, maintainer, labels, lifecycle',
@@ -812,7 +1006,9 @@ const commands = [
812
1006
 
813
1007
  export const catalog = commands;
814
1008
  export const groups = {
815
- logs: 'Search, count and rank logs; context around a log; volume trends; sources',
1009
+ fleet: 'One-call briefing of the whole fleet',
1010
+ logs: 'Search, count and rank logs; compare windows; context around a log; volume trends; sources',
1011
+ metrics: 'Rank devices by an aggregated metric across the fleet',
816
1012
  devices: 'Device list, details, lookup, reception overview',
817
1013
  metadata: 'Device metadata: read, edit (editor), import/export (admin)',
818
1014
  sessions: 'Your sessions and CLI tokens',
package/src/main.mjs CHANGED
@@ -14,6 +14,7 @@ const globalDefinitions = {
14
14
  profile: { type: 'string', placeholder: 'name', description: 'Saved login profile (default "default").' },
15
15
  output: { alias: 'o', type: 'string', choices: outputChoices, description: 'Output format. Use json or ndjson from agents and scripts.' },
16
16
  json: { type: 'boolean', description: 'Same as -o json.' },
17
+ select: { type: 'string', placeholder: 'fields', description: 'json/ndjson: keep only these comma-separated fields (dotted paths like metadata.role) of each item in the main list, plus top-level scalars. Example: --select agentId,displayName,status,memory' },
17
18
  quiet: { type: 'boolean', description: 'Suppress summaries and hints on stderr (errors are still printed).' },
18
19
  help: { alias: 'h', type: 'boolean', description: 'Show help.' },
19
20
  version: { alias: 'V', type: 'boolean', description: 'Show the version.' },
@@ -22,7 +23,7 @@ const specialCommands = {
22
23
  login: { summary: 'Approve in the browser with your intflow.ai Google Workspace account and store a 30-day CLI token', usage: 'sentinelctl login [--no-browser] [--name <client-name>]' },
23
24
  logout: { summary: 'Revoke the CLI token on the server and delete local credentials', usage: 'sentinelctl logout' },
24
25
  whoami: { summary: 'Show the server, user, role and token expiry', usage: 'sentinelctl whoami' },
25
- commands: { summary: 'List every command; with --json, a machine-readable catalog with options, roles and examples', usage: 'sentinelctl commands [--json]' },
26
+ commands: { summary: 'List every command; with --json, a machine-readable catalog with options, roles and examples (optionally for one group or command)', usage: 'sentinelctl commands [<group> [<command>]] [--json]' },
26
27
  help: { summary: 'Help: help <command>, help agents (guide for AI agents)', usage: 'sentinelctl help [<command> | agents]' },
27
28
  };
28
29
 
@@ -95,6 +96,7 @@ function overviewHelp() {
95
96
  '',
96
97
  'Getting started:',
97
98
  ' sentinelctl login approve once in the browser (valid for 30 days)',
99
+ ' sentinelctl fleet summary one-call briefing of the whole fleet',
98
100
  ' sentinelctl devices list --status offline',
99
101
  ' sentinelctl logs search -l error -r 1h',
100
102
  '',
@@ -130,35 +132,48 @@ const agentGuide = `Using sentinelctl from AI agents and scripts
130
132
  · 3 not logged in or insufficient role (do not retry). GET requests that hit 429 are retried
131
133
  automatically after Retry-After.
132
134
  4. \`sentinelctl commands --json\` returns every command with its options, choices, required role,
133
- whether it changes data, examples and output fields.
134
- 5. Typical investigation
135
+ whether it changes data, examples and output fields; \`commands logs --json\` or
136
+ \`commands "logs stats" --json\` returns only that part.
137
+ 5. Typical investigation (start broad, then narrow)
138
+ - overview: fleet summary -o json (~5KB: status, attention, busiest devices,
139
+ log levels, top/rising/new error patterns)
140
+ - what changed: logs stats -l error -r 24h --by message --compare --sort change -o json
141
+ - fleet metrics: metrics top memory -r 7d -o json (ranked devices + fleet distribution)
142
+ metrics top disk --agg last --above 85 · metrics top cpu --agg p95 --compare
143
+ metrics top reboots -r 7d --above 0
135
144
  - find a device: devices lookup <part of name> -> agentId
136
- - status: devices list --status offline -o json | devices overview <id> -o json
137
- - metrics: devices show <id> -r 24h -o json (series.cpu etc. are [epochSeconds, value])
138
- - logs: logs search -a <id> -l error -r 24h -o ndjson
145
+ - status: devices list --focus attention --all -o json --select agentId,displayName,status,attentionReasons
146
+ devices overview <id> -o json
147
+ - metrics: devices show <id> -r 24h -o json --series cpu,memory (series are [epochSeconds, value])
148
+ - logs: logs search -q "<searchText from logs stats>" -r 24h -o ndjson
149
+ logs search -a <id> -l error -r 24h -o ndjson
139
150
  then logs context -a <id> -s <source> -t <timestamp> for surrounding lines
140
151
  - counts/ranking: logs stats -l error -r 24h --by agent|source|message -o json
141
152
  - trend: logs histogram -a <id> -r 7d -o json
142
153
  6. Efficiency
143
- - To answer "how many" or "which devices/sources the most", use logs stats: one server-side
144
- aggregation with exact counts. Do not download thousands of logs to count them.
154
+ - Prefer aggregates: fleet summary, logs stats and metrics top answer "how many", "which the most"
155
+ and "what changed" in one server-side query. Do not download thousands of logs or call
156
+ devices show per device to compute fleet-wide numbers.
157
+ - Keep JSON small: --select <fields> keeps only the fields you need (devices list --all -o json is
158
+ ~700KB without it, ~20KB with a few fields); devices show --series none drops time series.
145
159
  - Narrow the window and filters (-a, -s, -l, -q) first. Use --pages N (max 40) to take only what you
146
160
  need; --all stops at 10,000 logs.
147
- - Each user can run 2 log queries at a time (web and CLI combined); run commands one after another.
161
+ - Each user can run 2 log queries and 2 metric aggregations at a time (web and CLI combined); run
162
+ commands one after another.
148
163
  - Do not use --follow (it never ends). Repeat a -r 15m search instead.
149
- - The server caches identical log searches for 5s, histograms and sources for 60s.
164
+ - The server caches identical log searches for 5s; stats, histograms, sources and metrics top for 60s.
150
165
  7. For commands that change data (metadata set/bulk/rollback/import, admin ...), run with --dry-run
151
166
  first when available, check the result, then apply.
152
167
  8. Server error messages are in Korean (shared with the web console); the "hint" field is English.`;
153
168
 
154
- function commandsCatalog() {
169
+ function commandsCatalog(selected = null) {
155
170
  return {
156
171
  version,
157
172
  globalOptions: globalDefinitions,
158
173
  exitCodes,
159
174
  commands: [
160
- ...Object.entries(specialCommands).map(([name, command]) => ({ command: name, usage: command.usage, summary: command.summary, role: null, mutates: name === 'logout' })),
161
- ...catalog.map((command) => ({
175
+ ...(selected ? [] : Object.entries(specialCommands).map(([name, command]) => ({ command: name, usage: command.usage, summary: command.summary, role: null, mutates: name === 'logout' }))),
176
+ ...(selected ?? catalog).map((command) => ({
162
177
  command: commandPath(command), usage: usageLine(command), summary: command.summary, description: command.description ?? null,
163
178
  role: command.role, roleNote: command.roleNote ?? null, mutates: command.mutates, args: command.args ?? [],
164
179
  options: Object.fromEntries(Object.entries(command.options).map(([name, spec]) => [name, {
@@ -329,10 +344,13 @@ export async function main(argv, {
329
344
  return 0;
330
345
  }
331
346
  if (name === 'commands') {
347
+ const scope = [subcommand, ...commandArguments].filter(Boolean).join(' ');
348
+ const selected = catalog.filter((item) => !scope || commandPath(item) === scope || item.group === scope);
349
+ if (scope && !selected.length) throw unknownCommandError(scope, [...Object.keys(groups), ...catalog.map(commandPath)], 'commands ');
332
350
  if (global.output === 'table') {
333
- out(catalog.map((item) => `${commandPath(item).padEnd(20)} ${item.role.padEnd(7)} ${item.mutates ? 'write' : 'read '} ${item.summary}`).join('\n'));
351
+ out(selected.map((item) => `${commandPath(item).padEnd(20)} ${item.role.padEnd(7)} ${item.mutates ? 'write' : 'read '} ${item.summary}`).join('\n'));
334
352
  } else {
335
- out(JSON.stringify(commandsCatalog(), null, global.output === 'json' ? 2 : 0));
353
+ out(JSON.stringify(commandsCatalog(scope ? selected : null), null, global.output === 'json' ? 2 : 0));
336
354
  }
337
355
  return 0;
338
356
  }
@@ -365,6 +383,7 @@ export async function main(argv, {
365
383
  path: credentialsPath(environment) });
366
384
  context = {
367
385
  global, connection, out, err, fetch, sleep, signal, openBrowser: open, width: stdout.columns ?? 120,
386
+ select: global.select ? global.select.split(',').map((field) => field.trim()).filter(Boolean) : null,
368
387
  hint: (text) => { if (!global.quiet) err(`hint: ${text}`); },
369
388
  note: (text) => { if (!global.quiet && global.output === 'table') err(text); },
370
389
  client: createClient({ server: connection.server, token: connection.token, fetch, sleep }),