ndi-cli 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ndi_cli/_render.py ADDED
@@ -0,0 +1,679 @@
1
+ """Compact, deterministic text renderings of the tool responses.
2
+
3
+ The audience is a coding agent reading terminal output: every fact the
4
+ response carries is either shown or explicitly counted, long free text is
5
+ kept whole (summaries and answers are the payload), and nothing is styled.
6
+ ``--json`` on any command bypasses all of this with the raw response.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import sys
13
+
14
+ import html
15
+ import re
16
+
17
+ from ndi_sdk.models.common import Coverage, Evidence, File
18
+ from ndi_sdk.models.document_ops import ExtractSchemaValidationResponse
19
+ from ndi_sdk.models.files import FileDetail
20
+ from ndi_sdk.models.jobs import (
21
+ ClassifyResult,
22
+ DeepSearchV2Result,
23
+ ExtractResult,
24
+ GroundResult,
25
+ IngestionResult,
26
+ IntelligentSearchResult,
27
+ Job,
28
+ ParseResult,
29
+ SplitResult,
30
+ )
31
+ from ndi_sdk.models.workspaces import Workspace, WorkspaceStats
32
+ from ndi_sdk.models.tools import (
33
+ FileMetadataToolResponse,
34
+ FolderMetadataResponse,
35
+ QaFileResponse,
36
+ ReadFileResponse,
37
+ RunSqlResponse,
38
+ SearchResponse,
39
+ )
40
+
41
+ _SNIPPET_LIMIT = 500
42
+
43
+
44
+ def _size(num_bytes: int) -> str:
45
+ value = float(num_bytes)
46
+ for unit in ("B", "KB", "MB", "GB"):
47
+ if value < 1024 or unit == "GB":
48
+ return f"{value:.1f}{unit}" if unit != "B" else f"{int(value)}B"
49
+ value /= 1024
50
+ return f"{int(value)}B"
51
+
52
+
53
+ def _clip(text: str, limit: int = _SNIPPET_LIMIT) -> str:
54
+ text = text.strip()
55
+ if len(text) <= limit:
56
+ return text
57
+ return text[:limit].rstrip() + f"… [{len(text) - limit} more chars, use --json for all]"
58
+
59
+
60
+ #: A well-formed closing tag — the trigger for markup collapsing. Prose that
61
+ #: merely contains ``<`` ("x < 5") never matches, so it rides through verbatim.
62
+ _CLOSING_TAG = re.compile(r"</[a-zA-Z][^>]*>")
63
+ _CELL_BOUNDARY = re.compile(r"</(?:td|th|tr)\s*>", re.IGNORECASE)
64
+ #: Tag-like spans only (a letter or ``/`` after ``<``), matching the other two
65
+ #: patterns — an in-cell inequality pair ("a < b and c > d") is content, not
66
+ #: markup, and must survive even after flattening has triggered.
67
+ _TAG = re.compile(r"</?[a-zA-Z][^>]*>")
68
+ #: An unterminated tag opened at the very end — the server's snippet window cut
69
+ #: mid-tag. Requires a letter after ``<``, so a bare trailing ``<`` or ``a < b``
70
+ #: math survives.
71
+ _TRUNCATED_TAG_TAIL = re.compile(r"</?[a-zA-Z][^>]*$")
72
+ _SEPARATOR_RUN = re.compile(r"\|(?:\s*\|)+")
73
+
74
+
75
+ def _flatten_markup(text: str) -> str:
76
+ """Collapse HTML table soup to readable cell text; markup-free text is untouched.
77
+
78
+ PDF-table snippets arrive as raw ``</td></tr><tr><td>…`` markup — roughly
79
+ half the snippet window was tags. Cell/row boundaries become ``" | "``,
80
+ every other tag becomes whitespace, entities are unescaped, and whitespace
81
+ is collapsed — so the clipped window carries content, not angle brackets.
82
+ Text mode only; ``--json`` stays raw.
83
+ """
84
+ if not _CLOSING_TAG.search(text):
85
+ return text
86
+ text = _TRUNCATED_TAG_TAIL.sub("", text)
87
+ text = _CELL_BOUNDARY.sub(" | ", text)
88
+ text = _TAG.sub(" ", text)
89
+ text = html.unescape(text)
90
+ text = re.sub(r"\s+", " ", text)
91
+ text = _SEPARATOR_RUN.sub("|", text)
92
+ return text.strip(" |")
93
+
94
+
95
+ def _coverage_lines(coverage: Coverage) -> list[str]:
96
+ if coverage.restricted_candidates <= 0:
97
+ return []
98
+ labels = f" (labels required: {', '.join(coverage.labels_required)})" if coverage.labels_required else ""
99
+ lines = [f"note: {coverage.restricted_candidates} candidate(s) withheld by access controls{labels}"]
100
+ if coverage.request_access_hint:
101
+ lines.append(f"access: {coverage.request_access_hint}")
102
+ return lines
103
+
104
+
105
+ def _categories(pairs: list, limit: int | None = None) -> str:
106
+ shown = pairs if limit is None else pairs[:limit]
107
+ rendered = ", ".join(f"{c.name} x{c.count}" for c in shown)
108
+ if limit is not None and len(pairs) > limit:
109
+ rendered += f", +{len(pairs) - limit} more"
110
+ return rendered
111
+
112
+
113
+ def _locator(locator) -> str:
114
+ """One compact position per locator kind — the part of a citation an agent follows."""
115
+ match locator.kind:
116
+ case "spreadsheet_range":
117
+ return f"{locator.sheet}!{locator.a1_range}"
118
+ case "text_range":
119
+ page = f"page {locator.page}, " if locator.page is not None else ""
120
+ return f"{page}chars {locator.char_start}-{locator.char_end}"
121
+ case "visual_region":
122
+ return f"page {locator.page}, bbox {locator.bbox}"
123
+ case "jsonl_record":
124
+ column = f" col {locator.column}" if locator.column else ""
125
+ return f"row {locator.row_offset}{column}"
126
+ case "audio_range":
127
+ return f"{locator.start_ms / 1000:.0f}s-{locator.end_ms / 1000:.0f}s"
128
+ return locator.kind
129
+
130
+
131
+ def _evidence_lines(hits: list[Evidence]) -> list[str]:
132
+ lines: list[str] = []
133
+ for index, hit in enumerate(hits, start=1):
134
+ where = f" [{hit.component}]" if hit.component else ""
135
+ page = f", page {hit.page}" if hit.page is not None else ""
136
+ at = f" @ {_locator(hit.locator)}" if hit.locator is not None else ""
137
+ lines.append(f"{index}. {hit.path}{where} (score {hit.relevance_score:.3f}{page}){at}")
138
+ if hit.snippet:
139
+ lines.append(f" {_clip(_flatten_markup(hit.snippet))}")
140
+ if hit.why:
141
+ lines.append(f" why: {_clip(hit.why, 300)}")
142
+ return lines
143
+
144
+
145
+ def folder_metadata(response: FolderMetadataResponse) -> str:
146
+ lines = [f"directory: {response.directory}", f"files in scope: {response.overall_files}"]
147
+ if response.categories:
148
+ lines.append(f"categories: {_categories(response.categories)}")
149
+ if response.ingestion_summary:
150
+ summary = ", ".join(f"{status} x{count}" for status, count in sorted(response.ingestion_summary.items()))
151
+ lines.append(f"ingestion: {summary}")
152
+ if response.subdirectories:
153
+ lines.append(f"subdirectories ({len(response.subdirectories)}):")
154
+ for sub in response.subdirectories:
155
+ failed = f" ({sub.failed_files} failed)" if sub.failed_files else ""
156
+ cats = f" categories: {_categories(sub.top_categories, limit=3)}" if sub.top_categories else ""
157
+ lines.append(f" {sub.name}/ {sub.overall_files} files{failed}{cats}")
158
+ if response.file_entries:
159
+ cut = " (truncated — resume with --start-after, or narrow)" if response.files_truncated else ""
160
+ lines.append(f"files listed ({len(response.file_entries)}){cut}:")
161
+ for entry in response.file_entries:
162
+ extras = " ".join(part for part in (entry.category, entry.kind) if part)
163
+ extras = f" [{extras}]" if extras else ""
164
+ uploaded = f" {entry.uploaded_at:%Y-%m-%d}" if entry.uploaded_at is not None else ""
165
+ lines.append(f" {entry.ingestion_status} {_size(entry.size_bytes):>9}{uploaded} {entry.path}{extras}")
166
+ elif response.files_truncated:
167
+ lines.append("files listed (0) — page cut before any entry; raise --page-size or narrow")
168
+ if response.hint:
169
+ lines.append(f"hint: {response.hint}")
170
+ if response.tables is not None:
171
+ cut = " (truncated — narrow the directory)" if response.tables_truncated else ""
172
+ lines.append(f"tables ({len(response.tables)}){cut}:")
173
+ lines.extend(_census_table_lines(response.tables))
174
+ if response.shared_columns:
175
+ sql_names = {(t.path, t.component): t.table_name for t in (response.tables or []) if t.table_name}
176
+ cut = " (list truncated)" if response.shared_columns_truncated else ""
177
+ lines.append(f"shared columns — UNVERIFIED join candidates, confirm with one run-sql join{cut}:")
178
+ for column in response.shared_columns:
179
+ sides = "; ".join(
180
+ _join_side(occ, sql_names.get((occ.path, occ.component))) for occ in column.appears_in[:_JOIN_EXAMPLES_MAX]
181
+ )
182
+ hidden = column.table_count - min(len(column.appears_in), _JOIN_EXAMPLES_MAX)
183
+ more = f"; +{hidden} more" if hidden > 0 else ""
184
+ lines.append(f" {column.name} in {column.table_count} tables: {sides}{more}")
185
+ lines.extend(_coverage_lines(response.coverage))
186
+ return "\n".join(lines)
187
+
188
+
189
+ #: Join-candidate examples shown per shared column in text mode.
190
+ _JOIN_EXAMPLES_MAX = 3
191
+
192
+ #: The partition vocabulary, mirroring je_testing's ``partitions.py``: a
193
+ #: plausible four-digit year (1900–2099) or a ``p``/``P`` part marker (``p1``,
194
+ #: ``P12``), alone between token boundaries. An arbitrary digit run
195
+ #: (``invoice_1001``) is NOT a partition value — masking every digit run
196
+ #: grouped unrelated ids into an invented population.
197
+ _PARTITION_TOKEN = re.compile(r"(?<![A-Za-z0-9])(?:(?:19|20)\d{2}|[pP]\d+)(?![A-Za-z0-9])")
198
+ _COLLISION_SUFFIX = re.compile(r"_\d+$")
199
+
200
+
201
+ def _join_side(occ, sql_name: str | None) -> str:
202
+ """One join-candidate occurrence — named by its sql table when the census knows it."""
203
+ name = sql_name or f"{occ.path}::{occ.component}"
204
+ if occ.distinct_count is not None and occ.row_count is not None:
205
+ return f"{name} (distinct≈{occ.distinct_count}/{occ.row_count} rows)"
206
+ return name
207
+
208
+
209
+ def _norm_name(text: str) -> str:
210
+ return re.sub(r"[^a-z0-9]+", "_", text.lower()).strip("_")
211
+
212
+
213
+ def _component_redundant(table) -> bool:
214
+ """True when the component restates the sql name (modulo case/sanitizing/collision suffix) or the file stem."""
215
+ if not table.table_name:
216
+ return False
217
+ component = _norm_name(table.component)
218
+ name = _norm_name(table.table_name)
219
+ stem = _norm_name(table.path.rsplit("/", 1)[-1].rsplit(".", 1)[0])
220
+ return component in (name, _COLLISION_SUFFIX.sub("", name), stem)
221
+
222
+
223
+ def _extent(table) -> str:
224
+ extent = f"{table.row_count} rows" if table.row_count is not None else "rows unknown"
225
+ if table.column_count is not None:
226
+ extent += f" x {table.column_count} cols"
227
+ return extent
228
+
229
+
230
+ def _family_key(table) -> tuple[str, str, str] | None:
231
+ """The partition-family a table belongs to, or None for a standalone table.
232
+
233
+ Siblings live in the SAME directory (real corpora nest files under numbered
234
+ directories, so directory digits never count as partition values) and
235
+ differ only in partition tokens — years or pN part markers, the
236
+ :data:`_PARTITION_TOKEN` vocabulary — with the SAME tokens in file stem and
237
+ sql name. The family line's shared directory + basename pattern plus a
238
+ member's values therefore reconstruct its exact path. A name whose only
239
+ digits are ids or a collision suffix carries no partition token and stays a
240
+ standalone row. A component that is not redundant carries information a
241
+ family row would hide, so it opts out.
242
+ """
243
+ if not table.table_name or not _component_redundant(table):
244
+ return None
245
+ values = _PARTITION_TOKEN.findall(table.table_name)
246
+ directory, _, basename = table.path.rpartition("/")
247
+ stem, dot, extension = basename.rpartition(".")
248
+ if not dot:
249
+ stem = basename
250
+ if not values or [v.casefold() for v in _PARTITION_TOKEN.findall(stem)] != [v.casefold() for v in values]:
251
+ return None
252
+ basename_pattern = _PARTITION_TOKEN.sub("*", stem) + (f".{extension}" if dot else "")
253
+ return (directory, basename_pattern, _PARTITION_TOKEN.sub("*", table.table_name))
254
+
255
+
256
+ def _census_table_lines(tables: list) -> list[str]:
257
+ """One line per standalone table; partition families grouped under one header.
258
+
259
+ Grouping SURFACES the partition structure (the family line names the stem
260
+ and every partition value) and never hides a member's sql name or
261
+ rows×cols — those stay printed per member. The redundant ``[component]``
262
+ middle name is dropped whenever the sql name or file stem already says it.
263
+ """
264
+ members_by_family: dict[tuple[str, str, str], list] = {}
265
+ for table in tables:
266
+ key = _family_key(table)
267
+ if key is not None:
268
+ members_by_family.setdefault(key, []).append(table)
269
+ families = {key: members for key, members in members_by_family.items() if len(members) > 1}
270
+ lines: list[str] = []
271
+ rendered: set[tuple[str, str, str]] = set()
272
+ for table in tables:
273
+ key = _family_key(table)
274
+ if key in families:
275
+ if key in rendered:
276
+ continue
277
+ rendered.add(key)
278
+ members = families[key]
279
+ values = ["/".join(_PARTITION_TOKEN.findall(member.table_name)) for member in members]
280
+ directory, basename_pattern, name_pattern = key
281
+ path_pattern = f"{directory}/{basename_pattern}" if directory else basename_pattern
282
+ lines.append(f" family {name_pattern} (partitions {', '.join(values)}) — {path_pattern}:")
283
+ lines.extend(f" {value}: {_extent(member)} sql: {member.table_name}" for value, member in zip(values, members))
284
+ continue
285
+ component = "" if _component_redundant(table) else f" [{table.component}]"
286
+ sql_name = f" sql: {table.table_name}" if table.table_name else ""
287
+ lines.append(f" {_extent(table)} {table.path}{component}{sql_name}")
288
+ return lines
289
+
290
+
291
+ def file_metadata(response: FileMetadataToolResponse) -> str:
292
+ facts = [f"status: {response.ingestion_status}", f"size: {_size(response.size_bytes)}"]
293
+ if response.kind:
294
+ facts.insert(0, f"kind: {response.kind}")
295
+ if response.page_count is not None:
296
+ facts.append(f"pages: {response.page_count}")
297
+ if response.tab_count is not None:
298
+ facts.append(f"tabs: {response.tab_count}")
299
+ if response.total_chars is not None:
300
+ facts.append(f"chars: {response.total_chars}")
301
+ elif response.total_approx_bytes is not None:
302
+ facts.append(f"~{_size(response.total_approx_bytes)} text")
303
+ lines = [f"path: {response.path}", " | ".join(facts)]
304
+ if response.ingestion_error:
305
+ lines.append(f"ingestion error: {response.ingestion_error}")
306
+ if response.summary:
307
+ lines.append(f"summary: {response.summary}")
308
+ if response.components:
309
+ lines.append(f"components ({len(response.components)}):")
310
+ for comp in response.components:
311
+ if comp.kind == "media":
312
+ seconds = comp.duration_ms / 1000
313
+ frames = f", {comp.frame_count} sampled frame(s)" if comp.frame_count else ""
314
+ lines.append(
315
+ f" [media/{comp.modality}] {seconds:.0f}s, {comp.speaker_count} speaker(s), "
316
+ f"{comp.speech_segment_count} speech segment(s){frames} — ask with ask-file"
317
+ )
318
+ if comp.summary:
319
+ lines.append(f" {comp.summary}")
320
+ continue
321
+ extent: list[str] = []
322
+ if comp.page_start is not None and comp.page_end is not None:
323
+ extent.append(f"pages {comp.page_start}-{comp.page_end}")
324
+ if comp.row_count is not None:
325
+ cols = f" x {comp.column_count} cols" if comp.column_count is not None else ""
326
+ extent.append(f"{comp.row_count} rows{cols}")
327
+ if comp.char_count is not None:
328
+ extent.append(f"{comp.char_count} chars")
329
+ elif comp.approx_bytes is not None:
330
+ extent.append(f"~{_size(comp.approx_bytes)}")
331
+ abilities = [name for name, flag in (("readable", comp.readable), ("queryable", comp.queryable)) if flag]
332
+ if comp.table_name:
333
+ abilities.append(f"sql table: {comp.table_name}")
334
+ tail = ", ".join(extent + (abilities or ["no content tools apply"]))
335
+ name = comp.name or "(unnamed)"
336
+ lines.append(f" [{comp.kind}] {name} — {tail}")
337
+ if comp.summary:
338
+ lines.append(f" {comp.summary}")
339
+ if comp.columns:
340
+ lines.extend(_column_lines(comp.columns, bool(comp.columns_truncated)))
341
+ if comp.error:
342
+ lines.append(f" error: {comp.error}")
343
+ return "\n".join(lines)
344
+
345
+
346
+ #: Census values shown per column in text mode — the full list rides --json.
347
+ _SHOWN_VALUES = 8
348
+
349
+
350
+ def _column_lines(columns: list, truncated: bool) -> list[str]:
351
+ """Render a tab's schema: one dense line per column when stats are present.
352
+
353
+ A schema without stats (a sidecar from before profiling measured them)
354
+ stays a single names+types line, as before.
355
+ """
356
+ if not any(col.null_count is not None or col.top_values or col.sample_values for col in columns):
357
+ rendered = ", ".join(f"{col.name} ({col.data_type})" for col in columns)
358
+ cut = " …schema truncated" if truncated else ""
359
+ return [f" columns ({len(columns)}): {rendered}{cut}"]
360
+ lines = [f" columns ({len(columns)}){' …schema truncated' if truncated else ''}:"]
361
+ # Only the server's own proof (all_null, measured against the profile's exact row
362
+ # count) groups a column as empty; comparing counts from two sources here could
363
+ # hide a populated column. getattr: an older server's schema has no such field.
364
+ empty = [col for col in columns if getattr(col, "all_null", None)]
365
+ if empty:
366
+ lines.append(" all-null columns (no values): " + ", ".join(f"{col.name} ({col.data_type})" for col in empty))
367
+ for col in columns:
368
+ if col in empty:
369
+ continue
370
+ facts: list[str] = []
371
+ if col.distinct_count is not None:
372
+ facts.append(f"distinct≈{col.distinct_count}")
373
+ if col.null_count:
374
+ facts.append(f"nulls {col.null_count}")
375
+ if col.sum_value is not None:
376
+ facts.append(f"sum {col.sum_value}")
377
+ if col.min_value is not None and col.max_value is not None:
378
+ facts.append(f"range {col.min_value} .. {col.max_value}")
379
+ line = f" {col.name} {col.data_type}" + (f" {', '.join(facts)}" if facts else "")
380
+ values = col.top_values or []
381
+ if values:
382
+ shown = ", ".join(f"{entry.value} x{entry.count}" for entry in values[:_SHOWN_VALUES])
383
+ more = len(values) - _SHOWN_VALUES
384
+ tail = f", +{more} more" if more > 0 else ""
385
+ coverage = (
386
+ f" (top {len(values)} cover {col.top_values_coverage:.0%})"
387
+ if col.top_values_coverage is not None and col.top_values_coverage < 1
388
+ else ""
389
+ )
390
+ line += f" values: {shown}{tail}{coverage}"
391
+ elif col.sample_values:
392
+ shown = ", ".join(col.sample_values[:_SHOWN_VALUES])
393
+ more = len(col.sample_values) - _SHOWN_VALUES
394
+ line += f" values: {shown}" + (f", +{more} more" if more > 0 else "")
395
+ lines.append(line)
396
+ return lines
397
+
398
+
399
+ def read_file_footer(response: ReadFileResponse) -> str:
400
+ parts = [f"read {response.path}"]
401
+ if response.pages_returned:
402
+ parts.append(f"pages {_page_ranges(response.pages_returned)}")
403
+ if response.components_returned:
404
+ parts.append(f"components: {', '.join(response.components_returned)}")
405
+ if response.components_omitted:
406
+ parts.append(f"omitted: {', '.join(response.components_omitted)}")
407
+ if response.truncated:
408
+ parts.append("TRUNCATED (raise --max-chars, or scope with --component/--pages)")
409
+ return " | ".join(parts)
410
+
411
+
412
+ def _page_ranges(pages: list[int]) -> str:
413
+ ranges: list[str] = []
414
+ start = prev = pages[0]
415
+ for page in pages[1:]:
416
+ if page == prev + 1:
417
+ prev = page
418
+ continue
419
+ ranges.append(f"{start}-{prev}" if start != prev else str(start))
420
+ start = prev = page
421
+ ranges.append(f"{start}-{prev}" if start != prev else str(start))
422
+ return ",".join(ranges)
423
+
424
+
425
+ def qa_file(response: QaFileResponse) -> str:
426
+ lines = [response.answer.strip(), "", f"engine: {response.engine}"]
427
+ if response.unavailable:
428
+ lines.append("unavailable:")
429
+ lines.extend(f" {item.path}: {item.error}" for item in response.unavailable)
430
+ if response.citations:
431
+ lines.append(f"citations ({len(response.citations)}):")
432
+ lines.extend(f" {line}" for line in _evidence_lines(response.citations))
433
+ return "\n".join(lines)
434
+
435
+
436
+ def run_sql(response: RunSqlResponse) -> str:
437
+ lines: list[str] = []
438
+ result = response.result
439
+ cut = ""
440
+ if result.preview_truncated:
441
+ # The download hint must not outrun the response: when retention
442
+ # failed there is no citation and no link to follow.
443
+ cut = (
444
+ ", preview truncated (full result via the download link or LIMIT/OFFSET)"
445
+ if response.result_citation is not None
446
+ else ", preview truncated (page it with LIMIT/OFFSET)"
447
+ )
448
+ dup = " (duplicate — served from the stored result)" if response.duplicate else ""
449
+ lines.append(f"result table {result.handle} — {result.row_count} rows{cut}{dup}")
450
+ if result.columns:
451
+ names = [col.name for col in result.columns]
452
+ lines.append(" " + " | ".join(f"{col.name} ({col.data_type})" for col in result.columns))
453
+ if len(set(names)) < len(names):
454
+ # Object rows keep one value per key, so a duplicated name loses a column
455
+ # with no visible trace against the header — say so instead.
456
+ lines.append(
457
+ " note: duplicate column names in the result — object rows keep only one value per name; alias duplicates (AS) to see both."
458
+ )
459
+ for row in result.rows:
460
+ # default=str keeps an unvalidated value from a future producer readable
461
+ # instead of crashing the render.
462
+ lines.append(" " + json.dumps({name: row.get(name, "") for name in names}, ensure_ascii=False, default=str))
463
+ lines.append(f"note: '{result.handle}' is a registered table — join or page it in your next run-sql call.")
464
+ citation = response.result_citation
465
+ # source_paths rides the response itself; older servers only set it on the
466
+ # citation, and agent-attributed statements have no citation at all.
467
+ sources = response.source_paths or (citation.source_paths if citation is not None else [])
468
+ if sources:
469
+ lines.append(f"sources: {', '.join(sources)}")
470
+ if citation is not None:
471
+ lines.append(f"download (parquet): {citation.result_file.download_url}")
472
+ if response.result_materialization_error:
473
+ lines.append(f"note: {response.result_materialization_error}")
474
+ return "\n".join(lines)
475
+
476
+
477
+ def search(response: SearchResponse) -> str:
478
+ if not response.hits:
479
+ lines = ["no hits"]
480
+ else:
481
+ lines = _evidence_lines(response.hits)
482
+ tail = f"candidates seen: {response.candidates_seen}"
483
+ if response.truncated:
484
+ tail += " (truncated upstream of ranking — a matching file may be missing; narrow with --path-prefix)"
485
+ lines.append(tail)
486
+ lines.extend(_coverage_lines(response.coverage))
487
+ return "\n".join(lines)
488
+
489
+
490
+ def parse_job(job: Job) -> str:
491
+ result = job.result
492
+ if not isinstance(result, ParseResult):
493
+ return job_line(job)
494
+ print(
495
+ f"{result.file_name} | {result.document.page_count} pages | lane {result.lane} | "
496
+ f"ocr_applied={result.ocr_applied} | {result.units} units",
497
+ file=sys.stderr,
498
+ )
499
+ document = result.document
500
+ if document.markdown or document.text:
501
+ return document.markdown or document.text or ""
502
+ # `--format blocks` asks for neither markdown nor text. Returning "" here
503
+ # threw away the whole result — a 0-byte --save after a billed job.
504
+ if document.blocks is not None:
505
+ return json.dumps([block.model_dump(mode="json", exclude_none=True) for block in document.blocks], indent=2)
506
+ return ""
507
+
508
+
509
+ def split_job(job: Job) -> str:
510
+ result = job.result
511
+ if not isinstance(result, SplitResult):
512
+ return job_line(job)
513
+ lines = [f"{result.file_name} | {result.page_count} pages | {len(result.segments)} segments | lane {result.lane}"]
514
+ for segment in result.segments:
515
+ label = segment.split_class.label if segment.split_class is not None else segment.status
516
+ pages = f"pages {segment.start_page}-{segment.end_page}" if segment.start_page is not None else "pages ?"
517
+ sheet = f" sheet {segment.sheet_name}" if segment.sheet_name else ""
518
+ lines.append(f" {segment.sequence}. {label} {pages}{sheet}")
519
+ if segment.content:
520
+ lines.append(f" {segment.content}")
521
+ return "\n".join(lines)
522
+
523
+
524
+ def classify_job(job: Job) -> str:
525
+ result = job.result
526
+ if not isinstance(result, ClassifyResult):
527
+ return job_line(job)
528
+ lines = [f"units {len(result.units)} | unknown {result.unknown_units} | pages_read {result.pages_read_fraction:.0%}"]
529
+ for unit in result.units:
530
+ pages = f" pages {unit.page_range.start}-{unit.page_range.end}" if unit.page_range is not None else ""
531
+ unknown = " unknown" if unit.unknown else ""
532
+ lines.append(f" {unit.granularity}{pages}{unknown}")
533
+ for label in unit.labels:
534
+ conf = f" {label.confidence:.2f}" if label.confidence is not None else ""
535
+ reason = f" — {label.reason}" if label.reason else ""
536
+ lines.append(f" {label.rank}. {label.label} ({label.class_id}){conf}{reason}")
537
+ return "\n".join(lines)
538
+
539
+
540
+ def extract_job(job: Job) -> str:
541
+ result = job.result
542
+ if not isinstance(result, ExtractResult):
543
+ return job_line(job)
544
+ lines = [json.dumps(result.data, indent=2, ensure_ascii=False, default=str)]
545
+ if result.fields:
546
+ lines.append("fields:")
547
+ for field in result.fields:
548
+ conf = f" {field.confidence:.2f}" if field.confidence is not None else ""
549
+ cites = f" citations {len(field.citations)}" if field.citations else ""
550
+ lines.append(f" {field.path}: {field.status}{conf}{cites}")
551
+ if result.warnings:
552
+ lines.append("warnings:")
553
+ lines.extend(f" {warning}" for warning in result.warnings)
554
+ return "\n".join(lines)
555
+
556
+
557
+ def ground_job(job: Job) -> str:
558
+ result = job.result
559
+ if not isinstance(result, GroundResult):
560
+ return job_line(job)
561
+ lines = [f"targets {len(result.targets)} | {result.units} units"]
562
+ for target in result.targets:
563
+ lines.append(f" {target.id}: {target.status}")
564
+ if target.error:
565
+ lines.append(f" error: {target.error}")
566
+ for match in target.matches:
567
+ conf = f" {match.confidence:.2f}" if match.confidence is not None else ""
568
+ lines.append(f" {match.rank}. {match.match_method}{conf} {match.matched_text}")
569
+ return "\n".join(lines)
570
+
571
+
572
+ def schema_validation(result: ExtractSchemaValidationResponse) -> str:
573
+ if result.valid:
574
+ return "schema valid"
575
+ lines = [f"schema invalid ({len(result.errors)} error(s))"]
576
+ lines.extend(f" {error.path}: {error.message}" for error in result.errors)
577
+ return "\n".join(lines)
578
+
579
+
580
+ def workspace(item: Workspace) -> str:
581
+ purpose = f" purpose={item.purpose}" if item.purpose else ""
582
+ return f"{item.workspace_id} {item.name} {item.domain_slug}@{item.domain_version} {item.status}{purpose}"
583
+
584
+
585
+ def workspace_page(page) -> str:
586
+ lines = [f"workspaces ({page.total_count}):"]
587
+ lines.extend(f" {workspace(item)}" for item in page.items)
588
+ if page.next_cursor:
589
+ lines.append(f"next: {page.next_cursor}")
590
+ return "\n".join(lines)
591
+
592
+
593
+ def workspace_stats(stats: WorkspaceStats) -> str:
594
+ statuses = ", ".join(
595
+ f"{status} x{count}" for status, count in sorted(stats.files_by_ingestion_status.items(), key=lambda pair: str(pair[0]))
596
+ )
597
+ lines = [f"files: {stats.files_total}"]
598
+ if statuses:
599
+ lines.append(f"ingestion: {statuses}")
600
+ lines.extend(
601
+ (
602
+ f"bytes: {stats.bytes_total}",
603
+ f"pages ingested: {stats.pages_ingested}",
604
+ f"index documents: {stats.index_documents}",
605
+ f"kg: {stats.kg_nodes} nodes / {stats.kg_edges} edges",
606
+ )
607
+ )
608
+ return "\n".join(lines)
609
+
610
+
611
+ def file_line(item: File) -> str:
612
+ return f"{item.ingestion_status} {_size(item.size_bytes):>9} {item.path} {item.file_id}"
613
+
614
+
615
+ def file_page(page) -> str:
616
+ lines = [f"files ({page.total_count}):"]
617
+ lines.extend(f" {file_line(item)}" for item in page.items)
618
+ if page.next_cursor:
619
+ lines.append(f"next: {page.next_cursor}")
620
+ return "\n".join(lines)
621
+
622
+
623
+ def file_detail(item: FileDetail) -> str:
624
+ lines = [file_line(item), f"download: {item.download_url}"]
625
+ if item.category:
626
+ lines.append(f"category: {item.category}")
627
+ return "\n".join(lines)
628
+
629
+
630
+ def job_line(job: Job) -> str:
631
+ name = f" {job.name}" if job.name else ""
632
+ return f"{job.status} {job.kind} {job.job_id}{name}"
633
+
634
+
635
+ def job_page(page) -> str:
636
+ lines = [f"jobs ({page.total_count}):"]
637
+ lines.extend(f" {job_line(job)}" for job in page.items)
638
+ if page.next_cursor:
639
+ lines.append(f"next: {page.next_cursor}")
640
+ return "\n".join(lines)
641
+
642
+
643
+ def ingestion_job(job: Job) -> str:
644
+ result = job.result
645
+ if not isinstance(result, IngestionResult):
646
+ return job_line(job)
647
+ lines = [
648
+ f"ingested {result.files_ingested} | skipped {result.files_skipped} | failed {result.files_failed} | {result.units_total} units"
649
+ ]
650
+ for outcome in result.outcomes:
651
+ extra = f" {outcome.error.message}" if outcome.error else ""
652
+ lines.append(f" {outcome.status} {outcome.path}{extra}")
653
+ return "\n".join(lines)
654
+
655
+
656
+ def search_job(job: Job) -> str:
657
+ result = job.result
658
+ if isinstance(result, DeepSearchV2Result):
659
+ return _search_result(result.answer, result.clarification, result.evidences, receipts=len(result.sql_receipts))
660
+ if isinstance(result, IntelligentSearchResult):
661
+ return _search_result(result.answer, result.clarification, result.evidences, receipts=0)
662
+ return job_line(job)
663
+
664
+
665
+ def _search_result(answer: str | None, clarification: str | None, evidences: list, *, receipts: int) -> str:
666
+ lines: list[str] = []
667
+ if clarification:
668
+ lines.append(f"clarification: {clarification}")
669
+ if answer:
670
+ lines.append(answer.strip())
671
+ lines.append(f"evidences ({len(evidences)}):")
672
+ for index, evidence in enumerate(evidences, start=1):
673
+ page = f", page {evidence.page}" if evidence.page is not None else ""
674
+ lines.append(f"{index}. {evidence.source_path}{page}")
675
+ if evidence.quote:
676
+ lines.append(f" {evidence.quote}")
677
+ if receipts:
678
+ lines.append(f"receipts: {receipts}")
679
+ return "\n".join(lines)