ndi-cli 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ndi_cli/__init__.py +9 -0
- ndi_cli/_cli.py +166 -0
- ndi_cli/_config.py +94 -0
- ndi_cli/_docops.py +443 -0
- ndi_cli/_jobs.py +106 -0
- ndi_cli/_local.py +130 -0
- ndi_cli/_login.py +47 -0
- ndi_cli/_output.py +128 -0
- ndi_cli/_render.py +679 -0
- ndi_cli/_sources.py +154 -0
- ndi_cli/_tools.py +210 -0
- ndi_cli/_workspace.py +226 -0
- ndi_cli/commands.py +23 -0
- ndi_cli-0.4.0.dist-info/METADATA +86 -0
- ndi_cli-0.4.0.dist-info/RECORD +18 -0
- ndi_cli-0.4.0.dist-info/WHEEL +4 -0
- ndi_cli-0.4.0.dist-info/entry_points.txt +3 -0
- ndi_cli-0.4.0.dist-info/licenses/LICENSE +21 -0
ndi_cli/_render.py
ADDED
|
@@ -0,0 +1,679 @@
|
|
|
1
|
+
"""Compact, deterministic text renderings of the tool responses.
|
|
2
|
+
|
|
3
|
+
The audience is a coding agent reading terminal output: every fact the
|
|
4
|
+
response carries is either shown or explicitly counted, long free text is
|
|
5
|
+
kept whole (summaries and answers are the payload), and nothing is styled.
|
|
6
|
+
``--json`` on any command bypasses all of this with the raw response.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
|
|
14
|
+
import html
|
|
15
|
+
import re
|
|
16
|
+
|
|
17
|
+
from ndi_sdk.models.common import Coverage, Evidence, File
|
|
18
|
+
from ndi_sdk.models.document_ops import ExtractSchemaValidationResponse
|
|
19
|
+
from ndi_sdk.models.files import FileDetail
|
|
20
|
+
from ndi_sdk.models.jobs import (
|
|
21
|
+
ClassifyResult,
|
|
22
|
+
DeepSearchV2Result,
|
|
23
|
+
ExtractResult,
|
|
24
|
+
GroundResult,
|
|
25
|
+
IngestionResult,
|
|
26
|
+
IntelligentSearchResult,
|
|
27
|
+
Job,
|
|
28
|
+
ParseResult,
|
|
29
|
+
SplitResult,
|
|
30
|
+
)
|
|
31
|
+
from ndi_sdk.models.workspaces import Workspace, WorkspaceStats
|
|
32
|
+
from ndi_sdk.models.tools import (
|
|
33
|
+
FileMetadataToolResponse,
|
|
34
|
+
FolderMetadataResponse,
|
|
35
|
+
QaFileResponse,
|
|
36
|
+
ReadFileResponse,
|
|
37
|
+
RunSqlResponse,
|
|
38
|
+
SearchResponse,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
_SNIPPET_LIMIT = 500
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _size(num_bytes: int) -> str:
|
|
45
|
+
value = float(num_bytes)
|
|
46
|
+
for unit in ("B", "KB", "MB", "GB"):
|
|
47
|
+
if value < 1024 or unit == "GB":
|
|
48
|
+
return f"{value:.1f}{unit}" if unit != "B" else f"{int(value)}B"
|
|
49
|
+
value /= 1024
|
|
50
|
+
return f"{int(value)}B"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _clip(text: str, limit: int = _SNIPPET_LIMIT) -> str:
|
|
54
|
+
text = text.strip()
|
|
55
|
+
if len(text) <= limit:
|
|
56
|
+
return text
|
|
57
|
+
return text[:limit].rstrip() + f"… [{len(text) - limit} more chars, use --json for all]"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
#: A well-formed closing tag — the trigger for markup collapsing. Prose that
|
|
61
|
+
#: merely contains ``<`` ("x < 5") never matches, so it rides through verbatim.
|
|
62
|
+
_CLOSING_TAG = re.compile(r"</[a-zA-Z][^>]*>")
|
|
63
|
+
_CELL_BOUNDARY = re.compile(r"</(?:td|th|tr)\s*>", re.IGNORECASE)
|
|
64
|
+
#: Tag-like spans only (a letter or ``/`` after ``<``), matching the other two
|
|
65
|
+
#: patterns — an in-cell inequality pair ("a < b and c > d") is content, not
|
|
66
|
+
#: markup, and must survive even after flattening has triggered.
|
|
67
|
+
_TAG = re.compile(r"</?[a-zA-Z][^>]*>")
|
|
68
|
+
#: An unterminated tag opened at the very end — the server's snippet window cut
|
|
69
|
+
#: mid-tag. Requires a letter after ``<``, so a bare trailing ``<`` or ``a < b``
|
|
70
|
+
#: math survives.
|
|
71
|
+
_TRUNCATED_TAG_TAIL = re.compile(r"</?[a-zA-Z][^>]*$")
|
|
72
|
+
_SEPARATOR_RUN = re.compile(r"\|(?:\s*\|)+")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _flatten_markup(text: str) -> str:
|
|
76
|
+
"""Collapse HTML table soup to readable cell text; markup-free text is untouched.
|
|
77
|
+
|
|
78
|
+
PDF-table snippets arrive as raw ``</td></tr><tr><td>…`` markup — roughly
|
|
79
|
+
half the snippet window was tags. Cell/row boundaries become ``" | "``,
|
|
80
|
+
every other tag becomes whitespace, entities are unescaped, and whitespace
|
|
81
|
+
is collapsed — so the clipped window carries content, not angle brackets.
|
|
82
|
+
Text mode only; ``--json`` stays raw.
|
|
83
|
+
"""
|
|
84
|
+
if not _CLOSING_TAG.search(text):
|
|
85
|
+
return text
|
|
86
|
+
text = _TRUNCATED_TAG_TAIL.sub("", text)
|
|
87
|
+
text = _CELL_BOUNDARY.sub(" | ", text)
|
|
88
|
+
text = _TAG.sub(" ", text)
|
|
89
|
+
text = html.unescape(text)
|
|
90
|
+
text = re.sub(r"\s+", " ", text)
|
|
91
|
+
text = _SEPARATOR_RUN.sub("|", text)
|
|
92
|
+
return text.strip(" |")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _coverage_lines(coverage: Coverage) -> list[str]:
|
|
96
|
+
if coverage.restricted_candidates <= 0:
|
|
97
|
+
return []
|
|
98
|
+
labels = f" (labels required: {', '.join(coverage.labels_required)})" if coverage.labels_required else ""
|
|
99
|
+
lines = [f"note: {coverage.restricted_candidates} candidate(s) withheld by access controls{labels}"]
|
|
100
|
+
if coverage.request_access_hint:
|
|
101
|
+
lines.append(f"access: {coverage.request_access_hint}")
|
|
102
|
+
return lines
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _categories(pairs: list, limit: int | None = None) -> str:
|
|
106
|
+
shown = pairs if limit is None else pairs[:limit]
|
|
107
|
+
rendered = ", ".join(f"{c.name} x{c.count}" for c in shown)
|
|
108
|
+
if limit is not None and len(pairs) > limit:
|
|
109
|
+
rendered += f", +{len(pairs) - limit} more"
|
|
110
|
+
return rendered
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _locator(locator) -> str:
|
|
114
|
+
"""One compact position per locator kind — the part of a citation an agent follows."""
|
|
115
|
+
match locator.kind:
|
|
116
|
+
case "spreadsheet_range":
|
|
117
|
+
return f"{locator.sheet}!{locator.a1_range}"
|
|
118
|
+
case "text_range":
|
|
119
|
+
page = f"page {locator.page}, " if locator.page is not None else ""
|
|
120
|
+
return f"{page}chars {locator.char_start}-{locator.char_end}"
|
|
121
|
+
case "visual_region":
|
|
122
|
+
return f"page {locator.page}, bbox {locator.bbox}"
|
|
123
|
+
case "jsonl_record":
|
|
124
|
+
column = f" col {locator.column}" if locator.column else ""
|
|
125
|
+
return f"row {locator.row_offset}{column}"
|
|
126
|
+
case "audio_range":
|
|
127
|
+
return f"{locator.start_ms / 1000:.0f}s-{locator.end_ms / 1000:.0f}s"
|
|
128
|
+
return locator.kind
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _evidence_lines(hits: list[Evidence]) -> list[str]:
|
|
132
|
+
lines: list[str] = []
|
|
133
|
+
for index, hit in enumerate(hits, start=1):
|
|
134
|
+
where = f" [{hit.component}]" if hit.component else ""
|
|
135
|
+
page = f", page {hit.page}" if hit.page is not None else ""
|
|
136
|
+
at = f" @ {_locator(hit.locator)}" if hit.locator is not None else ""
|
|
137
|
+
lines.append(f"{index}. {hit.path}{where} (score {hit.relevance_score:.3f}{page}){at}")
|
|
138
|
+
if hit.snippet:
|
|
139
|
+
lines.append(f" {_clip(_flatten_markup(hit.snippet))}")
|
|
140
|
+
if hit.why:
|
|
141
|
+
lines.append(f" why: {_clip(hit.why, 300)}")
|
|
142
|
+
return lines
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def folder_metadata(response: FolderMetadataResponse) -> str:
|
|
146
|
+
lines = [f"directory: {response.directory}", f"files in scope: {response.overall_files}"]
|
|
147
|
+
if response.categories:
|
|
148
|
+
lines.append(f"categories: {_categories(response.categories)}")
|
|
149
|
+
if response.ingestion_summary:
|
|
150
|
+
summary = ", ".join(f"{status} x{count}" for status, count in sorted(response.ingestion_summary.items()))
|
|
151
|
+
lines.append(f"ingestion: {summary}")
|
|
152
|
+
if response.subdirectories:
|
|
153
|
+
lines.append(f"subdirectories ({len(response.subdirectories)}):")
|
|
154
|
+
for sub in response.subdirectories:
|
|
155
|
+
failed = f" ({sub.failed_files} failed)" if sub.failed_files else ""
|
|
156
|
+
cats = f" categories: {_categories(sub.top_categories, limit=3)}" if sub.top_categories else ""
|
|
157
|
+
lines.append(f" {sub.name}/ {sub.overall_files} files{failed}{cats}")
|
|
158
|
+
if response.file_entries:
|
|
159
|
+
cut = " (truncated — resume with --start-after, or narrow)" if response.files_truncated else ""
|
|
160
|
+
lines.append(f"files listed ({len(response.file_entries)}){cut}:")
|
|
161
|
+
for entry in response.file_entries:
|
|
162
|
+
extras = " ".join(part for part in (entry.category, entry.kind) if part)
|
|
163
|
+
extras = f" [{extras}]" if extras else ""
|
|
164
|
+
uploaded = f" {entry.uploaded_at:%Y-%m-%d}" if entry.uploaded_at is not None else ""
|
|
165
|
+
lines.append(f" {entry.ingestion_status} {_size(entry.size_bytes):>9}{uploaded} {entry.path}{extras}")
|
|
166
|
+
elif response.files_truncated:
|
|
167
|
+
lines.append("files listed (0) — page cut before any entry; raise --page-size or narrow")
|
|
168
|
+
if response.hint:
|
|
169
|
+
lines.append(f"hint: {response.hint}")
|
|
170
|
+
if response.tables is not None:
|
|
171
|
+
cut = " (truncated — narrow the directory)" if response.tables_truncated else ""
|
|
172
|
+
lines.append(f"tables ({len(response.tables)}){cut}:")
|
|
173
|
+
lines.extend(_census_table_lines(response.tables))
|
|
174
|
+
if response.shared_columns:
|
|
175
|
+
sql_names = {(t.path, t.component): t.table_name for t in (response.tables or []) if t.table_name}
|
|
176
|
+
cut = " (list truncated)" if response.shared_columns_truncated else ""
|
|
177
|
+
lines.append(f"shared columns — UNVERIFIED join candidates, confirm with one run-sql join{cut}:")
|
|
178
|
+
for column in response.shared_columns:
|
|
179
|
+
sides = "; ".join(
|
|
180
|
+
_join_side(occ, sql_names.get((occ.path, occ.component))) for occ in column.appears_in[:_JOIN_EXAMPLES_MAX]
|
|
181
|
+
)
|
|
182
|
+
hidden = column.table_count - min(len(column.appears_in), _JOIN_EXAMPLES_MAX)
|
|
183
|
+
more = f"; +{hidden} more" if hidden > 0 else ""
|
|
184
|
+
lines.append(f" {column.name} in {column.table_count} tables: {sides}{more}")
|
|
185
|
+
lines.extend(_coverage_lines(response.coverage))
|
|
186
|
+
return "\n".join(lines)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
#: Join-candidate examples shown per shared column in text mode.
|
|
190
|
+
_JOIN_EXAMPLES_MAX = 3
|
|
191
|
+
|
|
192
|
+
#: The partition vocabulary, mirroring je_testing's ``partitions.py``: a
|
|
193
|
+
#: plausible four-digit year (1900–2099) or a ``p``/``P`` part marker (``p1``,
|
|
194
|
+
#: ``P12``), alone between token boundaries. An arbitrary digit run
|
|
195
|
+
#: (``invoice_1001``) is NOT a partition value — masking every digit run
|
|
196
|
+
#: grouped unrelated ids into an invented population.
|
|
197
|
+
_PARTITION_TOKEN = re.compile(r"(?<![A-Za-z0-9])(?:(?:19|20)\d{2}|[pP]\d+)(?![A-Za-z0-9])")
|
|
198
|
+
_COLLISION_SUFFIX = re.compile(r"_\d+$")
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _join_side(occ, sql_name: str | None) -> str:
|
|
202
|
+
"""One join-candidate occurrence — named by its sql table when the census knows it."""
|
|
203
|
+
name = sql_name or f"{occ.path}::{occ.component}"
|
|
204
|
+
if occ.distinct_count is not None and occ.row_count is not None:
|
|
205
|
+
return f"{name} (distinct≈{occ.distinct_count}/{occ.row_count} rows)"
|
|
206
|
+
return name
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _norm_name(text: str) -> str:
|
|
210
|
+
return re.sub(r"[^a-z0-9]+", "_", text.lower()).strip("_")
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _component_redundant(table) -> bool:
|
|
214
|
+
"""True when the component restates the sql name (modulo case/sanitizing/collision suffix) or the file stem."""
|
|
215
|
+
if not table.table_name:
|
|
216
|
+
return False
|
|
217
|
+
component = _norm_name(table.component)
|
|
218
|
+
name = _norm_name(table.table_name)
|
|
219
|
+
stem = _norm_name(table.path.rsplit("/", 1)[-1].rsplit(".", 1)[0])
|
|
220
|
+
return component in (name, _COLLISION_SUFFIX.sub("", name), stem)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _extent(table) -> str:
|
|
224
|
+
extent = f"{table.row_count} rows" if table.row_count is not None else "rows unknown"
|
|
225
|
+
if table.column_count is not None:
|
|
226
|
+
extent += f" x {table.column_count} cols"
|
|
227
|
+
return extent
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _family_key(table) -> tuple[str, str, str] | None:
|
|
231
|
+
"""The partition-family a table belongs to, or None for a standalone table.
|
|
232
|
+
|
|
233
|
+
Siblings live in the SAME directory (real corpora nest files under numbered
|
|
234
|
+
directories, so directory digits never count as partition values) and
|
|
235
|
+
differ only in partition tokens — years or pN part markers, the
|
|
236
|
+
:data:`_PARTITION_TOKEN` vocabulary — with the SAME tokens in file stem and
|
|
237
|
+
sql name. The family line's shared directory + basename pattern plus a
|
|
238
|
+
member's values therefore reconstruct its exact path. A name whose only
|
|
239
|
+
digits are ids or a collision suffix carries no partition token and stays a
|
|
240
|
+
standalone row. A component that is not redundant carries information a
|
|
241
|
+
family row would hide, so it opts out.
|
|
242
|
+
"""
|
|
243
|
+
if not table.table_name or not _component_redundant(table):
|
|
244
|
+
return None
|
|
245
|
+
values = _PARTITION_TOKEN.findall(table.table_name)
|
|
246
|
+
directory, _, basename = table.path.rpartition("/")
|
|
247
|
+
stem, dot, extension = basename.rpartition(".")
|
|
248
|
+
if not dot:
|
|
249
|
+
stem = basename
|
|
250
|
+
if not values or [v.casefold() for v in _PARTITION_TOKEN.findall(stem)] != [v.casefold() for v in values]:
|
|
251
|
+
return None
|
|
252
|
+
basename_pattern = _PARTITION_TOKEN.sub("*", stem) + (f".{extension}" if dot else "")
|
|
253
|
+
return (directory, basename_pattern, _PARTITION_TOKEN.sub("*", table.table_name))
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _census_table_lines(tables: list) -> list[str]:
|
|
257
|
+
"""One line per standalone table; partition families grouped under one header.
|
|
258
|
+
|
|
259
|
+
Grouping SURFACES the partition structure (the family line names the stem
|
|
260
|
+
and every partition value) and never hides a member's sql name or
|
|
261
|
+
rows×cols — those stay printed per member. The redundant ``[component]``
|
|
262
|
+
middle name is dropped whenever the sql name or file stem already says it.
|
|
263
|
+
"""
|
|
264
|
+
members_by_family: dict[tuple[str, str, str], list] = {}
|
|
265
|
+
for table in tables:
|
|
266
|
+
key = _family_key(table)
|
|
267
|
+
if key is not None:
|
|
268
|
+
members_by_family.setdefault(key, []).append(table)
|
|
269
|
+
families = {key: members for key, members in members_by_family.items() if len(members) > 1}
|
|
270
|
+
lines: list[str] = []
|
|
271
|
+
rendered: set[tuple[str, str, str]] = set()
|
|
272
|
+
for table in tables:
|
|
273
|
+
key = _family_key(table)
|
|
274
|
+
if key in families:
|
|
275
|
+
if key in rendered:
|
|
276
|
+
continue
|
|
277
|
+
rendered.add(key)
|
|
278
|
+
members = families[key]
|
|
279
|
+
values = ["/".join(_PARTITION_TOKEN.findall(member.table_name)) for member in members]
|
|
280
|
+
directory, basename_pattern, name_pattern = key
|
|
281
|
+
path_pattern = f"{directory}/{basename_pattern}" if directory else basename_pattern
|
|
282
|
+
lines.append(f" family {name_pattern} (partitions {', '.join(values)}) — {path_pattern}:")
|
|
283
|
+
lines.extend(f" {value}: {_extent(member)} sql: {member.table_name}" for value, member in zip(values, members))
|
|
284
|
+
continue
|
|
285
|
+
component = "" if _component_redundant(table) else f" [{table.component}]"
|
|
286
|
+
sql_name = f" sql: {table.table_name}" if table.table_name else ""
|
|
287
|
+
lines.append(f" {_extent(table)} {table.path}{component}{sql_name}")
|
|
288
|
+
return lines
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def file_metadata(response: FileMetadataToolResponse) -> str:
|
|
292
|
+
facts = [f"status: {response.ingestion_status}", f"size: {_size(response.size_bytes)}"]
|
|
293
|
+
if response.kind:
|
|
294
|
+
facts.insert(0, f"kind: {response.kind}")
|
|
295
|
+
if response.page_count is not None:
|
|
296
|
+
facts.append(f"pages: {response.page_count}")
|
|
297
|
+
if response.tab_count is not None:
|
|
298
|
+
facts.append(f"tabs: {response.tab_count}")
|
|
299
|
+
if response.total_chars is not None:
|
|
300
|
+
facts.append(f"chars: {response.total_chars}")
|
|
301
|
+
elif response.total_approx_bytes is not None:
|
|
302
|
+
facts.append(f"~{_size(response.total_approx_bytes)} text")
|
|
303
|
+
lines = [f"path: {response.path}", " | ".join(facts)]
|
|
304
|
+
if response.ingestion_error:
|
|
305
|
+
lines.append(f"ingestion error: {response.ingestion_error}")
|
|
306
|
+
if response.summary:
|
|
307
|
+
lines.append(f"summary: {response.summary}")
|
|
308
|
+
if response.components:
|
|
309
|
+
lines.append(f"components ({len(response.components)}):")
|
|
310
|
+
for comp in response.components:
|
|
311
|
+
if comp.kind == "media":
|
|
312
|
+
seconds = comp.duration_ms / 1000
|
|
313
|
+
frames = f", {comp.frame_count} sampled frame(s)" if comp.frame_count else ""
|
|
314
|
+
lines.append(
|
|
315
|
+
f" [media/{comp.modality}] {seconds:.0f}s, {comp.speaker_count} speaker(s), "
|
|
316
|
+
f"{comp.speech_segment_count} speech segment(s){frames} — ask with ask-file"
|
|
317
|
+
)
|
|
318
|
+
if comp.summary:
|
|
319
|
+
lines.append(f" {comp.summary}")
|
|
320
|
+
continue
|
|
321
|
+
extent: list[str] = []
|
|
322
|
+
if comp.page_start is not None and comp.page_end is not None:
|
|
323
|
+
extent.append(f"pages {comp.page_start}-{comp.page_end}")
|
|
324
|
+
if comp.row_count is not None:
|
|
325
|
+
cols = f" x {comp.column_count} cols" if comp.column_count is not None else ""
|
|
326
|
+
extent.append(f"{comp.row_count} rows{cols}")
|
|
327
|
+
if comp.char_count is not None:
|
|
328
|
+
extent.append(f"{comp.char_count} chars")
|
|
329
|
+
elif comp.approx_bytes is not None:
|
|
330
|
+
extent.append(f"~{_size(comp.approx_bytes)}")
|
|
331
|
+
abilities = [name for name, flag in (("readable", comp.readable), ("queryable", comp.queryable)) if flag]
|
|
332
|
+
if comp.table_name:
|
|
333
|
+
abilities.append(f"sql table: {comp.table_name}")
|
|
334
|
+
tail = ", ".join(extent + (abilities or ["no content tools apply"]))
|
|
335
|
+
name = comp.name or "(unnamed)"
|
|
336
|
+
lines.append(f" [{comp.kind}] {name} — {tail}")
|
|
337
|
+
if comp.summary:
|
|
338
|
+
lines.append(f" {comp.summary}")
|
|
339
|
+
if comp.columns:
|
|
340
|
+
lines.extend(_column_lines(comp.columns, bool(comp.columns_truncated)))
|
|
341
|
+
if comp.error:
|
|
342
|
+
lines.append(f" error: {comp.error}")
|
|
343
|
+
return "\n".join(lines)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
#: Census values shown per column in text mode — the full list rides --json.
|
|
347
|
+
_SHOWN_VALUES = 8
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _column_lines(columns: list, truncated: bool) -> list[str]:
|
|
351
|
+
"""Render a tab's schema: one dense line per column when stats are present.
|
|
352
|
+
|
|
353
|
+
A schema without stats (a sidecar from before profiling measured them)
|
|
354
|
+
stays a single names+types line, as before.
|
|
355
|
+
"""
|
|
356
|
+
if not any(col.null_count is not None or col.top_values or col.sample_values for col in columns):
|
|
357
|
+
rendered = ", ".join(f"{col.name} ({col.data_type})" for col in columns)
|
|
358
|
+
cut = " …schema truncated" if truncated else ""
|
|
359
|
+
return [f" columns ({len(columns)}): {rendered}{cut}"]
|
|
360
|
+
lines = [f" columns ({len(columns)}){' …schema truncated' if truncated else ''}:"]
|
|
361
|
+
# Only the server's own proof (all_null, measured against the profile's exact row
|
|
362
|
+
# count) groups a column as empty; comparing counts from two sources here could
|
|
363
|
+
# hide a populated column. getattr: an older server's schema has no such field.
|
|
364
|
+
empty = [col for col in columns if getattr(col, "all_null", None)]
|
|
365
|
+
if empty:
|
|
366
|
+
lines.append(" all-null columns (no values): " + ", ".join(f"{col.name} ({col.data_type})" for col in empty))
|
|
367
|
+
for col in columns:
|
|
368
|
+
if col in empty:
|
|
369
|
+
continue
|
|
370
|
+
facts: list[str] = []
|
|
371
|
+
if col.distinct_count is not None:
|
|
372
|
+
facts.append(f"distinct≈{col.distinct_count}")
|
|
373
|
+
if col.null_count:
|
|
374
|
+
facts.append(f"nulls {col.null_count}")
|
|
375
|
+
if col.sum_value is not None:
|
|
376
|
+
facts.append(f"sum {col.sum_value}")
|
|
377
|
+
if col.min_value is not None and col.max_value is not None:
|
|
378
|
+
facts.append(f"range {col.min_value} .. {col.max_value}")
|
|
379
|
+
line = f" {col.name} {col.data_type}" + (f" {', '.join(facts)}" if facts else "")
|
|
380
|
+
values = col.top_values or []
|
|
381
|
+
if values:
|
|
382
|
+
shown = ", ".join(f"{entry.value} x{entry.count}" for entry in values[:_SHOWN_VALUES])
|
|
383
|
+
more = len(values) - _SHOWN_VALUES
|
|
384
|
+
tail = f", +{more} more" if more > 0 else ""
|
|
385
|
+
coverage = (
|
|
386
|
+
f" (top {len(values)} cover {col.top_values_coverage:.0%})"
|
|
387
|
+
if col.top_values_coverage is not None and col.top_values_coverage < 1
|
|
388
|
+
else ""
|
|
389
|
+
)
|
|
390
|
+
line += f" values: {shown}{tail}{coverage}"
|
|
391
|
+
elif col.sample_values:
|
|
392
|
+
shown = ", ".join(col.sample_values[:_SHOWN_VALUES])
|
|
393
|
+
more = len(col.sample_values) - _SHOWN_VALUES
|
|
394
|
+
line += f" values: {shown}" + (f", +{more} more" if more > 0 else "")
|
|
395
|
+
lines.append(line)
|
|
396
|
+
return lines
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def read_file_footer(response: ReadFileResponse) -> str:
|
|
400
|
+
parts = [f"read {response.path}"]
|
|
401
|
+
if response.pages_returned:
|
|
402
|
+
parts.append(f"pages {_page_ranges(response.pages_returned)}")
|
|
403
|
+
if response.components_returned:
|
|
404
|
+
parts.append(f"components: {', '.join(response.components_returned)}")
|
|
405
|
+
if response.components_omitted:
|
|
406
|
+
parts.append(f"omitted: {', '.join(response.components_omitted)}")
|
|
407
|
+
if response.truncated:
|
|
408
|
+
parts.append("TRUNCATED (raise --max-chars, or scope with --component/--pages)")
|
|
409
|
+
return " | ".join(parts)
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _page_ranges(pages: list[int]) -> str:
|
|
413
|
+
ranges: list[str] = []
|
|
414
|
+
start = prev = pages[0]
|
|
415
|
+
for page in pages[1:]:
|
|
416
|
+
if page == prev + 1:
|
|
417
|
+
prev = page
|
|
418
|
+
continue
|
|
419
|
+
ranges.append(f"{start}-{prev}" if start != prev else str(start))
|
|
420
|
+
start = prev = page
|
|
421
|
+
ranges.append(f"{start}-{prev}" if start != prev else str(start))
|
|
422
|
+
return ",".join(ranges)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def qa_file(response: QaFileResponse) -> str:
|
|
426
|
+
lines = [response.answer.strip(), "", f"engine: {response.engine}"]
|
|
427
|
+
if response.unavailable:
|
|
428
|
+
lines.append("unavailable:")
|
|
429
|
+
lines.extend(f" {item.path}: {item.error}" for item in response.unavailable)
|
|
430
|
+
if response.citations:
|
|
431
|
+
lines.append(f"citations ({len(response.citations)}):")
|
|
432
|
+
lines.extend(f" {line}" for line in _evidence_lines(response.citations))
|
|
433
|
+
return "\n".join(lines)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def run_sql(response: RunSqlResponse) -> str:
|
|
437
|
+
lines: list[str] = []
|
|
438
|
+
result = response.result
|
|
439
|
+
cut = ""
|
|
440
|
+
if result.preview_truncated:
|
|
441
|
+
# The download hint must not outrun the response: when retention
|
|
442
|
+
# failed there is no citation and no link to follow.
|
|
443
|
+
cut = (
|
|
444
|
+
", preview truncated (full result via the download link or LIMIT/OFFSET)"
|
|
445
|
+
if response.result_citation is not None
|
|
446
|
+
else ", preview truncated (page it with LIMIT/OFFSET)"
|
|
447
|
+
)
|
|
448
|
+
dup = " (duplicate — served from the stored result)" if response.duplicate else ""
|
|
449
|
+
lines.append(f"result table {result.handle} — {result.row_count} rows{cut}{dup}")
|
|
450
|
+
if result.columns:
|
|
451
|
+
names = [col.name for col in result.columns]
|
|
452
|
+
lines.append(" " + " | ".join(f"{col.name} ({col.data_type})" for col in result.columns))
|
|
453
|
+
if len(set(names)) < len(names):
|
|
454
|
+
# Object rows keep one value per key, so a duplicated name loses a column
|
|
455
|
+
# with no visible trace against the header — say so instead.
|
|
456
|
+
lines.append(
|
|
457
|
+
" note: duplicate column names in the result — object rows keep only one value per name; alias duplicates (AS) to see both."
|
|
458
|
+
)
|
|
459
|
+
for row in result.rows:
|
|
460
|
+
# default=str keeps an unvalidated value from a future producer readable
|
|
461
|
+
# instead of crashing the render.
|
|
462
|
+
lines.append(" " + json.dumps({name: row.get(name, "") for name in names}, ensure_ascii=False, default=str))
|
|
463
|
+
lines.append(f"note: '{result.handle}' is a registered table — join or page it in your next run-sql call.")
|
|
464
|
+
citation = response.result_citation
|
|
465
|
+
# source_paths rides the response itself; older servers only set it on the
|
|
466
|
+
# citation, and agent-attributed statements have no citation at all.
|
|
467
|
+
sources = response.source_paths or (citation.source_paths if citation is not None else [])
|
|
468
|
+
if sources:
|
|
469
|
+
lines.append(f"sources: {', '.join(sources)}")
|
|
470
|
+
if citation is not None:
|
|
471
|
+
lines.append(f"download (parquet): {citation.result_file.download_url}")
|
|
472
|
+
if response.result_materialization_error:
|
|
473
|
+
lines.append(f"note: {response.result_materialization_error}")
|
|
474
|
+
return "\n".join(lines)
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
def search(response: SearchResponse) -> str:
|
|
478
|
+
if not response.hits:
|
|
479
|
+
lines = ["no hits"]
|
|
480
|
+
else:
|
|
481
|
+
lines = _evidence_lines(response.hits)
|
|
482
|
+
tail = f"candidates seen: {response.candidates_seen}"
|
|
483
|
+
if response.truncated:
|
|
484
|
+
tail += " (truncated upstream of ranking — a matching file may be missing; narrow with --path-prefix)"
|
|
485
|
+
lines.append(tail)
|
|
486
|
+
lines.extend(_coverage_lines(response.coverage))
|
|
487
|
+
return "\n".join(lines)
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def parse_job(job: Job) -> str:
|
|
491
|
+
result = job.result
|
|
492
|
+
if not isinstance(result, ParseResult):
|
|
493
|
+
return job_line(job)
|
|
494
|
+
print(
|
|
495
|
+
f"{result.file_name} | {result.document.page_count} pages | lane {result.lane} | "
|
|
496
|
+
f"ocr_applied={result.ocr_applied} | {result.units} units",
|
|
497
|
+
file=sys.stderr,
|
|
498
|
+
)
|
|
499
|
+
document = result.document
|
|
500
|
+
if document.markdown or document.text:
|
|
501
|
+
return document.markdown or document.text or ""
|
|
502
|
+
# `--format blocks` asks for neither markdown nor text. Returning "" here
|
|
503
|
+
# threw away the whole result — a 0-byte --save after a billed job.
|
|
504
|
+
if document.blocks is not None:
|
|
505
|
+
return json.dumps([block.model_dump(mode="json", exclude_none=True) for block in document.blocks], indent=2)
|
|
506
|
+
return ""
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def split_job(job: Job) -> str:
|
|
510
|
+
result = job.result
|
|
511
|
+
if not isinstance(result, SplitResult):
|
|
512
|
+
return job_line(job)
|
|
513
|
+
lines = [f"{result.file_name} | {result.page_count} pages | {len(result.segments)} segments | lane {result.lane}"]
|
|
514
|
+
for segment in result.segments:
|
|
515
|
+
label = segment.split_class.label if segment.split_class is not None else segment.status
|
|
516
|
+
pages = f"pages {segment.start_page}-{segment.end_page}" if segment.start_page is not None else "pages ?"
|
|
517
|
+
sheet = f" sheet {segment.sheet_name}" if segment.sheet_name else ""
|
|
518
|
+
lines.append(f" {segment.sequence}. {label} {pages}{sheet}")
|
|
519
|
+
if segment.content:
|
|
520
|
+
lines.append(f" {segment.content}")
|
|
521
|
+
return "\n".join(lines)
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def classify_job(job: Job) -> str:
|
|
525
|
+
result = job.result
|
|
526
|
+
if not isinstance(result, ClassifyResult):
|
|
527
|
+
return job_line(job)
|
|
528
|
+
lines = [f"units {len(result.units)} | unknown {result.unknown_units} | pages_read {result.pages_read_fraction:.0%}"]
|
|
529
|
+
for unit in result.units:
|
|
530
|
+
pages = f" pages {unit.page_range.start}-{unit.page_range.end}" if unit.page_range is not None else ""
|
|
531
|
+
unknown = " unknown" if unit.unknown else ""
|
|
532
|
+
lines.append(f" {unit.granularity}{pages}{unknown}")
|
|
533
|
+
for label in unit.labels:
|
|
534
|
+
conf = f" {label.confidence:.2f}" if label.confidence is not None else ""
|
|
535
|
+
reason = f" — {label.reason}" if label.reason else ""
|
|
536
|
+
lines.append(f" {label.rank}. {label.label} ({label.class_id}){conf}{reason}")
|
|
537
|
+
return "\n".join(lines)
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def extract_job(job: Job) -> str:
|
|
541
|
+
result = job.result
|
|
542
|
+
if not isinstance(result, ExtractResult):
|
|
543
|
+
return job_line(job)
|
|
544
|
+
lines = [json.dumps(result.data, indent=2, ensure_ascii=False, default=str)]
|
|
545
|
+
if result.fields:
|
|
546
|
+
lines.append("fields:")
|
|
547
|
+
for field in result.fields:
|
|
548
|
+
conf = f" {field.confidence:.2f}" if field.confidence is not None else ""
|
|
549
|
+
cites = f" citations {len(field.citations)}" if field.citations else ""
|
|
550
|
+
lines.append(f" {field.path}: {field.status}{conf}{cites}")
|
|
551
|
+
if result.warnings:
|
|
552
|
+
lines.append("warnings:")
|
|
553
|
+
lines.extend(f" {warning}" for warning in result.warnings)
|
|
554
|
+
return "\n".join(lines)
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
def ground_job(job: Job) -> str:
|
|
558
|
+
result = job.result
|
|
559
|
+
if not isinstance(result, GroundResult):
|
|
560
|
+
return job_line(job)
|
|
561
|
+
lines = [f"targets {len(result.targets)} | {result.units} units"]
|
|
562
|
+
for target in result.targets:
|
|
563
|
+
lines.append(f" {target.id}: {target.status}")
|
|
564
|
+
if target.error:
|
|
565
|
+
lines.append(f" error: {target.error}")
|
|
566
|
+
for match in target.matches:
|
|
567
|
+
conf = f" {match.confidence:.2f}" if match.confidence is not None else ""
|
|
568
|
+
lines.append(f" {match.rank}. {match.match_method}{conf} {match.matched_text}")
|
|
569
|
+
return "\n".join(lines)
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def schema_validation(result: ExtractSchemaValidationResponse) -> str:
|
|
573
|
+
if result.valid:
|
|
574
|
+
return "schema valid"
|
|
575
|
+
lines = [f"schema invalid ({len(result.errors)} error(s))"]
|
|
576
|
+
lines.extend(f" {error.path}: {error.message}" for error in result.errors)
|
|
577
|
+
return "\n".join(lines)
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def workspace(item: Workspace) -> str:
|
|
581
|
+
purpose = f" purpose={item.purpose}" if item.purpose else ""
|
|
582
|
+
return f"{item.workspace_id} {item.name} {item.domain_slug}@{item.domain_version} {item.status}{purpose}"
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def workspace_page(page) -> str:
|
|
586
|
+
lines = [f"workspaces ({page.total_count}):"]
|
|
587
|
+
lines.extend(f" {workspace(item)}" for item in page.items)
|
|
588
|
+
if page.next_cursor:
|
|
589
|
+
lines.append(f"next: {page.next_cursor}")
|
|
590
|
+
return "\n".join(lines)
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
def workspace_stats(stats: WorkspaceStats) -> str:
|
|
594
|
+
statuses = ", ".join(
|
|
595
|
+
f"{status} x{count}" for status, count in sorted(stats.files_by_ingestion_status.items(), key=lambda pair: str(pair[0]))
|
|
596
|
+
)
|
|
597
|
+
lines = [f"files: {stats.files_total}"]
|
|
598
|
+
if statuses:
|
|
599
|
+
lines.append(f"ingestion: {statuses}")
|
|
600
|
+
lines.extend(
|
|
601
|
+
(
|
|
602
|
+
f"bytes: {stats.bytes_total}",
|
|
603
|
+
f"pages ingested: {stats.pages_ingested}",
|
|
604
|
+
f"index documents: {stats.index_documents}",
|
|
605
|
+
f"kg: {stats.kg_nodes} nodes / {stats.kg_edges} edges",
|
|
606
|
+
)
|
|
607
|
+
)
|
|
608
|
+
return "\n".join(lines)
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def file_line(item: File) -> str:
|
|
612
|
+
return f"{item.ingestion_status} {_size(item.size_bytes):>9} {item.path} {item.file_id}"
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def file_page(page) -> str:
|
|
616
|
+
lines = [f"files ({page.total_count}):"]
|
|
617
|
+
lines.extend(f" {file_line(item)}" for item in page.items)
|
|
618
|
+
if page.next_cursor:
|
|
619
|
+
lines.append(f"next: {page.next_cursor}")
|
|
620
|
+
return "\n".join(lines)
|
|
621
|
+
|
|
622
|
+
|
|
623
|
+
def file_detail(item: FileDetail) -> str:
|
|
624
|
+
lines = [file_line(item), f"download: {item.download_url}"]
|
|
625
|
+
if item.category:
|
|
626
|
+
lines.append(f"category: {item.category}")
|
|
627
|
+
return "\n".join(lines)
|
|
628
|
+
|
|
629
|
+
|
|
630
|
+
def job_line(job: Job) -> str:
|
|
631
|
+
name = f" {job.name}" if job.name else ""
|
|
632
|
+
return f"{job.status} {job.kind} {job.job_id}{name}"
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def job_page(page) -> str:
|
|
636
|
+
lines = [f"jobs ({page.total_count}):"]
|
|
637
|
+
lines.extend(f" {job_line(job)}" for job in page.items)
|
|
638
|
+
if page.next_cursor:
|
|
639
|
+
lines.append(f"next: {page.next_cursor}")
|
|
640
|
+
return "\n".join(lines)
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def ingestion_job(job: Job) -> str:
|
|
644
|
+
result = job.result
|
|
645
|
+
if not isinstance(result, IngestionResult):
|
|
646
|
+
return job_line(job)
|
|
647
|
+
lines = [
|
|
648
|
+
f"ingested {result.files_ingested} | skipped {result.files_skipped} | failed {result.files_failed} | {result.units_total} units"
|
|
649
|
+
]
|
|
650
|
+
for outcome in result.outcomes:
|
|
651
|
+
extra = f" {outcome.error.message}" if outcome.error else ""
|
|
652
|
+
lines.append(f" {outcome.status} {outcome.path}{extra}")
|
|
653
|
+
return "\n".join(lines)
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
def search_job(job: Job) -> str:
|
|
657
|
+
result = job.result
|
|
658
|
+
if isinstance(result, DeepSearchV2Result):
|
|
659
|
+
return _search_result(result.answer, result.clarification, result.evidences, receipts=len(result.sql_receipts))
|
|
660
|
+
if isinstance(result, IntelligentSearchResult):
|
|
661
|
+
return _search_result(result.answer, result.clarification, result.evidences, receipts=0)
|
|
662
|
+
return job_line(job)
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
def _search_result(answer: str | None, clarification: str | None, evidences: list, *, receipts: int) -> str:
|
|
666
|
+
lines: list[str] = []
|
|
667
|
+
if clarification:
|
|
668
|
+
lines.append(f"clarification: {clarification}")
|
|
669
|
+
if answer:
|
|
670
|
+
lines.append(answer.strip())
|
|
671
|
+
lines.append(f"evidences ({len(evidences)}):")
|
|
672
|
+
for index, evidence in enumerate(evidences, start=1):
|
|
673
|
+
page = f", page {evidence.page}" if evidence.page is not None else ""
|
|
674
|
+
lines.append(f"{index}. {evidence.source_path}{page}")
|
|
675
|
+
if evidence.quote:
|
|
676
|
+
lines.append(f" {evidence.quote}")
|
|
677
|
+
if receipts:
|
|
678
|
+
lines.append(f"receipts: {receipts}")
|
|
679
|
+
return "\n".join(lines)
|