ndi-cli 0.8.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/PKG-INFO +15 -5
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/README.md +14 -4
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/pyproject.toml +1 -1
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/pyproject.toml.orig +1 -1
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/__init__.py +1 -1
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_jobs.py +26 -9
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_render.py +37 -3
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_sources.py +69 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_workspace.py +194 -9
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/LICENSE +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_cli.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_config.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_docops.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_entrypoint.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_local.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_login.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_output.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/_tools.py +0 -0
- {ndi_cli-0.8.0 → ndi_cli-0.9.0}/src/ndi_cli/commands.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ndi-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: NDI platform CLI: document operations, jobs, workspaces, and the agent workspace tools
|
|
5
5
|
Keywords: ndi,cli,document-intelligence,coding-agents
|
|
6
6
|
Author: Nace AI
|
|
@@ -50,14 +50,14 @@ config location. Workspace-scoped verbs also take `--workspace ID`.
|
|
|
50
50
|
| `ndi extract SOURCE -s SCHEMA` | Structured extract (`--validate` checks a schema with no job) |
|
|
51
51
|
| `ndi split SOURCE --class id:label` | Logical sections |
|
|
52
52
|
| `ndi classify SOURCE --class id:label` | Labels (refuses `jobid://`) |
|
|
53
|
-
| `ndi ground SOURCE --target id=TEXT` | Locate quoted text |
|
|
53
|
+
| `ndi ground SOURCE --target id=TEXT` | Locate quoted text (standalone quotes only; after search, poll `grounding_job_ids`) |
|
|
54
54
|
| `ndi job ID` / `ndi jobs` / `ndi cancel ID` | Inspect or cancel jobs |
|
|
55
55
|
| `ndi workspace create\|list\|get\|stats\|delete\|use` | Workspace lifecycle |
|
|
56
|
-
| `ndi files upload\|list\|get\|delete` | Workspace files (`--ingest`
|
|
56
|
+
| `ndi files upload\|list\|get\|delete` | Workspace files (a directory walks nested files; `--ingest` then queues ingestion) |
|
|
57
57
|
| `ndi ingest` | Queue ingestion |
|
|
58
|
-
| `ndi deep-search QUERY` / `ndi fact-search QUERY` | Agentic / single-shot search |
|
|
58
|
+
| `ndi deep-search QUERY` / `ndi fact-search QUERY` | Agentic / single-shot search; then `ndi job` on each printed `grounding_job_ids` entry |
|
|
59
59
|
| `ndi filtered-search QUERY` | Corpus filter/rank/count over the metadata catalog (`--cursor` pages an earlier run) |
|
|
60
|
-
| `ndi je-testing QUERY` | Journal-entry testing over a ledger package (`--path` to scope it, `--effort` for the
|
|
60
|
+
| `ndi je-testing QUERY` | Journal-entry testing over a ledger package (`--path` to scope it, `--reasoning-effort` for the thinking budget) |
|
|
61
61
|
| `ndi folder-metadata` / `file-metadata` / `read-file` / `ask-file` / `run-sql` / `hybrid-search` | Read-only workspace tools (the sandbox surface) |
|
|
62
62
|
|
|
63
63
|
## Sources
|
|
@@ -74,8 +74,15 @@ config location. Workspace-scoped verbs also take `--workspace ID`.
|
|
|
74
74
|
|
|
75
75
|
```bash
|
|
76
76
|
ndi parse a.pdf -o id | ndi extract - -s schema.json
|
|
77
|
+
ndi files upload ./corpus --ingest
|
|
77
78
|
```
|
|
78
79
|
|
|
80
|
+
`ndi files upload DIR` keeps relative paths under `--path` (default: the
|
|
81
|
+
directory name) and uploads in parallel (`-j`, default 8). Unsupported
|
|
82
|
+
files (and `.DS_Store`) are skipped; a warning on stderr lists them when
|
|
83
|
+
the run finishes. `--ingest` then queues ingestion in batches of 1000
|
|
84
|
+
file ids. A single file still requires `--path`.
|
|
85
|
+
|
|
79
86
|
## Output
|
|
80
87
|
|
|
81
88
|
Result content goes to **stdout**; status (`job <id> queued`, `saved …`) goes
|
|
@@ -94,6 +101,9 @@ URLs automatically.
|
|
|
94
101
|
API failures exit 1. Usage / config errors exit 2. Schema-validation
|
|
95
102
|
failures also show up to five field constraints, each limited to 500
|
|
96
103
|
characters; request input and unrelated error-detail fields are omitted.
|
|
104
|
+
Unsupported inputs print `error: 422 [unsupported_file_type] ...` to stderr and
|
|
105
|
+
exit 1. `ndi upload` prints no handle, document operations print no job ID, and
|
|
106
|
+
`ndi files upload --ingest` creates no workspace file or ingestion job.
|
|
97
107
|
|
|
98
108
|
See the [CLI guide](https://docs.ndi.nace.ai/guides/cli) for the full command
|
|
99
109
|
reference. `SKILL.md` is the agent-facing guide to the six workspace tools.
|
|
@@ -26,14 +26,14 @@ config location. Workspace-scoped verbs also take `--workspace ID`.
|
|
|
26
26
|
| `ndi extract SOURCE -s SCHEMA` | Structured extract (`--validate` checks a schema with no job) |
|
|
27
27
|
| `ndi split SOURCE --class id:label` | Logical sections |
|
|
28
28
|
| `ndi classify SOURCE --class id:label` | Labels (refuses `jobid://`) |
|
|
29
|
-
| `ndi ground SOURCE --target id=TEXT` | Locate quoted text |
|
|
29
|
+
| `ndi ground SOURCE --target id=TEXT` | Locate quoted text (standalone quotes only; after search, poll `grounding_job_ids`) |
|
|
30
30
|
| `ndi job ID` / `ndi jobs` / `ndi cancel ID` | Inspect or cancel jobs |
|
|
31
31
|
| `ndi workspace create\|list\|get\|stats\|delete\|use` | Workspace lifecycle |
|
|
32
|
-
| `ndi files upload\|list\|get\|delete` | Workspace files (`--ingest`
|
|
32
|
+
| `ndi files upload\|list\|get\|delete` | Workspace files (a directory walks nested files; `--ingest` then queues ingestion) |
|
|
33
33
|
| `ndi ingest` | Queue ingestion |
|
|
34
|
-
| `ndi deep-search QUERY` / `ndi fact-search QUERY` | Agentic / single-shot search |
|
|
34
|
+
| `ndi deep-search QUERY` / `ndi fact-search QUERY` | Agentic / single-shot search; then `ndi job` on each printed `grounding_job_ids` entry |
|
|
35
35
|
| `ndi filtered-search QUERY` | Corpus filter/rank/count over the metadata catalog (`--cursor` pages an earlier run) |
|
|
36
|
-
| `ndi je-testing QUERY` | Journal-entry testing over a ledger package (`--path` to scope it, `--effort` for the
|
|
36
|
+
| `ndi je-testing QUERY` | Journal-entry testing over a ledger package (`--path` to scope it, `--reasoning-effort` for the thinking budget) |
|
|
37
37
|
| `ndi folder-metadata` / `file-metadata` / `read-file` / `ask-file` / `run-sql` / `hybrid-search` | Read-only workspace tools (the sandbox surface) |
|
|
38
38
|
|
|
39
39
|
## Sources
|
|
@@ -50,8 +50,15 @@ config location. Workspace-scoped verbs also take `--workspace ID`.
|
|
|
50
50
|
|
|
51
51
|
```bash
|
|
52
52
|
ndi parse a.pdf -o id | ndi extract - -s schema.json
|
|
53
|
+
ndi files upload ./corpus --ingest
|
|
53
54
|
```
|
|
54
55
|
|
|
56
|
+
`ndi files upload DIR` keeps relative paths under `--path` (default: the
|
|
57
|
+
directory name) and uploads in parallel (`-j`, default 8). Unsupported
|
|
58
|
+
files (and `.DS_Store`) are skipped; a warning on stderr lists them when
|
|
59
|
+
the run finishes. `--ingest` then queues ingestion in batches of 1000
|
|
60
|
+
file ids. A single file still requires `--path`.
|
|
61
|
+
|
|
55
62
|
## Output
|
|
56
63
|
|
|
57
64
|
Result content goes to **stdout**; status (`job <id> queued`, `saved …`) goes
|
|
@@ -70,6 +77,9 @@ URLs automatically.
|
|
|
70
77
|
API failures exit 1. Usage / config errors exit 2. Schema-validation
|
|
71
78
|
failures also show up to five field constraints, each limited to 500
|
|
72
79
|
characters; request input and unrelated error-detail fields are omitted.
|
|
80
|
+
Unsupported inputs print `error: 422 [unsupported_file_type] ...` to stderr and
|
|
81
|
+
exit 1. `ndi upload` prints no handle, document operations print no job ID, and
|
|
82
|
+
`ndi files upload --ingest` creates no workspace file or ingestion job.
|
|
73
83
|
|
|
74
84
|
See the [CLI guide](https://docs.ndi.nace.ai/guides/cli) for the full command
|
|
75
85
|
reference. `SKILL.md` is the agent-facing guide to the six workspace tools.
|
|
@@ -36,18 +36,17 @@ def add_parsers(sub: argparse._SubParsersAction, common: argparse.ArgumentParser
|
|
|
36
36
|
cancel.set_defaults(run=_run_cancel, family="job", needs_workspace=False, needs_client=True)
|
|
37
37
|
|
|
38
38
|
|
|
39
|
-
def
|
|
39
|
+
def wait_job(
|
|
40
40
|
client: NdiClient,
|
|
41
41
|
job: Job,
|
|
42
42
|
args: argparse.Namespace,
|
|
43
43
|
*,
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
44
|
+
print_timeout_id: bool = True,
|
|
45
|
+
) -> tuple[Job | None, int]:
|
|
46
|
+
"""Wait for ``job``; status lines go to stderr. Does not render the result."""
|
|
47
47
|
print(f"job {job.job_id} {job.status}", file=sys.stderr)
|
|
48
48
|
if getattr(args, "async_submit", False):
|
|
49
|
-
|
|
50
|
-
return 0
|
|
49
|
+
return job, 0
|
|
51
50
|
try:
|
|
52
51
|
if job.is_terminal:
|
|
53
52
|
check_terminal(job, raise_on_failure=True)
|
|
@@ -56,12 +55,30 @@ def wait_and_report(
|
|
|
56
55
|
except JobFailedError as exc:
|
|
57
56
|
code = f" [{exc.job.error.code}]" if exc.job.error and exc.job.error.code else ""
|
|
58
57
|
print(f"error:{code} {exc}", file=sys.stderr)
|
|
59
|
-
return 1
|
|
58
|
+
return exc.job, 1
|
|
60
59
|
except JobTimeoutError as exc:
|
|
61
60
|
print(f"error: {exc}", file=sys.stderr)
|
|
62
|
-
|
|
63
|
-
|
|
61
|
+
if print_timeout_id:
|
|
62
|
+
print(exc.job.job_id)
|
|
63
|
+
return exc.job, 1
|
|
64
64
|
print(f"job {job.job_id} {job.status}", file=sys.stderr)
|
|
65
|
+
return job, 0
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def wait_and_report(
|
|
69
|
+
client: NdiClient,
|
|
70
|
+
job: Job,
|
|
71
|
+
args: argparse.Namespace,
|
|
72
|
+
*,
|
|
73
|
+
auto: Callable[[Job], str],
|
|
74
|
+
stem: str,
|
|
75
|
+
) -> int:
|
|
76
|
+
job, code = wait_job(client, job, args)
|
|
77
|
+
if code != 0 or job is None:
|
|
78
|
+
return code or 1
|
|
79
|
+
if getattr(args, "async_submit", False):
|
|
80
|
+
print(job.job_id)
|
|
81
|
+
return 0
|
|
65
82
|
return render_job(job, args, auto=auto, stem=stem)
|
|
66
83
|
|
|
67
84
|
|
|
@@ -10,6 +10,7 @@ from __future__ import annotations
|
|
|
10
10
|
|
|
11
11
|
import json
|
|
12
12
|
import sys
|
|
13
|
+
import uuid
|
|
13
14
|
|
|
14
15
|
import html
|
|
15
16
|
import re
|
|
@@ -658,9 +659,21 @@ def ingestion_job(job: Job) -> str:
|
|
|
658
659
|
def search_job(job: Job) -> str:
|
|
659
660
|
result = job.result
|
|
660
661
|
if isinstance(result, DeepSearchV2Result):
|
|
661
|
-
return _search_result(
|
|
662
|
+
return _search_result(
|
|
663
|
+
result.answer,
|
|
664
|
+
result.clarification,
|
|
665
|
+
result.evidences,
|
|
666
|
+
receipts=len(result.sql_receipts),
|
|
667
|
+
grounding_job_ids=result.grounding_job_ids,
|
|
668
|
+
)
|
|
662
669
|
if isinstance(result, IntelligentSearchResult):
|
|
663
|
-
return _search_result(
|
|
670
|
+
return _search_result(
|
|
671
|
+
result.answer,
|
|
672
|
+
result.clarification,
|
|
673
|
+
result.evidences,
|
|
674
|
+
receipts=0,
|
|
675
|
+
grounding_job_ids=result.grounding_job_ids,
|
|
676
|
+
)
|
|
664
677
|
return job_line(job)
|
|
665
678
|
|
|
666
679
|
|
|
@@ -704,6 +717,7 @@ def je_testing_job(job: Job) -> str:
|
|
|
704
717
|
lines.append(f"quality: {result.quality.status}")
|
|
705
718
|
if result.exhausted:
|
|
706
719
|
lines.append("exhausted: step budget ran out before the procedure finished")
|
|
720
|
+
lines.extend(_grounding_job_lines(result.grounding_job_ids))
|
|
707
721
|
return "\n".join(lines)
|
|
708
722
|
|
|
709
723
|
|
|
@@ -832,7 +846,26 @@ def filtered_search_job(job: Job) -> str:
|
|
|
832
846
|
return "\n".join(lines)
|
|
833
847
|
|
|
834
848
|
|
|
835
|
-
def
|
|
849
|
+
def _grounding_job_lines(ids: list[uuid.UUID] | None) -> list[str]:
|
|
850
|
+
"""Tell the reader which Ground jobs to poll after this search result."""
|
|
851
|
+
if ids is None:
|
|
852
|
+
return []
|
|
853
|
+
if not ids:
|
|
854
|
+
return ["grounding_job_ids: none planned"]
|
|
855
|
+
lines = ["grounding_job_ids (poll each with ndi job <id>):"]
|
|
856
|
+
for item in ids:
|
|
857
|
+
lines.append(f" {item}")
|
|
858
|
+
return lines
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
def _search_result(
|
|
862
|
+
answer: str | None,
|
|
863
|
+
clarification: str | None,
|
|
864
|
+
evidences: list,
|
|
865
|
+
*,
|
|
866
|
+
receipts: int,
|
|
867
|
+
grounding_job_ids: list[uuid.UUID] | None = None,
|
|
868
|
+
) -> str:
|
|
836
869
|
lines: list[str] = []
|
|
837
870
|
if clarification:
|
|
838
871
|
lines.append(f"clarification: {clarification}")
|
|
@@ -846,4 +879,5 @@ def _search_result(answer: str | None, clarification: str | None, evidences: lis
|
|
|
846
879
|
lines.append(f" {evidence.quote}")
|
|
847
880
|
if receipts:
|
|
848
881
|
lines.append(f"receipts: {receipts}")
|
|
882
|
+
lines.extend(_grounding_job_lines(grounding_job_ids))
|
|
849
883
|
return "\n".join(lines)
|
|
@@ -9,6 +9,7 @@ from __future__ import annotations
|
|
|
9
9
|
import re
|
|
10
10
|
import sys
|
|
11
11
|
import uuid
|
|
12
|
+
from dataclasses import dataclass
|
|
12
13
|
from pathlib import Path
|
|
13
14
|
|
|
14
15
|
from ndi_sdk.client import NdiClient
|
|
@@ -49,6 +50,38 @@ SUPPORTED_SUFFIXES = frozenset(
|
|
|
49
50
|
".mov",
|
|
50
51
|
}
|
|
51
52
|
)
|
|
53
|
+
# Workspace file upload admits more suffixes than document-ops parse: HTML,
|
|
54
|
+
# plaintext, markdown, email, extra media, and the JSON family that ingest
|
|
55
|
+
# routes even when the console picker omits them.
|
|
56
|
+
WORKSPACE_UPLOAD_SUFFIXES = SUPPORTED_SUFFIXES | {
|
|
57
|
+
".html",
|
|
58
|
+
".htm",
|
|
59
|
+
".txt",
|
|
60
|
+
".log",
|
|
61
|
+
".md",
|
|
62
|
+
".mdx",
|
|
63
|
+
".eml",
|
|
64
|
+
".msg",
|
|
65
|
+
".oft",
|
|
66
|
+
".xml",
|
|
67
|
+
".docm",
|
|
68
|
+
".bmp",
|
|
69
|
+
".flac",
|
|
70
|
+
".ogg",
|
|
71
|
+
".oga",
|
|
72
|
+
".mpga",
|
|
73
|
+
".avi",
|
|
74
|
+
".mkv",
|
|
75
|
+
}
|
|
76
|
+
_SKIP_NAMES = frozenset({".ds_store"})
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@dataclass(frozen=True)
|
|
80
|
+
class WorkspaceUploadSelection:
|
|
81
|
+
"""Local files a workspace folder upload will send, versus ones it skips."""
|
|
82
|
+
|
|
83
|
+
accepted: list[Path]
|
|
84
|
+
skipped: list[Path]
|
|
52
85
|
|
|
53
86
|
|
|
54
87
|
class SourceError(ValueError):
|
|
@@ -76,6 +109,42 @@ def collect_specs(spec: str) -> list[str]:
|
|
|
76
109
|
return [spec]
|
|
77
110
|
|
|
78
111
|
|
|
112
|
+
def collect_workspace_uploads(root: Path) -> WorkspaceUploadSelection:
|
|
113
|
+
"""Split a directory into uploadable files and ones the workspace API will refuse."""
|
|
114
|
+
children = sorted((child for child in root.rglob("*") if child.is_file()), key=lambda path: str(path))
|
|
115
|
+
accepted: list[Path] = []
|
|
116
|
+
skipped: list[Path] = []
|
|
117
|
+
for child in children:
|
|
118
|
+
if child.name.lower() in _SKIP_NAMES or child.suffix.lower() not in WORKSPACE_UPLOAD_SUFFIXES:
|
|
119
|
+
skipped.append(child)
|
|
120
|
+
else:
|
|
121
|
+
accepted.append(child)
|
|
122
|
+
return WorkspaceUploadSelection(accepted=accepted, skipped=skipped)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def workspace_dest_path(relative: Path, *, prefix: str | None) -> str:
|
|
126
|
+
"""Join a workspace prefix with a path relative to the uploaded directory."""
|
|
127
|
+
rel = relative.as_posix().lstrip("/")
|
|
128
|
+
if not prefix or prefix.strip("/") in ("", "."):
|
|
129
|
+
return rel
|
|
130
|
+
return f"{prefix.strip('/')}/{rel}"
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def folder_upload_prefix(root: Path, path_arg: str | None) -> str:
|
|
134
|
+
"""Workspace destination prefix for a directory upload.
|
|
135
|
+
|
|
136
|
+
``--path`` wins. Otherwise the directory's name. ``.`` / ``..`` have no
|
|
137
|
+
usable ``Path.name``, so those resolve to the real folder name instead of
|
|
138
|
+
uploading into the workspace root.
|
|
139
|
+
"""
|
|
140
|
+
if path_arg is not None:
|
|
141
|
+
return path_arg
|
|
142
|
+
name = root.name
|
|
143
|
+
if name in ("", ".", ".."):
|
|
144
|
+
name = root.resolve().name
|
|
145
|
+
return name
|
|
146
|
+
|
|
147
|
+
|
|
79
148
|
def resolve_source(
|
|
80
149
|
spec: str,
|
|
81
150
|
*,
|
|
@@ -3,17 +3,27 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import argparse
|
|
6
|
+
import json
|
|
6
7
|
import sys
|
|
8
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
9
|
+
from dataclasses import dataclass
|
|
7
10
|
from pathlib import Path
|
|
8
11
|
|
|
9
12
|
from ndi_sdk.client import NdiClient
|
|
10
13
|
from ndi_sdk.credentials import write_config
|
|
11
|
-
from ndi_sdk.
|
|
14
|
+
from ndi_sdk.errors import ErrorCode, JobFailedError, JobTimeoutError, NdiError, NdiStatusError
|
|
15
|
+
from ndi_sdk.models.jobs import Job, UploadFileResult
|
|
16
|
+
from ndi_sdk.resources.jobs import check_terminal
|
|
12
17
|
|
|
13
18
|
from ndi_cli import _render
|
|
14
19
|
from ndi_cli._config import ConfigError, config_path, read_config_values, resolve_settings
|
|
15
|
-
from ndi_cli._jobs import wait_and_report
|
|
16
|
-
from ndi_cli._output import add_job_output_args, emit_text, output_mode
|
|
20
|
+
from ndi_cli._jobs import wait_and_report, wait_job
|
|
21
|
+
from ndi_cli._output import add_job_output_args, emit_text, output_mode, _content_failure
|
|
22
|
+
from ndi_cli._sources import collect_workspace_uploads, folder_upload_prefix, workspace_dest_path
|
|
23
|
+
|
|
24
|
+
DEFAULT_UPLOAD_PARALLEL = 8
|
|
25
|
+
INGEST_FILE_ID_BATCH = 1000
|
|
26
|
+
_SKIP_WARNING_LIST_CAP = 20
|
|
17
27
|
|
|
18
28
|
|
|
19
29
|
def add_parsers(sub: argparse._SubParsersAction, common: argparse.ArgumentParser) -> None:
|
|
@@ -50,11 +60,22 @@ def add_parsers(sub: argparse._SubParsersAction, common: argparse.ArgumentParser
|
|
|
50
60
|
files = sub.add_parser("files", help="Upload, list, inspect, or delete workspace files.")
|
|
51
61
|
fs = files.add_subparsers(dest="files_command", required=True)
|
|
52
62
|
|
|
53
|
-
f_upload = fs.add_parser("upload", parents=[common], help="Upload a file into a workspace.")
|
|
54
|
-
f_upload.add_argument("file")
|
|
55
|
-
f_upload.add_argument(
|
|
63
|
+
f_upload = fs.add_parser("upload", parents=[common], help="Upload a file or folder into a workspace.")
|
|
64
|
+
f_upload.add_argument("file", help="Local file, or a directory to walk recursively.")
|
|
65
|
+
f_upload.add_argument(
|
|
66
|
+
"--path",
|
|
67
|
+
default=None,
|
|
68
|
+
help="Workspace destination path (required for a file). For a directory, a prefix; default is the directory name.",
|
|
69
|
+
)
|
|
56
70
|
f_upload.add_argument("--on-conflict", choices=("reject", "new_version"), default="reject")
|
|
57
71
|
f_upload.add_argument("--ingest", action="store_true", help="Queue ingestion after the upload.")
|
|
72
|
+
f_upload.add_argument(
|
|
73
|
+
"-j",
|
|
74
|
+
type=int,
|
|
75
|
+
default=DEFAULT_UPLOAD_PARALLEL,
|
|
76
|
+
dest="parallel",
|
|
77
|
+
help=f"Parallel uploads for a directory (default {DEFAULT_UPLOAD_PARALLEL}).",
|
|
78
|
+
)
|
|
58
79
|
add_job_output_args(f_upload)
|
|
59
80
|
f_upload.set_defaults(run=_run_files_upload, family="workspace", needs_workspace=True, needs_client=True)
|
|
60
81
|
|
|
@@ -113,12 +134,12 @@ def add_parsers(sub: argparse._SubParsersAction, common: argparse.ArgumentParser
|
|
|
113
134
|
|
|
114
135
|
jet = sub.add_parser("je-testing", parents=[common], help="Run journal-entry testing over a ledger package.")
|
|
115
136
|
jet.add_argument("query")
|
|
116
|
-
jet.add_argument("--effort", choices=("low", "medium", "high"), default=None)
|
|
137
|
+
jet.add_argument("--reasoning-effort", choices=("none", "minimal", "low", "medium", "high"), default=None)
|
|
117
138
|
jet.add_argument("--prefix", default=None, dest="path_prefix")
|
|
118
139
|
jet.add_argument("--path", action="append", dest="paths", default=None, help="Scope to this file; repeatable.")
|
|
119
140
|
jet.add_argument("--context", default=None)
|
|
120
141
|
# No --session-id, for the reason deep-search omits it: a shell has nowhere
|
|
121
|
-
# to keep a thread. Omitting --effort
|
|
142
|
+
# to keep a thread. Omitting --reasoning-effort uses the service default.
|
|
122
143
|
add_job_output_args(jet)
|
|
123
144
|
jet.set_defaults(run=_run_je_testing, family="workspace", needs_workspace=True, needs_client=True)
|
|
124
145
|
|
|
@@ -186,9 +207,14 @@ def _run_ws_use(args: argparse.Namespace) -> int:
|
|
|
186
207
|
|
|
187
208
|
def _run_files_upload(client: NdiClient, workspace_id: str, args: argparse.Namespace) -> int:
|
|
188
209
|
path = Path(args.file)
|
|
210
|
+
if path.is_dir():
|
|
211
|
+
return _run_folder_upload(client, workspace_id, args, path)
|
|
189
212
|
if not path.is_file():
|
|
190
213
|
print(f"error: file not found: {args.file}", file=sys.stderr)
|
|
191
214
|
return 2
|
|
215
|
+
if not args.path:
|
|
216
|
+
print("error: --path is required when uploading a single file", file=sys.stderr)
|
|
217
|
+
return 2
|
|
192
218
|
dest = args.path.lstrip("/")
|
|
193
219
|
if args.ingest:
|
|
194
220
|
result = client.upload_and_ingest(
|
|
@@ -203,6 +229,165 @@ def _run_files_upload(client: NdiClient, workspace_id: str, args: argparse.Names
|
|
|
203
229
|
return _finish_job(client, job, args, _render.job_line, path.stem)
|
|
204
230
|
|
|
205
231
|
|
|
232
|
+
@dataclass(frozen=True)
|
|
233
|
+
class _UploadOutcome:
|
|
234
|
+
kind: str
|
|
235
|
+
dest: str
|
|
236
|
+
file_id: str | None = None
|
|
237
|
+
error: str | None = None
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _run_folder_upload(client: NdiClient, workspace_id: str, args: argparse.Namespace, root: Path) -> int:
|
|
241
|
+
selection = collect_workspace_uploads(root)
|
|
242
|
+
skipped = [_rel(root, path) for path in selection.skipped]
|
|
243
|
+
if not selection.accepted:
|
|
244
|
+
_print_skip_warning(skipped)
|
|
245
|
+
print(f"error: no supported files under {args.file}", file=sys.stderr)
|
|
246
|
+
return 2
|
|
247
|
+
prefix = folder_upload_prefix(root, args.path)
|
|
248
|
+
workers = max(1, args.parallel)
|
|
249
|
+
outcomes: list[_UploadOutcome] = []
|
|
250
|
+
with ThreadPoolExecutor(max_workers=workers) as pool:
|
|
251
|
+
futures = [
|
|
252
|
+
pool.submit(
|
|
253
|
+
_upload_one,
|
|
254
|
+
client,
|
|
255
|
+
workspace_id,
|
|
256
|
+
local,
|
|
257
|
+
workspace_dest_path(local.relative_to(root), prefix=prefix),
|
|
258
|
+
args,
|
|
259
|
+
)
|
|
260
|
+
for local in selection.accepted
|
|
261
|
+
]
|
|
262
|
+
for future in as_completed(futures):
|
|
263
|
+
outcomes.append(future.result())
|
|
264
|
+
uploaded = [item for item in outcomes if item.kind == "ok"]
|
|
265
|
+
skipped.extend(item.dest for item in outcomes if item.kind == "skipped")
|
|
266
|
+
failed = [item for item in outcomes if item.kind == "failed"]
|
|
267
|
+
json_mode = output_mode(args) == "json"
|
|
268
|
+
if not json_mode:
|
|
269
|
+
print(f"uploaded {len(uploaded)}")
|
|
270
|
+
for item in sorted(uploaded, key=lambda row: row.dest):
|
|
271
|
+
print(f" {item.dest}")
|
|
272
|
+
for item in failed:
|
|
273
|
+
print(f"error: {item.dest}: {item.error}", file=sys.stderr)
|
|
274
|
+
_print_skip_warning(skipped)
|
|
275
|
+
ingestions: list[Job] | None = None
|
|
276
|
+
code = 1 if failed else 0
|
|
277
|
+
if not failed and args.ingest:
|
|
278
|
+
file_ids = [item.file_id for item in uploaded if item.file_id]
|
|
279
|
+
if not file_ids:
|
|
280
|
+
if json_mode:
|
|
281
|
+
_emit_folder_upload_json(uploaded, skipped, failed, ingestions=[], args=args)
|
|
282
|
+
print("error: no uploaded file ids to ingest", file=sys.stderr)
|
|
283
|
+
return 1
|
|
284
|
+
ingestions, code = _ingest_file_ids(client, workspace_id, args, file_ids, report=not json_mode)
|
|
285
|
+
if json_mode:
|
|
286
|
+
emit_code = _emit_folder_upload_json(uploaded, skipped, failed, ingestions, args)
|
|
287
|
+
return emit_code or code
|
|
288
|
+
return code
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _upload_one(
|
|
292
|
+
client: NdiClient,
|
|
293
|
+
workspace_id: str,
|
|
294
|
+
local: Path,
|
|
295
|
+
dest: str,
|
|
296
|
+
args: argparse.Namespace,
|
|
297
|
+
) -> _UploadOutcome:
|
|
298
|
+
try:
|
|
299
|
+
job = client.files.upload(workspace_id, local, path=dest, on_conflict=args.on_conflict)
|
|
300
|
+
if not getattr(args, "async_submit", False):
|
|
301
|
+
if job.is_terminal:
|
|
302
|
+
check_terminal(job, raise_on_failure=True)
|
|
303
|
+
else:
|
|
304
|
+
job = client.jobs.wait(job.job_id, timeout=args.timeout)
|
|
305
|
+
return _UploadOutcome(kind="ok", dest=dest, file_id=_uploaded_file_id(job))
|
|
306
|
+
except NdiStatusError as exc:
|
|
307
|
+
if exc.status_code == 422 and exc.code == ErrorCode.UNSUPPORTED_FILE_TYPE:
|
|
308
|
+
return _UploadOutcome(kind="skipped", dest=dest, error=exc.message)
|
|
309
|
+
return _UploadOutcome(kind="failed", dest=dest, error=str(exc))
|
|
310
|
+
except (NdiError, JobFailedError, JobTimeoutError) as exc:
|
|
311
|
+
return _UploadOutcome(kind="failed", dest=dest, error=str(exc))
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _uploaded_file_id(job: Job) -> str | None:
|
|
315
|
+
result = job.result
|
|
316
|
+
if isinstance(result, UploadFileResult):
|
|
317
|
+
return str(result.file.file_id)
|
|
318
|
+
return None
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _emit_folder_upload_json(
|
|
322
|
+
uploaded: list[_UploadOutcome],
|
|
323
|
+
skipped: list[str],
|
|
324
|
+
failed: list[_UploadOutcome],
|
|
325
|
+
ingestions: list[Job] | None,
|
|
326
|
+
args: argparse.Namespace,
|
|
327
|
+
) -> int:
|
|
328
|
+
payload: dict = {
|
|
329
|
+
"uploaded": [{"path": item.dest, "file_id": item.file_id} for item in sorted(uploaded, key=lambda row: row.dest)],
|
|
330
|
+
"skipped": sorted(skipped),
|
|
331
|
+
"failed": [{"path": item.dest, "error": item.error} for item in failed],
|
|
332
|
+
}
|
|
333
|
+
if ingestions is not None:
|
|
334
|
+
payload["ingestions"] = [json.loads(job.model_dump_json(exclude_none=True)) for job in ingestions]
|
|
335
|
+
return emit_text(json.dumps(payload, indent=2), args, default_stem="upload", suffix=".json")
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _ingest_file_ids(
|
|
339
|
+
client: NdiClient,
|
|
340
|
+
workspace_id: str,
|
|
341
|
+
args: argparse.Namespace,
|
|
342
|
+
file_ids: list[str],
|
|
343
|
+
*,
|
|
344
|
+
report: bool = True,
|
|
345
|
+
) -> tuple[list[Job], int]:
|
|
346
|
+
jobs: list[Job] = []
|
|
347
|
+
code = 0
|
|
348
|
+
for start in range(0, len(file_ids), INGEST_FILE_ID_BATCH):
|
|
349
|
+
chunk = file_ids[start : start + INGEST_FILE_ID_BATCH]
|
|
350
|
+
try:
|
|
351
|
+
job = client.ingestion.ingest(workspace_id, file_ids=chunk)
|
|
352
|
+
except NdiError as exc:
|
|
353
|
+
if report:
|
|
354
|
+
raise
|
|
355
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
356
|
+
return jobs, 1
|
|
357
|
+
if report:
|
|
358
|
+
chunk_code = _finish_job(client, job, args, _render.ingestion_job, "ingest")
|
|
359
|
+
if chunk_code != 0:
|
|
360
|
+
code = chunk_code
|
|
361
|
+
continue
|
|
362
|
+
waited, wait_code = wait_job(client, job, args, print_timeout_id=False)
|
|
363
|
+
if waited is not None:
|
|
364
|
+
jobs.append(waited)
|
|
365
|
+
code = code or wait_code or _content_failure(waited)
|
|
366
|
+
elif wait_code:
|
|
367
|
+
code = wait_code
|
|
368
|
+
return jobs, code
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _print_skip_warning(skipped: list[str]) -> None:
|
|
372
|
+
if not skipped:
|
|
373
|
+
return
|
|
374
|
+
ordered = sorted(skipped)
|
|
375
|
+
print(f"warning: skipped {len(ordered)} unsupported file(s):", file=sys.stderr)
|
|
376
|
+
shown = ordered[:_SKIP_WARNING_LIST_CAP]
|
|
377
|
+
for name in shown:
|
|
378
|
+
print(f" {name}", file=sys.stderr)
|
|
379
|
+
remaining = len(ordered) - len(shown)
|
|
380
|
+
if remaining:
|
|
381
|
+
print(f" ... and {remaining} more", file=sys.stderr)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _rel(root: Path, path: Path) -> str:
|
|
385
|
+
try:
|
|
386
|
+
return path.relative_to(root).as_posix()
|
|
387
|
+
except ValueError:
|
|
388
|
+
return str(path)
|
|
389
|
+
|
|
390
|
+
|
|
206
391
|
def _run_files_list(client: NdiClient, workspace_id: str, args: argparse.Namespace) -> int:
|
|
207
392
|
page = client.files.list(workspace_id, path_prefix=args.path_prefix, ingestion_status=args.statuses)
|
|
208
393
|
return _dump_or_text(page, args, text=_render.file_page(page), stem="files")
|
|
@@ -283,7 +468,7 @@ def _run_je_testing(client: NdiClient, workspace_id: str, args: argparse.Namespa
|
|
|
283
468
|
workspace_id,
|
|
284
469
|
query=args.query,
|
|
285
470
|
context=args.context,
|
|
286
|
-
|
|
471
|
+
reasoning_effort=args.reasoning_effort,
|
|
287
472
|
path_prefix=args.path_prefix,
|
|
288
473
|
paths=args.paths,
|
|
289
474
|
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|