reindex-cli 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. reindex_cli/__init__.py +6 -0
  2. reindex_cli/__main__.py +3 -0
  3. reindex_cli/api.py +69 -0
  4. reindex_cli/archives.py +97 -0
  5. reindex_cli/bundled_skills/reindex-create/SKILL.md +12 -0
  6. reindex_cli/bundled_skills/reindex-data/SKILL.md +22 -0
  7. reindex_cli/bundled_skills/reindex-scan/SKILL.md +13 -0
  8. reindex_cli/cli.py +176 -0
  9. reindex_cli/collection/__init__.py +9 -0
  10. reindex_cli/collection/resolver.py +51 -0
  11. reindex_cli/collection/state.py +88 -0
  12. reindex_cli/config.py +49 -0
  13. reindex_cli/errors.py +10 -0
  14. reindex_cli/get_ops.py +183 -0
  15. reindex_cli/manifest/__init__.py +4 -0
  16. reindex_cli/manifest/models.py +31 -0
  17. reindex_cli/manifest/parser.py +211 -0
  18. reindex_cli/manifest/yaml_support.py +52 -0
  19. reindex_cli/package/__init__.py +15 -0
  20. reindex_cli/package/cards.py +41 -0
  21. reindex_cli/package/renderer.py +190 -0
  22. reindex_cli/package/validation.py +211 -0
  23. reindex_cli/parsers/__init__.py +1 -0
  24. reindex_cli/parsers/common.py +44 -0
  25. reindex_cli/parsers/csv_parser.py +76 -0
  26. reindex_cli/parsers/docling_pdf.py +222 -0
  27. reindex_cli/parsers/generic.py +25 -0
  28. reindex_cli/parsers/markdown.py +38 -0
  29. reindex_cli/parsers/profiles.py +77 -0
  30. reindex_cli/parsers/registry.py +49 -0
  31. reindex_cli/pipeline/__init__.py +3 -0
  32. reindex_cli/pipeline/assembly.py +172 -0
  33. reindex_cli/pipeline/discovery.py +73 -0
  34. reindex_cli/pipeline/inspection.py +124 -0
  35. reindex_cli/pipeline/models.py +102 -0
  36. reindex_cli/pipeline/parsing.py +63 -0
  37. reindex_cli/pipeline/planning.py +54 -0
  38. reindex_cli/pipeline/publish.py +26 -0
  39. reindex_cli/pipeline/runner.py +188 -0
  40. reindex_cli/remote_ops.py +109 -0
  41. reindex_cli/remote_state.py +39 -0
  42. reindex_cli/skills.py +92 -0
  43. reindex_cli/util.py +50 -0
  44. reindex_cli-0.3.0.dist-info/METADATA +94 -0
  45. reindex_cli-0.3.0.dist-info/RECORD +48 -0
  46. reindex_cli-0.3.0.dist-info/WHEEL +5 -0
  47. reindex_cli-0.3.0.dist-info/entry_points.txt +3 -0
  48. reindex_cli-0.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,6 @@
1
+ from importlib.metadata import PackageNotFoundError, version
2
+
3
+ try:
4
+ __version__ = version("reindex-cli")
5
+ except PackageNotFoundError: # Source-only execution without an installed distribution.
6
+ __version__ = "0.0.0+local"
@@ -0,0 +1,3 @@
1
+ from reindex_cli.cli import main
2
+
3
+ raise SystemExit(main())
reindex_cli/api.py ADDED
@@ -0,0 +1,69 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from pathlib import Path
5
+
6
+ import httpx
7
+
8
+ from reindex_cli.errors import ReIndexError
9
+
10
+
11
+ class ApiClient:
12
+ def __init__(self, base_url: str, timeout: float = 1800.0) -> None:
13
+ self.base_url = base_url.rstrip("/")
14
+ self.timeout = timeout
15
+
16
+ def health(self) -> dict:
17
+ with httpx.Client(base_url=self.base_url, timeout=30.0) as client:
18
+ return self._json(client.get("/health"))
19
+
20
+ def push(self, name: str, package: Path, sources: Path) -> dict:
21
+ with (
22
+ package.open("rb") as package_stream,
23
+ sources.open("rb") as source_stream,
24
+ httpx.Client(base_url=self.base_url, timeout=self.timeout) as client,
25
+ ):
26
+ response = client.post(
27
+ "/v1/push",
28
+ data={"name": name},
29
+ files={
30
+ "package": ("package.zip", package_stream, "application/zip"),
31
+ "sources": ("sources.zip", source_stream, "application/zip"),
32
+ },
33
+ )
34
+ return self._json(response)
35
+
36
+ def json(self, path: str, payload: dict) -> dict:
37
+ with httpx.Client(base_url=self.base_url, timeout=self.timeout) as client:
38
+ return self._json(client.post(path, json=payload))
39
+
40
+ def bytes(self, path: str, payload: dict) -> tuple[bytes, dict[str, str]]:
41
+ with httpx.Client(base_url=self.base_url, timeout=self.timeout) as client:
42
+ response = client.post(path, json=payload)
43
+ self._raise(response)
44
+ return response.content, dict(response.headers)
45
+
46
+ def _json(self, response: httpx.Response) -> dict:
47
+ self._raise(response)
48
+ try:
49
+ value = response.json()
50
+ except json.JSONDecodeError as error:
51
+ raise ReIndexError("Server returned invalid JSON") from error
52
+ if not isinstance(value, dict):
53
+ raise ReIndexError("Server returned an invalid response")
54
+ return value
55
+
56
+ @staticmethod
57
+ def _raise(response: httpx.Response) -> None:
58
+ if response.is_success:
59
+ return
60
+ try:
61
+ body = response.json()
62
+ detail = body.get("error", body)
63
+ message = detail.get("message") if isinstance(detail, dict) else None
64
+ except (json.JSONDecodeError, AttributeError):
65
+ message = None
66
+ raise ReIndexError(
67
+ f"Server request failed ({response.status_code}): "
68
+ f"{message or response.text or response.reason_phrase}"
69
+ )
@@ -0,0 +1,97 @@
1
+ from __future__ import annotations
2
+
3
+ import tempfile
4
+ import zipfile
5
+ from pathlib import Path, PurePosixPath
6
+
7
+ from reindex_cli.errors import ReIndexError
8
+ from reindex_cli.package.cards import parse_card
9
+ from reindex_cli.util import sha256_file
10
+
11
+
12
+ def build_push_archives(context) -> tuple[Path, Path, tempfile.TemporaryDirectory]:
13
+ temporary = tempfile.TemporaryDirectory(prefix="rei-push-")
14
+ root = Path(temporary.name)
15
+ package_zip = root / "package.zip"
16
+ source_zip = root / "sources.zip"
17
+ with zipfile.ZipFile(package_zip, "w", zipfile.ZIP_DEFLATED) as bundle:
18
+ for file in sorted(context.output_dir.rglob("*")):
19
+ if file.is_file():
20
+ relative = Path(context.output_dir.name) / file.relative_to(
21
+ context.output_dir
22
+ )
23
+ bundle.write(file, relative.as_posix())
24
+ references = raw_references(context.output_dir)
25
+ with zipfile.ZipFile(source_zip, "w", zipfile.ZIP_DEFLATED) as bundle:
26
+ for relative, expected in sorted(references.items()):
27
+ source = context.root / relative
28
+ if not source.is_file() or sha256_file(source) != expected:
29
+ temporary.cleanup()
30
+ raise ReIndexError(f"Missing or changed source: {relative}")
31
+ bundle.write(source, relative)
32
+ return package_zip, source_zip, temporary
33
+
34
+
35
+ def raw_references(package: Path) -> dict[str, str]:
36
+ result: dict[str, str] = {}
37
+ for card_path in package.rglob("*.node.md"):
38
+ metadata, _body = parse_card(card_path)
39
+ values = [
40
+ metadata.get("source"),
41
+ metadata.get("content"),
42
+ *metadata.get("assets", []),
43
+ ]
44
+ for value in values:
45
+ if not isinstance(value, dict):
46
+ continue
47
+ uri = str(value.get("uri", ""))
48
+ if not uri.startswith("raw://"):
49
+ continue
50
+ relative = _safe(uri.removeprefix("raw://"))
51
+ digest = str(value.get("sha256", ""))
52
+ if relative in result and result[relative] != digest:
53
+ raise ReIndexError(f"Conflicting source hashes: {relative}")
54
+ result[relative] = digest
55
+ return result
56
+
57
+
58
+ def extract_node_archive(content: bytes, target: Path) -> int:
59
+ target.mkdir(parents=True, exist_ok=True)
60
+ archive_path = target / ".download.zip"
61
+ archive_path.write_bytes(content)
62
+ count = 0
63
+ try:
64
+ with zipfile.ZipFile(archive_path) as bundle:
65
+ names: set[str] = set()
66
+ for item in bundle.infolist():
67
+ path = PurePosixPath(item.filename)
68
+ if (
69
+ item.filename in names
70
+ or path.is_absolute()
71
+ or ".." in path.parts
72
+ or not item.filename.endswith(".node.md")
73
+ or (item.external_attr >> 16) & 0o170000 == 0o120000
74
+ ):
75
+ raise ReIndexError(
76
+ f"Unsafe or non-Node pull entry: {item.filename}"
77
+ )
78
+ names.add(item.filename)
79
+ if item.is_dir():
80
+ continue
81
+ destination = target.joinpath(*path.parts)
82
+ destination.parent.mkdir(parents=True, exist_ok=True)
83
+ destination.write_bytes(bundle.read(item))
84
+ parse_card(destination)
85
+ count += 1
86
+ finally:
87
+ archive_path.unlink(missing_ok=True)
88
+ if not (target / "index.node.md").is_file():
89
+ raise ReIndexError("Pulled Node tree has no root index.node.md")
90
+ return count
91
+
92
+
93
+ def _safe(value: str) -> str:
94
+ path = PurePosixPath(value)
95
+ if path.is_absolute() or ".." in path.parts or not path.parts:
96
+ raise ReIndexError(f"Unsafe raw path: {value}")
97
+ return path.as_posix()
@@ -0,0 +1,12 @@
1
+ ---
2
+ name: reindex-create
3
+ description: Initialize a ReIndex project when the user asks to set up ReIndex or run rei init/create.
4
+ ---
5
+
6
+ # ReIndex create
7
+
8
+ 1. Resolve the exact directory named by the user.
9
+ 2. Prefer `rei init <directory> --agent <current-agent>` for normal setup. It is idempotent and manages skills.
10
+ 3. Use `rei create <directory>` only when identity-only initialization was explicitly requested.
11
+ 4. Report the Collection name and whether it was created or reused. UUID is internal and normally omitted.
12
+ 5. Do not scan, push, delete source files, or replace `.rei/collection.json` during initialization.
@@ -0,0 +1,22 @@
1
+ ---
2
+ name: reindex-data
3
+ description: Push, pull, search, and get exact ReIndex data when the user asks to publish or use remote ReIndex knowledge.
4
+ ---
5
+
6
+ # ReIndex data
7
+
8
+ ## Publish
9
+
10
+ 1. Run `rei check <path>` before `rei push <path>`; scan first if stale.
11
+ 2. `rei push` synchronously sends the validated package and exactly referenced sources.
12
+ 3. Report the user-facing Collection name and ready status, not internal UUID details.
13
+
14
+ ## Find and fetch
15
+
16
+ 1. Use `rei search "<question>"` before downloading large files.
17
+ 2. Select the result whose Evidence supports the task.
18
+ 3. Use the result's Node path with `rei get <node-path> --target content`; use `source` only when original bytes are required.
19
+ 4. Use `rei get raw://<path>` for an explicitly named raw source.
20
+ 5. Answer from the fetched file and cite the Collection name and Node or raw path.
21
+
22
+ `rei pull <name>` downloads the complete Node tree only. It intentionally excludes source, content and assets; fetch exact resources later with `rei get`.
@@ -0,0 +1,13 @@
1
+ ---
2
+ name: reindex-scan
3
+ description: Scan raw files into ReIndex when the user asks to scan, compile, ingest, or convert local data.
4
+ ---
5
+
6
+ # ReIndex scan
7
+
8
+ 1. Run `rei inspect <path>` and review effective inputs, relationships and changes.
9
+ 2. Apply only evidence-backed manifest corrections; never delete raw files just to pass validation.
10
+ 3. Run `rei scan <path>` and review changes, warnings and generated Node cards.
11
+ 4. Edit Markdown card bodies only. The CLI owns YAML frontmatter.
12
+ 5. Run `rei check <path>` after manual card edits.
13
+ 6. Report the Collection name, Node count, warnings and package location. Passing checks is not human approval.
reindex_cli/cli.py ADDED
@@ -0,0 +1,176 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from pathlib import Path
7
+
8
+ import httpx
9
+
10
+ from reindex_cli import __version__
11
+ from reindex_cli.collection import create_collection, resolve_collection
12
+ from reindex_cli.config import set_api_url
13
+ from reindex_cli.errors import ReIndexError
14
+ from reindex_cli.skills import AGENTS, manage_skills
15
+
16
+
17
+ def build_parser() -> argparse.ArgumentParser:
18
+ parser = argparse.ArgumentParser(
19
+ prog="rei", description="Build and use ReIndex knowledge Collections."
20
+ )
21
+ parser.add_argument("--version", action="version", version=__version__)
22
+ commands = parser.add_subparsers(dest="command", required=True)
23
+ init = commands.add_parser("init", help="Initialize ReIndex and Agent skills.")
24
+ init.add_argument("path", type=Path, nargs="?", default=Path.cwd())
25
+ init.add_argument("--name")
26
+ init.add_argument("--agent", choices=AGENTS, default="codex")
27
+ init.add_argument("--codex-home", type=Path)
28
+ create = commands.add_parser("create", help="Create only local identity state.")
29
+ create.add_argument("path", type=Path)
30
+ create.add_argument("--name")
31
+ rename = commands.add_parser("rename", help="Change the Collection name.")
32
+ rename.add_argument("path", type=Path)
33
+ rename.add_argument("name")
34
+ inspect = commands.add_parser("inspect", help="Inspect inputs without writing.")
35
+ inspect.add_argument("path", type=Path)
36
+ scan = commands.add_parser("scan", help="Compile a validated local package.")
37
+ scan.add_argument("path", type=Path)
38
+ scan.add_argument("--collection-root", type=Path)
39
+ check = commands.add_parser("check", help="Validate the current package.")
40
+ check.add_argument("path", type=Path)
41
+ skills = commands.add_parser("skills", help="Manage bundled Agent skills.")
42
+ skill_commands = skills.add_subparsers(dest="skill_command", required=True)
43
+ for name in ("install", "update"):
44
+ command = skill_commands.add_parser(name)
45
+ command.add_argument("--agent", choices=AGENTS, default="codex")
46
+ command.add_argument("--workspace-root", type=Path, default=Path.cwd())
47
+ command.add_argument("--codex-home", type=Path)
48
+ command.add_argument("--force", action="store_true")
49
+ api = commands.add_parser("set-api", help="Persist the default API URL.")
50
+ api.add_argument("url")
51
+ push = commands.add_parser(
52
+ "push", help="Synchronously publish package and sources."
53
+ )
54
+ push.add_argument("path", type=Path, nargs="?", default=Path.cwd())
55
+ push.add_argument("--api-url")
56
+ pull = commands.add_parser("pull", help="Download a Node-only ReIndex tree.")
57
+ pull.add_argument("name")
58
+ pull.add_argument("--output", type=Path)
59
+ pull.add_argument("--api-url")
60
+ pull.add_argument("--force", action="store_true")
61
+ search = commands.add_parser("search", help="Search a remote Collection.")
62
+ search.add_argument("query")
63
+ search.add_argument("--path", type=Path, default=Path.cwd())
64
+ search.add_argument("--remote")
65
+ search.add_argument("--api-url")
66
+ search.add_argument(
67
+ "--mode", choices=("lexical", "semantic", "hybrid"), default="lexical"
68
+ )
69
+ search.add_argument("--limit", type=int, default=10)
70
+ get = commands.add_parser("get", help="Reuse or fetch one exact resource.")
71
+ get.add_argument("reference")
72
+ get.add_argument("--path", type=Path, default=Path.cwd())
73
+ get.add_argument("--remote")
74
+ get.add_argument("--api-url")
75
+ get.add_argument("--target", choices=("card", "source", "content", "asset"))
76
+ get.add_argument("--asset-ordinal", type=int)
77
+ get.add_argument("--output", type=Path)
78
+ return parser
79
+
80
+
81
+ def main(argv: list[str] | None = None) -> int:
82
+ try:
83
+ args = build_parser().parse_args(argv)
84
+ output = _execute(args)
85
+ print(json.dumps(output, ensure_ascii=False, indent=2))
86
+ return 0
87
+ except (ReIndexError, OSError, ValueError, httpx.HTTPError) as error:
88
+ print(
89
+ json.dumps({"status": "error", "error": str(error)}, ensure_ascii=False),
90
+ file=sys.stderr,
91
+ )
92
+ return 1
93
+
94
+
95
+ def _execute(args) -> dict:
96
+ if args.command == "init":
97
+ root = args.path.expanduser().resolve()
98
+ collection = create_collection(root, args.name)
99
+ skills = manage_skills(
100
+ args.agent,
101
+ root,
102
+ update=True,
103
+ codex_home=args.codex_home,
104
+ )
105
+ return {
106
+ "status": "ready",
107
+ **collection,
108
+ "collection_id": collection["id"],
109
+ "skills": [vars(item) for item in skills],
110
+ }
111
+ if args.command in {"create", "rename"}:
112
+ name = args.name
113
+ result = create_collection(args.path.expanduser().resolve(), name)
114
+ return {
115
+ "status": "ready",
116
+ **result,
117
+ "collection_id": result["id"],
118
+ "renamed": args.command == "rename",
119
+ }
120
+ if args.command == "skills":
121
+ results = manage_skills(
122
+ args.agent,
123
+ args.workspace_root.expanduser().resolve(),
124
+ update=args.skill_command == "update",
125
+ force=args.force,
126
+ codex_home=args.codex_home,
127
+ )
128
+ return {"status": "ready", "skills": [vars(item) for item in results]}
129
+ if args.command == "set-api":
130
+ return {"status": "ready", "api_url": set_api_url(args.url)}
131
+ if args.command in {"push", "pull", "search"}:
132
+ from reindex_cli.remote_ops import (
133
+ pull_collection,
134
+ push_collection,
135
+ search_remote,
136
+ )
137
+
138
+ if args.command == "push":
139
+ return push_collection(args.path, args.api_url)
140
+ if args.command == "pull":
141
+ return pull_collection(
142
+ args.name,
143
+ args.output or Path.cwd() / args.name,
144
+ args.api_url,
145
+ force=args.force,
146
+ )
147
+ if args.command == "search":
148
+ return search_remote(
149
+ args.query, args.path, args.remote, args.api_url, args.mode, args.limit
150
+ )
151
+ if args.command == "get":
152
+ from reindex_cli.get_ops import get_resource
153
+
154
+ return get_resource(
155
+ args.reference,
156
+ args.path,
157
+ target=args.target,
158
+ asset_ordinal=args.asset_ordinal,
159
+ output=args.output,
160
+ remote=args.remote,
161
+ api_url=args.api_url,
162
+ )
163
+ from reindex_cli.pipeline.runner import (
164
+ check_collection,
165
+ inspect_collection,
166
+ run_scan,
167
+ )
168
+
169
+ context = resolve_collection(
170
+ args.path, args.collection_root if args.command == "scan" else None
171
+ )
172
+ if args.command == "inspect":
173
+ return inspect_collection(context)
174
+ if args.command == "scan":
175
+ return run_scan(context)
176
+ return check_collection(context)
@@ -0,0 +1,9 @@
1
+ from .resolver import CollectionContext, resolve_collection
2
+ from .state import create_collection, load_collection
3
+
4
+ __all__ = [
5
+ "CollectionContext",
6
+ "create_collection",
7
+ "load_collection",
8
+ "resolve_collection",
9
+ ]
@@ -0,0 +1,51 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass
4
+ from pathlib import Path
5
+
6
+ from reindex_cli.collection.state import load_collection
7
+ from reindex_cli.errors import ReIndexError
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class CollectionContext:
12
+ root: Path
13
+ scope: Path
14
+ scope_relative: str
15
+ collection_id: str
16
+ output_dir: Path
17
+ state: dict
18
+
19
+
20
+ def resolve_collection(
21
+ path: Path, explicit_root: Path | None = None
22
+ ) -> CollectionContext:
23
+ requested = path.expanduser().resolve()
24
+ if not requested.exists():
25
+ raise ReIndexError(f"Scan path does not exist: {requested}")
26
+ root = (
27
+ explicit_root.expanduser().resolve() if explicit_root else _find_root(requested)
28
+ )
29
+ if root is None:
30
+ raise ReIndexError(f"No .rei/collection.json found above: {requested}")
31
+ if not root.is_dir():
32
+ raise ReIndexError(f"Collection root is not a directory: {root}")
33
+ try:
34
+ relative = requested.relative_to(root)
35
+ except ValueError as error:
36
+ raise ReIndexError(
37
+ f"Scan path is outside Collection root: {requested}"
38
+ ) from error
39
+ state = load_collection(root)
40
+ output = root / "reIndex" / state["output_dir"]
41
+ return CollectionContext(
42
+ root, requested, relative.as_posix(), state["id"], output, state
43
+ )
44
+
45
+
46
+ def _find_root(requested: Path) -> Path | None:
47
+ start = requested if requested.is_dir() else requested.parent
48
+ for candidate in (start, *start.parents):
49
+ if (candidate / ".rei" / "collection.json").is_file():
50
+ return candidate
51
+ return None
@@ -0,0 +1,88 @@
1
+ from __future__ import annotations
2
+
3
+ from datetime import UTC, datetime
4
+ from pathlib import Path
5
+ from uuid import uuid4
6
+
7
+ from reindex_cli.errors import ReIndexError
8
+ from reindex_cli.util import atomic_json, load_json, slugify
9
+
10
+ COLLECTION_SPEC = "reindex/collection@1.0"
11
+ IDENTITY_FILE = "node-identities.json"
12
+
13
+
14
+ def create_collection(directory: Path, name: str | None = None) -> dict:
15
+ root = directory.expanduser().resolve()
16
+ if not root.is_dir():
17
+ raise ReIndexError(f"Collection directory does not exist: {root}")
18
+ state_path = root / ".rei" / "collection.json"
19
+ if state_path.exists():
20
+ stored = load_json(state_path, {})
21
+ state = load_collection(root)
22
+ if name is not None:
23
+ state["name"] = collection_name(name)
24
+ if not isinstance(stored, dict) or stored.get("name") != state["name"]:
25
+ atomic_json(state_path, state)
26
+ return {**state, "created": False}
27
+ parent = _find_parent_collection(root.parent)
28
+ if parent is not None:
29
+ raise ReIndexError(f"Collection is nested inside existing Collection: {parent}")
30
+ collection_id = str(uuid4())
31
+ normalized_name = collection_name(name or root.name)
32
+ state = {
33
+ "spec": COLLECTION_SPEC,
34
+ "id": collection_id,
35
+ "name": normalized_name,
36
+ "created_at": datetime.now(UTC).isoformat(),
37
+ "output_dir": f"{collection_id}--{slugify(root.name, 'collection')}",
38
+ }
39
+ atomic_json(state_path, state)
40
+ atomic_json(
41
+ root / ".rei" / IDENTITY_FILE,
42
+ {"spec": "reindex/node-identities@1.0", "nodes": {}},
43
+ )
44
+ agent_path = root / ".rei" / "agent" / "collection.md"
45
+ agent_path.parent.mkdir(parents=True, exist_ok=True)
46
+ agent_path.write_text(
47
+ f"# {root.name}\n\nCollection ID: `{collection_id}`.\n\n"
48
+ "Review the root `reIndex.md` against real files before scanning.\n",
49
+ encoding="utf-8",
50
+ newline="\n",
51
+ )
52
+ return {**state, "created": True}
53
+
54
+
55
+ def load_collection(root: Path) -> dict:
56
+ path = root / ".rei" / "collection.json"
57
+ try:
58
+ state = load_json(path, None)
59
+ except (OSError, ValueError) as error:
60
+ raise ReIndexError(f"Invalid Collection state: {path}") from error
61
+ if not isinstance(state, dict) or state.get("spec") != COLLECTION_SPEC:
62
+ raise ReIndexError(f"Unsupported Collection state: {path}")
63
+ if not all(
64
+ isinstance(state.get(key), str) and state[key] for key in ("id", "output_dir")
65
+ ):
66
+ raise ReIndexError(f"Incomplete Collection state: {path}")
67
+ state.setdefault("name", collection_name(root.name))
68
+ return state
69
+
70
+
71
+ def collection_name(value: str) -> str:
72
+ normalized = slugify(value, "collection")[:80].rstrip("-")
73
+ if not normalized:
74
+ raise ReIndexError("Collection name must not be empty")
75
+ return normalized
76
+
77
+
78
+ def identity_path(root: Path) -> Path:
79
+ current = root / ".rei" / IDENTITY_FILE
80
+ legacy = root / ".rei" / "identities.json"
81
+ return current if current.exists() or not legacy.exists() else legacy
82
+
83
+
84
+ def _find_parent_collection(start: Path) -> Path | None:
85
+ for candidate in (start, *start.parents):
86
+ if (candidate / ".rei" / "collection.json").is_file():
87
+ return candidate
88
+ return None
reindex_cli/config.py ADDED
@@ -0,0 +1,49 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ from pathlib import Path
6
+
7
+ from reindex_cli.util import atomic_json
8
+
9
+ DEFAULT_API_URL = "http://127.0.0.1:8000"
10
+
11
+
12
+ def config_dir() -> Path:
13
+ if value := os.getenv("REINDEX_CONFIG_HOME"):
14
+ return Path(value).expanduser().resolve()
15
+ base = Path(os.getenv("XDG_CONFIG_HOME", Path.home() / ".config"))
16
+ return base.expanduser().resolve() / "reindex"
17
+
18
+
19
+ def cache_dir() -> Path:
20
+ if value := os.getenv("REINDEX_CACHE_HOME"):
21
+ return Path(value).expanduser().resolve()
22
+ base = Path(os.getenv("XDG_CACHE_HOME", Path.home() / ".cache"))
23
+ return base.expanduser().resolve() / "reindex"
24
+
25
+
26
+ def get_api_url(explicit: str | None = None) -> str:
27
+ if explicit:
28
+ return normalize_url(explicit)
29
+ if value := os.getenv("REINDEX_API_URL"):
30
+ return normalize_url(value)
31
+ path = config_dir() / "config.json"
32
+ if path.is_file():
33
+ value = json.loads(path.read_text(encoding="utf-8")).get("api_url")
34
+ if isinstance(value, str) and value.strip():
35
+ return normalize_url(value)
36
+ return DEFAULT_API_URL
37
+
38
+
39
+ def set_api_url(value: str) -> str:
40
+ url = normalize_url(value)
41
+ atomic_json(config_dir() / "config.json", {"api_url": url})
42
+ return url
43
+
44
+
45
+ def normalize_url(value: str) -> str:
46
+ result = value.strip().rstrip("/")
47
+ if not result.startswith(("http://", "https://")):
48
+ raise ValueError("API URL must start with http:// or https://")
49
+ return result
reindex_cli/errors.py ADDED
@@ -0,0 +1,10 @@
1
+ class ReIndexError(RuntimeError):
2
+ """Expected CLI failure with a user-facing message."""
3
+
4
+
5
+ class ManifestError(ReIndexError):
6
+ """The authoring manifest is invalid."""
7
+
8
+
9
+ class PackageError(ReIndexError):
10
+ """The generated package violates the ReIndex protocol."""