theurian 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- theurian/__init__.py +14 -0
- theurian/application/__init__.py +11 -0
- theurian/application/index_builder.py +197 -0
- theurian/application/ingestion_service.py +255 -0
- theurian/application/migration_engine.py +492 -0
- theurian/application/project_service.py +1028 -0
- theurian/application/retrieval_service.py +894 -0
- theurian/application/setup_context.py +92 -0
- theurian/application/setup_service.py +294 -0
- theurian/application/setup_steps.py +821 -0
- theurian/application/setup_withholding.py +162 -0
- theurian/application/visibility.py +224 -0
- theurian/cli/__init__.py +0 -0
- theurian/cli/auth_commands.py +103 -0
- theurian/cli/commands.py +1300 -0
- theurian/cli/context.py +194 -0
- theurian/cli/index_commands.py +495 -0
- theurian/cli/main.py +179 -0
- theurian/cli/setup_commands.py +389 -0
- theurian/daemon/__init__.py +11 -0
- theurian/daemon/instance.py +236 -0
- theurian/daemon/runner.py +135 -0
- theurian/daemon/server.py +171 -0
- theurian/domain/__init__.py +6 -0
- theurian/domain/chunking.py +261 -0
- theurian/domain/compatibility.py +471 -0
- theurian/domain/context.py +61 -0
- theurian/domain/enums.py +237 -0
- theurian/domain/errors.py +140 -0
- theurian/domain/identifiers.py +199 -0
- theurian/domain/ingestion.py +131 -0
- theurian/domain/knowledge.py +352 -0
- theurian/domain/migration.py +447 -0
- theurian/domain/ports/__init__.py +74 -0
- theurian/domain/ports/authorization.py +47 -0
- theurian/domain/ports/canonical_store.py +213 -0
- theurian/domain/ports/daemon_manager.py +125 -0
- theurian/domain/ports/determinism.py +42 -0
- theurian/domain/ports/embedding.py +46 -0
- theurian/domain/ports/index_store.py +197 -0
- theurian/domain/ports/mcp_client_config.py +82 -0
- theurian/domain/ports/object_store.py +38 -0
- theurian/domain/ports/reranking.py +42 -0
- theurian/domain/ports/review_provider.py +48 -0
- theurian/domain/ports/secret_store.py +49 -0
- theurian/domain/ports/source_parser.py +71 -0
- theurian/domain/ports/specification_provider.py +30 -0
- theurian/domain/ports/summarization.py +55 -0
- theurian/domain/ports/vector_store.py +58 -0
- theurian/domain/project.py +106 -0
- theurian/domain/ranking.py +328 -0
- theurian/domain/retrieval.py +129 -0
- theurian/domain/review.py +229 -0
- theurian/domain/setup.py +338 -0
- theurian/domain/specification.py +162 -0
- theurian/domain/state.py +183 -0
- theurian/domain/values.py +220 -0
- theurian/indexing/__init__.py +21 -0
- theurian/infrastructure/__init__.py +9 -0
- theurian/infrastructure/claude/__init__.py +5 -0
- theurian/infrastructure/claude/mcp_config.py +304 -0
- theurian/infrastructure/determinism.py +87 -0
- theurian/infrastructure/embedding/__init__.py +9 -0
- theurian/infrastructure/embedding/hashing.py +125 -0
- theurian/infrastructure/filesystem/__init__.py +9 -0
- theurian/infrastructure/filesystem/migration_loader.py +336 -0
- theurian/infrastructure/filesystem/parsers/markdown.py +243 -0
- theurian/infrastructure/filesystem/parsers/openapi.py +255 -0
- theurian/infrastructure/filesystem/parsers/registry.py +122 -0
- theurian/infrastructure/filesystem/parsers/structured.py +160 -0
- theurian/infrastructure/git/__init__.py +9 -0
- theurian/infrastructure/github/__init__.py +10 -0
- theurian/infrastructure/raptor/__init__.py +24 -0
- theurian/infrastructure/secrets/__init__.py +10 -0
- theurian/infrastructure/secrets/file_store.py +122 -0
- theurian/infrastructure/services/__init__.py +57 -0
- theurian/infrastructure/services/launchagent.py +299 -0
- theurian/infrastructure/services/runner.py +115 -0
- theurian/infrastructure/services/systemd_user.py +432 -0
- theurian/infrastructure/sqlite/__init__.py +9 -0
- theurian/infrastructure/sqlite/connection.py +317 -0
- theurian/infrastructure/sqlite/index_query.py +327 -0
- theurian/infrastructure/sqlite/index_scan.py +353 -0
- theurian/infrastructure/sqlite/index_schema.py +160 -0
- theurian/infrastructure/sqlite/index_store.py +1158 -0
- theurian/infrastructure/sqlite/schema.py +237 -0
- theurian/infrastructure/sqlite/store.py +918 -0
- theurian/infrastructure/vector/__init__.py +21 -0
- theurian/ingestion/__init__.py +9 -0
- theurian/mcp/__init__.py +12 -0
- theurian/mcp/results.py +80 -0
- theurian/mcp/search.py +777 -0
- theurian/mcp/tools.py +636 -0
- theurian/migrations/__init__.py +10 -0
- theurian/normalization/__init__.py +10 -0
- theurian/normalization/projection.py +153 -0
- theurian/observability/__init__.py +6 -0
- theurian/retrieval/__init__.py +28 -0
- theurian/review/__init__.py +9 -0
- theurian/schemas/README.md +118 -0
- theurian/schemas/cli/version.schema.json +27 -0
- theurian/schemas/config/project-config.schema.json +157 -0
- theurian/schemas/knowledge/retrieval-result.schema.json +111 -0
- theurian/schemas/mcp/knowledge-search-response.schema.json +38 -0
- theurian/schemas/mcp/project-list-response.schema.json +48 -0
- theurian/schemas/mcp/retrieval-metadata.schema.json +154 -0
- theurian/schemas/mcp/tool-context.schema.json +32 -0
- theurian/schemas/migrations/migration.schema.json +358 -0
- theurian/schemas/protocol/compatibility.schema.json +41 -0
- theurian/security/__init__.py +30 -0
- theurian/security/env_file.py +31 -0
- theurian/security/paths.py +142 -0
- theurian/security/tokens.py +112 -0
- theurian/security/yaml_loading.py +105 -0
- theurian/specification/__init__.py +7 -0
- theurian/traceability/__init__.py +13 -0
- theurian-0.1.0.dev0.dist-info/METADATA +46 -0
- theurian-0.1.0.dev0.dist-info/RECORD +121 -0
- theurian-0.1.0.dev0.dist-info/WHEEL +4 -0
- theurian-0.1.0.dev0.dist-info/entry_points.txt +2 -0
- theurian-0.1.0.dev0.dist-info/licenses/LICENSE +202 -0
theurian/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Theurian: Git-native engineering knowledge for AI agents.
|
|
2
|
+
|
|
3
|
+
The public surface for external consumers is the ``theurian`` CLI, the MCP
|
|
4
|
+
server, and the JSON Schemas under ``schemas/``. Python modules inside this
|
|
5
|
+
package are internal and carry no stability guarantee -- notably, the Claude Code
|
|
6
|
+
plugin must never import them (ADR-0001, CP-2).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from theurian.domain.compatibility import CURRENT_PROTOCOL_VERSION
|
|
10
|
+
|
|
11
|
+
__version__ = "0.1.0.dev0"
|
|
12
|
+
__protocol_version__ = CURRENT_PROTOCOL_VERSION
|
|
13
|
+
|
|
14
|
+
__all__ = ["__protocol_version__", "__version__"]
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""Application layer: use cases and orchestration.
|
|
2
|
+
|
|
3
|
+
Depends on :mod:`theurian.domain` only. Adapters arrive by constructor
|
|
4
|
+
injection; nothing here names a concrete implementation (ADR-0003).
|
|
5
|
+
|
|
6
|
+
Milestone 1 onward: ``SetupService``, ``MigrationService``, ``IndexingService``,
|
|
7
|
+
``RetrievalService``, ``ReviewService``, ``TraceabilityService``.
|
|
8
|
+
|
|
9
|
+
``SetupService`` is shared by ``theurian setup`` and ``/theurian:setup``. There
|
|
10
|
+
is exactly one implementation of setup, because two would drift (FR-L1).
|
|
11
|
+
"""
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
"""Turning a canonical state into an index build (FR-R2, FR-R3, ADR-0022).
|
|
2
|
+
|
|
3
|
+
Split out of :mod:`theurian.application.retrieval_service` for the reason
|
|
4
|
+
:mod:`theurian.application.visibility` was split out of it before: that file had
|
|
5
|
+
grown past the size at which it can be read in one sitting, and this is a seam
|
|
6
|
+
rather than a cut. It described itself as "three use cases over one index file",
|
|
7
|
+
and this is the one that *writes*. Everything left there reads.
|
|
8
|
+
|
|
9
|
+
The seam is real rather than arithmetic. Nothing here consults a
|
|
10
|
+
:class:`~theurian.application.visibility.Visibility`, because at build time there
|
|
11
|
+
is no caller to be visible *to*: the filter that applies is
|
|
12
|
+
:func:`~theurian.domain.enums.may_surface` against the operator's
|
|
13
|
+
``include_unapproved``, and it decides what is written rather than what is shown.
|
|
14
|
+
The equality property the query side exists to hold has no counterpart here.
|
|
15
|
+
|
|
16
|
+
Takes its collaborators by injection, so a build is testable without a database
|
|
17
|
+
and without an embedding provider.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import asyncio
|
|
23
|
+
from collections.abc import Callable, Sequence
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import Final, final
|
|
27
|
+
|
|
28
|
+
from theurian.domain.chunking import IndexableChunk, chunk_document
|
|
29
|
+
from theurian.domain.context import RequestContext
|
|
30
|
+
from theurian.domain.enums import may_surface
|
|
31
|
+
from theurian.domain.identifiers import ProjectId
|
|
32
|
+
from theurian.domain.ports.canonical_store import CanonicalReadSession
|
|
33
|
+
from theurian.domain.ports.embedding import EmbeddingProvider
|
|
34
|
+
from theurian.domain.ports.index_store import IndexStore
|
|
35
|
+
|
|
36
|
+
#: Chunks per embedding request. An API-backed provider caps request size, and a
|
|
37
|
+
#: local one gains nothing from an unbounded batch -- while an unbounded batch
|
|
38
|
+
#: holds the whole corpus and all its vectors in memory at once.
|
|
39
|
+
EMBED_BATCH: Final = 128
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True, slots=True)
|
|
43
|
+
class IndexRequest:
|
|
44
|
+
"""What to index, and where to put it."""
|
|
45
|
+
|
|
46
|
+
database: Path
|
|
47
|
+
index_path: Path
|
|
48
|
+
project_id: str
|
|
49
|
+
state_hash: str
|
|
50
|
+
index_build_id: str
|
|
51
|
+
#: Whether unapproved revisions are written at all. Off by default, so an
|
|
52
|
+
#: operator who never opts in has a hard guarantee that no draft is in the
|
|
53
|
+
#: file — not merely that a query filter is expected to hold.
|
|
54
|
+
include_unapproved: bool = False
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@final
|
|
58
|
+
class IndexBuilder:
|
|
59
|
+
"""Turns a canonical state into an index build."""
|
|
60
|
+
|
|
61
|
+
def __init__(
|
|
62
|
+
self,
|
|
63
|
+
*,
|
|
64
|
+
store_factory: Callable[[Path], CanonicalReadSession],
|
|
65
|
+
index_factory: Callable[[Path], IndexStore],
|
|
66
|
+
embedder: EmbeddingProvider | None = None,
|
|
67
|
+
) -> None:
|
|
68
|
+
self._store_factory = store_factory
|
|
69
|
+
self._index_factory = index_factory
|
|
70
|
+
self._embedder = embedder
|
|
71
|
+
|
|
72
|
+
def build(self, request: IndexRequest) -> dict[str, object]:
|
|
73
|
+
"""Write a new index file from a canonical state.
|
|
74
|
+
|
|
75
|
+
Unapproved revisions are written only when asked for, and `rejected`
|
|
76
|
+
never is.
|
|
77
|
+
|
|
78
|
+
The obvious simplification — index everything, filter at query time —
|
|
79
|
+
was tried and reverted. It makes `includeUnapproved=True` a single
|
|
80
|
+
boolean that reaches content the team decided must not be followed, and
|
|
81
|
+
it removes the operator's ability to guarantee that a draft is not in
|
|
82
|
+
the file at all. The cost is that `includeUnapproved=True` cannot return
|
|
83
|
+
rows that were never written, which is reported rather than hidden:
|
|
84
|
+
`indexesUnapproved` says whether this build can answer such a query.
|
|
85
|
+
|
|
86
|
+
Either a whole index file or none: a build that fails part-way deletes
|
|
87
|
+
what it wrote. That guarantee used to live only in the CLI while
|
|
88
|
+
:meth:`_embed`'s docstring asserted it here, so any other caller of
|
|
89
|
+
`build` — a daemon, a test, a future scheduled rebuild — got a
|
|
90
|
+
half-written file and the promise that it could not happen.
|
|
91
|
+
|
|
92
|
+
**What it wrote, and nothing else.** `IndexStore.create` refuses to
|
|
93
|
+
overwrite an existing file and raises; the cleanup then unlinked the very
|
|
94
|
+
file it had just been refused permission to touch — which is the file
|
|
95
|
+
`active-index.json` names, so a build against an already-taken path
|
|
96
|
+
deleted the published index and left the pointer aimed at nothing. Not
|
|
97
|
+
reachable from `theurian index build`, which mints a fresh ULID per
|
|
98
|
+
build, but `build` is a public application-layer API and `create`'s
|
|
99
|
+
contract says an existing file is left alone.
|
|
100
|
+
"""
|
|
101
|
+
# Sampled before `create` rather than inferred from the exception: only a
|
|
102
|
+
# path this call brought into existence may be removed by it.
|
|
103
|
+
preexisting = request.index_path.exists()
|
|
104
|
+
try:
|
|
105
|
+
return self._build(request)
|
|
106
|
+
except Exception:
|
|
107
|
+
# A partial index is worse than none. It looks complete, ranks the
|
|
108
|
+
# fraction it holds, and never surfaces the rest -- which reads as a
|
|
109
|
+
# relevance problem rather than a build failure, and so does not get
|
|
110
|
+
# investigated. Nothing this build wrote is published until it
|
|
111
|
+
# returns, so deleting here loses no index a search could have used.
|
|
112
|
+
if not preexisting:
|
|
113
|
+
request.index_path.unlink(missing_ok=True)
|
|
114
|
+
raise
|
|
115
|
+
|
|
116
|
+
def _build(self, request: IndexRequest) -> dict[str, object]:
|
|
117
|
+
index = self._index_factory(request.index_path)
|
|
118
|
+
index.create(index_build_id=request.index_build_id, state_hash=request.state_hash)
|
|
119
|
+
|
|
120
|
+
context = RequestContext(project_id=ProjectId(request.project_id))
|
|
121
|
+
indexable: list[IndexableChunk] = []
|
|
122
|
+
|
|
123
|
+
with self._store_factory(request.database) as store:
|
|
124
|
+
for item in store.list_items(context):
|
|
125
|
+
# The same authority the search paths consult. Inlined here as
|
|
126
|
+
# two comparisons until `may_surface` moved to the domain, which
|
|
127
|
+
# is one copy of a security rule too many.
|
|
128
|
+
if not may_surface(item.status, include_unapproved=request.include_unapproved):
|
|
129
|
+
continue
|
|
130
|
+
if item.current_revision_id is None:
|
|
131
|
+
continue
|
|
132
|
+
revision = store.get_revision(context, item.current_revision_id)
|
|
133
|
+
if revision is None: # pragma: no cover - the pointer is a foreign key
|
|
134
|
+
continue
|
|
135
|
+
|
|
136
|
+
# The title is prepended to the body before splitting so that a
|
|
137
|
+
# query matching only the title still finds the document. A
|
|
138
|
+
# separately indexed title field would need its own retriever and
|
|
139
|
+
# its own fusion weight for the same effect.
|
|
140
|
+
body = f"{revision.title}\n\n{revision.body}"
|
|
141
|
+
for chunk in chunk_document(revision.revision_id.value, body):
|
|
142
|
+
indexable.append(
|
|
143
|
+
IndexableChunk(
|
|
144
|
+
chunk=chunk,
|
|
145
|
+
project_id=request.project_id,
|
|
146
|
+
item_id=item.item_id.value,
|
|
147
|
+
revision_id=revision.revision_id.value,
|
|
148
|
+
status=item.status.value,
|
|
149
|
+
sensitivity=revision.metadata.sensitivity.value,
|
|
150
|
+
trust_level=revision.metadata.trust_level.value,
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
index.add_chunks(indexable)
|
|
155
|
+
embedded = self._embed(index, indexable)
|
|
156
|
+
|
|
157
|
+
return {
|
|
158
|
+
"indexBuildId": request.index_build_id,
|
|
159
|
+
"stateHash": request.state_hash,
|
|
160
|
+
"indexPath": str(request.index_path),
|
|
161
|
+
"chunks": len(indexable),
|
|
162
|
+
"embeddings": embedded,
|
|
163
|
+
"embeddingModel": self._embedder.model_id if self._embedder else "",
|
|
164
|
+
"indexesUnapproved": request.include_unapproved,
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
def _embed(self, index: IndexStore, indexable: Sequence[IndexableChunk]) -> int:
|
|
168
|
+
"""Embed every chunk, or none.
|
|
169
|
+
|
|
170
|
+
Batched, because a real provider caps request size and a local one gains
|
|
171
|
+
nothing from an unbounded batch.
|
|
172
|
+
|
|
173
|
+
A partial embedding is worse than none: the dense retriever would rank
|
|
174
|
+
the embedded half and silently never surface the rest, which looks like
|
|
175
|
+
a relevance problem rather than a build problem. :meth:`build` discards
|
|
176
|
+
the whole index file if any batch raises, so a partial one never exists
|
|
177
|
+
to be published.
|
|
178
|
+
"""
|
|
179
|
+
if self._embedder is None or not indexable:
|
|
180
|
+
return 0
|
|
181
|
+
|
|
182
|
+
embedded = 0
|
|
183
|
+
for start in range(0, len(indexable), EMBED_BATCH):
|
|
184
|
+
batch = indexable[start : start + EMBED_BATCH]
|
|
185
|
+
vectors = asyncio.run(self._embedder.embed(tuple(c.chunk.text for c in batch)))
|
|
186
|
+
index.add_embeddings(
|
|
187
|
+
[(c.chunk.chunk_id, v) for c, v in zip(batch, vectors, strict=True)]
|
|
188
|
+
)
|
|
189
|
+
embedded += len(vectors)
|
|
190
|
+
|
|
191
|
+
index.record_embedding_model(
|
|
192
|
+
model_id=self._embedder.model_id, dimension=self._embedder.dimension
|
|
193
|
+
)
|
|
194
|
+
return embedded
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
__all__ = ["EMBED_BATCH", "IndexBuilder", "IndexRequest"]
|
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""Source ingestion (FR-S1 .. FR-S6, ADR-0010).
|
|
2
|
+
|
|
3
|
+
Walks a project's knowledge directory, dispatches each file to a parser, and
|
|
4
|
+
normalizes the result into the Canonical Layer.
|
|
5
|
+
|
|
6
|
+
Two properties shape the design:
|
|
7
|
+
|
|
8
|
+
**Failure is per document.** A malformed YAML file among two hundred must not
|
|
9
|
+
make the other 199 unavailable, so a parse failure becomes a value in the report
|
|
10
|
+
rather than an exception that unwinds the walk.
|
|
11
|
+
|
|
12
|
+
**Unchanged files cost one hash.** Touching a file without changing it must not
|
|
13
|
+
trigger a reparse and a reindex, so content hashes from the previous run are
|
|
14
|
+
compared before any parsing happens.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from collections.abc import Iterator
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from pathlib import Path, PurePosixPath
|
|
22
|
+
from typing import Protocol
|
|
23
|
+
|
|
24
|
+
from theurian.domain.errors import (
|
|
25
|
+
InputTooLargeError,
|
|
26
|
+
PathEscapeError,
|
|
27
|
+
TheurianError,
|
|
28
|
+
)
|
|
29
|
+
from theurian.domain.ingestion import (
|
|
30
|
+
IngestedDocument,
|
|
31
|
+
IngestionReport,
|
|
32
|
+
ParseFailure,
|
|
33
|
+
ParseWarning,
|
|
34
|
+
)
|
|
35
|
+
from theurian.domain.knowledge import SourceAnchor
|
|
36
|
+
from theurian.domain.ports import NormalizedDocument, SourceParser
|
|
37
|
+
from theurian.domain.values import ContentHash, MediaType
|
|
38
|
+
from theurian.normalization.projection import project
|
|
39
|
+
from theurian.security.paths import MAX_SOURCE_FILE_BYTES, read_source_file
|
|
40
|
+
|
|
41
|
+
#: Directories under `.theurian/` that hold ingestible sources. Derived
|
|
42
|
+
#: directories are excluded: ingesting `generated/` would feed Theurian's own
|
|
43
|
+
#: output back into itself as though it were a source (ADR-0004).
|
|
44
|
+
INGESTIBLE_SUBDIRECTORIES: tuple[str, ...] = ("knowledge", "specifications")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class ParserResolver(Protocol):
|
|
48
|
+
"""Resolves a document to its media type and parser.
|
|
49
|
+
|
|
50
|
+
A Protocol rather than a concrete registry so the application layer never
|
|
51
|
+
imports an adapter (ADR-0003).
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
def detect(self, path: PurePosixPath, data: bytes) -> MediaType | None: ...
|
|
55
|
+
def for_media_type(self, media_type: MediaType) -> SourceParser | None: ...
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class WarningSource(Protocol):
|
|
59
|
+
"""Parsers that surface warnings alongside a normalized document.
|
|
60
|
+
|
|
61
|
+
``NormalizedDocument`` carries no warning field, and widening the
|
|
62
|
+
``SourceParser`` port for one parser's benefit would push a Markdown concern
|
|
63
|
+
into every future adapter. Parsers that have warnings expose them here
|
|
64
|
+
instead, and the service asks only those that do (ADR-0019).
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
def warnings_for(self, text: str) -> tuple[ParseWarning, ...]: ...
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True, slots=True)
|
|
71
|
+
class IngestionRequest:
|
|
72
|
+
"""One ingestion run's inputs."""
|
|
73
|
+
|
|
74
|
+
project_root: Path
|
|
75
|
+
knowledge_dir: Path
|
|
76
|
+
#: Content hashes from the previous run, keyed by project-relative path.
|
|
77
|
+
#: Anything matching is reported unchanged and never reparsed.
|
|
78
|
+
known_hashes: dict[str, str]
|
|
79
|
+
#: Commit the sources were read at, so every anchor pins an immutable
|
|
80
|
+
#: object rather than a path that may since have moved (FR-S3).
|
|
81
|
+
commit_sha: str | None = None
|
|
82
|
+
repository: str | None = None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class IngestionService:
|
|
86
|
+
"""Normalizes a project's sources into canonical documents."""
|
|
87
|
+
|
|
88
|
+
def __init__(self, resolver: ParserResolver) -> None:
|
|
89
|
+
self._resolver = resolver
|
|
90
|
+
|
|
91
|
+
def ingest(self, request: IngestionRequest) -> IngestionReport:
|
|
92
|
+
"""Walk and normalize every ingestible source.
|
|
93
|
+
|
|
94
|
+
Never raises for a document-level problem. A caller inspects
|
|
95
|
+
``report.failures`` and decides; the run itself always completes.
|
|
96
|
+
"""
|
|
97
|
+
report = IngestionReport()
|
|
98
|
+
|
|
99
|
+
for path in self._discover(request.knowledge_dir):
|
|
100
|
+
relative = path.relative_to(request.project_root).as_posix()
|
|
101
|
+
try:
|
|
102
|
+
self._ingest_one(relative, request, report)
|
|
103
|
+
except TheurianError as exc:
|
|
104
|
+
# Security and limit refusals are per-document too. One hostile
|
|
105
|
+
# file must not prevent the rest of the tree from being read.
|
|
106
|
+
report.failures.append(ParseFailure(path=relative, reason=str(exc)))
|
|
107
|
+
|
|
108
|
+
return report
|
|
109
|
+
|
|
110
|
+
def _ingest_one(
|
|
111
|
+
self, relative: str, request: IngestionRequest, report: IngestionReport
|
|
112
|
+
) -> None:
|
|
113
|
+
try:
|
|
114
|
+
data = read_source_file(request.project_root, PurePosixPath(relative))
|
|
115
|
+
except (PathEscapeError, InputTooLargeError) as exc:
|
|
116
|
+
report.failures.append(ParseFailure(path=relative, reason=str(exc)))
|
|
117
|
+
return
|
|
118
|
+
except OSError as exc:
|
|
119
|
+
report.failures.append(ParseFailure(path=relative, reason=f"unreadable: {exc}"))
|
|
120
|
+
return
|
|
121
|
+
|
|
122
|
+
content_hash = ContentHash.of_bytes(data)
|
|
123
|
+
if request.known_hashes.get(relative) == content_hash.value:
|
|
124
|
+
# The cheap early exit: touching a file without changing it costs
|
|
125
|
+
# one hash, not a reparse and a reindex.
|
|
126
|
+
report.unchanged.append(relative)
|
|
127
|
+
return
|
|
128
|
+
|
|
129
|
+
media_type = self._resolver.detect(PurePosixPath(relative), data)
|
|
130
|
+
if media_type is None:
|
|
131
|
+
report.skipped.append(relative)
|
|
132
|
+
return
|
|
133
|
+
|
|
134
|
+
parser = self._resolver.for_media_type(media_type)
|
|
135
|
+
if parser is None:
|
|
136
|
+
report.failures.append(
|
|
137
|
+
ParseFailure(
|
|
138
|
+
path=relative,
|
|
139
|
+
reason=f"no parser registered for {media_type}",
|
|
140
|
+
media_type=media_type.value,
|
|
141
|
+
)
|
|
142
|
+
)
|
|
143
|
+
return
|
|
144
|
+
|
|
145
|
+
anchor = _anchor(relative, request)
|
|
146
|
+
|
|
147
|
+
try:
|
|
148
|
+
normalized = parser.parse(data, media_type=media_type, anchor=anchor)
|
|
149
|
+
except (ValueError, InputTooLargeError) as exc:
|
|
150
|
+
report.failures.append(
|
|
151
|
+
ParseFailure(path=relative, reason=str(exc), media_type=media_type.value)
|
|
152
|
+
)
|
|
153
|
+
return
|
|
154
|
+
|
|
155
|
+
report.documents.append(
|
|
156
|
+
_to_document(
|
|
157
|
+
normalized,
|
|
158
|
+
path=relative,
|
|
159
|
+
source_hash=content_hash,
|
|
160
|
+
parser=parser,
|
|
161
|
+
warnings=_warnings(parser, data),
|
|
162
|
+
)
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
def _discover(self, knowledge_dir: Path) -> Iterator[Path]:
|
|
166
|
+
"""Yield candidate source files in a stable order.
|
|
167
|
+
|
|
168
|
+
Sorted so a failure reports the same first offender on every run rather
|
|
169
|
+
than whichever the filesystem happened to yield first.
|
|
170
|
+
"""
|
|
171
|
+
for subdirectory in INGESTIBLE_SUBDIRECTORIES:
|
|
172
|
+
root = knowledge_dir / subdirectory
|
|
173
|
+
if not root.is_dir():
|
|
174
|
+
continue
|
|
175
|
+
for path in sorted(root.rglob("*")):
|
|
176
|
+
if not path.is_file() or path.name.startswith("."):
|
|
177
|
+
continue
|
|
178
|
+
# A symlink is refused later by read_source_file if it escapes;
|
|
179
|
+
# skipping obviously-oversized files here avoids reading them.
|
|
180
|
+
if path.stat().st_size > MAX_SOURCE_FILE_BYTES:
|
|
181
|
+
continue
|
|
182
|
+
yield path
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _warnings(parser: SourceParser, data: bytes) -> tuple[ParseWarning, ...]:
|
|
186
|
+
"""Collect warnings from parsers that produce them."""
|
|
187
|
+
collector = getattr(parser, "warnings_for", None)
|
|
188
|
+
if collector is None:
|
|
189
|
+
return ()
|
|
190
|
+
try:
|
|
191
|
+
text = data.decode("utf-8")
|
|
192
|
+
except UnicodeDecodeError: # pragma: no cover - parse would have failed first
|
|
193
|
+
return ()
|
|
194
|
+
result: tuple[ParseWarning, ...] = collector(text)
|
|
195
|
+
return result
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _anchor(relative: str, request: IngestionRequest) -> SourceAnchor:
|
|
199
|
+
"""Build the anchor that makes this document traceable (FR-S3).
|
|
200
|
+
|
|
201
|
+
A commit SHA plus a path pins an immutable Git object, so the anchor still
|
|
202
|
+
resolves after the file is edited, moved, or deleted.
|
|
203
|
+
"""
|
|
204
|
+
return SourceAnchor(
|
|
205
|
+
provider="git" if request.commit_sha else "filesystem",
|
|
206
|
+
source_uri=f"git://{request.repository or 'local'}/{relative}"
|
|
207
|
+
if request.commit_sha
|
|
208
|
+
else f"file://{relative}",
|
|
209
|
+
repository=request.repository,
|
|
210
|
+
commit_sha=request.commit_sha,
|
|
211
|
+
file_path=relative,
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _to_document(
|
|
216
|
+
normalized: NormalizedDocument,
|
|
217
|
+
*,
|
|
218
|
+
path: str,
|
|
219
|
+
source_hash: ContentHash,
|
|
220
|
+
parser: SourceParser,
|
|
221
|
+
warnings: tuple[ParseWarning, ...],
|
|
222
|
+
) -> IngestedDocument:
|
|
223
|
+
projection: str | None = None
|
|
224
|
+
if normalized.structured is not None:
|
|
225
|
+
projection = project(normalized.structured)
|
|
226
|
+
|
|
227
|
+
return IngestedDocument(
|
|
228
|
+
path=path,
|
|
229
|
+
title=normalized.title,
|
|
230
|
+
body=normalized.body,
|
|
231
|
+
content_type=normalized.content_type,
|
|
232
|
+
content_hash=normalized.content_hash,
|
|
233
|
+
source_hash=source_hash,
|
|
234
|
+
anchors=normalized.anchors,
|
|
235
|
+
parser_id=parser.parser_id,
|
|
236
|
+
structured=normalized.structured,
|
|
237
|
+
text_projection=projection,
|
|
238
|
+
warnings=warnings,
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def manifest_from(report: IngestionReport, previous: dict[str, str]) -> dict[str, str]:
|
|
243
|
+
"""Build the content-hash manifest for the next run's early exit.
|
|
244
|
+
|
|
245
|
+
Carries forward hashes for unchanged files, because those were never
|
|
246
|
+
reparsed and so are absent from ``report.documents``. Dropping them would
|
|
247
|
+
make every second run a full reparse.
|
|
248
|
+
"""
|
|
249
|
+
manifest = {path: previous[path] for path in report.unchanged if path in previous}
|
|
250
|
+
for document in report.documents:
|
|
251
|
+
# The *source* hash, matching what the early exit computes. Storing the
|
|
252
|
+
# body hash instead makes every Markdown file with front matter reparse
|
|
253
|
+
# on every run, because its body and its bytes differ.
|
|
254
|
+
manifest[document.path] = document.source_hash.value
|
|
255
|
+
return manifest
|