theurian 0.1.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. theurian/__init__.py +14 -0
  2. theurian/application/__init__.py +11 -0
  3. theurian/application/index_builder.py +197 -0
  4. theurian/application/ingestion_service.py +255 -0
  5. theurian/application/migration_engine.py +492 -0
  6. theurian/application/project_service.py +1028 -0
  7. theurian/application/retrieval_service.py +894 -0
  8. theurian/application/setup_context.py +92 -0
  9. theurian/application/setup_service.py +294 -0
  10. theurian/application/setup_steps.py +821 -0
  11. theurian/application/setup_withholding.py +162 -0
  12. theurian/application/visibility.py +224 -0
  13. theurian/cli/__init__.py +0 -0
  14. theurian/cli/auth_commands.py +103 -0
  15. theurian/cli/commands.py +1300 -0
  16. theurian/cli/context.py +194 -0
  17. theurian/cli/index_commands.py +495 -0
  18. theurian/cli/main.py +179 -0
  19. theurian/cli/setup_commands.py +389 -0
  20. theurian/daemon/__init__.py +11 -0
  21. theurian/daemon/instance.py +236 -0
  22. theurian/daemon/runner.py +135 -0
  23. theurian/daemon/server.py +171 -0
  24. theurian/domain/__init__.py +6 -0
  25. theurian/domain/chunking.py +261 -0
  26. theurian/domain/compatibility.py +471 -0
  27. theurian/domain/context.py +61 -0
  28. theurian/domain/enums.py +237 -0
  29. theurian/domain/errors.py +140 -0
  30. theurian/domain/identifiers.py +199 -0
  31. theurian/domain/ingestion.py +131 -0
  32. theurian/domain/knowledge.py +352 -0
  33. theurian/domain/migration.py +447 -0
  34. theurian/domain/ports/__init__.py +74 -0
  35. theurian/domain/ports/authorization.py +47 -0
  36. theurian/domain/ports/canonical_store.py +213 -0
  37. theurian/domain/ports/daemon_manager.py +125 -0
  38. theurian/domain/ports/determinism.py +42 -0
  39. theurian/domain/ports/embedding.py +46 -0
  40. theurian/domain/ports/index_store.py +197 -0
  41. theurian/domain/ports/mcp_client_config.py +82 -0
  42. theurian/domain/ports/object_store.py +38 -0
  43. theurian/domain/ports/reranking.py +42 -0
  44. theurian/domain/ports/review_provider.py +48 -0
  45. theurian/domain/ports/secret_store.py +49 -0
  46. theurian/domain/ports/source_parser.py +71 -0
  47. theurian/domain/ports/specification_provider.py +30 -0
  48. theurian/domain/ports/summarization.py +55 -0
  49. theurian/domain/ports/vector_store.py +58 -0
  50. theurian/domain/project.py +106 -0
  51. theurian/domain/ranking.py +328 -0
  52. theurian/domain/retrieval.py +129 -0
  53. theurian/domain/review.py +229 -0
  54. theurian/domain/setup.py +338 -0
  55. theurian/domain/specification.py +162 -0
  56. theurian/domain/state.py +183 -0
  57. theurian/domain/values.py +220 -0
  58. theurian/indexing/__init__.py +21 -0
  59. theurian/infrastructure/__init__.py +9 -0
  60. theurian/infrastructure/claude/__init__.py +5 -0
  61. theurian/infrastructure/claude/mcp_config.py +304 -0
  62. theurian/infrastructure/determinism.py +87 -0
  63. theurian/infrastructure/embedding/__init__.py +9 -0
  64. theurian/infrastructure/embedding/hashing.py +125 -0
  65. theurian/infrastructure/filesystem/__init__.py +9 -0
  66. theurian/infrastructure/filesystem/migration_loader.py +336 -0
  67. theurian/infrastructure/filesystem/parsers/markdown.py +243 -0
  68. theurian/infrastructure/filesystem/parsers/openapi.py +255 -0
  69. theurian/infrastructure/filesystem/parsers/registry.py +122 -0
  70. theurian/infrastructure/filesystem/parsers/structured.py +160 -0
  71. theurian/infrastructure/git/__init__.py +9 -0
  72. theurian/infrastructure/github/__init__.py +10 -0
  73. theurian/infrastructure/raptor/__init__.py +24 -0
  74. theurian/infrastructure/secrets/__init__.py +10 -0
  75. theurian/infrastructure/secrets/file_store.py +122 -0
  76. theurian/infrastructure/services/__init__.py +57 -0
  77. theurian/infrastructure/services/launchagent.py +299 -0
  78. theurian/infrastructure/services/runner.py +115 -0
  79. theurian/infrastructure/services/systemd_user.py +432 -0
  80. theurian/infrastructure/sqlite/__init__.py +9 -0
  81. theurian/infrastructure/sqlite/connection.py +317 -0
  82. theurian/infrastructure/sqlite/index_query.py +327 -0
  83. theurian/infrastructure/sqlite/index_scan.py +353 -0
  84. theurian/infrastructure/sqlite/index_schema.py +160 -0
  85. theurian/infrastructure/sqlite/index_store.py +1158 -0
  86. theurian/infrastructure/sqlite/schema.py +237 -0
  87. theurian/infrastructure/sqlite/store.py +918 -0
  88. theurian/infrastructure/vector/__init__.py +21 -0
  89. theurian/ingestion/__init__.py +9 -0
  90. theurian/mcp/__init__.py +12 -0
  91. theurian/mcp/results.py +80 -0
  92. theurian/mcp/search.py +777 -0
  93. theurian/mcp/tools.py +636 -0
  94. theurian/migrations/__init__.py +10 -0
  95. theurian/normalization/__init__.py +10 -0
  96. theurian/normalization/projection.py +153 -0
  97. theurian/observability/__init__.py +6 -0
  98. theurian/retrieval/__init__.py +28 -0
  99. theurian/review/__init__.py +9 -0
  100. theurian/schemas/README.md +118 -0
  101. theurian/schemas/cli/version.schema.json +27 -0
  102. theurian/schemas/config/project-config.schema.json +157 -0
  103. theurian/schemas/knowledge/retrieval-result.schema.json +111 -0
  104. theurian/schemas/mcp/knowledge-search-response.schema.json +38 -0
  105. theurian/schemas/mcp/project-list-response.schema.json +48 -0
  106. theurian/schemas/mcp/retrieval-metadata.schema.json +154 -0
  107. theurian/schemas/mcp/tool-context.schema.json +32 -0
  108. theurian/schemas/migrations/migration.schema.json +358 -0
  109. theurian/schemas/protocol/compatibility.schema.json +41 -0
  110. theurian/security/__init__.py +30 -0
  111. theurian/security/env_file.py +31 -0
  112. theurian/security/paths.py +142 -0
  113. theurian/security/tokens.py +112 -0
  114. theurian/security/yaml_loading.py +105 -0
  115. theurian/specification/__init__.py +7 -0
  116. theurian/traceability/__init__.py +13 -0
  117. theurian-0.1.0.dev0.dist-info/METADATA +46 -0
  118. theurian-0.1.0.dev0.dist-info/RECORD +121 -0
  119. theurian-0.1.0.dev0.dist-info/WHEEL +4 -0
  120. theurian-0.1.0.dev0.dist-info/entry_points.txt +2 -0
  121. theurian-0.1.0.dev0.dist-info/licenses/LICENSE +202 -0
theurian/__init__.py ADDED
@@ -0,0 +1,14 @@
1
+ """Theurian: Git-native engineering knowledge for AI agents.
2
+
3
+ The public surface for external consumers is the ``theurian`` CLI, the MCP
4
+ server, and the JSON Schemas under ``schemas/``. Python modules inside this
5
+ package are internal and carry no stability guarantee -- notably, the Claude Code
6
+ plugin must never import them (ADR-0001, CP-2).
7
+ """
8
+
9
+ from theurian.domain.compatibility import CURRENT_PROTOCOL_VERSION
10
+
11
+ __version__ = "0.1.0.dev0"
12
+ __protocol_version__ = CURRENT_PROTOCOL_VERSION
13
+
14
+ __all__ = ["__protocol_version__", "__version__"]
@@ -0,0 +1,11 @@
1
+ """Application layer: use cases and orchestration.
2
+
3
+ Depends on :mod:`theurian.domain` only. Adapters arrive by constructor
4
+ injection; nothing here names a concrete implementation (ADR-0003).
5
+
6
+ Milestone 1 onward: ``SetupService``, ``MigrationService``, ``IndexingService``,
7
+ ``RetrievalService``, ``ReviewService``, ``TraceabilityService``.
8
+
9
+ ``SetupService`` is shared by ``theurian setup`` and ``/theurian:setup``. There
10
+ is exactly one implementation of setup, because two would drift (FR-L1).
11
+ """
@@ -0,0 +1,197 @@
1
+ """Turning a canonical state into an index build (FR-R2, FR-R3, ADR-0022).
2
+
3
+ Split out of :mod:`theurian.application.retrieval_service` for the reason
4
+ :mod:`theurian.application.visibility` was split out of it before: that file had
5
+ grown past the size at which it can be read in one sitting, and this is a seam
6
+ rather than a cut. It described itself as "three use cases over one index file",
7
+ and this is the one that *writes*. Everything left there reads.
8
+
9
+ The seam is real rather than arithmetic. Nothing here consults a
10
+ :class:`~theurian.application.visibility.Visibility`, because at build time there
11
+ is no caller to be visible *to*: the filter that applies is
12
+ :func:`~theurian.domain.enums.may_surface` against the operator's
13
+ ``include_unapproved``, and it decides what is written rather than what is shown.
14
+ The equality property the query side exists to hold has no counterpart here.
15
+
16
+ Takes its collaborators by injection, so a build is testable without a database
17
+ and without an embedding provider.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import asyncio
23
+ from collections.abc import Callable, Sequence
24
+ from dataclasses import dataclass
25
+ from pathlib import Path
26
+ from typing import Final, final
27
+
28
+ from theurian.domain.chunking import IndexableChunk, chunk_document
29
+ from theurian.domain.context import RequestContext
30
+ from theurian.domain.enums import may_surface
31
+ from theurian.domain.identifiers import ProjectId
32
+ from theurian.domain.ports.canonical_store import CanonicalReadSession
33
+ from theurian.domain.ports.embedding import EmbeddingProvider
34
+ from theurian.domain.ports.index_store import IndexStore
35
+
36
+ #: Chunks per embedding request. An API-backed provider caps request size, and a
37
+ #: local one gains nothing from an unbounded batch -- while an unbounded batch
38
+ #: holds the whole corpus and all its vectors in memory at once.
39
+ EMBED_BATCH: Final = 128
40
+
41
+
42
+ @dataclass(frozen=True, slots=True)
43
+ class IndexRequest:
44
+ """What to index, and where to put it."""
45
+
46
+ database: Path
47
+ index_path: Path
48
+ project_id: str
49
+ state_hash: str
50
+ index_build_id: str
51
+ #: Whether unapproved revisions are written at all. Off by default, so an
52
+ #: operator who never opts in has a hard guarantee that no draft is in the
53
+ #: file — not merely that a query filter is expected to hold.
54
+ include_unapproved: bool = False
55
+
56
+
57
+ @final
58
+ class IndexBuilder:
59
+ """Turns a canonical state into an index build."""
60
+
61
+ def __init__(
62
+ self,
63
+ *,
64
+ store_factory: Callable[[Path], CanonicalReadSession],
65
+ index_factory: Callable[[Path], IndexStore],
66
+ embedder: EmbeddingProvider | None = None,
67
+ ) -> None:
68
+ self._store_factory = store_factory
69
+ self._index_factory = index_factory
70
+ self._embedder = embedder
71
+
72
+ def build(self, request: IndexRequest) -> dict[str, object]:
73
+ """Write a new index file from a canonical state.
74
+
75
+ Unapproved revisions are written only when asked for, and `rejected`
76
+ never is.
77
+
78
+ The obvious simplification — index everything, filter at query time —
79
+ was tried and reverted. It makes `includeUnapproved=True` a single
80
+ boolean that reaches content the team decided must not be followed, and
81
+ it removes the operator's ability to guarantee that a draft is not in
82
+ the file at all. The cost is that `includeUnapproved=True` cannot return
83
+ rows that were never written, which is reported rather than hidden:
84
+ `indexesUnapproved` says whether this build can answer such a query.
85
+
86
+ Either a whole index file or none: a build that fails part-way deletes
87
+ what it wrote. That guarantee used to live only in the CLI while
88
+ :meth:`_embed`'s docstring asserted it here, so any other caller of
89
+ `build` — a daemon, a test, a future scheduled rebuild — got a
90
+ half-written file and the promise that it could not happen.
91
+
92
+ **What it wrote, and nothing else.** `IndexStore.create` refuses to
93
+ overwrite an existing file and raises; the cleanup then unlinked the very
94
+ file it had just been refused permission to touch — which is the file
95
+ `active-index.json` names, so a build against an already-taken path
96
+ deleted the published index and left the pointer aimed at nothing. Not
97
+ reachable from `theurian index build`, which mints a fresh ULID per
98
+ build, but `build` is a public application-layer API and `create`'s
99
+ contract says an existing file is left alone.
100
+ """
101
+ # Sampled before `create` rather than inferred from the exception: only a
102
+ # path this call brought into existence may be removed by it.
103
+ preexisting = request.index_path.exists()
104
+ try:
105
+ return self._build(request)
106
+ except Exception:
107
+ # A partial index is worse than none. It looks complete, ranks the
108
+ # fraction it holds, and never surfaces the rest -- which reads as a
109
+ # relevance problem rather than a build failure, and so does not get
110
+ # investigated. Nothing this build wrote is published until it
111
+ # returns, so deleting here loses no index a search could have used.
112
+ if not preexisting:
113
+ request.index_path.unlink(missing_ok=True)
114
+ raise
115
+
116
+ def _build(self, request: IndexRequest) -> dict[str, object]:
117
+ index = self._index_factory(request.index_path)
118
+ index.create(index_build_id=request.index_build_id, state_hash=request.state_hash)
119
+
120
+ context = RequestContext(project_id=ProjectId(request.project_id))
121
+ indexable: list[IndexableChunk] = []
122
+
123
+ with self._store_factory(request.database) as store:
124
+ for item in store.list_items(context):
125
+ # The same authority the search paths consult. Inlined here as
126
+ # two comparisons until `may_surface` moved to the domain, which
127
+ # is one copy of a security rule too many.
128
+ if not may_surface(item.status, include_unapproved=request.include_unapproved):
129
+ continue
130
+ if item.current_revision_id is None:
131
+ continue
132
+ revision = store.get_revision(context, item.current_revision_id)
133
+ if revision is None: # pragma: no cover - the pointer is a foreign key
134
+ continue
135
+
136
+ # The title is prepended to the body before splitting so that a
137
+ # query matching only the title still finds the document. A
138
+ # separately indexed title field would need its own retriever and
139
+ # its own fusion weight for the same effect.
140
+ body = f"{revision.title}\n\n{revision.body}"
141
+ for chunk in chunk_document(revision.revision_id.value, body):
142
+ indexable.append(
143
+ IndexableChunk(
144
+ chunk=chunk,
145
+ project_id=request.project_id,
146
+ item_id=item.item_id.value,
147
+ revision_id=revision.revision_id.value,
148
+ status=item.status.value,
149
+ sensitivity=revision.metadata.sensitivity.value,
150
+ trust_level=revision.metadata.trust_level.value,
151
+ )
152
+ )
153
+
154
+ index.add_chunks(indexable)
155
+ embedded = self._embed(index, indexable)
156
+
157
+ return {
158
+ "indexBuildId": request.index_build_id,
159
+ "stateHash": request.state_hash,
160
+ "indexPath": str(request.index_path),
161
+ "chunks": len(indexable),
162
+ "embeddings": embedded,
163
+ "embeddingModel": self._embedder.model_id if self._embedder else "",
164
+ "indexesUnapproved": request.include_unapproved,
165
+ }
166
+
167
+ def _embed(self, index: IndexStore, indexable: Sequence[IndexableChunk]) -> int:
168
+ """Embed every chunk, or none.
169
+
170
+ Batched, because a real provider caps request size and a local one gains
171
+ nothing from an unbounded batch.
172
+
173
+ A partial embedding is worse than none: the dense retriever would rank
174
+ the embedded half and silently never surface the rest, which looks like
175
+ a relevance problem rather than a build problem. :meth:`build` discards
176
+ the whole index file if any batch raises, so a partial one never exists
177
+ to be published.
178
+ """
179
+ if self._embedder is None or not indexable:
180
+ return 0
181
+
182
+ embedded = 0
183
+ for start in range(0, len(indexable), EMBED_BATCH):
184
+ batch = indexable[start : start + EMBED_BATCH]
185
+ vectors = asyncio.run(self._embedder.embed(tuple(c.chunk.text for c in batch)))
186
+ index.add_embeddings(
187
+ [(c.chunk.chunk_id, v) for c, v in zip(batch, vectors, strict=True)]
188
+ )
189
+ embedded += len(vectors)
190
+
191
+ index.record_embedding_model(
192
+ model_id=self._embedder.model_id, dimension=self._embedder.dimension
193
+ )
194
+ return embedded
195
+
196
+
197
+ __all__ = ["EMBED_BATCH", "IndexBuilder", "IndexRequest"]
@@ -0,0 +1,255 @@
1
+ """Source ingestion (FR-S1 .. FR-S6, ADR-0010).
2
+
3
+ Walks a project's knowledge directory, dispatches each file to a parser, and
4
+ normalizes the result into the Canonical Layer.
5
+
6
+ Two properties shape the design:
7
+
8
+ **Failure is per document.** A malformed YAML file among two hundred must not
9
+ make the other 199 unavailable, so a parse failure becomes a value in the report
10
+ rather than an exception that unwinds the walk.
11
+
12
+ **Unchanged files cost one hash.** Touching a file without changing it must not
13
+ trigger a reparse and a reindex, so content hashes from the previous run are
14
+ compared before any parsing happens.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from collections.abc import Iterator
20
+ from dataclasses import dataclass
21
+ from pathlib import Path, PurePosixPath
22
+ from typing import Protocol
23
+
24
+ from theurian.domain.errors import (
25
+ InputTooLargeError,
26
+ PathEscapeError,
27
+ TheurianError,
28
+ )
29
+ from theurian.domain.ingestion import (
30
+ IngestedDocument,
31
+ IngestionReport,
32
+ ParseFailure,
33
+ ParseWarning,
34
+ )
35
+ from theurian.domain.knowledge import SourceAnchor
36
+ from theurian.domain.ports import NormalizedDocument, SourceParser
37
+ from theurian.domain.values import ContentHash, MediaType
38
+ from theurian.normalization.projection import project
39
+ from theurian.security.paths import MAX_SOURCE_FILE_BYTES, read_source_file
40
+
41
+ #: Directories under `.theurian/` that hold ingestible sources. Derived
42
+ #: directories are excluded: ingesting `generated/` would feed Theurian's own
43
+ #: output back into itself as though it were a source (ADR-0004).
44
+ INGESTIBLE_SUBDIRECTORIES: tuple[str, ...] = ("knowledge", "specifications")
45
+
46
+
47
+ class ParserResolver(Protocol):
48
+ """Resolves a document to its media type and parser.
49
+
50
+ A Protocol rather than a concrete registry so the application layer never
51
+ imports an adapter (ADR-0003).
52
+ """
53
+
54
+ def detect(self, path: PurePosixPath, data: bytes) -> MediaType | None: ...
55
+ def for_media_type(self, media_type: MediaType) -> SourceParser | None: ...
56
+
57
+
58
+ class WarningSource(Protocol):
59
+ """Parsers that surface warnings alongside a normalized document.
60
+
61
+ ``NormalizedDocument`` carries no warning field, and widening the
62
+ ``SourceParser`` port for one parser's benefit would push a Markdown concern
63
+ into every future adapter. Parsers that have warnings expose them here
64
+ instead, and the service asks only those that do (ADR-0019).
65
+ """
66
+
67
+ def warnings_for(self, text: str) -> tuple[ParseWarning, ...]: ...
68
+
69
+
70
+ @dataclass(frozen=True, slots=True)
71
+ class IngestionRequest:
72
+ """One ingestion run's inputs."""
73
+
74
+ project_root: Path
75
+ knowledge_dir: Path
76
+ #: Content hashes from the previous run, keyed by project-relative path.
77
+ #: Anything matching is reported unchanged and never reparsed.
78
+ known_hashes: dict[str, str]
79
+ #: Commit the sources were read at, so every anchor pins an immutable
80
+ #: object rather than a path that may since have moved (FR-S3).
81
+ commit_sha: str | None = None
82
+ repository: str | None = None
83
+
84
+
85
+ class IngestionService:
86
+ """Normalizes a project's sources into canonical documents."""
87
+
88
+ def __init__(self, resolver: ParserResolver) -> None:
89
+ self._resolver = resolver
90
+
91
+ def ingest(self, request: IngestionRequest) -> IngestionReport:
92
+ """Walk and normalize every ingestible source.
93
+
94
+ Never raises for a document-level problem. A caller inspects
95
+ ``report.failures`` and decides; the run itself always completes.
96
+ """
97
+ report = IngestionReport()
98
+
99
+ for path in self._discover(request.knowledge_dir):
100
+ relative = path.relative_to(request.project_root).as_posix()
101
+ try:
102
+ self._ingest_one(relative, request, report)
103
+ except TheurianError as exc:
104
+ # Security and limit refusals are per-document too. One hostile
105
+ # file must not prevent the rest of the tree from being read.
106
+ report.failures.append(ParseFailure(path=relative, reason=str(exc)))
107
+
108
+ return report
109
+
110
+ def _ingest_one(
111
+ self, relative: str, request: IngestionRequest, report: IngestionReport
112
+ ) -> None:
113
+ try:
114
+ data = read_source_file(request.project_root, PurePosixPath(relative))
115
+ except (PathEscapeError, InputTooLargeError) as exc:
116
+ report.failures.append(ParseFailure(path=relative, reason=str(exc)))
117
+ return
118
+ except OSError as exc:
119
+ report.failures.append(ParseFailure(path=relative, reason=f"unreadable: {exc}"))
120
+ return
121
+
122
+ content_hash = ContentHash.of_bytes(data)
123
+ if request.known_hashes.get(relative) == content_hash.value:
124
+ # The cheap early exit: touching a file without changing it costs
125
+ # one hash, not a reparse and a reindex.
126
+ report.unchanged.append(relative)
127
+ return
128
+
129
+ media_type = self._resolver.detect(PurePosixPath(relative), data)
130
+ if media_type is None:
131
+ report.skipped.append(relative)
132
+ return
133
+
134
+ parser = self._resolver.for_media_type(media_type)
135
+ if parser is None:
136
+ report.failures.append(
137
+ ParseFailure(
138
+ path=relative,
139
+ reason=f"no parser registered for {media_type}",
140
+ media_type=media_type.value,
141
+ )
142
+ )
143
+ return
144
+
145
+ anchor = _anchor(relative, request)
146
+
147
+ try:
148
+ normalized = parser.parse(data, media_type=media_type, anchor=anchor)
149
+ except (ValueError, InputTooLargeError) as exc:
150
+ report.failures.append(
151
+ ParseFailure(path=relative, reason=str(exc), media_type=media_type.value)
152
+ )
153
+ return
154
+
155
+ report.documents.append(
156
+ _to_document(
157
+ normalized,
158
+ path=relative,
159
+ source_hash=content_hash,
160
+ parser=parser,
161
+ warnings=_warnings(parser, data),
162
+ )
163
+ )
164
+
165
+ def _discover(self, knowledge_dir: Path) -> Iterator[Path]:
166
+ """Yield candidate source files in a stable order.
167
+
168
+ Sorted so a failure reports the same first offender on every run rather
169
+ than whichever the filesystem happened to yield first.
170
+ """
171
+ for subdirectory in INGESTIBLE_SUBDIRECTORIES:
172
+ root = knowledge_dir / subdirectory
173
+ if not root.is_dir():
174
+ continue
175
+ for path in sorted(root.rglob("*")):
176
+ if not path.is_file() or path.name.startswith("."):
177
+ continue
178
+ # A symlink is refused later by read_source_file if it escapes;
179
+ # skipping obviously-oversized files here avoids reading them.
180
+ if path.stat().st_size > MAX_SOURCE_FILE_BYTES:
181
+ continue
182
+ yield path
183
+
184
+
185
+ def _warnings(parser: SourceParser, data: bytes) -> tuple[ParseWarning, ...]:
186
+ """Collect warnings from parsers that produce them."""
187
+ collector = getattr(parser, "warnings_for", None)
188
+ if collector is None:
189
+ return ()
190
+ try:
191
+ text = data.decode("utf-8")
192
+ except UnicodeDecodeError: # pragma: no cover - parse would have failed first
193
+ return ()
194
+ result: tuple[ParseWarning, ...] = collector(text)
195
+ return result
196
+
197
+
198
+ def _anchor(relative: str, request: IngestionRequest) -> SourceAnchor:
199
+ """Build the anchor that makes this document traceable (FR-S3).
200
+
201
+ A commit SHA plus a path pins an immutable Git object, so the anchor still
202
+ resolves after the file is edited, moved, or deleted.
203
+ """
204
+ return SourceAnchor(
205
+ provider="git" if request.commit_sha else "filesystem",
206
+ source_uri=f"git://{request.repository or 'local'}/{relative}"
207
+ if request.commit_sha
208
+ else f"file://{relative}",
209
+ repository=request.repository,
210
+ commit_sha=request.commit_sha,
211
+ file_path=relative,
212
+ )
213
+
214
+
215
+ def _to_document(
216
+ normalized: NormalizedDocument,
217
+ *,
218
+ path: str,
219
+ source_hash: ContentHash,
220
+ parser: SourceParser,
221
+ warnings: tuple[ParseWarning, ...],
222
+ ) -> IngestedDocument:
223
+ projection: str | None = None
224
+ if normalized.structured is not None:
225
+ projection = project(normalized.structured)
226
+
227
+ return IngestedDocument(
228
+ path=path,
229
+ title=normalized.title,
230
+ body=normalized.body,
231
+ content_type=normalized.content_type,
232
+ content_hash=normalized.content_hash,
233
+ source_hash=source_hash,
234
+ anchors=normalized.anchors,
235
+ parser_id=parser.parser_id,
236
+ structured=normalized.structured,
237
+ text_projection=projection,
238
+ warnings=warnings,
239
+ )
240
+
241
+
242
+ def manifest_from(report: IngestionReport, previous: dict[str, str]) -> dict[str, str]:
243
+ """Build the content-hash manifest for the next run's early exit.
244
+
245
+ Carries forward hashes for unchanged files, because those were never
246
+ reparsed and so are absent from ``report.documents``. Dropping them would
247
+ make every second run a full reparse.
248
+ """
249
+ manifest = {path: previous[path] for path in report.unchanged if path in previous}
250
+ for document in report.documents:
251
+ # The *source* hash, matching what the early exit computes. Storing the
252
+ # body hash instead makes every Markdown file with front matter reparse
253
+ # on every run, because its body and its bytes differ.
254
+ manifest[document.path] = document.source_hash.value
255
+ return manifest