open-code-review-toolkit 0.4.7__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/PKG-INFO +4 -4
  2. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/README.md +3 -3
  3. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/_version.py +2 -2
  4. open_code_review_toolkit-0.6.0/src/ocr_toolkit/common/filesystem.py +19 -0
  5. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/collect.py +25 -9
  6. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/collectors/__init__.py +24 -0
  7. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/collectors/graphs.py +413 -0
  8. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/collectors/orchestration.py +530 -0
  9. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/collectors/projections.py +117 -0
  10. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/collectors/registry.py +138 -0
  11. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/collectors/sources.py +72 -0
  12. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/__init__.py +1 -0
  13. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/ansible/__init__.py +1 -0
  14. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/go_manifests.py → open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/go.py +1 -1
  15. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/javascript_manifests.py → open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/javascript.py +1 -1
  16. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/composer_manifests.py → open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/php.py +1 -1
  17. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/python_manifests.py → open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/python.py +1 -1
  18. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/__init__.py +42 -0
  19. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/contracts.py +81 -0
  20. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/detection.py +635 -0
  21. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/providers/__init__.py +8 -0
  22. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/providers/frontend.py +35 -0
  23. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/providers/go.py +22 -0
  24. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/providers/php.py +25 -0
  25. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/providers/python.py +19 -0
  26. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/registry.py +165 -0
  27. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/schema.py +275 -0
  28. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/frameworks/templates.py +144 -0
  29. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/infrastructure.py +1 -1
  30. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/mcp.py +88 -20
  31. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/model.py +51 -2
  32. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/__init__.py +26 -0
  33. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/contracts.py +92 -0
  34. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/decisions.py +176 -0
  35. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/guidance.py +146 -0
  36. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/registry.py +20 -0
  37. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/schema.py +217 -0
  38. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/policy/scopes.py +81 -0
  39. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/project.py +217 -0
  40. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/store/__init__.py +9 -0
  41. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/store/atomic.py +46 -0
  42. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/store/contracts.py +74 -0
  43. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/store/core.py +341 -0
  44. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/store/readback.py +242 -0
  45. open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/store/values.py +73 -0
  46. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/ocr_result.py +2 -15
  47. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/formatting.py +130 -48
  48. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/settings.py +11 -0
  49. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/workflow.py +8 -1
  50. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/preflight.py +1 -1
  51. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/collectors.py +0 -867
  52. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/project.py +0 -127
  53. open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/store.py +0 -448
  54. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/.gitignore +0 -0
  55. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/LICENSE +0 -0
  56. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/pyproject.toml +0 -0
  57. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/__init__.py +0 -0
  58. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/cli.py +0 -0
  59. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/common/__init__.py +0 -0
  60. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/common/git.py +0 -0
  61. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/common/language.py +0 -0
  62. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/common/markdown.py +0 -0
  63. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/common/redaction.py +0 -0
  64. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/config_writer.py +0 -0
  65. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/configure.py +0 -0
  66. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/__init__.py +0 -0
  67. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/__main__.py +0 -0
  68. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/artifacts.py +0 -0
  69. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/categorize.py +0 -0
  70. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/coverage.py +0 -0
  71. /open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/ansible_requirements.py → /open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/ansible/requirements.py +0 -0
  72. /open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/ansible.py → /open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/ansible/topology.py +0 -0
  73. /open_code_review_toolkit-0.4.7/src/ocr_toolkit/evidence/manifest_model.py → /open_code_review_toolkit-0.6.0/src/ocr_toolkit/evidence/ecosystems/contracts.py +0 -0
  74. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/invocation.py +0 -0
  75. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/evidence/repository.py +0 -0
  76. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/mcp_config.py +0 -0
  77. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/__init__.py +0 -0
  78. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/__main__.py +0 -0
  79. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/approval.py +0 -0
  80. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/comments.py +0 -0
  81. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/gitlab.py +0 -0
  82. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/gitlab_approval.py +0 -0
  83. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/markers.py +0 -0
  84. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/payloads.py +0 -0
  85. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/result.py +0 -0
  86. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/snapshot.py +0 -0
  87. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/posting/suggestions.py +0 -0
  88. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/providers/__init__.py +0 -0
  89. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/providers/gitlab.py +0 -0
  90. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/py.typed +0 -0
  91. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/result_contract.py +0 -0
  92. {open_code_review_toolkit-0.4.7 → open_code_review_toolkit-0.6.0}/src/ocr_toolkit/review_runner.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: open-code-review-toolkit
3
- Version: 0.4.7
3
+ Version: 0.6.0
4
4
  Summary: Unofficial GitLab CI integration layer for Open Code Review
5
5
  Project-URL: Homepage, https://github.com/xeonvs/open-code-review-toolkit
6
6
  Project-URL: Repository, https://github.com/xeonvs/open-code-review-toolkit
@@ -243,7 +243,7 @@ ocr --version
243
243
  ocr-ci --help
244
244
  ```
245
245
 
246
- The current compatibility target is OCR `1.9.1`. CI should pin the release and verify its published checksum before execution.
246
+ The exact recommended OCR release and its verified asset checksums live in the [versioned compatibility manifest](compatibility/ocr-support.json). CI should pin that release and checksum before execution.
247
247
  The [versioned compatibility policy](docs/compatibility.md) records tested assets and evidence and describes the conservative Dependabot-like qualification workflow for later upstream releases.
248
248
  Review output defaults to English. `OCR_REVIEW_LANGUAGE` accepts another explicit language name when a project needs localized review output; for example, `OCR_REVIEW_LANGUAGE=Russian`.
249
249
 
@@ -262,11 +262,11 @@ must remain comment-only. GitLab approval rules and protected-branch policy
262
262
  remain authoritative. The toolkit only adds an eligible approval; it never
263
263
  removes an existing approval when a later review is ineligible or disabled.
264
264
 
265
- Project-wide accepted tradeoffs can be recorded separately in `.opencodereview/accepted-decisions.md`; the evidence collector supplies target-ref decisions to OCR and never lets a source change self-authorize its own review. See [Accepted project decisions](docs/configuration.md#accepted-project-decisions) for the entry format, inline marker convention, security boundary, and limitations.
265
+ Accepted tradeoffs can be recorded in `.opencodereview/accepted-decisions.md`; the evidence collector supplies only applicable target-ref decisions and never lets a source change self-authorize its review. Root and nested target `AGENTS.md`/`CLAUDE.md` guidance is similarly exposed through the existing evidence MCP with deterministic scope and precedence, while any guidance touched by the merge request is excluded. See [Accepted project decisions](docs/configuration.md#accepted-project-decisions) and [Target project guidance](docs/configuration.md#target-project-guidance) for formats and trust boundaries.
266
266
 
267
267
  ## Project architecture
268
268
 
269
- The shipped Repository Evidence Engine reads immutable base/head Git objects, stores bounded typed facts and deltas, creates the compact bootstrap used by OCR, and exposes detailed evidence through the mandatory built-in read-only MCP server. Reviewed external stdio or native HTTPS MCP servers compose alongside it without replacing the built-in evidence boundary.
269
+ The shipped Repository Evidence Engine reads immutable base/head Git objects, stores bounded typed facts and deltas, creates the compact bootstrap used by OCR, and exposes detailed facts, scoped completeness, and base/head changes through the mandatory built-in read-only MCP server. Reviewed external stdio or native HTTPS MCP servers compose alongside it without replacing the built-in evidence boundary.
270
270
 
271
271
  - [Toolkit strategy](docs/engineering/toolkit_strategy.md) - durable product boundaries, architecture, invariants, and non-goals.
272
272
  - [Roadmap](ROADMAP.md) - milestone status, dependencies, outcomes, and completion signals.
@@ -17,7 +17,7 @@ ocr --version
17
17
  ocr-ci --help
18
18
  ```
19
19
 
20
- The current compatibility target is OCR `1.9.1`. CI should pin the release and verify its published checksum before execution.
20
+ The exact recommended OCR release and its verified asset checksums live in the [versioned compatibility manifest](compatibility/ocr-support.json). CI should pin that release and checksum before execution.
21
21
  The [versioned compatibility policy](docs/compatibility.md) records tested assets and evidence and describes the conservative Dependabot-like qualification workflow for later upstream releases.
22
22
  Review output defaults to English. `OCR_REVIEW_LANGUAGE` accepts another explicit language name when a project needs localized review output; for example, `OCR_REVIEW_LANGUAGE=Russian`.
23
23
 
@@ -36,11 +36,11 @@ must remain comment-only. GitLab approval rules and protected-branch policy
36
36
  remain authoritative. The toolkit only adds an eligible approval; it never
37
37
  removes an existing approval when a later review is ineligible or disabled.
38
38
 
39
- Project-wide accepted tradeoffs can be recorded separately in `.opencodereview/accepted-decisions.md`; the evidence collector supplies target-ref decisions to OCR and never lets a source change self-authorize its own review. See [Accepted project decisions](docs/configuration.md#accepted-project-decisions) for the entry format, inline marker convention, security boundary, and limitations.
39
+ Accepted tradeoffs can be recorded in `.opencodereview/accepted-decisions.md`; the evidence collector supplies only applicable target-ref decisions and never lets a source change self-authorize its review. Root and nested target `AGENTS.md`/`CLAUDE.md` guidance is similarly exposed through the existing evidence MCP with deterministic scope and precedence, while any guidance touched by the merge request is excluded. See [Accepted project decisions](docs/configuration.md#accepted-project-decisions) and [Target project guidance](docs/configuration.md#target-project-guidance) for formats and trust boundaries.
40
40
 
41
41
  ## Project architecture
42
42
 
43
- The shipped Repository Evidence Engine reads immutable base/head Git objects, stores bounded typed facts and deltas, creates the compact bootstrap used by OCR, and exposes detailed evidence through the mandatory built-in read-only MCP server. Reviewed external stdio or native HTTPS MCP servers compose alongside it without replacing the built-in evidence boundary.
43
+ The shipped Repository Evidence Engine reads immutable base/head Git objects, stores bounded typed facts and deltas, creates the compact bootstrap used by OCR, and exposes detailed facts, scoped completeness, and base/head changes through the mandatory built-in read-only MCP server. Reviewed external stdio or native HTTPS MCP servers compose alongside it without replacing the built-in evidence boundary.
44
44
 
45
45
  - [Toolkit strategy](docs/engineering/toolkit_strategy.md) - durable product boundaries, architecture, invariants, and non-goals.
46
46
  - [Roadmap](ROADMAP.md) - milestone status, dependencies, outcomes, and completion signals.
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.4.7'
22
- __version_tuple__ = version_tuple = (0, 4, 7)
21
+ __version__ = version = '0.6.0'
22
+ __version_tuple__ = version_tuple = (0, 6, 0)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -0,0 +1,19 @@
1
+ """Small cross-platform filesystem durability helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import errno
6
+ import os
7
+
8
+
9
+ def fsync_directory(descriptor: int) -> None:
10
+ """Persist a directory-entry change when the platform supports it."""
11
+
12
+ try:
13
+ os.fsync(descriptor)
14
+ except OSError as exc:
15
+ # Some supported filesystems reject directory fsync even though the
16
+ # atomic rename itself succeeded. Ignore only documented unsupported
17
+ # descriptor/filesystem cases; propagate genuine durability failures.
18
+ if exc.errno not in {errno.EINVAL, errno.ENOTSUP, errno.EBADF}:
19
+ raise
@@ -86,14 +86,9 @@ def collect_repository_evidence(
86
86
  coverage=tuple(head_coverage),
87
87
  )
88
88
  all_coverage = tuple((*base.coverage, *head.coverage))
89
- store = EvidenceStore(base=base, head=head, deltas=file_deltas(base, head))
89
+ snapshot_deltas = file_deltas(base, head)
90
+ store = EvidenceStore(base=base, head=head)
90
91
  typed_facts = [*base_facts, *head_facts]
91
- store.deltas = tuple(
92
- sorted(
93
- (*store.deltas, *fact_deltas(typed_facts), *coverage_deltas(all_coverage)),
94
- key=lambda item: (item.kind, item.component, item.identity),
95
- )
96
- )
97
92
  rejected_snapshot_records = [
98
93
  record for record in (*base.records, *head.records) if not store.add(record)
99
94
  ]
@@ -117,14 +112,35 @@ def collect_repository_evidence(
117
112
  record.id,
118
113
  ),
119
114
  )
115
+ exhausted_kinds: set[str] = set()
120
116
  for record in ordered_typed_facts:
117
+ if record.kind in exhausted_kinds:
118
+ continue
121
119
  if not store.add(record):
122
120
  if record.component == "ansible" and record.kind.startswith("ansible."):
123
121
  raise EvidenceStoreError(
124
122
  "Ansible topology facts exceed the atomic evidence store limits"
125
123
  )
126
- store.add_diagnostic("typed evidence was truncated by store limits")
127
- break
124
+ limit_state = store.record_limit_state(record.kind)
125
+ if limit_state == "global":
126
+ store.add_diagnostic("typed evidence was truncated by the global store limit")
127
+ break
128
+ if limit_state == "kind":
129
+ # A per-kind omission must not suppress later independent domains.
130
+ store.add_diagnostic(f"typed {record.kind} evidence was truncated by store limits")
131
+ exhausted_kinds.add(record.kind)
132
+ # Deltas are projections of canonical accepted store records, never raw facts
133
+ # or references to values that redaction, deduplication, or a budget omitted.
134
+ store.deltas = tuple(
135
+ sorted(
136
+ (
137
+ *snapshot_deltas,
138
+ *fact_deltas(store.records),
139
+ *coverage_deltas(all_coverage),
140
+ ),
141
+ key=lambda item: (item.kind, item.component, item.identity),
142
+ )
143
+ )
128
144
  categories = categorize_paths(list(changed))
129
145
  categories_truncated = False
130
146
  for category, paths in sorted(categories.items()):
@@ -0,0 +1,24 @@
1
+ """Bounded immutable source collection with explicit responsibility modules."""
2
+
3
+ from ocr_toolkit.evidence.collectors.graphs import (
4
+ MAX_MANIFEST_INCLUDE_DIAGNOSTICS,
5
+ MAX_MANIFEST_INCLUDE_EDGES,
6
+ MAX_MANIFEST_INCLUDE_FILES,
7
+ )
8
+ from ocr_toolkit.evidence.collectors.orchestration import (
9
+ MAX_TOPOLOGY_FACTS_PER_KIND,
10
+ collect_ref_facts,
11
+ )
12
+ from ocr_toolkit.evidence.collectors.projections import fact_deltas
13
+ from ocr_toolkit.evidence.collectors.registry import manifest_collector, parse_manifest
14
+
15
+ __all__ = [
16
+ "MAX_MANIFEST_INCLUDE_DIAGNOSTICS",
17
+ "MAX_MANIFEST_INCLUDE_EDGES",
18
+ "MAX_MANIFEST_INCLUDE_FILES",
19
+ "MAX_TOPOLOGY_FACTS_PER_KIND",
20
+ "collect_ref_facts",
21
+ "fact_deltas",
22
+ "manifest_collector",
23
+ "parse_manifest",
24
+ ]
@@ -0,0 +1,413 @@
1
+ """Read bounded local manifest include graphs from immutable repository objects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from dataclasses import dataclass
7
+ from pathlib import PurePosixPath
8
+
9
+ from ocr_toolkit.evidence.ecosystems.ansible.requirements import parse_galaxy_requirements
10
+ from ocr_toolkit.evidence.ecosystems.python import parse_requirements
11
+ from ocr_toolkit.evidence.repository import GitRepositoryReader, RepositoryObject
12
+
13
+ MAX_MANIFEST_INCLUDE_FILES = 32
14
+ MAX_MANIFEST_INCLUDE_DEPTH = 8
15
+ MAX_MANIFEST_INCLUDE_DIAGNOSTICS = 64
16
+ MAX_MANIFEST_INCLUDE_EDGES = 4_096
17
+
18
+
19
+ @dataclass(frozen=True, slots=True)
20
+ class ManifestBlobSet:
21
+ """Return immutable Galaxy blobs, diagnostics, and affected graph roots."""
22
+
23
+ blobs: dict[str, bytes]
24
+ galaxy_paths: tuple[str, ...]
25
+ diagnostics: tuple[str, ...]
26
+ degraded_roots: tuple[tuple[str, str], ...] = ()
27
+
28
+
29
+ @dataclass(frozen=True, slots=True)
30
+ class PythonRequirementBlobSet:
31
+ """Return immutable requirements blobs, diagnostics, and affected roots."""
32
+
33
+ blobs: dict[str, bytes]
34
+ requirement_paths: tuple[str, ...]
35
+ diagnostics: tuple[str, ...]
36
+ degraded_roots: tuple[tuple[str, str], ...] = ()
37
+
38
+
39
+ def resolve_manifest_include(
40
+ path: str, include_path: str, *, suffixes: tuple[str, ...]
41
+ ) -> str | None:
42
+ """Resolve a local manifest include inside the immutable repository tree."""
43
+
44
+ if (
45
+ not include_path
46
+ or include_path.startswith(("/", "~"))
47
+ or "\x00" in include_path
48
+ or ":" in include_path
49
+ or "\\" in include_path
50
+ ):
51
+ return None
52
+ parts: list[str] = []
53
+ for part in (PurePosixPath(path).parent / include_path).parts:
54
+ if part in {"", "."}:
55
+ continue
56
+ if part == "..":
57
+ if not parts:
58
+ return None
59
+ parts.pop()
60
+ else:
61
+ parts.append(part)
62
+ resolved = "/".join(parts)
63
+ return resolved if resolved.casefold().endswith(suffixes) else None
64
+
65
+
66
+ def include_cycle_diagnostics(edges: Mapping[str, tuple[str, ...]]) -> tuple[str, ...]:
67
+ """Describe one canonical closing edge per cyclic Galaxy component."""
68
+
69
+ nodes = set(edges)
70
+ nodes.update(target for targets in edges.values() for target in targets)
71
+ visited: set[str] = set()
72
+ finish_order: list[str] = []
73
+ for root in sorted(nodes):
74
+ if root in visited:
75
+ continue
76
+ visited.add(root)
77
+ traversal: list[tuple[str, bool]] = [(root, False)]
78
+ while traversal:
79
+ path, expanded = traversal.pop()
80
+ if expanded:
81
+ finish_order.append(path)
82
+ continue
83
+ traversal.append((path, True))
84
+ for target in reversed(sorted(edges.get(path, ()))):
85
+ if target not in visited:
86
+ visited.add(target)
87
+ traversal.append((target, False))
88
+
89
+ reverse_edges: dict[str, list[str]] = {path: [] for path in nodes}
90
+ for path, targets in edges.items():
91
+ for target in targets:
92
+ reverse_edges[target].append(path)
93
+ components: list[tuple[str, ...]] = []
94
+ assigned: set[str] = set()
95
+ for root in reversed(finish_order):
96
+ if root in assigned:
97
+ continue
98
+ component: list[str] = []
99
+ component_stack = [root]
100
+ assigned.add(root)
101
+ while component_stack:
102
+ path = component_stack.pop()
103
+ component.append(path)
104
+ for source in reversed(sorted(reverse_edges[path])):
105
+ if source not in assigned:
106
+ assigned.add(source)
107
+ component_stack.append(source)
108
+ components.append(tuple(component))
109
+
110
+ diagnostics: list[str] = []
111
+
112
+ def component_key(path: str) -> tuple[int, str, str]:
113
+ """Order graph paths by repository depth and stable spelling."""
114
+
115
+ return path.count("/"), path.casefold(), path
116
+
117
+ for component in sorted(components, key=lambda item: min(component_key(path) for path in item)):
118
+ members = set(component)
119
+ anchor = min(component, key=component_key)
120
+ if len(component) == 1 and anchor not in edges.get(anchor, ()):
121
+ continue
122
+ # Every predecessor inside a strongly connected component closes a
123
+ # path back to the canonical anchor; selecting one makes diagnostics
124
+ # independent from root discovery and traversal order.
125
+ source = min(
126
+ (path for path in members if anchor in edges.get(path, ())),
127
+ key=component_key,
128
+ )
129
+ diagnostics.append(f"{source}: Ansible Galaxy include cycle skipped: {anchor}")
130
+ return tuple(diagnostics)
131
+
132
+
133
+ def bound_include_diagnostics(
134
+ diagnostics: list[str],
135
+ *,
136
+ truncation_notice: str = "Ansible Galaxy include diagnostics were truncated",
137
+ ) -> tuple[str, ...]:
138
+ """Cap graph diagnostics and retain one explicit truncation notice."""
139
+
140
+ if len(diagnostics) <= MAX_MANIFEST_INCLUDE_DIAGNOSTICS:
141
+ return tuple(diagnostics)
142
+ return (
143
+ *diagnostics[: MAX_MANIFEST_INCLUDE_DIAGNOSTICS - 1],
144
+ truncation_notice,
145
+ )
146
+
147
+
148
+ def roots_reaching_graph_degradation(
149
+ roots: tuple[str, ...],
150
+ edges: Mapping[str, tuple[str, ...]],
151
+ degraded_paths: Mapping[str, str],
152
+ ) -> tuple[tuple[str, str], ...]:
153
+ """Return roots whose accepted graph reaches a bounded degraded source."""
154
+
155
+ supported_reasons = {"bounded-source-omission", "include-graph-truncation"}
156
+ if any(reason not in supported_reasons for reason in degraded_paths.values()):
157
+ raise ValueError("include graph has an unsupported degradation reason")
158
+ affected: list[tuple[str, str]] = []
159
+ for root in sorted(set(roots)):
160
+ pending = [root]
161
+ visited: set[str] = set()
162
+ reasons: set[str] = set()
163
+ while pending:
164
+ path = pending.pop()
165
+ if path in visited:
166
+ continue
167
+ visited.add(path)
168
+ reason = degraded_paths.get(path)
169
+ if reason is not None:
170
+ reasons.add(reason)
171
+ pending.extend(reversed(edges.get(path, ())))
172
+ if reasons:
173
+ # A bounded omission is stronger than a traversal/item limit because
174
+ # the source itself was never parsed.
175
+ reason = (
176
+ "bounded-source-omission"
177
+ if "bounded-source-omission" in reasons
178
+ else "include-graph-truncation"
179
+ )
180
+ affected.append((root, reason))
181
+ return tuple(affected)
182
+
183
+
184
+ def read_manifest_graph(
185
+ reader: GitRepositoryReader,
186
+ entries_by_path: Mapping[str, RepositoryObject],
187
+ initial_paths: tuple[str, ...],
188
+ initial_blobs: dict[str, bytes],
189
+ ) -> ManifestBlobSet:
190
+ """Read bounded Galaxy includes in one immutable Git batch per graph depth."""
191
+
192
+ blobs = dict(initial_blobs)
193
+ diagnostics: list[str] = []
194
+ visited: set[str] = set()
195
+ admitted = set(initial_paths)
196
+ root_paths = set(initial_paths)
197
+ edges: dict[str, list[str]] = {}
198
+ degraded_paths: dict[str, str] = {}
199
+ pending = [(path, "") for path in initial_paths]
200
+ included_files = 0
201
+ file_limit_reported = False
202
+ included_edges = 0
203
+ edge_limit_reported = False
204
+ for depth in range(MAX_MANIFEST_INCLUDE_DEPTH + 1):
205
+ if not pending:
206
+ break
207
+ level_sources: dict[str, list[str]] = {}
208
+ for path, included_from in pending:
209
+ level_sources.setdefault(path, []).append(included_from)
210
+ level = [
211
+ (path, tuple(dict.fromkeys(level_sources[path]))) for path in sorted(level_sources)
212
+ ]
213
+ pending = []
214
+ to_read: list[RepositoryObject] = []
215
+ process_paths: list[str] = []
216
+ for path, included_from_values in level:
217
+ if path in visited:
218
+ entry = entries_by_path.get(path)
219
+ if entry is None or entry.is_symlink or entry.is_submodule:
220
+ for source in included_from_values:
221
+ diagnostics.append(
222
+ f"{source or path}: Ansible Galaxy include is missing: {path}"
223
+ )
224
+ continue
225
+ included_from = next((value for value in included_from_values if value), "")
226
+ entry = entries_by_path.get(path)
227
+ if entry is None or entry.is_symlink or entry.is_submodule:
228
+ for source in included_from_values:
229
+ diagnostics.append(f"{source}: Ansible Galaxy include is missing: {path}")
230
+ visited.add(path)
231
+ continue
232
+ if path not in root_paths and path not in admitted:
233
+ if included_files >= MAX_MANIFEST_INCLUDE_FILES:
234
+ if not file_limit_reported:
235
+ diagnostics.append(
236
+ f"{included_from}: Ansible Galaxy includes were truncated after "
237
+ f"{MAX_MANIFEST_INCLUDE_FILES} files"
238
+ )
239
+ file_limit_reported = True
240
+ degraded_paths[path] = "include-graph-truncation"
241
+ continue
242
+ admitted.add(path)
243
+ included_files += 1
244
+ if path not in blobs:
245
+ to_read.append(entry)
246
+ process_paths.append(path)
247
+ if to_read:
248
+ read = reader.read_candidate_blobs(tuple(to_read))
249
+ blobs.update(read.blobs)
250
+ diagnostics.extend(read.diagnostics)
251
+ for path in (entry.path for entry in to_read if entry.path not in read.blobs):
252
+ degraded_paths[path] = "bounded-source-omission"
253
+ for path in process_paths:
254
+ visited.add(path)
255
+ blob = blobs.get(path)
256
+ if blob is None:
257
+ continue
258
+ try:
259
+ parsed = parse_galaxy_requirements(blob.decode("utf-8"))
260
+ if any("truncated" in notice for notice in parsed.notices):
261
+ degraded_paths[path] = "include-graph-truncation"
262
+ except UnicodeDecodeError:
263
+ diagnostics.append(f"{path}: Ansible Galaxy include is not UTF-8")
264
+ continue
265
+ for include_path in parsed.include_paths:
266
+ resolved = resolve_manifest_include(path, include_path, suffixes=(".yml", ".yaml"))
267
+ if resolved is None:
268
+ diagnostics.append(f"{path}: invalid Ansible Galaxy include skipped")
269
+ continue
270
+ if included_edges >= MAX_MANIFEST_INCLUDE_EDGES:
271
+ if not edge_limit_reported:
272
+ diagnostics.append(
273
+ f"{path}: Ansible Galaxy include graph was truncated after "
274
+ f"{MAX_MANIFEST_INCLUDE_EDGES} edges"
275
+ )
276
+ edge_limit_reported = True
277
+ degraded_paths[path] = "include-graph-truncation"
278
+ continue
279
+ included_edges += 1
280
+ edges.setdefault(path, []).append(resolved)
281
+ if depth >= MAX_MANIFEST_INCLUDE_DEPTH:
282
+ diagnostics.append(
283
+ f"{path}: Ansible Galaxy include depth exceeded at {resolved}"
284
+ )
285
+ degraded_paths[path] = "include-graph-truncation"
286
+ else:
287
+ pending.append((resolved, path))
288
+ normalized_edges = {path: tuple(dict.fromkeys(targets)) for path, targets in edges.items()}
289
+ diagnostics.extend(include_cycle_diagnostics(normalized_edges))
290
+ galaxy_paths = tuple(sorted(path for path in visited if path in blobs))
291
+ return ManifestBlobSet(
292
+ blobs,
293
+ galaxy_paths,
294
+ bound_include_diagnostics(diagnostics),
295
+ roots_reaching_graph_degradation(initial_paths, normalized_edges, degraded_paths),
296
+ )
297
+
298
+
299
+ def read_python_requirement_graph(
300
+ reader: GitRepositoryReader,
301
+ entries_by_path: Mapping[str, RepositoryObject],
302
+ initial_paths: tuple[str, ...],
303
+ initial_blobs: dict[str, bytes],
304
+ ) -> PythonRequirementBlobSet:
305
+ """Read bounded local requirements includes from one immutable Git ref."""
306
+
307
+ blobs = dict(initial_blobs)
308
+ diagnostics: list[str] = []
309
+ visited: set[str] = set()
310
+ admitted = set(initial_paths)
311
+ edges: dict[str, list[str]] = {}
312
+ degraded_paths: dict[str, str] = {}
313
+ pending = [(path, "") for path in initial_paths]
314
+ included_files = 0
315
+ included_edges = 0
316
+ file_limit_reported = False
317
+ edge_limit_reported = False
318
+ for depth in range(MAX_MANIFEST_INCLUDE_DEPTH + 1):
319
+ if not pending:
320
+ break
321
+ level_sources: dict[str, list[str]] = {}
322
+ for path, included_from in pending:
323
+ level_sources.setdefault(path, []).append(included_from)
324
+ pending = []
325
+ to_read: list[RepositoryObject] = []
326
+ process_paths: list[str] = []
327
+ for path in sorted(level_sources):
328
+ if path in visited:
329
+ entry = entries_by_path.get(path)
330
+ if entry is None or entry.is_symlink or entry.is_submodule:
331
+ for include_source in tuple(dict.fromkeys(level_sources[path])):
332
+ diagnostics.append(
333
+ f"{include_source or path}: Python requirements include is missing: "
334
+ f"{path}"
335
+ )
336
+ continue
337
+ sources = tuple(dict.fromkeys(level_sources[path]))
338
+ source = next((value for value in sources if value), path)
339
+ entry = entries_by_path.get(path)
340
+ if entry is None or entry.is_symlink or entry.is_submodule:
341
+ for include_source in sources:
342
+ diagnostics.append(
343
+ f"{include_source or path}: Python requirements include is missing: {path}"
344
+ )
345
+ visited.add(path)
346
+ continue
347
+ if path not in admitted:
348
+ if included_files >= MAX_MANIFEST_INCLUDE_FILES:
349
+ if not file_limit_reported:
350
+ diagnostics.append(
351
+ f"{source}: Python requirements includes were truncated after "
352
+ f"{MAX_MANIFEST_INCLUDE_FILES} files"
353
+ )
354
+ file_limit_reported = True
355
+ degraded_paths[path] = "include-graph-truncation"
356
+ continue
357
+ admitted.add(path)
358
+ included_files += 1
359
+ to_read.append(entry)
360
+ process_paths.append(path)
361
+ if to_read:
362
+ read = reader.read_candidate_blobs(tuple(sorted(to_read, key=lambda item: item.path)))
363
+ blobs.update(read.blobs)
364
+ diagnostics.extend(read.diagnostics)
365
+ for path in (entry.path for entry in to_read if entry.path not in read.blobs):
366
+ degraded_paths[path] = "bounded-source-omission"
367
+ for path in process_paths:
368
+ visited.add(path)
369
+ if path not in blobs:
370
+ continue
371
+ try:
372
+ parsed = parse_requirements(blobs[path].decode("utf-8"))
373
+ if any("truncated" in notice for notice in parsed.notices):
374
+ degraded_paths[path] = "include-graph-truncation"
375
+ except UnicodeDecodeError:
376
+ diagnostics.append(f"{path}: Python requirements include is not UTF-8")
377
+ continue
378
+ for include_path in parsed.include_paths:
379
+ resolved = resolve_manifest_include(path, include_path, suffixes=(".txt", ".in"))
380
+ if resolved is None:
381
+ diagnostics.append(
382
+ f"{path}: Python requirements include is outside the supported tree"
383
+ )
384
+ continue
385
+ if included_edges >= MAX_MANIFEST_INCLUDE_EDGES:
386
+ if not edge_limit_reported:
387
+ diagnostics.append(
388
+ "Python requirements include graph was truncated after "
389
+ f"{MAX_MANIFEST_INCLUDE_EDGES} edges"
390
+ )
391
+ edge_limit_reported = True
392
+ degraded_paths[path] = "include-graph-truncation"
393
+ continue
394
+ included_edges += 1
395
+ edges.setdefault(path, []).append(resolved)
396
+ if depth == MAX_MANIFEST_INCLUDE_DEPTH:
397
+ diagnostics.append(
398
+ f"{path}: Python requirements include depth exceeded at {resolved}"
399
+ )
400
+ degraded_paths[path] = "include-graph-truncation"
401
+ else:
402
+ pending.append((resolved, path))
403
+ requirement_paths = tuple(sorted(path for path in visited if path in blobs))
404
+ normalized_edges = {path: tuple(dict.fromkeys(targets)) for path, targets in edges.items()}
405
+ return PythonRequirementBlobSet(
406
+ blobs,
407
+ requirement_paths,
408
+ bound_include_diagnostics(
409
+ diagnostics,
410
+ truncation_notice="Python requirements include diagnostics were truncated",
411
+ ),
412
+ roots_reaching_graph_degradation(initial_paths, normalized_edges, degraded_paths),
413
+ )