plm-knowledge 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,92 @@
1
+ """TracePulse PLM Knowledge — Wave 6.5 Conv B.
2
+
3
+ PRD §3 #6 product-line carve. Hosts the 4-sub-layer knowledge stack:
4
+
5
+ * **Foundation** — `01_Dictionnaries/**` (146 .md files at repo root)
6
+ + `plm_taxonomy` + `plm_evaluation_grids`. Path stays at repo root
7
+ through Wave 6.5 per Decision #171 = (c) defer relocation to Wave 7;
8
+ the ``foundation_loader`` module is a thin reader.
9
+ * **Rule & Policy** — ``issue_classification`` + ``business_validator``
10
+ (lifted via `git mv` from the monolith).
11
+ * **Retrieval** — ``rag_service`` + ``knowledge_pack_loader`` +
12
+ ``knowledge_search_engine`` (BM25; lifted via `git mv` from the
13
+ monolith).
14
+ * **Operational** — V1 placeholder per D-LOCKED-15 + Decision #172.
15
+
16
+ Public surface:
17
+ * The 5 lifted modules at top level (``rag_service``,
18
+ ``issue_classification``, ``business_validator``,
19
+ ``knowledge_pack_loader``, ``knowledge_search_engine``).
20
+ * ``foundation_loader`` — Foundation layer thin reader.
21
+ * ``operational`` — Operational layer V1 placeholder.
22
+ * ``lifecycle`` — 4-state FSM aligned verbatim with PRD §8.11.3
23
+ (Decision #167 mirrored).
24
+ * ``manifests`` — YAML manifest loader (Decision #168 precedent =
25
+ YAML + filesystem).
26
+ * ``contracts`` — wire-contract Pydantic v2 frozen models for the
27
+ public surface (Decision #160 interface-freeze pattern).
28
+ * ``registry`` — manifest+lifecycle registry exposed to the kernel
29
+ (capability registry callers).
30
+
31
+ Re-export shims at every monolith path keep legacy callers working
32
+ through Wave 7 Extraction (Wave 1 W1.4 RunTaskTracker template + Wave
33
+ 6.5 Conv A precedent).
34
+ """
35
+
36
+ from plm_knowledge.contracts import (
37
+ KNOWLEDGE_PACKAGE_VERSION,
38
+ KnowledgeKind,
39
+ KnowledgeLifecycleState,
40
+ KnowledgePackManifest,
41
+ )
42
+ from plm_knowledge.lifecycle import (
43
+ LIFECYCLE_STATES,
44
+ LIFECYCLE_TRANSITIONS,
45
+ InvalidLifecycleTransition,
46
+ can_transition,
47
+ is_terminal,
48
+ next_state,
49
+ )
50
+ from plm_knowledge.manifests import (
51
+ MANIFESTS_DIR,
52
+ ManifestLoadError,
53
+ load_all,
54
+ load_manifest,
55
+ )
56
+ from plm_knowledge.registry import (
57
+ KnowledgePackRegistry,
58
+ get_knowledge_registry,
59
+ )
60
+ from plm_knowledge.foundation_loader import (
61
+ list_foundation_documents,
62
+ resolve_foundation_root,
63
+ )
64
+ from plm_knowledge.operational import (
65
+ OPERATIONAL_LAYER_VERSION,
66
+ OperationalAck,
67
+ record_operational_event,
68
+ )
69
+
70
+ __all__ = [
71
+ "KNOWLEDGE_PACKAGE_VERSION",
72
+ "KnowledgeKind",
73
+ "KnowledgeLifecycleState",
74
+ "KnowledgePackManifest",
75
+ "LIFECYCLE_STATES",
76
+ "LIFECYCLE_TRANSITIONS",
77
+ "InvalidLifecycleTransition",
78
+ "can_transition",
79
+ "is_terminal",
80
+ "next_state",
81
+ "MANIFESTS_DIR",
82
+ "ManifestLoadError",
83
+ "load_all",
84
+ "load_manifest",
85
+ "KnowledgePackRegistry",
86
+ "get_knowledge_registry",
87
+ "list_foundation_documents",
88
+ "resolve_foundation_root",
89
+ "OPERATIONAL_LAYER_VERSION",
90
+ "OperationalAck",
91
+ "record_operational_event",
92
+ ]
@@ -0,0 +1,279 @@
1
+ """
2
+ Business Validator Agent
3
+ Validates generated BPMN processes against original source data.
4
+ """
5
+
6
+ import json
7
+ from typing import Dict, Any, Optional
8
+
9
+ from plm_shared.agents.base_agent import BaseAgent
10
+ from plm_shared.protocols.services.llm import llm_service
11
+
12
+
13
+ # Define the response schema for validation as a plain dict
14
+ VALIDATION_SCHEMA = {
15
+ "type": "object",
16
+ "properties": {
17
+ "isValid": {
18
+ "type": "boolean",
19
+ "description": "True ONLY if the process perfectly matches ALL source data patterns and is syntactically perfect."
20
+ },
21
+ "score": {
22
+ "type": "number",
23
+ "description": "A confidence score from 0 to 100 based on data coverage."
24
+ },
25
+ "keywords": {
26
+ "type": "array",
27
+ "items": {"type": "string"},
28
+ "description": "3 to 10 short thematic keywords (domains, roles, themes) for the process. Lowercase. No duplicates."
29
+ },
30
+ "criticalIssues": {
31
+ "type": "array",
32
+ "items": {"type": "string"},
33
+ "description": "List of logical errors, missing activities from CSV, BPMN violations, or BPMN elements missing a precise expert description in documentation/persona/system."
34
+ },
35
+ "improvementSuggestions": {
36
+ "type": "string",
37
+ "description": "Technical instructions for the analyst to fix the issues, including completing expert-level documentation for each BPMN element."
38
+ }
39
+ },
40
+ "required": ["isValid", "score", "criticalIssues", "improvementSuggestions", "keywords"]
41
+ }
42
+
43
+
44
+ SYSTEM_INSTRUCTION = """You are a Senior Process Auditor & BPMN Validator.
45
+ Your goal is to find ANY discrepancy between the raw source data (CSV/Text) and the AI-generated model.
46
+
47
+ STRICT CRITERIA:
48
+ 1. EXHAUSTIVE COVERAGE: Every unique activity found in the CSV 'Activity' or 'Event' column MUST be represented as a Task in the BPMN.
49
+ 2. SEQUENCE INTEGRITY: Timestamps in CSV must justify the flow. If 'Invoice' always follows 'Receipt', the BPMN must show this.
50
+ 3. BPMN 2.0 SYNTAX:
51
+ - Must have exactly ONE Start Event.
52
+ - Must have at least ONE End Event.
53
+ - All nodes must be connected (no islands).
54
+ - Gateways must have clear labels for their outgoing flows.
55
+ 4. METRICS REALISM: Execution counts in 'miningData' must sum up logically (e.g., sum of branches = entry count).
56
+ 5. BUSINESS METADATA COMPLETENESS: Every step in the BPMN diagram (tasks, start/end events, gateways) MUST have a precise, expert-level description in <bpmn:documentation>. The description must be specific to the element (what it represents in the process, business meaning), not generic. For each <bpmn:task>: require <bpmn:documentation>; tasks that involve a responsible role or an IT system must have <bpmn:extensionElements> with <persona> and/or <system>. For start event: documentation should state what triggers the process. For end event(s): what outcome is reached. For gateways: what decision or split is represented. If any element lacks a precise, expert description or missing persona/system where relevant, add a critical issue and in improvementSuggestions ask to complete these properties with expert-level wording.
57
+
58
+ BE PEDANTIC. If a single task is missing or a flow is illogical, set isValid to false.
59
+
60
+ KEYWORDS REQUIREMENT:
61
+ - Provide 3 to 10 short thematic keywords for the process.
62
+ - Derive them from the source data and the generated BPMN (domains, roles, themes).
63
+ - Use lowercase, avoid duplicates.
64
+ """
65
+
66
+
67
+ class BusinessValidatorAgent(BaseAgent):
68
+ """
69
+ Business Validator Agent
70
+
71
+ Validates generated BPMN processes against original source data.
72
+ Acts as a Senior Process Auditor checking:
73
+
74
+ 1. EXHAUSTIVE COVERAGE: Every activity in source data must be in BPMN
75
+ 2. SEQUENCE INTEGRITY: Timestamps must justify the flow
76
+ 3. BPMN 2.0 SYNTAX: Valid structure, connected nodes
77
+ 4. METRICS REALISM: Execution counts add up logically
78
+ 5. BUSINESS METADATA: Precise descriptions and persona/system assignments
79
+
80
+ Input:
81
+ - original_content: Original CSV or text source data
82
+ - generated_process: Generated process dictionary to validate
83
+
84
+ Output:
85
+ - isValid: Boolean validation result
86
+ - score: Confidence score (0-100)
87
+ - keywords: Extracted thematic keywords
88
+ - criticalIssues: List of found issues
89
+ - improvementSuggestions: Instructions for fixing issues
90
+ """
91
+
92
+ def __init__(self):
93
+ super().__init__(
94
+ name="business_validator",
95
+ description="Validates generated BPMN processes against original source data"
96
+ )
97
+
98
+ def get_output_schema(self) -> Dict[str, Any]:
99
+ """Define business validator output schema."""
100
+ return {
101
+ "type": "object",
102
+ "properties": {
103
+ "isValid": {"type": "boolean"},
104
+ "score": {"type": "number"},
105
+ "keywords": {"type": "array", "items": {"type": "string"}},
106
+ "criticalIssues": {"type": "array", "items": {"type": "string"}},
107
+ "improvementSuggestions": {"type": "string"},
108
+ },
109
+ "required": ["isValid", "score", "criticalIssues", "improvementSuggestions", "keywords"],
110
+ }
111
+
112
+ def get_required_fields(self) -> list:
113
+ """Return required input fields."""
114
+ return ["original_content", "generated_process"]
115
+
116
+ def validate_input(self, input_data: Dict[str, Any]) -> bool:
117
+ """Validate business validator input."""
118
+ if "original_content" not in input_data:
119
+ raise ValueError("Missing required field: original_content")
120
+
121
+ if "generated_process" not in input_data:
122
+ raise ValueError("Missing required field: generated_process")
123
+
124
+ original_content = input_data.get("original_content")
125
+ if not original_content or not str(original_content).strip():
126
+ raise ValueError("original_content must be non-empty string")
127
+
128
+ generated_process = input_data.get("generated_process")
129
+ if not generated_process or not isinstance(generated_process, (dict, str)):
130
+ raise ValueError("generated_process must be dict or string")
131
+
132
+ return True
133
+
134
+ async def execute(self, input_data: Dict[str, Any]) -> Dict[str, Any]:
135
+ """Execute business validation workflow."""
136
+ self.validate_input(input_data)
137
+
138
+ original_content = input_data.get("original_content")
139
+ generated_process = input_data.get("generated_process")
140
+
141
+ return await self._validate_process(original_content, generated_process)
142
+
143
+ async def _validate_process(
144
+ self,
145
+ original_content: str,
146
+ generated_process: Dict[str, Any]
147
+ ) -> Dict[str, Any]:
148
+ """
149
+ Validate a generated BPMN process against original source data using ADK multi-LLM.
150
+
151
+ Args:
152
+ original_content: Original CSV or text content
153
+ generated_process: Generated process dictionary to validate
154
+
155
+ Returns:
156
+ Dictionary containing isValid, score, criticalIssues, improvementSuggestions, keywords
157
+
158
+ Raises:
159
+ Exception: If validation fails
160
+ """
161
+ try:
162
+ print(f"[SCAN] Validating with provider: {llm_service.provider}")
163
+
164
+ # Phase 1: Preprocess large CSV to avoid token overflow
165
+ source_content = await self._preprocess_source_data(original_content)
166
+
167
+ # Phase 2: Build validation prompt
168
+ prompt = self._build_validation_prompt(source_content, generated_process)
169
+
170
+ # Phase 3: Call ADK service
171
+ result = await llm_service.generate_structured(
172
+ system_prompt=SYSTEM_INSTRUCTION,
173
+ user_prompt=prompt,
174
+ output_schema=VALIDATION_SCHEMA,
175
+ temperature=0.3
176
+ )
177
+
178
+ print(f"[OK] Validation complete - Score: {result.get('score', 0)}%")
179
+ return result
180
+
181
+ except Exception as error:
182
+ print(f"[ERR] Validator Agent Error: {error}")
183
+ import traceback
184
+ traceback.print_exc()
185
+ # Return permissive validation on error
186
+ return self._get_default_validation()
187
+
188
+ async def _preprocess_source_data(self, original_content: str) -> str:
189
+ """
190
+ Preprocess large CSV to avoid token overflow.
191
+
192
+ Args:
193
+ original_content: Original source data (CSV or text)
194
+
195
+ Returns:
196
+ Preprocessed source data or original if small enough
197
+ """
198
+ if len(original_content) > 50000 and "Employee_ID" in original_content:
199
+ print("[SCAN] Applying CSV preprocessing for validation...")
200
+ try:
201
+ from plm_shared.utils.csv_preprocessor import smart_csv_preprocessor
202
+ return smart_csv_preprocessor(original_content)
203
+ except ImportError:
204
+ print("[WARN] CSV preprocessor not available, using original content")
205
+ return original_content[:50000]
206
+
207
+ return original_content
208
+
209
+ def _build_validation_prompt(
210
+ self,
211
+ source_content: str,
212
+ generated_process: Dict[str, Any]
213
+ ) -> str:
214
+ """
215
+ Build validation prompt combining source data and generated process.
216
+
217
+ Args:
218
+ source_content: Preprocessed source data
219
+ generated_process: Generated process to validate
220
+
221
+ Returns:
222
+ Formatted validation prompt
223
+ """
224
+ return f"""[SOURCE_DATA_EXTRACT]
225
+ {source_content[:10000]}
226
+
227
+ [GENERATED_BPMN_AND_METRICS]
228
+ {json.dumps(generated_process) if isinstance(generated_process, dict) else str(generated_process)}
229
+ """
230
+
231
+ def _get_default_validation(self) -> Dict[str, Any]:
232
+ """
233
+ Return permissive default validation result on error.
234
+
235
+ Returns:
236
+ Default validation dictionary
237
+ """
238
+ return {
239
+ "isValid": True,
240
+ "score": 100,
241
+ "keywords": [],
242
+ "criticalIssues": [],
243
+ "improvementSuggestions": ""
244
+ }
245
+
246
+
247
+ # Singleton instance
248
+ business_validator_agent = BusinessValidatorAgent()
249
+
250
+
251
+ # ==================== BACKWARD COMPATIBILITY ====================
252
+
253
+ async def validate_process(
254
+ original_content: str,
255
+ generated_process: Dict[str, Any]
256
+ ) -> Dict[str, Any]:
257
+ """
258
+ Backward compatibility wrapper for original function.
259
+
260
+ Validate a generated BPMN process against original source data using ADK multi-LLM
261
+
262
+ Args:
263
+ original_content: Original CSV or text content
264
+ generated_process: Generated process dictionary to validate
265
+
266
+ Returns:
267
+ Dictionary containing isValid, score, criticalIssues, improvementSuggestions
268
+
269
+ Raises:
270
+ Exception: If validation fails
271
+
272
+ Delegates to business_validator_agent singleton.
273
+ """
274
+ result = await business_validator_agent.execute({
275
+ "original_content": original_content,
276
+ "generated_process": generated_process,
277
+ })
278
+
279
+ return result
@@ -0,0 +1,178 @@
1
+ """Wire-contract Pydantic v2 frozen models for the public surface.
2
+
3
+ Authored at Wave 6.5 Conv B per Decision #160 (interface-freeze pattern,
4
+ P-3 lock) — every cross-package symbol exposed by this sibling-package
5
+ becomes a frozen v2 model so Wave 7 Extraction is mechanical (filter-
6
+ repo + pin update; no behaviour rewrite).
7
+
8
+ Public surface (frozen V1):
9
+ * ``KNOWLEDGE_PACKAGE_VERSION`` — the wire-contract version string.
10
+ Bump via dedicated migration story; out-of-band bumps trip
11
+ ``test_contracts_version_pinned``.
12
+ * ``KnowledgeKind`` — enum of knowledge-pack kinds carved at Wave 6.5
13
+ Conv B (covers all 4 PRD §3 #6 sub-layers).
14
+ * ``KnowledgeLifecycleState`` — enum mirroring
15
+ ``lifecycle.LIFECYCLE_STATES`` (4 states, PRD §8.11.3 verbatim).
16
+ * ``KnowledgePackManifest`` — the YAML manifest shape; one file per
17
+ knowledge pack under ``manifests/*.yaml``.
18
+
19
+ Pydantic v2 ``model_config = ConfigDict(frozen=True)`` enforces
20
+ immutability after construction — the registry and downstream consumers
21
+ (kernel manifest harvest, capability registry) treat manifests as
22
+ read-only descriptors. Mutations require constructing a new instance.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ from enum import Enum
27
+ from typing import List, Optional
28
+
29
+ from pydantic import BaseModel, ConfigDict, Field
30
+
31
+ # Wire-contract version. Authored at Wave 6.5 Conv B. Subsequent bumps
32
+ # trace through a dedicated story + ADR per the interface-freeze
33
+ # discipline (Decision #160). Pinned to the same wave-tag string used
34
+ # by the Wave 6.5 Conv A `plm-skill-packages` carve (`v1.0.0-wave6.5`)
35
+ # so the two siblings share a freeze checkpoint.
36
+ KNOWLEDGE_PACKAGE_VERSION: str = "v1.0.0-wave6.5"
37
+
38
+
39
+ class KnowledgeKind(str, Enum):
40
+ """Knowledge-pack kinds carved into plm-knowledge at Wave 6.5 Conv B.
41
+
42
+ Mirrors PRD §3 #6 product-line scope across the 4 named sub-layers:
43
+ Foundation, Rule & Policy, Retrieval, Operational. Wave 7+ may
44
+ extend this enum via a wire-contract bump
45
+ (``KNOWLEDGE_PACKAGE_VERSION``).
46
+ """
47
+
48
+ FOUNDATION_DICTIONARY = "foundation_dictionary"
49
+ """Foundation layer — `01_Dictionnaries/**` markdown corpus."""
50
+
51
+ FOUNDATION_TAXONOMY = "foundation_taxonomy"
52
+ """Foundation layer — `plm_taxonomy` definitions."""
53
+
54
+ FOUNDATION_EVALUATION_GRID = "foundation_evaluation_grid"
55
+ """Foundation layer — `plm_evaluation_grids` rubrics."""
56
+
57
+ RULE_POLICY = "rule_policy"
58
+ """Rule & Policy layer — issue classification + business validator."""
59
+
60
+ RETRIEVAL_RAG = "retrieval_rag"
61
+ """Retrieval layer — Azure Assistants RAG service."""
62
+
63
+ RETRIEVAL_BM25 = "retrieval_bm25"
64
+ """Retrieval layer — BM25 search engine over knowledge chunks."""
65
+
66
+ OPERATIONAL = "operational"
67
+ """Operational layer — V1 placeholder per D-LOCKED-15."""
68
+
69
+
70
+ class KnowledgeLifecycleState(str, Enum):
71
+ """4-state lifecycle FSM mirror — PRD §8.11.3 verbatim (Decision #167).
72
+
73
+ Mirrors ``plm_knowledge.lifecycle.LIFECYCLE_STATES`` 1:1; the enum
74
+ exists to give Pydantic a constrained validator for the
75
+ ``lifecycle_state`` field of ``KnowledgePackManifest`` without a
76
+ free-form string.
77
+ """
78
+
79
+ DRAFT = "draft"
80
+ RELEASED = "released"
81
+ DEPRECATED = "deprecated"
82
+ RETIRED = "retired"
83
+
84
+
85
+ class KnowledgePackManifest(BaseModel):
86
+ """YAML knowledge-pack manifest — one file per pack under ``manifests/*.yaml``.
87
+
88
+ Mirrors the Wave 6.5 Conv A `plm_skill_packages.contracts.SkillManifest`
89
+ template (Decision #168 = YAML + filesystem precedent) — manifests
90
+ load at boot via ``manifests.load_all()``, matching the
91
+ ``litellm_config/model_list.yaml`` catalog-as-truth pattern. No SQL,
92
+ no admin UI in V1.
93
+
94
+ Frozen-v2 model — mutations require constructing a new instance.
95
+ """
96
+
97
+ model_config = ConfigDict(frozen=True, extra="forbid")
98
+
99
+ pack_id: str = Field(
100
+ ...,
101
+ description="Stable identifier (e.g. 'foundation.dictionnaries'). Matches the YAML file basename.",
102
+ min_length=3,
103
+ max_length=128,
104
+ )
105
+ """Stable identifier; the basename of the manifest YAML file (less
106
+ extension). Format: ``<sub_layer>.<pack_name>``."""
107
+
108
+ kind: KnowledgeKind = Field(
109
+ ...,
110
+ description="Knowledge-pack kind enum (sub-layer assignment).",
111
+ )
112
+
113
+ title: str = Field(
114
+ ...,
115
+ description="Human-readable title shown in admin UI / capability registry.",
116
+ min_length=3,
117
+ max_length=200,
118
+ )
119
+
120
+ description: str = Field(
121
+ ...,
122
+ description="One-line description of the knowledge pack's surface.",
123
+ min_length=5,
124
+ max_length=500,
125
+ )
126
+
127
+ version: str = Field(
128
+ ...,
129
+ description="Knowledge-pack version (semver-style). Bumps trace through a dedicated story.",
130
+ pattern=r"^[0-9]+\.[0-9]+\.[0-9]+(?:-[A-Za-z0-9.-]+)?$",
131
+ )
132
+
133
+ lifecycle_state: KnowledgeLifecycleState = Field(
134
+ ...,
135
+ description="Current lifecycle state (PRD §8.11.3 4-state).",
136
+ )
137
+
138
+ capability_id: Optional[str] = Field(
139
+ default=None,
140
+ description="Optional pointer into plm_shared.capability_registry.",
141
+ )
142
+
143
+ autonomy_level: Optional[str] = Field(
144
+ default=None,
145
+ description="Optional pointer into plm_shared autonomy levels (CR.10).",
146
+ )
147
+
148
+ entry_module: str = Field(
149
+ ...,
150
+ description="Dotted module path inside plm_knowledge (e.g. 'plm_knowledge.rag_service').",
151
+ pattern=r"^plm_knowledge(?:\.[A-Za-z_][A-Za-z0-9_]*)+$",
152
+ )
153
+
154
+ entry_callable: Optional[str] = Field(
155
+ default=None,
156
+ description="Optional callable name within the entry module (function or class). Foundation packs may omit this when the entry module is a static asset directory.",
157
+ max_length=128,
158
+ )
159
+
160
+ tags: List[str] = Field(
161
+ default_factory=list,
162
+ description="Optional tags for capability registry filtering.",
163
+ max_length=32,
164
+ )
165
+
166
+ notes: Optional[str] = Field(
167
+ default=None,
168
+ description="Free-form notes; not parsed by runtime.",
169
+ max_length=2000,
170
+ )
171
+
172
+
173
+ __all__ = [
174
+ "KNOWLEDGE_PACKAGE_VERSION",
175
+ "KnowledgeKind",
176
+ "KnowledgeLifecycleState",
177
+ "KnowledgePackManifest",
178
+ ]
@@ -0,0 +1,113 @@
1
+ """Foundation layer — thin reader for `01_Dictionnaries/`.
2
+
3
+ Wave 6.5 Conv B per Decision #171 = (c) "defer relocation to Wave 7".
4
+ The 146 markdown files at `01_Dictionnaries/` (repo root) stay in place
5
+ through Wave 6.5; this loader resolves the path through the existing
6
+ backend `config.dictionnaries_path` setting (already configurable) so
7
+ no consumer needs a path change at this conv.
8
+
9
+ Public surface:
10
+ * ``resolve_foundation_root(override=None)`` — returns the directory
11
+ Path. With ``override=None``, resolves via the backend
12
+ ``config.settings.dictionnaries_path`` (default `../../01_Dictionnaries`
13
+ relative to the backend module). The override exists for tests +
14
+ admin tooling that want to point at a snapshot directory.
15
+ * ``list_foundation_documents(override=None)`` — enumerate the
16
+ `*.md` files under the resolved root in stable lexicographic order.
17
+ Returns an empty list (with a warning) if the root is missing —
18
+ matches the boot-time pattern in `02_App/backend/main.py:306`
19
+ ("PLM Knowledge Engine (BM25 — non-bloquant si 01_Dictionnaries
20
+ absent)").
21
+
22
+ Wave 7 Extraction will physically relocate the directory into
23
+ `02_App/plm-knowledge/foundation/01_Dictionnaries/` and the resolution
24
+ function will switch to a package-relative anchor.
25
+ """
26
+ from __future__ import annotations
27
+
28
+ import logging
29
+ from pathlib import Path
30
+ from typing import List, Optional
31
+
32
+ logger = logging.getLogger(__name__)
33
+
34
+
35
+ # Default repo-root path. The backend `config.settings.dictionnaries_path`
36
+ # resolves this same string relative to the backend CWD; here we keep an
37
+ # absolute fallback for sibling-package callers that don't have the
38
+ # backend `config` module loaded (e.g. standalone admin tooling).
39
+ _DEFAULT_REPO_ROOT_FALLBACK: Path = (
40
+ Path(__file__).resolve().parents[3] / "01_Dictionnaries"
41
+ )
42
+
43
+
44
+ def resolve_foundation_root(override: Optional[Path] = None) -> Path:
45
+ """Resolve the Foundation layer root directory.
46
+
47
+ Resolution order:
48
+ 1. Explicit ``override`` Path (test + admin tooling escape hatch).
49
+ 2. Backend ``config.settings.dictionnaries_path`` if importable.
50
+ 3. Repo-root fallback ``../../../../01_Dictionnaries`` relative
51
+ to this module.
52
+
53
+ Returns the resolved Path even if the directory does not exist on
54
+ disk; callers that need a stricter contract can call
55
+ ``Path.exists()`` + branch on absence.
56
+ """
57
+ if override is not None:
58
+ return Path(override).resolve()
59
+ # Wave 6.7 Conv A: prefer the plm_shared.protocols settings proxy so
60
+ # the resolution path does not depend on ``02_App/backend/`` at
61
+ # import time. The proxy raises ``ProtocolNotRegistered`` on
62
+ # attribute access when the host has not wired the DI registry
63
+ # (e.g. truly standalone foundation_loader tests); fall back to the
64
+ # legacy ``config`` import in that case, then to the absolute repo-
65
+ # root fallback.
66
+ configured = None
67
+ try:
68
+ from plm_shared.protocols.services.settings import settings as _shared_settings
69
+
70
+ configured = getattr(_shared_settings, "dictionnaries_path", None)
71
+ except Exception:
72
+ configured = None
73
+ if configured is None:
74
+ try:
75
+ from plm_shared.settings import settings # type: ignore[import-not-found]
76
+
77
+ configured = getattr(settings, "dictionnaries_path", None)
78
+ except ImportError:
79
+ configured = None
80
+ if configured is None:
81
+ return _DEFAULT_REPO_ROOT_FALLBACK
82
+ return Path(configured).resolve()
83
+
84
+
85
+ def list_foundation_documents(override: Optional[Path] = None) -> List[Path]:
86
+ """Enumerate ``*.md`` files under the Foundation root.
87
+
88
+ Returns an empty list (with a warning logged) if the root is
89
+ missing — matches the non-blocking startup pattern in
90
+ `backend/main.py:306`. Sub-directories are walked recursively;
91
+ output is in stable lexicographic order.
92
+ """
93
+ root = resolve_foundation_root(override)
94
+ if not root.exists():
95
+ logger.warning(
96
+ "foundation_root_missing path=%s — returning empty document list",
97
+ root,
98
+ )
99
+ return []
100
+ if not root.is_dir():
101
+ logger.warning(
102
+ "foundation_root_not_a_directory path=%s — returning empty document list",
103
+ root,
104
+ )
105
+ return []
106
+ documents = sorted(root.rglob("*.md"))
107
+ return documents
108
+
109
+
110
+ __all__ = [
111
+ "resolve_foundation_root",
112
+ "list_foundation_documents",
113
+ ]