plm-knowledge 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- plm_knowledge/__init__.py +92 -0
- plm_knowledge/business_validator.py +279 -0
- plm_knowledge/contracts.py +178 -0
- plm_knowledge/foundation_loader.py +113 -0
- plm_knowledge/issue_classification.py +155 -0
- plm_knowledge/knowledge_pack_loader.py +177 -0
- plm_knowledge/knowledge_search_engine.py +140 -0
- plm_knowledge/lifecycle.py +115 -0
- plm_knowledge/manifests/foundation.dictionnaries.yaml +17 -0
- plm_knowledge/manifests/foundation.evaluation_grids.yaml +16 -0
- plm_knowledge/manifests/foundation.taxonomy.yaml +15 -0
- plm_knowledge/manifests/operational.placeholder.yaml +18 -0
- plm_knowledge/manifests/retrieval.bm25.yaml +20 -0
- plm_knowledge/manifests/retrieval.rag_service.yaml +21 -0
- plm_knowledge/manifests/rule.business_validator.yaml +20 -0
- plm_knowledge/manifests/rule.issue_classification.yaml +19 -0
- plm_knowledge/manifests.py +108 -0
- plm_knowledge/operational.py +96 -0
- plm_knowledge/rag_service.py +371 -0
- plm_knowledge/registry.py +112 -0
- plm_knowledge-1.0.0.dist-info/METADATA +93 -0
- plm_knowledge-1.0.0.dist-info/RECORD +24 -0
- plm_knowledge-1.0.0.dist-info/WHEEL +5 -0
- plm_knowledge-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""TracePulse PLM Knowledge — Wave 6.5 Conv B.
|
|
2
|
+
|
|
3
|
+
PRD §3 #6 product-line carve. Hosts the 4-sub-layer knowledge stack:
|
|
4
|
+
|
|
5
|
+
* **Foundation** — `01_Dictionnaries/**` (146 .md files at repo root)
|
|
6
|
+
+ `plm_taxonomy` + `plm_evaluation_grids`. Path stays at repo root
|
|
7
|
+
through Wave 6.5 per Decision #171 = (c) defer relocation to Wave 7;
|
|
8
|
+
the ``foundation_loader`` module is a thin reader.
|
|
9
|
+
* **Rule & Policy** — ``issue_classification`` + ``business_validator``
|
|
10
|
+
(lifted via `git mv` from the monolith).
|
|
11
|
+
* **Retrieval** — ``rag_service`` + ``knowledge_pack_loader`` +
|
|
12
|
+
``knowledge_search_engine`` (BM25; lifted via `git mv` from the
|
|
13
|
+
monolith).
|
|
14
|
+
* **Operational** — V1 placeholder per D-LOCKED-15 + Decision #172.
|
|
15
|
+
|
|
16
|
+
Public surface:
|
|
17
|
+
* The 5 lifted modules at top level (``rag_service``,
|
|
18
|
+
``issue_classification``, ``business_validator``,
|
|
19
|
+
``knowledge_pack_loader``, ``knowledge_search_engine``).
|
|
20
|
+
* ``foundation_loader`` — Foundation layer thin reader.
|
|
21
|
+
* ``operational`` — Operational layer V1 placeholder.
|
|
22
|
+
* ``lifecycle`` — 4-state FSM aligned verbatim with PRD §8.11.3
|
|
23
|
+
(Decision #167 mirrored).
|
|
24
|
+
* ``manifests`` — YAML manifest loader (Decision #168 precedent =
|
|
25
|
+
YAML + filesystem).
|
|
26
|
+
* ``contracts`` — wire-contract Pydantic v2 frozen models for the
|
|
27
|
+
public surface (Decision #160 interface-freeze pattern).
|
|
28
|
+
* ``registry`` — manifest+lifecycle registry exposed to the kernel
|
|
29
|
+
(capability registry callers).
|
|
30
|
+
|
|
31
|
+
Re-export shims at every monolith path keep legacy callers working
|
|
32
|
+
through Wave 7 Extraction (Wave 1 W1.4 RunTaskTracker template + Wave
|
|
33
|
+
6.5 Conv A precedent).
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from plm_knowledge.contracts import (
|
|
37
|
+
KNOWLEDGE_PACKAGE_VERSION,
|
|
38
|
+
KnowledgeKind,
|
|
39
|
+
KnowledgeLifecycleState,
|
|
40
|
+
KnowledgePackManifest,
|
|
41
|
+
)
|
|
42
|
+
from plm_knowledge.lifecycle import (
|
|
43
|
+
LIFECYCLE_STATES,
|
|
44
|
+
LIFECYCLE_TRANSITIONS,
|
|
45
|
+
InvalidLifecycleTransition,
|
|
46
|
+
can_transition,
|
|
47
|
+
is_terminal,
|
|
48
|
+
next_state,
|
|
49
|
+
)
|
|
50
|
+
from plm_knowledge.manifests import (
|
|
51
|
+
MANIFESTS_DIR,
|
|
52
|
+
ManifestLoadError,
|
|
53
|
+
load_all,
|
|
54
|
+
load_manifest,
|
|
55
|
+
)
|
|
56
|
+
from plm_knowledge.registry import (
|
|
57
|
+
KnowledgePackRegistry,
|
|
58
|
+
get_knowledge_registry,
|
|
59
|
+
)
|
|
60
|
+
from plm_knowledge.foundation_loader import (
|
|
61
|
+
list_foundation_documents,
|
|
62
|
+
resolve_foundation_root,
|
|
63
|
+
)
|
|
64
|
+
from plm_knowledge.operational import (
|
|
65
|
+
OPERATIONAL_LAYER_VERSION,
|
|
66
|
+
OperationalAck,
|
|
67
|
+
record_operational_event,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
__all__ = [
|
|
71
|
+
"KNOWLEDGE_PACKAGE_VERSION",
|
|
72
|
+
"KnowledgeKind",
|
|
73
|
+
"KnowledgeLifecycleState",
|
|
74
|
+
"KnowledgePackManifest",
|
|
75
|
+
"LIFECYCLE_STATES",
|
|
76
|
+
"LIFECYCLE_TRANSITIONS",
|
|
77
|
+
"InvalidLifecycleTransition",
|
|
78
|
+
"can_transition",
|
|
79
|
+
"is_terminal",
|
|
80
|
+
"next_state",
|
|
81
|
+
"MANIFESTS_DIR",
|
|
82
|
+
"ManifestLoadError",
|
|
83
|
+
"load_all",
|
|
84
|
+
"load_manifest",
|
|
85
|
+
"KnowledgePackRegistry",
|
|
86
|
+
"get_knowledge_registry",
|
|
87
|
+
"list_foundation_documents",
|
|
88
|
+
"resolve_foundation_root",
|
|
89
|
+
"OPERATIONAL_LAYER_VERSION",
|
|
90
|
+
"OperationalAck",
|
|
91
|
+
"record_operational_event",
|
|
92
|
+
]
|
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Business Validator Agent
|
|
3
|
+
Validates generated BPMN processes against original source data.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
from typing import Dict, Any, Optional
|
|
8
|
+
|
|
9
|
+
from plm_shared.agents.base_agent import BaseAgent
|
|
10
|
+
from plm_shared.protocols.services.llm import llm_service
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# Define the response schema for validation as a plain dict
|
|
14
|
+
VALIDATION_SCHEMA = {
|
|
15
|
+
"type": "object",
|
|
16
|
+
"properties": {
|
|
17
|
+
"isValid": {
|
|
18
|
+
"type": "boolean",
|
|
19
|
+
"description": "True ONLY if the process perfectly matches ALL source data patterns and is syntactically perfect."
|
|
20
|
+
},
|
|
21
|
+
"score": {
|
|
22
|
+
"type": "number",
|
|
23
|
+
"description": "A confidence score from 0 to 100 based on data coverage."
|
|
24
|
+
},
|
|
25
|
+
"keywords": {
|
|
26
|
+
"type": "array",
|
|
27
|
+
"items": {"type": "string"},
|
|
28
|
+
"description": "3 to 10 short thematic keywords (domains, roles, themes) for the process. Lowercase. No duplicates."
|
|
29
|
+
},
|
|
30
|
+
"criticalIssues": {
|
|
31
|
+
"type": "array",
|
|
32
|
+
"items": {"type": "string"},
|
|
33
|
+
"description": "List of logical errors, missing activities from CSV, BPMN violations, or BPMN elements missing a precise expert description in documentation/persona/system."
|
|
34
|
+
},
|
|
35
|
+
"improvementSuggestions": {
|
|
36
|
+
"type": "string",
|
|
37
|
+
"description": "Technical instructions for the analyst to fix the issues, including completing expert-level documentation for each BPMN element."
|
|
38
|
+
}
|
|
39
|
+
},
|
|
40
|
+
"required": ["isValid", "score", "criticalIssues", "improvementSuggestions", "keywords"]
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
SYSTEM_INSTRUCTION = """You are a Senior Process Auditor & BPMN Validator.
|
|
45
|
+
Your goal is to find ANY discrepancy between the raw source data (CSV/Text) and the AI-generated model.
|
|
46
|
+
|
|
47
|
+
STRICT CRITERIA:
|
|
48
|
+
1. EXHAUSTIVE COVERAGE: Every unique activity found in the CSV 'Activity' or 'Event' column MUST be represented as a Task in the BPMN.
|
|
49
|
+
2. SEQUENCE INTEGRITY: Timestamps in CSV must justify the flow. If 'Invoice' always follows 'Receipt', the BPMN must show this.
|
|
50
|
+
3. BPMN 2.0 SYNTAX:
|
|
51
|
+
- Must have exactly ONE Start Event.
|
|
52
|
+
- Must have at least ONE End Event.
|
|
53
|
+
- All nodes must be connected (no islands).
|
|
54
|
+
- Gateways must have clear labels for their outgoing flows.
|
|
55
|
+
4. METRICS REALISM: Execution counts in 'miningData' must sum up logically (e.g., sum of branches = entry count).
|
|
56
|
+
5. BUSINESS METADATA COMPLETENESS: Every step in the BPMN diagram (tasks, start/end events, gateways) MUST have a precise, expert-level description in <bpmn:documentation>. The description must be specific to the element (what it represents in the process, business meaning), not generic. For each <bpmn:task>: require <bpmn:documentation>; tasks that involve a responsible role or an IT system must have <bpmn:extensionElements> with <persona> and/or <system>. For start event: documentation should state what triggers the process. For end event(s): what outcome is reached. For gateways: what decision or split is represented. If any element lacks a precise, expert description or missing persona/system where relevant, add a critical issue and in improvementSuggestions ask to complete these properties with expert-level wording.
|
|
57
|
+
|
|
58
|
+
BE PEDANTIC. If a single task is missing or a flow is illogical, set isValid to false.
|
|
59
|
+
|
|
60
|
+
KEYWORDS REQUIREMENT:
|
|
61
|
+
- Provide 3 to 10 short thematic keywords for the process.
|
|
62
|
+
- Derive them from the source data and the generated BPMN (domains, roles, themes).
|
|
63
|
+
- Use lowercase, avoid duplicates.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class BusinessValidatorAgent(BaseAgent):
|
|
68
|
+
"""
|
|
69
|
+
Business Validator Agent
|
|
70
|
+
|
|
71
|
+
Validates generated BPMN processes against original source data.
|
|
72
|
+
Acts as a Senior Process Auditor checking:
|
|
73
|
+
|
|
74
|
+
1. EXHAUSTIVE COVERAGE: Every activity in source data must be in BPMN
|
|
75
|
+
2. SEQUENCE INTEGRITY: Timestamps must justify the flow
|
|
76
|
+
3. BPMN 2.0 SYNTAX: Valid structure, connected nodes
|
|
77
|
+
4. METRICS REALISM: Execution counts add up logically
|
|
78
|
+
5. BUSINESS METADATA: Precise descriptions and persona/system assignments
|
|
79
|
+
|
|
80
|
+
Input:
|
|
81
|
+
- original_content: Original CSV or text source data
|
|
82
|
+
- generated_process: Generated process dictionary to validate
|
|
83
|
+
|
|
84
|
+
Output:
|
|
85
|
+
- isValid: Boolean validation result
|
|
86
|
+
- score: Confidence score (0-100)
|
|
87
|
+
- keywords: Extracted thematic keywords
|
|
88
|
+
- criticalIssues: List of found issues
|
|
89
|
+
- improvementSuggestions: Instructions for fixing issues
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __init__(self):
|
|
93
|
+
super().__init__(
|
|
94
|
+
name="business_validator",
|
|
95
|
+
description="Validates generated BPMN processes against original source data"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
def get_output_schema(self) -> Dict[str, Any]:
|
|
99
|
+
"""Define business validator output schema."""
|
|
100
|
+
return {
|
|
101
|
+
"type": "object",
|
|
102
|
+
"properties": {
|
|
103
|
+
"isValid": {"type": "boolean"},
|
|
104
|
+
"score": {"type": "number"},
|
|
105
|
+
"keywords": {"type": "array", "items": {"type": "string"}},
|
|
106
|
+
"criticalIssues": {"type": "array", "items": {"type": "string"}},
|
|
107
|
+
"improvementSuggestions": {"type": "string"},
|
|
108
|
+
},
|
|
109
|
+
"required": ["isValid", "score", "criticalIssues", "improvementSuggestions", "keywords"],
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
def get_required_fields(self) -> list:
|
|
113
|
+
"""Return required input fields."""
|
|
114
|
+
return ["original_content", "generated_process"]
|
|
115
|
+
|
|
116
|
+
def validate_input(self, input_data: Dict[str, Any]) -> bool:
|
|
117
|
+
"""Validate business validator input."""
|
|
118
|
+
if "original_content" not in input_data:
|
|
119
|
+
raise ValueError("Missing required field: original_content")
|
|
120
|
+
|
|
121
|
+
if "generated_process" not in input_data:
|
|
122
|
+
raise ValueError("Missing required field: generated_process")
|
|
123
|
+
|
|
124
|
+
original_content = input_data.get("original_content")
|
|
125
|
+
if not original_content or not str(original_content).strip():
|
|
126
|
+
raise ValueError("original_content must be non-empty string")
|
|
127
|
+
|
|
128
|
+
generated_process = input_data.get("generated_process")
|
|
129
|
+
if not generated_process or not isinstance(generated_process, (dict, str)):
|
|
130
|
+
raise ValueError("generated_process must be dict or string")
|
|
131
|
+
|
|
132
|
+
return True
|
|
133
|
+
|
|
134
|
+
async def execute(self, input_data: Dict[str, Any]) -> Dict[str, Any]:
|
|
135
|
+
"""Execute business validation workflow."""
|
|
136
|
+
self.validate_input(input_data)
|
|
137
|
+
|
|
138
|
+
original_content = input_data.get("original_content")
|
|
139
|
+
generated_process = input_data.get("generated_process")
|
|
140
|
+
|
|
141
|
+
return await self._validate_process(original_content, generated_process)
|
|
142
|
+
|
|
143
|
+
async def _validate_process(
|
|
144
|
+
self,
|
|
145
|
+
original_content: str,
|
|
146
|
+
generated_process: Dict[str, Any]
|
|
147
|
+
) -> Dict[str, Any]:
|
|
148
|
+
"""
|
|
149
|
+
Validate a generated BPMN process against original source data using ADK multi-LLM.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
original_content: Original CSV or text content
|
|
153
|
+
generated_process: Generated process dictionary to validate
|
|
154
|
+
|
|
155
|
+
Returns:
|
|
156
|
+
Dictionary containing isValid, score, criticalIssues, improvementSuggestions, keywords
|
|
157
|
+
|
|
158
|
+
Raises:
|
|
159
|
+
Exception: If validation fails
|
|
160
|
+
"""
|
|
161
|
+
try:
|
|
162
|
+
print(f"[SCAN] Validating with provider: {llm_service.provider}")
|
|
163
|
+
|
|
164
|
+
# Phase 1: Preprocess large CSV to avoid token overflow
|
|
165
|
+
source_content = await self._preprocess_source_data(original_content)
|
|
166
|
+
|
|
167
|
+
# Phase 2: Build validation prompt
|
|
168
|
+
prompt = self._build_validation_prompt(source_content, generated_process)
|
|
169
|
+
|
|
170
|
+
# Phase 3: Call ADK service
|
|
171
|
+
result = await llm_service.generate_structured(
|
|
172
|
+
system_prompt=SYSTEM_INSTRUCTION,
|
|
173
|
+
user_prompt=prompt,
|
|
174
|
+
output_schema=VALIDATION_SCHEMA,
|
|
175
|
+
temperature=0.3
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
print(f"[OK] Validation complete - Score: {result.get('score', 0)}%")
|
|
179
|
+
return result
|
|
180
|
+
|
|
181
|
+
except Exception as error:
|
|
182
|
+
print(f"[ERR] Validator Agent Error: {error}")
|
|
183
|
+
import traceback
|
|
184
|
+
traceback.print_exc()
|
|
185
|
+
# Return permissive validation on error
|
|
186
|
+
return self._get_default_validation()
|
|
187
|
+
|
|
188
|
+
async def _preprocess_source_data(self, original_content: str) -> str:
|
|
189
|
+
"""
|
|
190
|
+
Preprocess large CSV to avoid token overflow.
|
|
191
|
+
|
|
192
|
+
Args:
|
|
193
|
+
original_content: Original source data (CSV or text)
|
|
194
|
+
|
|
195
|
+
Returns:
|
|
196
|
+
Preprocessed source data or original if small enough
|
|
197
|
+
"""
|
|
198
|
+
if len(original_content) > 50000 and "Employee_ID" in original_content:
|
|
199
|
+
print("[SCAN] Applying CSV preprocessing for validation...")
|
|
200
|
+
try:
|
|
201
|
+
from plm_shared.utils.csv_preprocessor import smart_csv_preprocessor
|
|
202
|
+
return smart_csv_preprocessor(original_content)
|
|
203
|
+
except ImportError:
|
|
204
|
+
print("[WARN] CSV preprocessor not available, using original content")
|
|
205
|
+
return original_content[:50000]
|
|
206
|
+
|
|
207
|
+
return original_content
|
|
208
|
+
|
|
209
|
+
def _build_validation_prompt(
|
|
210
|
+
self,
|
|
211
|
+
source_content: str,
|
|
212
|
+
generated_process: Dict[str, Any]
|
|
213
|
+
) -> str:
|
|
214
|
+
"""
|
|
215
|
+
Build validation prompt combining source data and generated process.
|
|
216
|
+
|
|
217
|
+
Args:
|
|
218
|
+
source_content: Preprocessed source data
|
|
219
|
+
generated_process: Generated process to validate
|
|
220
|
+
|
|
221
|
+
Returns:
|
|
222
|
+
Formatted validation prompt
|
|
223
|
+
"""
|
|
224
|
+
return f"""[SOURCE_DATA_EXTRACT]
|
|
225
|
+
{source_content[:10000]}
|
|
226
|
+
|
|
227
|
+
[GENERATED_BPMN_AND_METRICS]
|
|
228
|
+
{json.dumps(generated_process) if isinstance(generated_process, dict) else str(generated_process)}
|
|
229
|
+
"""
|
|
230
|
+
|
|
231
|
+
def _get_default_validation(self) -> Dict[str, Any]:
|
|
232
|
+
"""
|
|
233
|
+
Return permissive default validation result on error.
|
|
234
|
+
|
|
235
|
+
Returns:
|
|
236
|
+
Default validation dictionary
|
|
237
|
+
"""
|
|
238
|
+
return {
|
|
239
|
+
"isValid": True,
|
|
240
|
+
"score": 100,
|
|
241
|
+
"keywords": [],
|
|
242
|
+
"criticalIssues": [],
|
|
243
|
+
"improvementSuggestions": ""
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
# Singleton instance
|
|
248
|
+
business_validator_agent = BusinessValidatorAgent()
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
# ==================== BACKWARD COMPATIBILITY ====================
|
|
252
|
+
|
|
253
|
+
async def validate_process(
|
|
254
|
+
original_content: str,
|
|
255
|
+
generated_process: Dict[str, Any]
|
|
256
|
+
) -> Dict[str, Any]:
|
|
257
|
+
"""
|
|
258
|
+
Backward compatibility wrapper for original function.
|
|
259
|
+
|
|
260
|
+
Validate a generated BPMN process against original source data using ADK multi-LLM
|
|
261
|
+
|
|
262
|
+
Args:
|
|
263
|
+
original_content: Original CSV or text content
|
|
264
|
+
generated_process: Generated process dictionary to validate
|
|
265
|
+
|
|
266
|
+
Returns:
|
|
267
|
+
Dictionary containing isValid, score, criticalIssues, improvementSuggestions
|
|
268
|
+
|
|
269
|
+
Raises:
|
|
270
|
+
Exception: If validation fails
|
|
271
|
+
|
|
272
|
+
Delegates to business_validator_agent singleton.
|
|
273
|
+
"""
|
|
274
|
+
result = await business_validator_agent.execute({
|
|
275
|
+
"original_content": original_content,
|
|
276
|
+
"generated_process": generated_process,
|
|
277
|
+
})
|
|
278
|
+
|
|
279
|
+
return result
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""Wire-contract Pydantic v2 frozen models for the public surface.
|
|
2
|
+
|
|
3
|
+
Authored at Wave 6.5 Conv B per Decision #160 (interface-freeze pattern,
|
|
4
|
+
P-3 lock) — every cross-package symbol exposed by this sibling-package
|
|
5
|
+
becomes a frozen v2 model so Wave 7 Extraction is mechanical (filter-
|
|
6
|
+
repo + pin update; no behaviour rewrite).
|
|
7
|
+
|
|
8
|
+
Public surface (frozen V1):
|
|
9
|
+
* ``KNOWLEDGE_PACKAGE_VERSION`` — the wire-contract version string.
|
|
10
|
+
Bump via dedicated migration story; out-of-band bumps trip
|
|
11
|
+
``test_contracts_version_pinned``.
|
|
12
|
+
* ``KnowledgeKind`` — enum of knowledge-pack kinds carved at Wave 6.5
|
|
13
|
+
Conv B (covers all 4 PRD §3 #6 sub-layers).
|
|
14
|
+
* ``KnowledgeLifecycleState`` — enum mirroring
|
|
15
|
+
``lifecycle.LIFECYCLE_STATES`` (4 states, PRD §8.11.3 verbatim).
|
|
16
|
+
* ``KnowledgePackManifest`` — the YAML manifest shape; one file per
|
|
17
|
+
knowledge pack under ``manifests/*.yaml``.
|
|
18
|
+
|
|
19
|
+
Pydantic v2 ``model_config = ConfigDict(frozen=True)`` enforces
|
|
20
|
+
immutability after construction — the registry and downstream consumers
|
|
21
|
+
(kernel manifest harvest, capability registry) treat manifests as
|
|
22
|
+
read-only descriptors. Mutations require constructing a new instance.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from enum import Enum
|
|
27
|
+
from typing import List, Optional
|
|
28
|
+
|
|
29
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
30
|
+
|
|
31
|
+
# Wire-contract version. Authored at Wave 6.5 Conv B. Subsequent bumps
|
|
32
|
+
# trace through a dedicated story + ADR per the interface-freeze
|
|
33
|
+
# discipline (Decision #160). Pinned to the same wave-tag string used
|
|
34
|
+
# by the Wave 6.5 Conv A `plm-skill-packages` carve (`v1.0.0-wave6.5`)
|
|
35
|
+
# so the two siblings share a freeze checkpoint.
|
|
36
|
+
KNOWLEDGE_PACKAGE_VERSION: str = "v1.0.0-wave6.5"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class KnowledgeKind(str, Enum):
|
|
40
|
+
"""Knowledge-pack kinds carved into plm-knowledge at Wave 6.5 Conv B.
|
|
41
|
+
|
|
42
|
+
Mirrors PRD §3 #6 product-line scope across the 4 named sub-layers:
|
|
43
|
+
Foundation, Rule & Policy, Retrieval, Operational. Wave 7+ may
|
|
44
|
+
extend this enum via a wire-contract bump
|
|
45
|
+
(``KNOWLEDGE_PACKAGE_VERSION``).
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
FOUNDATION_DICTIONARY = "foundation_dictionary"
|
|
49
|
+
"""Foundation layer — `01_Dictionnaries/**` markdown corpus."""
|
|
50
|
+
|
|
51
|
+
FOUNDATION_TAXONOMY = "foundation_taxonomy"
|
|
52
|
+
"""Foundation layer — `plm_taxonomy` definitions."""
|
|
53
|
+
|
|
54
|
+
FOUNDATION_EVALUATION_GRID = "foundation_evaluation_grid"
|
|
55
|
+
"""Foundation layer — `plm_evaluation_grids` rubrics."""
|
|
56
|
+
|
|
57
|
+
RULE_POLICY = "rule_policy"
|
|
58
|
+
"""Rule & Policy layer — issue classification + business validator."""
|
|
59
|
+
|
|
60
|
+
RETRIEVAL_RAG = "retrieval_rag"
|
|
61
|
+
"""Retrieval layer — Azure Assistants RAG service."""
|
|
62
|
+
|
|
63
|
+
RETRIEVAL_BM25 = "retrieval_bm25"
|
|
64
|
+
"""Retrieval layer — BM25 search engine over knowledge chunks."""
|
|
65
|
+
|
|
66
|
+
OPERATIONAL = "operational"
|
|
67
|
+
"""Operational layer — V1 placeholder per D-LOCKED-15."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class KnowledgeLifecycleState(str, Enum):
|
|
71
|
+
"""4-state lifecycle FSM mirror — PRD §8.11.3 verbatim (Decision #167).
|
|
72
|
+
|
|
73
|
+
Mirrors ``plm_knowledge.lifecycle.LIFECYCLE_STATES`` 1:1; the enum
|
|
74
|
+
exists to give Pydantic a constrained validator for the
|
|
75
|
+
``lifecycle_state`` field of ``KnowledgePackManifest`` without a
|
|
76
|
+
free-form string.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
DRAFT = "draft"
|
|
80
|
+
RELEASED = "released"
|
|
81
|
+
DEPRECATED = "deprecated"
|
|
82
|
+
RETIRED = "retired"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class KnowledgePackManifest(BaseModel):
|
|
86
|
+
"""YAML knowledge-pack manifest — one file per pack under ``manifests/*.yaml``.
|
|
87
|
+
|
|
88
|
+
Mirrors the Wave 6.5 Conv A `plm_skill_packages.contracts.SkillManifest`
|
|
89
|
+
template (Decision #168 = YAML + filesystem precedent) — manifests
|
|
90
|
+
load at boot via ``manifests.load_all()``, matching the
|
|
91
|
+
``litellm_config/model_list.yaml`` catalog-as-truth pattern. No SQL,
|
|
92
|
+
no admin UI in V1.
|
|
93
|
+
|
|
94
|
+
Frozen-v2 model — mutations require constructing a new instance.
|
|
95
|
+
"""
|
|
96
|
+
|
|
97
|
+
model_config = ConfigDict(frozen=True, extra="forbid")
|
|
98
|
+
|
|
99
|
+
pack_id: str = Field(
|
|
100
|
+
...,
|
|
101
|
+
description="Stable identifier (e.g. 'foundation.dictionnaries'). Matches the YAML file basename.",
|
|
102
|
+
min_length=3,
|
|
103
|
+
max_length=128,
|
|
104
|
+
)
|
|
105
|
+
"""Stable identifier; the basename of the manifest YAML file (less
|
|
106
|
+
extension). Format: ``<sub_layer>.<pack_name>``."""
|
|
107
|
+
|
|
108
|
+
kind: KnowledgeKind = Field(
|
|
109
|
+
...,
|
|
110
|
+
description="Knowledge-pack kind enum (sub-layer assignment).",
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
title: str = Field(
|
|
114
|
+
...,
|
|
115
|
+
description="Human-readable title shown in admin UI / capability registry.",
|
|
116
|
+
min_length=3,
|
|
117
|
+
max_length=200,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
description: str = Field(
|
|
121
|
+
...,
|
|
122
|
+
description="One-line description of the knowledge pack's surface.",
|
|
123
|
+
min_length=5,
|
|
124
|
+
max_length=500,
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
version: str = Field(
|
|
128
|
+
...,
|
|
129
|
+
description="Knowledge-pack version (semver-style). Bumps trace through a dedicated story.",
|
|
130
|
+
pattern=r"^[0-9]+\.[0-9]+\.[0-9]+(?:-[A-Za-z0-9.-]+)?$",
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
lifecycle_state: KnowledgeLifecycleState = Field(
|
|
134
|
+
...,
|
|
135
|
+
description="Current lifecycle state (PRD §8.11.3 4-state).",
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
capability_id: Optional[str] = Field(
|
|
139
|
+
default=None,
|
|
140
|
+
description="Optional pointer into plm_shared.capability_registry.",
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
autonomy_level: Optional[str] = Field(
|
|
144
|
+
default=None,
|
|
145
|
+
description="Optional pointer into plm_shared autonomy levels (CR.10).",
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
entry_module: str = Field(
|
|
149
|
+
...,
|
|
150
|
+
description="Dotted module path inside plm_knowledge (e.g. 'plm_knowledge.rag_service').",
|
|
151
|
+
pattern=r"^plm_knowledge(?:\.[A-Za-z_][A-Za-z0-9_]*)+$",
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
entry_callable: Optional[str] = Field(
|
|
155
|
+
default=None,
|
|
156
|
+
description="Optional callable name within the entry module (function or class). Foundation packs may omit this when the entry module is a static asset directory.",
|
|
157
|
+
max_length=128,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
tags: List[str] = Field(
|
|
161
|
+
default_factory=list,
|
|
162
|
+
description="Optional tags for capability registry filtering.",
|
|
163
|
+
max_length=32,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
notes: Optional[str] = Field(
|
|
167
|
+
default=None,
|
|
168
|
+
description="Free-form notes; not parsed by runtime.",
|
|
169
|
+
max_length=2000,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
__all__ = [
|
|
174
|
+
"KNOWLEDGE_PACKAGE_VERSION",
|
|
175
|
+
"KnowledgeKind",
|
|
176
|
+
"KnowledgeLifecycleState",
|
|
177
|
+
"KnowledgePackManifest",
|
|
178
|
+
]
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Foundation layer — thin reader for `01_Dictionnaries/`.
|
|
2
|
+
|
|
3
|
+
Wave 6.5 Conv B per Decision #171 = (c) "defer relocation to Wave 7".
|
|
4
|
+
The 146 markdown files at `01_Dictionnaries/` (repo root) stay in place
|
|
5
|
+
through Wave 6.5; this loader resolves the path through the existing
|
|
6
|
+
backend `config.dictionnaries_path` setting (already configurable) so
|
|
7
|
+
no consumer needs a path change at this conv.
|
|
8
|
+
|
|
9
|
+
Public surface:
|
|
10
|
+
* ``resolve_foundation_root(override=None)`` — returns the directory
|
|
11
|
+
Path. With ``override=None``, resolves via the backend
|
|
12
|
+
``config.settings.dictionnaries_path`` (default `../../01_Dictionnaries`
|
|
13
|
+
relative to the backend module). The override exists for tests +
|
|
14
|
+
admin tooling that want to point at a snapshot directory.
|
|
15
|
+
* ``list_foundation_documents(override=None)`` — enumerate the
|
|
16
|
+
`*.md` files under the resolved root in stable lexicographic order.
|
|
17
|
+
Returns an empty list (with a warning) if the root is missing —
|
|
18
|
+
matches the boot-time pattern in `02_App/backend/main.py:306`
|
|
19
|
+
("PLM Knowledge Engine (BM25 — non-bloquant si 01_Dictionnaries
|
|
20
|
+
absent)").
|
|
21
|
+
|
|
22
|
+
Wave 7 Extraction will physically relocate the directory into
|
|
23
|
+
`02_App/plm-knowledge/foundation/01_Dictionnaries/` and the resolution
|
|
24
|
+
function will switch to a package-relative anchor.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import logging
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
from typing import List, Optional
|
|
31
|
+
|
|
32
|
+
logger = logging.getLogger(__name__)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# Default repo-root path. The backend `config.settings.dictionnaries_path`
|
|
36
|
+
# resolves this same string relative to the backend CWD; here we keep an
|
|
37
|
+
# absolute fallback for sibling-package callers that don't have the
|
|
38
|
+
# backend `config` module loaded (e.g. standalone admin tooling).
|
|
39
|
+
_DEFAULT_REPO_ROOT_FALLBACK: Path = (
|
|
40
|
+
Path(__file__).resolve().parents[3] / "01_Dictionnaries"
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def resolve_foundation_root(override: Optional[Path] = None) -> Path:
|
|
45
|
+
"""Resolve the Foundation layer root directory.
|
|
46
|
+
|
|
47
|
+
Resolution order:
|
|
48
|
+
1. Explicit ``override`` Path (test + admin tooling escape hatch).
|
|
49
|
+
2. Backend ``config.settings.dictionnaries_path`` if importable.
|
|
50
|
+
3. Repo-root fallback ``../../../../01_Dictionnaries`` relative
|
|
51
|
+
to this module.
|
|
52
|
+
|
|
53
|
+
Returns the resolved Path even if the directory does not exist on
|
|
54
|
+
disk; callers that need a stricter contract can call
|
|
55
|
+
``Path.exists()`` + branch on absence.
|
|
56
|
+
"""
|
|
57
|
+
if override is not None:
|
|
58
|
+
return Path(override).resolve()
|
|
59
|
+
# Wave 6.7 Conv A: prefer the plm_shared.protocols settings proxy so
|
|
60
|
+
# the resolution path does not depend on ``02_App/backend/`` at
|
|
61
|
+
# import time. The proxy raises ``ProtocolNotRegistered`` on
|
|
62
|
+
# attribute access when the host has not wired the DI registry
|
|
63
|
+
# (e.g. truly standalone foundation_loader tests); fall back to the
|
|
64
|
+
# legacy ``config`` import in that case, then to the absolute repo-
|
|
65
|
+
# root fallback.
|
|
66
|
+
configured = None
|
|
67
|
+
try:
|
|
68
|
+
from plm_shared.protocols.services.settings import settings as _shared_settings
|
|
69
|
+
|
|
70
|
+
configured = getattr(_shared_settings, "dictionnaries_path", None)
|
|
71
|
+
except Exception:
|
|
72
|
+
configured = None
|
|
73
|
+
if configured is None:
|
|
74
|
+
try:
|
|
75
|
+
from plm_shared.settings import settings # type: ignore[import-not-found]
|
|
76
|
+
|
|
77
|
+
configured = getattr(settings, "dictionnaries_path", None)
|
|
78
|
+
except ImportError:
|
|
79
|
+
configured = None
|
|
80
|
+
if configured is None:
|
|
81
|
+
return _DEFAULT_REPO_ROOT_FALLBACK
|
|
82
|
+
return Path(configured).resolve()
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def list_foundation_documents(override: Optional[Path] = None) -> List[Path]:
|
|
86
|
+
"""Enumerate ``*.md`` files under the Foundation root.
|
|
87
|
+
|
|
88
|
+
Returns an empty list (with a warning logged) if the root is
|
|
89
|
+
missing — matches the non-blocking startup pattern in
|
|
90
|
+
`backend/main.py:306`. Sub-directories are walked recursively;
|
|
91
|
+
output is in stable lexicographic order.
|
|
92
|
+
"""
|
|
93
|
+
root = resolve_foundation_root(override)
|
|
94
|
+
if not root.exists():
|
|
95
|
+
logger.warning(
|
|
96
|
+
"foundation_root_missing path=%s — returning empty document list",
|
|
97
|
+
root,
|
|
98
|
+
)
|
|
99
|
+
return []
|
|
100
|
+
if not root.is_dir():
|
|
101
|
+
logger.warning(
|
|
102
|
+
"foundation_root_not_a_directory path=%s — returning empty document list",
|
|
103
|
+
root,
|
|
104
|
+
)
|
|
105
|
+
return []
|
|
106
|
+
documents = sorted(root.rglob("*.md"))
|
|
107
|
+
return documents
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
__all__ = [
|
|
111
|
+
"resolve_foundation_root",
|
|
112
|
+
"list_foundation_documents",
|
|
113
|
+
]
|