plm-knowledge 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. plm_knowledge-1.0.0/PKG-INFO +93 -0
  2. plm_knowledge-1.0.0/README.md +75 -0
  3. plm_knowledge-1.0.0/plm_knowledge/__init__.py +92 -0
  4. plm_knowledge-1.0.0/plm_knowledge/business_validator.py +279 -0
  5. plm_knowledge-1.0.0/plm_knowledge/contracts.py +178 -0
  6. plm_knowledge-1.0.0/plm_knowledge/foundation_loader.py +113 -0
  7. plm_knowledge-1.0.0/plm_knowledge/issue_classification.py +155 -0
  8. plm_knowledge-1.0.0/plm_knowledge/knowledge_pack_loader.py +177 -0
  9. plm_knowledge-1.0.0/plm_knowledge/knowledge_search_engine.py +140 -0
  10. plm_knowledge-1.0.0/plm_knowledge/lifecycle.py +115 -0
  11. plm_knowledge-1.0.0/plm_knowledge/manifests/foundation.dictionnaries.yaml +17 -0
  12. plm_knowledge-1.0.0/plm_knowledge/manifests/foundation.evaluation_grids.yaml +16 -0
  13. plm_knowledge-1.0.0/plm_knowledge/manifests/foundation.taxonomy.yaml +15 -0
  14. plm_knowledge-1.0.0/plm_knowledge/manifests/operational.placeholder.yaml +18 -0
  15. plm_knowledge-1.0.0/plm_knowledge/manifests/retrieval.bm25.yaml +20 -0
  16. plm_knowledge-1.0.0/plm_knowledge/manifests/retrieval.rag_service.yaml +21 -0
  17. plm_knowledge-1.0.0/plm_knowledge/manifests/rule.business_validator.yaml +20 -0
  18. plm_knowledge-1.0.0/plm_knowledge/manifests/rule.issue_classification.yaml +19 -0
  19. plm_knowledge-1.0.0/plm_knowledge/manifests.py +108 -0
  20. plm_knowledge-1.0.0/plm_knowledge/operational.py +96 -0
  21. plm_knowledge-1.0.0/plm_knowledge/rag_service.py +371 -0
  22. plm_knowledge-1.0.0/plm_knowledge/registry.py +112 -0
  23. plm_knowledge-1.0.0/plm_knowledge.egg-info/PKG-INFO +93 -0
  24. plm_knowledge-1.0.0/plm_knowledge.egg-info/SOURCES.txt +35 -0
  25. plm_knowledge-1.0.0/plm_knowledge.egg-info/dependency_links.txt +1 -0
  26. plm_knowledge-1.0.0/plm_knowledge.egg-info/requires.txt +10 -0
  27. plm_knowledge-1.0.0/plm_knowledge.egg-info/top_level.txt +1 -0
  28. plm_knowledge-1.0.0/pyproject.toml +153 -0
  29. plm_knowledge-1.0.0/setup.cfg +4 -0
  30. plm_knowledge-1.0.0/tests/test_contracts.py +113 -0
  31. plm_knowledge-1.0.0/tests/test_foundation_loader.py +60 -0
  32. plm_knowledge-1.0.0/tests/test_lifecycle.py +88 -0
  33. plm_knowledge-1.0.0/tests/test_manifests.py +125 -0
  34. plm_knowledge-1.0.0/tests/test_operational.py +48 -0
  35. plm_knowledge-1.0.0/tests/test_public_surface.py +47 -0
  36. plm_knowledge-1.0.0/tests/test_registry.py +123 -0
  37. plm_knowledge-1.0.0/tests/test_smoke.py +53 -0
@@ -0,0 +1,93 @@
1
+ Metadata-Version: 2.4
2
+ Name: plm-knowledge
3
+ Version: 1.0.0
4
+ Summary: TracePulse PLM Knowledge product-line — BM25 retrieval, RAG, business rules, lifecycle FSM, knowledge-pack manifests (Wave 6.5 Conv B / PRD §3 #6). Full description in README.
5
+ Author: TracePulse
6
+ License: Proprietary
7
+ Requires-Python: >=3.12
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: plm-shared
10
+ Requires-Dist: pydantic>=2.5.0
11
+ Requires-Dist: PyYAML>=6.0
12
+ Requires-Dist: rank-bm25>=0.2.2
13
+ Requires-Dist: openai>=1.16.0
14
+ Requires-Dist: httpx>=0.27.0
15
+ Provides-Extra: test
16
+ Requires-Dist: pytest; extra == "test"
17
+ Requires-Dist: pytest-asyncio; extra == "test"
18
+
19
+ # plm-knowledge
20
+
21
+ TracePulse PLM Knowledge product-line. Wave 6.5 Conv B sibling-package
22
+ carve per PRD §3 #6.
23
+
24
+ ## Scope (PRD §3 #6 verbatim — 4 sub-layers)
25
+
26
+ | Sub-layer | Source | Carve disposition |
27
+ |-----------|--------|-------------------|
28
+ | **Foundation** | `01_Dictionnaries/**` (146 .md files at repo root) + `plm_taxonomy` + `plm_evaluation_grids` | **Path stays at repo root through Wave 6.5** per Decision #171 = (c) defer relocation to Wave 7. `plm_knowledge.foundation_loader` is a thin reader that resolves the path through `config.dictionnaries_path` (already configurable); no `git mv` of the dictionary tree. |
29
+ | **Rule & Policy** | `02_App/backend/services/issue_classification.py` + `02_App/backend/agents/business_validator.py` | Lifted via `git mv` to `02_App/plm-knowledge/plm_knowledge/{issue_classification,business_validator}.py`. Re-export shims at every old monolith path. |
30
+ | **Retrieval** | `02_App/backend/services/rag_service.py` + `02_App/backend/skills/knowledge_pack_loader.py` + `02_App/backend/skills/knowledge_search_engine.py` (BM25) | Lifted via `git mv` to `02_App/plm-knowledge/plm_knowledge/{rag_service,knowledge_pack_loader,knowledge_search_engine}.py`. Re-export shims at every old monolith path. |
31
+ | **Operational** | V1 placeholder per D-LOCKED-15 + Decision #172 = (a) keep in-process state V1 | `plm_knowledge.operational` exposes a stable but inert public surface so downstream callers can target the future operational telemetry hooks without an interface bump. |
32
+
33
+ ## Public surface (frozen V1 — `KNOWLEDGE_PACKAGE_VERSION = "v1.0.0-wave6.5"`)
34
+
35
+ * The 5 lifted modules at top level of the package (`rag_service`,
36
+ `issue_classification`, `business_validator`, `knowledge_pack_loader`,
37
+ `knowledge_search_engine`).
38
+ * `plm_knowledge.foundation_loader` — thin reader for the Foundation
39
+ layer (path resolution + .md enumeration).
40
+ * `plm_knowledge.operational` — V1 placeholder (per D-LOCKED-15 +
41
+ Decision #172).
42
+ * `plm_knowledge.lifecycle` — 4-state FSM aligned verbatim with PRD
43
+ §8.11.3 (mirror of Wave 6.5 Conv A `plm-skill-packages` per Decision
44
+ #167).
45
+ * `plm_knowledge.manifests` — YAML manifest loader (mirror of Decision
46
+ #168 precedent — YAML + filesystem; no SQL, no admin UI in V1).
47
+ * `plm_knowledge.contracts` — wire-contract Pydantic v2 frozen models
48
+ for the public surface (Decision #160 interface-freeze pattern).
49
+ * `plm_knowledge.registry` — manifest+lifecycle registry exposed to
50
+ the kernel (capability registry callers).
51
+
52
+ ## Lifecycle FSM (PRD §8.11.3 verbatim — Decision #167 mirrored)
53
+
54
+ ```
55
+ draft ──► released ──► deprecated ──► retired
56
+ │
57
+ └──► (terminal)
58
+ ```
59
+
60
+ States are **strictly forward-flowing**; the FSM rejects skip
61
+ transitions, out-of-terminal moves, and unknown states. Mirrors the
62
+ 4-state contract in `plm_shared.capability_registry`.
63
+
64
+ ## Transitional dependency note (Wave 6.5 → Wave 7 Extraction)
65
+
66
+ Lifted code retains its existing `from services.X` / `from agents.base_agent` /
67
+ `from config import settings` imports. These resolve at runtime in the
68
+ backend process via PYTHONPATH (also via `pip install -e ../plm-knowledge`
69
+ declared in `02_App/backend/requirements.txt`); sibling tests configure
70
+ `tests/conftest.py` to extend `sys.path`. Wave 7 Extraction physically
71
+ separates the repos and at that point inbound deps refactor to
72
+ ports/Protocols. Pattern matches Wave 6.5 Conv A `plm-skill-packages`
73
+ precedent verbatim.
74
+
75
+ ## Re-export shims
76
+
77
+ The following monolith paths are thin re-export shims pointing to the
78
+ canonical sibling-package modules. They keep legacy callers working
79
+ through Wave 7 Extraction (drop after 90-day soak per US-DC.1).
80
+
81
+ * `02_App/backend/services/rag_service.py`
82
+ * `02_App/backend/services/issue_classification.py`
83
+ * `02_App/backend/agents/business_validator.py`
84
+ * `02_App/backend/skills/knowledge_pack_loader.py`
85
+ * `02_App/backend/skills/knowledge_search_engine.py`
86
+
87
+ ## References
88
+
89
+ * PRD §3 #6 — `plm-knowledge` product-line definition
90
+ * PRD §8.11.3 — 4-state capability lifecycle FSM
91
+ * WAVE-6-READINESS-SWEEP.md §11.2 — Wave 6.5 conv-by-conv
92
+ * W6.5-CONV-A-FOLLOWUPS.md — Wave 6.5 Conv B warm-up
93
+ * Decisions #170-#173 — Wave 6.5 Conv B Q-W6.5B-* answers
@@ -0,0 +1,75 @@
1
+ # plm-knowledge
2
+
3
+ TracePulse PLM Knowledge product-line. Wave 6.5 Conv B sibling-package
4
+ carve per PRD §3 #6.
5
+
6
+ ## Scope (PRD §3 #6 verbatim — 4 sub-layers)
7
+
8
+ | Sub-layer | Source | Carve disposition |
9
+ |-----------|--------|-------------------|
10
+ | **Foundation** | `01_Dictionnaries/**` (146 .md files at repo root) + `plm_taxonomy` + `plm_evaluation_grids` | **Path stays at repo root through Wave 6.5** per Decision #171 = (c) defer relocation to Wave 7. `plm_knowledge.foundation_loader` is a thin reader that resolves the path through `config.dictionnaries_path` (already configurable); no `git mv` of the dictionary tree. |
11
+ | **Rule & Policy** | `02_App/backend/services/issue_classification.py` + `02_App/backend/agents/business_validator.py` | Lifted via `git mv` to `02_App/plm-knowledge/plm_knowledge/{issue_classification,business_validator}.py`. Re-export shims at every old monolith path. |
12
+ | **Retrieval** | `02_App/backend/services/rag_service.py` + `02_App/backend/skills/knowledge_pack_loader.py` + `02_App/backend/skills/knowledge_search_engine.py` (BM25) | Lifted via `git mv` to `02_App/plm-knowledge/plm_knowledge/{rag_service,knowledge_pack_loader,knowledge_search_engine}.py`. Re-export shims at every old monolith path. |
13
+ | **Operational** | V1 placeholder per D-LOCKED-15 + Decision #172 = (a) keep in-process state V1 | `plm_knowledge.operational` exposes a stable but inert public surface so downstream callers can target the future operational telemetry hooks without an interface bump. |
14
+
15
+ ## Public surface (frozen V1 — `KNOWLEDGE_PACKAGE_VERSION = "v1.0.0-wave6.5"`)
16
+
17
+ * The 5 lifted modules at top level of the package (`rag_service`,
18
+ `issue_classification`, `business_validator`, `knowledge_pack_loader`,
19
+ `knowledge_search_engine`).
20
+ * `plm_knowledge.foundation_loader` — thin reader for the Foundation
21
+ layer (path resolution + .md enumeration).
22
+ * `plm_knowledge.operational` — V1 placeholder (per D-LOCKED-15 +
23
+ Decision #172).
24
+ * `plm_knowledge.lifecycle` — 4-state FSM aligned verbatim with PRD
25
+ §8.11.3 (mirror of Wave 6.5 Conv A `plm-skill-packages` per Decision
26
+ #167).
27
+ * `plm_knowledge.manifests` — YAML manifest loader (mirror of Decision
28
+ #168 precedent — YAML + filesystem; no SQL, no admin UI in V1).
29
+ * `plm_knowledge.contracts` — wire-contract Pydantic v2 frozen models
30
+ for the public surface (Decision #160 interface-freeze pattern).
31
+ * `plm_knowledge.registry` — manifest+lifecycle registry exposed to
32
+ the kernel (capability registry callers).
33
+
34
+ ## Lifecycle FSM (PRD §8.11.3 verbatim — Decision #167 mirrored)
35
+
36
+ ```
37
+ draft ──► released ──► deprecated ──► retired
38
+ │
39
+ └──► (terminal)
40
+ ```
41
+
42
+ States are **strictly forward-flowing**; the FSM rejects skip
43
+ transitions, out-of-terminal moves, and unknown states. Mirrors the
44
+ 4-state contract in `plm_shared.capability_registry`.
45
+
46
+ ## Transitional dependency note (Wave 6.5 → Wave 7 Extraction)
47
+
48
+ Lifted code retains its existing `from services.X` / `from agents.base_agent` /
49
+ `from config import settings` imports. These resolve at runtime in the
50
+ backend process via PYTHONPATH (also via `pip install -e ../plm-knowledge`
51
+ declared in `02_App/backend/requirements.txt`); sibling tests configure
52
+ `tests/conftest.py` to extend `sys.path`. Wave 7 Extraction physically
53
+ separates the repos and at that point inbound deps refactor to
54
+ ports/Protocols. Pattern matches Wave 6.5 Conv A `plm-skill-packages`
55
+ precedent verbatim.
56
+
57
+ ## Re-export shims
58
+
59
+ The following monolith paths are thin re-export shims pointing to the
60
+ canonical sibling-package modules. They keep legacy callers working
61
+ through Wave 7 Extraction (drop after 90-day soak per US-DC.1).
62
+
63
+ * `02_App/backend/services/rag_service.py`
64
+ * `02_App/backend/services/issue_classification.py`
65
+ * `02_App/backend/agents/business_validator.py`
66
+ * `02_App/backend/skills/knowledge_pack_loader.py`
67
+ * `02_App/backend/skills/knowledge_search_engine.py`
68
+
69
+ ## References
70
+
71
+ * PRD §3 #6 — `plm-knowledge` product-line definition
72
+ * PRD §8.11.3 — 4-state capability lifecycle FSM
73
+ * WAVE-6-READINESS-SWEEP.md §11.2 — Wave 6.5 conv-by-conv
74
+ * W6.5-CONV-A-FOLLOWUPS.md — Wave 6.5 Conv B warm-up
75
+ * Decisions #170-#173 — Wave 6.5 Conv B Q-W6.5B-* answers
@@ -0,0 +1,92 @@
1
+ """TracePulse PLM Knowledge — Wave 6.5 Conv B.
2
+
3
+ PRD §3 #6 product-line carve. Hosts the 4-sub-layer knowledge stack:
4
+
5
+ * **Foundation** — `01_Dictionnaries/**` (146 .md files at repo root)
6
+ + `plm_taxonomy` + `plm_evaluation_grids`. Path stays at repo root
7
+ through Wave 6.5 per Decision #171 = (c) defer relocation to Wave 7;
8
+ the ``foundation_loader`` module is a thin reader.
9
+ * **Rule & Policy** — ``issue_classification`` + ``business_validator``
10
+ (lifted via `git mv` from the monolith).
11
+ * **Retrieval** — ``rag_service`` + ``knowledge_pack_loader`` +
12
+ ``knowledge_search_engine`` (BM25; lifted via `git mv` from the
13
+ monolith).
14
+ * **Operational** — V1 placeholder per D-LOCKED-15 + Decision #172.
15
+
16
+ Public surface:
17
+ * The 5 lifted modules at top level (``rag_service``,
18
+ ``issue_classification``, ``business_validator``,
19
+ ``knowledge_pack_loader``, ``knowledge_search_engine``).
20
+ * ``foundation_loader`` — Foundation layer thin reader.
21
+ * ``operational`` — Operational layer V1 placeholder.
22
+ * ``lifecycle`` — 4-state FSM aligned verbatim with PRD §8.11.3
23
+ (Decision #167 mirrored).
24
+ * ``manifests`` — YAML manifest loader (Decision #168 precedent =
25
+ YAML + filesystem).
26
+ * ``contracts`` — wire-contract Pydantic v2 frozen models for the
27
+ public surface (Decision #160 interface-freeze pattern).
28
+ * ``registry`` — manifest+lifecycle registry exposed to the kernel
29
+ (capability registry callers).
30
+
31
+ Re-export shims at every monolith path keep legacy callers working
32
+ through Wave 7 Extraction (Wave 1 W1.4 RunTaskTracker template + Wave
33
+ 6.5 Conv A precedent).
34
+ """
35
+
36
+ from plm_knowledge.contracts import (
37
+ KNOWLEDGE_PACKAGE_VERSION,
38
+ KnowledgeKind,
39
+ KnowledgeLifecycleState,
40
+ KnowledgePackManifest,
41
+ )
42
+ from plm_knowledge.lifecycle import (
43
+ LIFECYCLE_STATES,
44
+ LIFECYCLE_TRANSITIONS,
45
+ InvalidLifecycleTransition,
46
+ can_transition,
47
+ is_terminal,
48
+ next_state,
49
+ )
50
+ from plm_knowledge.manifests import (
51
+ MANIFESTS_DIR,
52
+ ManifestLoadError,
53
+ load_all,
54
+ load_manifest,
55
+ )
56
+ from plm_knowledge.registry import (
57
+ KnowledgePackRegistry,
58
+ get_knowledge_registry,
59
+ )
60
+ from plm_knowledge.foundation_loader import (
61
+ list_foundation_documents,
62
+ resolve_foundation_root,
63
+ )
64
+ from plm_knowledge.operational import (
65
+ OPERATIONAL_LAYER_VERSION,
66
+ OperationalAck,
67
+ record_operational_event,
68
+ )
69
+
70
+ __all__ = [
71
+ "KNOWLEDGE_PACKAGE_VERSION",
72
+ "KnowledgeKind",
73
+ "KnowledgeLifecycleState",
74
+ "KnowledgePackManifest",
75
+ "LIFECYCLE_STATES",
76
+ "LIFECYCLE_TRANSITIONS",
77
+ "InvalidLifecycleTransition",
78
+ "can_transition",
79
+ "is_terminal",
80
+ "next_state",
81
+ "MANIFESTS_DIR",
82
+ "ManifestLoadError",
83
+ "load_all",
84
+ "load_manifest",
85
+ "KnowledgePackRegistry",
86
+ "get_knowledge_registry",
87
+ "list_foundation_documents",
88
+ "resolve_foundation_root",
89
+ "OPERATIONAL_LAYER_VERSION",
90
+ "OperationalAck",
91
+ "record_operational_event",
92
+ ]
@@ -0,0 +1,279 @@
1
+ """
2
+ Business Validator Agent
3
+ Validates generated BPMN processes against original source data.
4
+ """
5
+
6
+ import json
7
+ from typing import Dict, Any, Optional
8
+
9
+ from plm_shared.agents.base_agent import BaseAgent
10
+ from plm_shared.protocols.services.llm import llm_service
11
+
12
+
13
+ # Define the response schema for validation as a plain dict
14
+ VALIDATION_SCHEMA = {
15
+ "type": "object",
16
+ "properties": {
17
+ "isValid": {
18
+ "type": "boolean",
19
+ "description": "True ONLY if the process perfectly matches ALL source data patterns and is syntactically perfect."
20
+ },
21
+ "score": {
22
+ "type": "number",
23
+ "description": "A confidence score from 0 to 100 based on data coverage."
24
+ },
25
+ "keywords": {
26
+ "type": "array",
27
+ "items": {"type": "string"},
28
+ "description": "3 to 10 short thematic keywords (domains, roles, themes) for the process. Lowercase. No duplicates."
29
+ },
30
+ "criticalIssues": {
31
+ "type": "array",
32
+ "items": {"type": "string"},
33
+ "description": "List of logical errors, missing activities from CSV, BPMN violations, or BPMN elements missing a precise expert description in documentation/persona/system."
34
+ },
35
+ "improvementSuggestions": {
36
+ "type": "string",
37
+ "description": "Technical instructions for the analyst to fix the issues, including completing expert-level documentation for each BPMN element."
38
+ }
39
+ },
40
+ "required": ["isValid", "score", "criticalIssues", "improvementSuggestions", "keywords"]
41
+ }
42
+
43
+
44
+ SYSTEM_INSTRUCTION = """You are a Senior Process Auditor & BPMN Validator.
45
+ Your goal is to find ANY discrepancy between the raw source data (CSV/Text) and the AI-generated model.
46
+
47
+ STRICT CRITERIA:
48
+ 1. EXHAUSTIVE COVERAGE: Every unique activity found in the CSV 'Activity' or 'Event' column MUST be represented as a Task in the BPMN.
49
+ 2. SEQUENCE INTEGRITY: Timestamps in CSV must justify the flow. If 'Invoice' always follows 'Receipt', the BPMN must show this.
50
+ 3. BPMN 2.0 SYNTAX:
51
+ - Must have exactly ONE Start Event.
52
+ - Must have at least ONE End Event.
53
+ - All nodes must be connected (no islands).
54
+ - Gateways must have clear labels for their outgoing flows.
55
+ 4. METRICS REALISM: Execution counts in 'miningData' must sum up logically (e.g., sum of branches = entry count).
56
+ 5. BUSINESS METADATA COMPLETENESS: Every step in the BPMN diagram (tasks, start/end events, gateways) MUST have a precise, expert-level description in <bpmn:documentation>. The description must be specific to the element (what it represents in the process, business meaning), not generic. For each <bpmn:task>: require <bpmn:documentation>; tasks that involve a responsible role or an IT system must have <bpmn:extensionElements> with <persona> and/or <system>. For start event: documentation should state what triggers the process. For end event(s): what outcome is reached. For gateways: what decision or split is represented. If any element lacks a precise, expert description or missing persona/system where relevant, add a critical issue and in improvementSuggestions ask to complete these properties with expert-level wording.
57
+
58
+ BE PEDANTIC. If a single task is missing or a flow is illogical, set isValid to false.
59
+
60
+ KEYWORDS REQUIREMENT:
61
+ - Provide 3 to 10 short thematic keywords for the process.
62
+ - Derive them from the source data and the generated BPMN (domains, roles, themes).
63
+ - Use lowercase, avoid duplicates.
64
+ """
65
+
66
+
67
+ class BusinessValidatorAgent(BaseAgent):
68
+ """
69
+ Business Validator Agent
70
+
71
+ Validates generated BPMN processes against original source data.
72
+ Acts as a Senior Process Auditor checking:
73
+
74
+ 1. EXHAUSTIVE COVERAGE: Every activity in source data must be in BPMN
75
+ 2. SEQUENCE INTEGRITY: Timestamps must justify the flow
76
+ 3. BPMN 2.0 SYNTAX: Valid structure, connected nodes
77
+ 4. METRICS REALISM: Execution counts add up logically
78
+ 5. BUSINESS METADATA: Precise descriptions and persona/system assignments
79
+
80
+ Input:
81
+ - original_content: Original CSV or text source data
82
+ - generated_process: Generated process dictionary to validate
83
+
84
+ Output:
85
+ - isValid: Boolean validation result
86
+ - score: Confidence score (0-100)
87
+ - keywords: Extracted thematic keywords
88
+ - criticalIssues: List of found issues
89
+ - improvementSuggestions: Instructions for fixing issues
90
+ """
91
+
92
+ def __init__(self):
93
+ super().__init__(
94
+ name="business_validator",
95
+ description="Validates generated BPMN processes against original source data"
96
+ )
97
+
98
+ def get_output_schema(self) -> Dict[str, Any]:
99
+ """Define business validator output schema."""
100
+ return {
101
+ "type": "object",
102
+ "properties": {
103
+ "isValid": {"type": "boolean"},
104
+ "score": {"type": "number"},
105
+ "keywords": {"type": "array", "items": {"type": "string"}},
106
+ "criticalIssues": {"type": "array", "items": {"type": "string"}},
107
+ "improvementSuggestions": {"type": "string"},
108
+ },
109
+ "required": ["isValid", "score", "criticalIssues", "improvementSuggestions", "keywords"],
110
+ }
111
+
112
+ def get_required_fields(self) -> list:
113
+ """Return required input fields."""
114
+ return ["original_content", "generated_process"]
115
+
116
+ def validate_input(self, input_data: Dict[str, Any]) -> bool:
117
+ """Validate business validator input."""
118
+ if "original_content" not in input_data:
119
+ raise ValueError("Missing required field: original_content")
120
+
121
+ if "generated_process" not in input_data:
122
+ raise ValueError("Missing required field: generated_process")
123
+
124
+ original_content = input_data.get("original_content")
125
+ if not original_content or not str(original_content).strip():
126
+ raise ValueError("original_content must be non-empty string")
127
+
128
+ generated_process = input_data.get("generated_process")
129
+ if not generated_process or not isinstance(generated_process, (dict, str)):
130
+ raise ValueError("generated_process must be dict or string")
131
+
132
+ return True
133
+
134
+ async def execute(self, input_data: Dict[str, Any]) -> Dict[str, Any]:
135
+ """Execute business validation workflow."""
136
+ self.validate_input(input_data)
137
+
138
+ original_content = input_data.get("original_content")
139
+ generated_process = input_data.get("generated_process")
140
+
141
+ return await self._validate_process(original_content, generated_process)
142
+
143
+ async def _validate_process(
144
+ self,
145
+ original_content: str,
146
+ generated_process: Dict[str, Any]
147
+ ) -> Dict[str, Any]:
148
+ """
149
+ Validate a generated BPMN process against original source data using ADK multi-LLM.
150
+
151
+ Args:
152
+ original_content: Original CSV or text content
153
+ generated_process: Generated process dictionary to validate
154
+
155
+ Returns:
156
+ Dictionary containing isValid, score, criticalIssues, improvementSuggestions, keywords
157
+
158
+ Raises:
159
+ Exception: If validation fails
160
+ """
161
+ try:
162
+ print(f"[SCAN] Validating with provider: {llm_service.provider}")
163
+
164
+ # Phase 1: Preprocess large CSV to avoid token overflow
165
+ source_content = await self._preprocess_source_data(original_content)
166
+
167
+ # Phase 2: Build validation prompt
168
+ prompt = self._build_validation_prompt(source_content, generated_process)
169
+
170
+ # Phase 3: Call ADK service
171
+ result = await llm_service.generate_structured(
172
+ system_prompt=SYSTEM_INSTRUCTION,
173
+ user_prompt=prompt,
174
+ output_schema=VALIDATION_SCHEMA,
175
+ temperature=0.3
176
+ )
177
+
178
+ print(f"[OK] Validation complete - Score: {result.get('score', 0)}%")
179
+ return result
180
+
181
+ except Exception as error:
182
+ print(f"[ERR] Validator Agent Error: {error}")
183
+ import traceback
184
+ traceback.print_exc()
185
+ # Return permissive validation on error
186
+ return self._get_default_validation()
187
+
188
+ async def _preprocess_source_data(self, original_content: str) -> str:
189
+ """
190
+ Preprocess large CSV to avoid token overflow.
191
+
192
+ Args:
193
+ original_content: Original source data (CSV or text)
194
+
195
+ Returns:
196
+ Preprocessed source data or original if small enough
197
+ """
198
+ if len(original_content) > 50000 and "Employee_ID" in original_content:
199
+ print("[SCAN] Applying CSV preprocessing for validation...")
200
+ try:
201
+ from plm_shared.utils.csv_preprocessor import smart_csv_preprocessor
202
+ return smart_csv_preprocessor(original_content)
203
+ except ImportError:
204
+ print("[WARN] CSV preprocessor not available, using original content")
205
+ return original_content[:50000]
206
+
207
+ return original_content
208
+
209
+ def _build_validation_prompt(
210
+ self,
211
+ source_content: str,
212
+ generated_process: Dict[str, Any]
213
+ ) -> str:
214
+ """
215
+ Build validation prompt combining source data and generated process.
216
+
217
+ Args:
218
+ source_content: Preprocessed source data
219
+ generated_process: Generated process to validate
220
+
221
+ Returns:
222
+ Formatted validation prompt
223
+ """
224
+ return f"""[SOURCE_DATA_EXTRACT]
225
+ {source_content[:10000]}
226
+
227
+ [GENERATED_BPMN_AND_METRICS]
228
+ {json.dumps(generated_process) if isinstance(generated_process, dict) else str(generated_process)}
229
+ """
230
+
231
+ def _get_default_validation(self) -> Dict[str, Any]:
232
+ """
233
+ Return permissive default validation result on error.
234
+
235
+ Returns:
236
+ Default validation dictionary
237
+ """
238
+ return {
239
+ "isValid": True,
240
+ "score": 100,
241
+ "keywords": [],
242
+ "criticalIssues": [],
243
+ "improvementSuggestions": ""
244
+ }
245
+
246
+
247
+ # Singleton instance
248
+ business_validator_agent = BusinessValidatorAgent()
249
+
250
+
251
+ # ==================== BACKWARD COMPATIBILITY ====================
252
+
253
+ async def validate_process(
254
+ original_content: str,
255
+ generated_process: Dict[str, Any]
256
+ ) -> Dict[str, Any]:
257
+ """
258
+ Backward compatibility wrapper for original function.
259
+
260
+ Validate a generated BPMN process against original source data using ADK multi-LLM
261
+
262
+ Args:
263
+ original_content: Original CSV or text content
264
+ generated_process: Generated process dictionary to validate
265
+
266
+ Returns:
267
+ Dictionary containing isValid, score, criticalIssues, improvementSuggestions
268
+
269
+ Raises:
270
+ Exception: If validation fails
271
+
272
+ Delegates to business_validator_agent singleton.
273
+ """
274
+ result = await business_validator_agent.execute({
275
+ "original_content": original_content,
276
+ "generated_process": generated_process,
277
+ })
278
+
279
+ return result