structverify 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. structverify/__init__.py +83 -0
  2. structverify/adaptation/__init__.py +0 -0
  3. structverify/adaptation/adapter_trainer.py +341 -0
  4. structverify/adaptation/feedback_store.py +31 -0
  5. structverify/adaptation/kosis_crawler.py +317 -0
  6. structverify/adaptation/sample_builder.py +149 -0
  7. structverify/adaptation/synthetic_generator.py +320 -0
  8. structverify/adaptation/update_embeddings.py +178 -0
  9. structverify/agent/__init__.py +21 -0
  10. structverify/agent/builder_agent.py +226 -0
  11. structverify/agent/conformance_agent.py +171 -0
  12. structverify/agent/dependency_planner.py +151 -0
  13. structverify/agent/indexing_agent.py +153 -0
  14. structverify/agent/indexing_planner.py +169 -0
  15. structverify/agent/integration_example.py +182 -0
  16. structverify/agent/loop.py +1165 -0
  17. structverify/agent/memory.py +207 -0
  18. structverify/agent/planner.py +817 -0
  19. structverify/agent/prompts/__init__.py +15 -0
  20. structverify/agent/prompts/planner_prompts.py +219 -0
  21. structverify/agent/prompts/reflect_prompts.py +387 -0
  22. structverify/agent/reflect.py +227 -0
  23. structverify/agent/runtime_agent.py +1272 -0
  24. structverify/agent/schemas.py +262 -0
  25. structverify/agent/source_profiler.py +229 -0
  26. structverify/agent/tools/__init__.py +64 -0
  27. structverify/agent/tools/base.py +222 -0
  28. structverify/agent/tools/calculate.py +244 -0
  29. structverify/agent/tools/catalog_search.py +859 -0
  30. structverify/agent/tools/deep_explore.py +293 -0
  31. structverify/agent/tools/explore_catalog.py +423 -0
  32. structverify/agent/tools/fetch_evidence.py +922 -0
  33. structverify/agent/tools/finish.py +423 -0
  34. structverify/agent/tools/meta_explore.py +267 -0
  35. structverify/agent/tools/query_rewriter.py +134 -0
  36. structverify/agent/tools/read_original.py +144 -0
  37. structverify/agent/tools/replan.py +365 -0
  38. structverify/agent/workspace.py +958 -0
  39. structverify/api.py +804 -0
  40. structverify/config/default.yaml +350 -0
  41. structverify/core/__init__.py +0 -0
  42. structverify/core/config_loader.py +30 -0
  43. structverify/core/pipeline.py +280 -0
  44. structverify/core/schemas.py +362 -0
  45. structverify/detection/__init__.py +26 -0
  46. structverify/detection/_config.py +163 -0
  47. structverify/detection/_llm.py +24 -0
  48. structverify/detection/candidate/__init__.py +1 -0
  49. structverify/detection/candidate/heuristic.py +60 -0
  50. structverify/detection/candidate/llm.py +51 -0
  51. structverify/detection/candidate_scorer.py +81 -0
  52. structverify/detection/claim_detector.py +164 -0
  53. structverify/detection/claims/__init__.py +1 -0
  54. structverify/detection/claims/worthiness.py +142 -0
  55. structverify/detection/domain/__init__.py +1 -0
  56. structverify/detection/domain/classify.py +84 -0
  57. structverify/detection/domain/preview.py +36 -0
  58. structverify/detection/domain/registry.py +99 -0
  59. structverify/detection/domain_classifier.py +75 -0
  60. structverify/detection/prompts/__init__.py +1 -0
  61. structverify/detection/prompts/candidate.py +38 -0
  62. structverify/detection/prompts/claim_worthiness.py +48 -0
  63. structverify/detection/prompts/domain.py +41 -0
  64. structverify/detection/prompts/schema.py +508 -0
  65. structverify/detection/prompts_loader.py +167 -0
  66. structverify/detection/schema/__init__.py +1 -0
  67. structverify/detection/schema/expand.py +83 -0
  68. structverify/detection/schema/induce.py +441 -0
  69. structverify/detection/schema/regenerate.py +162 -0
  70. structverify/detection/schema/temporal_hints.py +130 -0
  71. structverify/detection/schema/validate.py +193 -0
  72. structverify/detection/schema_inductor.py +112 -0
  73. structverify/detection/synthetic_generator.py +270 -0
  74. structverify/explanation/__init__.py +0 -0
  75. structverify/explanation/_config.py +18 -0
  76. structverify/explanation/_llm.py +25 -0
  77. structverify/explanation/explainer.py +183 -0
  78. structverify/explanation/fallback.py +29 -0
  79. structverify/explanation/formatters.py +75 -0
  80. structverify/explanation/prompts/__init__.py +1 -0
  81. structverify/explanation/prompts/match.py +27 -0
  82. structverify/explanation/prompts/mismatch.py +20 -0
  83. structverify/explanation/prompts/multihop.py +16 -0
  84. structverify/explanation/prompts/unverifiable.py +17 -0
  85. structverify/graph/__init__.py +0 -0
  86. structverify/graph/claim_graph.py +226 -0
  87. structverify/graph/document_graph.py +487 -0
  88. structverify/graph/graph_builder.py +238 -0
  89. structverify/graph/graph_multihop.py +335 -0
  90. structverify/graph/graph_store.py +281 -0
  91. structverify/graph/provenance.py +52 -0
  92. structverify/memory/__init__.py +44 -0
  93. structverify/memory/agent_memory.py +142 -0
  94. structverify/memory/embedder.py +69 -0
  95. structverify/memory/exemplar_store.py +241 -0
  96. structverify/memory/normalizer.py +91 -0
  97. structverify/memory/schema.py +119 -0
  98. structverify/memory/storage/__init__.py +29 -0
  99. structverify/memory/storage/jsonl_store.py +117 -0
  100. structverify/memory/working_memory.py +370 -0
  101. structverify/preprocessing/Dockerfile.scraper +27 -0
  102. structverify/preprocessing/__init__.py +0 -0
  103. structverify/preprocessing/extractor.py +574 -0
  104. structverify/preprocessing/pdf/__init__.py +16 -0
  105. structverify/preprocessing/pdf/fields.py +95 -0
  106. structverify/preprocessing/pdf/markdown.py +107 -0
  107. structverify/preprocessing/pdf/models.py +34 -0
  108. structverify/preprocessing/pdf/ocr.py +172 -0
  109. structverify/preprocessing/pdf/pipeline.py +74 -0
  110. structverify/preprocessing/pdf/reader.py +119 -0
  111. structverify/preprocessing/pdf/scoring.py +61 -0
  112. structverify/preprocessing/scraper_sandbox.py +561 -0
  113. structverify/preprocessing/segmenter.py +48 -0
  114. structverify/preprocessing/sir_builder.py +240 -0
  115. structverify/progress.py +591 -0
  116. structverify/retrieval/__init__.py +0 -0
  117. structverify/retrieval/base.py +208 -0
  118. structverify/retrieval/base_connector.py +85 -0
  119. structverify/retrieval/catalog_ranker.py +300 -0
  120. structverify/retrieval/catalog_search.py +583 -0
  121. structverify/retrieval/chunking.py +92 -0
  122. structverify/retrieval/custom_csv_source.py +386 -0
  123. structverify/retrieval/custom_db_source.py +396 -0
  124. structverify/retrieval/custom_docs_source.py +152 -0
  125. structverify/retrieval/dimension_resolver.py +281 -0
  126. structverify/retrieval/evidence_subgraph.py +63 -0
  127. structverify/retrieval/kosis_connector.py +1192 -0
  128. structverify/retrieval/kosis_relevance.py +142 -0
  129. structverify/retrieval/kosis_source.py +1541 -0
  130. structverify/retrieval/query_builder.py +72 -0
  131. structverify/retrieval/registry.py +133 -0
  132. structverify/retrieval/relevance_judge.py +141 -0
  133. structverify/retrieval/row_matcher.py +267 -0
  134. structverify/storage/__init__.py +0 -0
  135. structverify/storage/db_manager.py +157 -0
  136. structverify/storage/dwh_manager.py +92 -0
  137. structverify/storage/init_db.py +99 -0
  138. structverify/storage/raw_storage.py +29 -0
  139. structverify/training/__init__.py +26 -0
  140. structverify/training/curator.py +124 -0
  141. structverify/training/dataset.py +134 -0
  142. structverify/training/doctor.py +99 -0
  143. structverify/training/evalgate.py +96 -0
  144. structverify/training/generate.py +101 -0
  145. structverify/training/loop.py +116 -0
  146. structverify/training/recipe/train_mlx.py +99 -0
  147. structverify/training/recipe/train_qlora.py +104 -0
  148. structverify/training/tasks.py +79 -0
  149. structverify/utils/__init__.py +0 -0
  150. structverify/utils/embedding_client.py +248 -0
  151. structverify/utils/llm_client.py +809 -0
  152. structverify/utils/logger.py +81 -0
  153. structverify/verification/__init__.py +0 -0
  154. structverify/verification/_config.py +45 -0
  155. structverify/verification/adapters.py +405 -0
  156. structverify/verification/conformance.py +117 -0
  157. structverify/verification/decide_verdict.py +216 -0
  158. structverify/verification/decide_verdict_agent.py +454 -0
  159. structverify/verification/growth_diff.py +267 -0
  160. structverify/verification/row_match.py +345 -0
  161. structverify/verification/units.py +64 -0
  162. structverify/verification/verdict_thresholds.py +232 -0
  163. structverify/verification/verifier.py +84 -0
  164. structverify-0.3.0.dist-info/METADATA +903 -0
  165. structverify-0.3.0.dist-info/RECORD +168 -0
  166. structverify-0.3.0.dist-info/WHEEL +5 -0
  167. structverify-0.3.0.dist-info/licenses/LICENSE +21 -0
  168. structverify-0.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,72 @@
1
+ """
2
+
3
+ retrieval/query_builder.py — 검색 쿼리 생성 (Schema → ConnectorQuery)
4
+
5
+ [김예슬 - 2026-04-30 / v2]
6
+ - build_query()에 raw_claim 추가: LLM Agent가 catalog 검색 시 원문 맥락 활용
7
+ - ConnectorQuery.extra_params에 embedding_text 추가:
8
+ pgvector 유사도 검색에 사용할 텍스트 (indicator + population + time_period 조합)
9
+
10
+ [참고] ProgramFC (Pan et al., NAACL 2023) — https://github.com/mbzuai-nlp/ProgramFC
11
+ structured representation → data source query 변환 구조 참고
12
+ """
13
+ from __future__ import annotations
14
+ from structverify.core.schemas import Claim
15
+ from structverify.retrieval.base_connector import ConnectorQuery
16
+
17
+
18
+ def build_query(claim: Claim) -> ConnectorQuery:
19
+ """
20
+ Claim Schema → 커넥터 검색 쿼리 생성
21
+
22
+ [v4 김예슬] extra_params에 추가:
23
+ - raw_claim: 원문 주장 (LLM Agent catalog 검색 시 맥락)
24
+ - embedding_text: pgvector 유사도 검색용 텍스트
25
+ """
26
+ s = claim.schema
27
+ # [v3 김예슬] context_text 있으면 raw_claim에 포함 → 검색 맥락 강화
28
+ context = getattr(claim, "context_text", None) or claim.claim_text
29
+
30
+ if not s:
31
+ return ConnectorQuery(
32
+ keyword=claim.claim_text[:50],
33
+ extra_params={
34
+ "raw_claim": context,
35
+ "embedding_text": context[:200],
36
+ },
37
+ )
38
+
39
+ # keyword: source_reference(출처기관) + indicator + population 조합
40
+ # 출처기관이 있으면 KOSIS 검색 범위를 해당 기관으로 좁힐 수 있음
41
+ parts = [p for p in [s.source_reference, s.indicator, s.population] if p]
42
+ keyword = " ".join(parts) or claim.claim_text[:50]
43
+
44
+ # "카테고리 > indicator | 항목: indicator | 분류: population | 단위: unit"
45
+ # KOSIS catalog의 임베딩 인덱싱 포맷과 정렬되어 유사도 검색 정확도 향상.
46
+ emb_cat = s.parent_path or ""
47
+ emb_indicator = s.indicator or ""
48
+ emb_population = s.population or ""
49
+ emb_unit = s.unit or ""
50
+ if emb_indicator:
51
+ embedding_text = (
52
+ f"{emb_cat} > {emb_indicator} | "
53
+ f"항목: {emb_indicator} | "
54
+ f"분류: {emb_population} | "
55
+ f"단위: {emb_unit}"
56
+ ).strip()
57
+ else:
58
+ embedding_text = keyword
59
+
60
+ return ConnectorQuery(
61
+ keyword=keyword,
62
+ indicator=s.indicator,
63
+ time_period=s.time_period,
64
+ population=s.population,
65
+ extra_params={
66
+ "raw_claim": context,
67
+ "embedding_text": embedding_text,
68
+ "source_org": s.source_reference,
69
+ # [v6.11] parent_path 전달 → catalog_search가 LLM 호출 없이 활용
70
+ "parent_path": s.parent_path,
71
+ },
72
+ )
@@ -0,0 +1,133 @@
1
+ """
2
+ structverify.retrieval.registry — DataSource 동적 등록.
3
+
4
+ 회사가 *자체 데이터 소스*를 등록 가능:
5
+
6
+ # company_package/sales_source.py
7
+ from structverify.retrieval.base import BaseDataSource
8
+ from structverify.retrieval.registry import register_datasource
9
+
10
+ @register_datasource("my_sales_db")
11
+ class MySalesDB(BaseDataSource):
12
+ async def search_catalog(self, query, ...):
13
+ ...
14
+ async def fetch_evidence(self, candidate_id, ...):
15
+ ...
16
+
17
+ # config/default.yaml
18
+ data_sources:
19
+ enabled: ["my_sales_db"]
20
+ my_sales_db:
21
+ dsn: "postgresql://..."
22
+
23
+ Agent loop (Phase D)에서 이 registry를 통해 source 인스턴스 생성:
24
+
25
+ from structverify.retrieval.registry import build_datasource
26
+ source = build_datasource("my_sales_db", config={"dsn": "..."})
27
+ """
28
+ from __future__ import annotations
29
+
30
+ from structverify.utils.logger import get_logger
31
+ from typing import Callable, Type
32
+
33
+ from .base import BaseDataSource
34
+
35
+ logger = get_logger(__name__)
36
+
37
+
38
+ # ── Registry 본체 ────────────────────────────────────────────────
39
+
40
+ _REGISTRY: dict[str, Type[BaseDataSource]] = {}
41
+
42
+
43
+ def register_datasource(
44
+ name: str,
45
+ ) -> Callable[[Type[BaseDataSource]], Type[BaseDataSource]]:
46
+ """
47
+ DataSource 클래스를 등록하는 데코레이터.
48
+
49
+ Args:
50
+ name: config.data_sources.enabled 에 들어갈 이름.
51
+
52
+ Example:
53
+ @register_datasource("kosis")
54
+ class KOSISSource(BaseDataSource):
55
+ ...
56
+ """
57
+ def decorator(cls: Type[BaseDataSource]) -> Type[BaseDataSource]:
58
+ if not issubclass(cls, BaseDataSource):
59
+ raise TypeError(
60
+ f"{cls.__name__} must subclass BaseDataSource to be registered"
61
+ )
62
+ if name in _REGISTRY:
63
+ logger.warning(
64
+ f"[registry] DataSource '{name}' already registered, overwriting "
65
+ f"({_REGISTRY[name].__name__} → {cls.__name__})"
66
+ )
67
+ _REGISTRY[name] = cls
68
+ cls.name = name
69
+ logger.info(f"[registry] DataSource 등록: {name} ({cls.__name__})")
70
+ return cls
71
+ return decorator
72
+
73
+
74
+ def get_datasource_class(name: str) -> Type[BaseDataSource]:
75
+ """등록된 클래스 반환. 없으면 KeyError."""
76
+ if name not in _REGISTRY:
77
+ available = list(_REGISTRY.keys())
78
+ raise KeyError(
79
+ f"DataSource '{name}' not registered. "
80
+ f"Available: {available}. "
81
+ f"Register via @register_datasource('{name}')."
82
+ )
83
+ return _REGISTRY[name]
84
+
85
+
86
+ def list_datasources() -> list[str]:
87
+ """등록된 모든 DataSource 이름."""
88
+ return sorted(_REGISTRY.keys())
89
+
90
+
91
+ def build_datasource(name: str, config: dict | None = None) -> BaseDataSource:
92
+ """Config에서 datasource 인스턴스 생성.
93
+
94
+ Args:
95
+ name: 등록된 이름.
96
+ config: 클래스 __init__에 전달될 kwargs.
97
+ None이면 인자 없이 생성.
98
+
99
+ Example:
100
+ source = build_datasource("kosis", config={"api_key": "..."})
101
+ """
102
+ cls = get_datasource_class(name)
103
+ if config:
104
+ return cls(**config)
105
+ return cls()
106
+
107
+
108
+ # ── 편의 함수 ─────────────────────────────────────────────────────
109
+
110
+ def build_all_enabled(config: dict) -> list[BaseDataSource]:
111
+ """
112
+ config['data_sources'] 섹션에서 *enabled* 된 모든 source 빌드.
113
+
114
+ Args:
115
+ config: config.data_sources 섹션 전체 dict.
116
+ 예: {"enabled": ["kosis"], "kosis": {"api_key": "..."}, ...}
117
+
118
+ Returns:
119
+ 활성화된 DataSource 인스턴스 리스트.
120
+ """
121
+ enabled = config.get("enabled", [])
122
+ sources: list[BaseDataSource] = []
123
+ for name in enabled:
124
+ try:
125
+ source_config = config.get(name, {})
126
+ sources.append(build_datasource(name, source_config))
127
+ except KeyError as e:
128
+ logger.warning(
129
+ f"[registry] enabled={name!r} but not registered — skipped. {e}"
130
+ )
131
+ except Exception as e:
132
+ logger.error(f"[registry] build_datasource({name!r}) failed: {e}")
133
+ return sources
@@ -0,0 +1,141 @@
1
+ """
2
+ structverify.retrieval.relevance_judge — P32: LLM 기반 표 관련성 판단.
3
+
4
+ 배경:
5
+ kosis_connector._is_table_relevant()는 *토큰 매칭 룰베이스*라 sweet spot 밖
6
+ 케이스에 false negative 발생. 예:
7
+ - indicator: "체외 충격파 쇄석술 장비 수"
8
+ - table_name: "시군구(서울인천경기강원)별 주요 의료장비 현황"
9
+ - 룰: "장비"(2글자, len<3 룰에 막힘), "쇄석술"(table에 없음) → 거부 ❌
10
+ - 의미적으로는 *정답 표* (장비 row가 의료장비 분류에 포함될 가능성 높음)
11
+
12
+ 설계:
13
+ 룰베이스를 *fast-path*로 유지하되, 거부 결정 시 *LLM에게 한 번 더 위임*.
14
+ LLM은 claim 원문 + schema + table_name을 보고 *의미적 매칭* 판단.
15
+ 룰 *통과* 시엔 LLM 호출 없음 (속도).
16
+
17
+ - 룰 통과 → 그대로 진행 (LLM 호출 X)
18
+ - 룰 거부 → LLM 가드 호출 → True/False → 최종 판단
19
+
20
+ config:
21
+ kosis.relevance_guard.{enabled, llm_fallback, model_tier}
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import json
26
+ import re
27
+ from typing import Any
28
+
29
+ from structverify.utils.logger import get_logger
30
+
31
+ logger = get_logger(__name__)
32
+
33
+
34
+ def _build_prompt(
35
+ claim_text: str,
36
+ indicator: str,
37
+ population: str,
38
+ parent_path: str,
39
+ table_name: str,
40
+ ) -> str:
41
+ return f"""당신은 KOSIS 통계표 관련성 reviewer입니다. 사용자 claim의 검증에
42
+ *이 표가 의미적으로 적합한가*를 한 줄로 판단하세요.
43
+
44
+ [사용자 claim]
45
+ - 원문: {claim_text or '(없음)'}
46
+ - 검증 지표 (indicator): {indicator or '(없음)'}
47
+ - 대상 집단/지역 (population): {population or '(없음)'}
48
+ - 카테고리 경로 (parent_path): {parent_path or '(없음)'}
49
+
50
+ [후보 KOSIS 통계표]
51
+ - 표 이름: {table_name}
52
+
53
+ [판단 기준]
54
+ - 표 이름이 indicator를 *포함하거나*, indicator의 *상위 분류* (예: "체외 충격파 쇄석술 장비"의 상위 = "의료장비")로 *직접 매칭*되면 → relevant.
55
+ - 표 이름이 population (예: "강원도", "시도별")을 포함/매칭하면 → relevant 가능성 ↑.
56
+ - 표 이름이 더 *광범위*해도 (예: "주요 의료장비"가 "체외 충격파 쇄석술 장비"의 상위) row 검색으로 정답 찾을 가능성 있으면 → relevant.
57
+ - 외국 국가/해외 통계 / 장래 추계·전망 / 다른 도메인 (인구↔경제↔기상)이면 → NOT relevant.
58
+ - parent_path와 표 이름이 같은 분류 트리에 있으면 → relevant.
59
+
60
+ [응답 형식 — JSON only]
61
+ {{
62
+ "relevant": true | false,
63
+ "reason": "한 줄 이유"
64
+ }}
65
+ """
66
+
67
+
68
+ def _parse(raw: str) -> tuple[bool | None, str]:
69
+ """LLM 응답 파싱. (relevant, reason). None이면 파싱 실패."""
70
+ try:
71
+ m = re.search(r"\{[^{}]*\}", raw, re.DOTALL)
72
+ if not m:
73
+ return None, ""
74
+ data = json.loads(m.group(0))
75
+ rel = data.get("relevant")
76
+ reason = str(data.get("reason") or "")
77
+ if isinstance(rel, bool):
78
+ return rel, reason
79
+ if isinstance(rel, str):
80
+ return rel.strip().lower() in ("true", "yes", "y", "1"), reason
81
+ return None, reason
82
+ except Exception as e:
83
+ logger.debug(f"[relevance_judge] 파싱 실패: {e}")
84
+ return None, ""
85
+
86
+
87
+ async def is_table_relevant_semantic(
88
+ *,
89
+ claim_text: str,
90
+ indicator: str,
91
+ population: str,
92
+ parent_path: str,
93
+ table_name: str,
94
+ config: dict | None,
95
+ ) -> tuple[bool | None, str]:
96
+ """LLM 기반 표 관련성 판단.
97
+
98
+ Args:
99
+ claim_text: claim 원문 (1~3문장).
100
+ indicator: schema.indicator.
101
+ population: schema.population (지역/집단).
102
+ parent_path: schema.parent_path (계층 카테고리).
103
+ table_name: KOSIS 표 이름.
104
+ config: 전체 config dict.
105
+
106
+ Returns:
107
+ (relevant, reason).
108
+ relevant=True: 의미적 매칭 OK, fetch 진행 권장.
109
+ relevant=False: 매칭 안 됨, 거부 권장.
110
+ relevant=None: LLM 호출 실패/파싱 실패 → 호출자가 보수적으로 처리.
111
+ """
112
+ if not table_name or (not indicator and not claim_text):
113
+ return None, "input 부족"
114
+
115
+ _cfg = (config or {}).get("kosis") or {}
116
+ _rg = _cfg.get("relevance_guard") or {}
117
+ model_tier = str(_rg.get("model_tier") or "light").strip().lower()
118
+
119
+ prompt = _build_prompt(claim_text, indicator, population, parent_path, table_name)
120
+ from structverify.utils.llm_client import LLMClient
121
+ llm = LLMClient(config=(config or {}).get("llm") or {})
122
+ try:
123
+ raw = await llm.generate(
124
+ prompt=prompt,
125
+ system_prompt="KOSIS 표 관련성 reviewer. JSON만 응답.",
126
+ model_tier=model_tier,
127
+ )
128
+ except Exception as e:
129
+ logger.warning(f"[relevance_judge] LLM 호출 실패: {e}")
130
+ return None, f"llm_error: {e}"
131
+
132
+ rel, reason = _parse(raw)
133
+ if rel is None:
134
+ logger.info(f"[relevance_judge] 파싱 실패 raw={raw[:200]!r}")
135
+ return None, "parse_failed"
136
+
137
+ logger.info(
138
+ f"[relevance_judge] table={table_name!r} vs indicator={indicator!r}: "
139
+ f"relevant={rel} reason={reason[:120]!r}"
140
+ )
141
+ return rel, reason
@@ -0,0 +1,267 @@
1
+ """
2
+ structverify.retrieval.row_matcher — P33c: LLM 기반 row indicator 매칭.
3
+
4
+ 배경:
5
+ _select_best_row()의 _ind_match()는 정규화 후 정확 일치/substring 양방향 비교
6
+ 하는 룰베이스. KOSIS 표가 *계층적 분류*를 가져서 row의 ITM_NM이 *상위 카테고리*
7
+ (예: "진단방사선·특수의료장비")인데 사용자 indicator가 *세부 항목*(예:
8
+ "체외 충격파 쇄석술 장비")이면 직접 매칭이 안 됨.
9
+
10
+ LLM이 rows의 unique ITM_NM / C1_NM~C4_NM 값들을 보고 indicator와
11
+ *의미적으로 매칭*되는 컬럼 값을 식별해 그 값으로 row 필터링.
12
+
13
+ 용도:
14
+ _select_best_row()의 indicator pre-filter가 0건일 때만 호출 (룰 통과 시엔
15
+ 비용 절감 위해 LLM 호출 X). config.kosis.relevance_guard.row_match_llm_fallback
16
+ 이 true일 때만 활성화.
17
+
18
+ 반환:
19
+ 매칭 row list (1개 이상) 또는 빈 list. 빈 list면 진짜 그 표에 indicator가
20
+ 없는 것이므로 호출자가 None 반환.
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import json
25
+ import re
26
+ from typing import Any
27
+
28
+ from structverify.utils.logger import get_logger
29
+
30
+ logger = get_logger(__name__)
31
+
32
+
33
+ _INDICATOR_FIELDS = ("ITM_NM", "C1_NM", "C2_NM", "C3_NM", "C4_NM")
34
+
35
+
36
+ def _extract_unique_values(rows: list[dict], field: str, max_n: int = 80) -> list[str]:
37
+ """rows의 특정 컬럼에서 unique 값 list 추출 (LLM에 노출용)."""
38
+ seen: set[str] = set()
39
+ out: list[str] = []
40
+ for r in rows:
41
+ v = r.get(field)
42
+ if not v:
43
+ continue
44
+ s = str(v).strip()
45
+ if s and s not in seen:
46
+ seen.add(s)
47
+ out.append(s)
48
+ if len(out) >= max_n:
49
+ break
50
+ return out
51
+
52
+
53
+ def _build_prompt(
54
+ indicator: str,
55
+ claim_text: str,
56
+ parent_path: str,
57
+ population: str,
58
+ field_values: dict[str, list[str]],
59
+ ) -> str:
60
+ """LLM prompt — unique 컬럼 값들 보고 indicator와 매칭되는 값 추천."""
61
+ lines = []
62
+ for f, vals in field_values.items():
63
+ if not vals:
64
+ continue
65
+ # 너무 길면 자름
66
+ shown = vals[:50]
67
+ more = f" (외 {len(vals)-len(shown)}개)" if len(vals) > len(shown) else ""
68
+ lines.append(f"- {f}: {', '.join(repr(v) for v in shown)}{more}")
69
+ field_block = "\n".join(lines) if lines else "(분류 컬럼 값 없음)"
70
+
71
+ return f"""당신은 KOSIS 통계표 row 매칭 reviewer입니다. 사용자 indicator가
72
+ 표 안 *어떤 row 분류 값*에 해당하는지 의미적으로 판단하세요.
73
+
74
+ [사용자 검색 의도]
75
+ - indicator: {indicator!r}
76
+ - claim 원문: {claim_text or '(없음)'}
77
+ - parent_path: {parent_path or '(없음)'}
78
+ - population: {population or '(없음)'}
79
+
80
+ [표 안 row의 분류 컬럼 unique 값 list]
81
+ {field_block}
82
+
83
+ [판단 기준]
84
+ 1. indicator의 *핵심 키워드*가 어느 컬럼의 어떤 값에 *직접/유사 매칭*되는가? (예: indicator="체외 충격파 쇄석술 장비" → ITM_NM 또는 C2_NM에 "체외 충격파 쇄석술기" 또는 "ESWL" 같은 값이 있으면 그것).
85
+ 2. parent_path(계층)에 비추어 *상위 카테고리* row (예: "진단방사선·특수의료장비" 합계)는 매칭이 *아니다*. *세부 항목 row*가 필요.
86
+ 3. claim 원문에 "보유대수", "장비 수" 같은 *수량 표현*은 indicator의 일반어이므로 row 분류 값과 직접 비교 불필요.
87
+ 4. 어느 컬럼에도 indicator와 의미적 매칭이 없으면 → 그 표에 indicator 없는 것 → "matches": [] 반환.
88
+
89
+ ★ field 라벨 정확성:
90
+ - "field"는 *반드시* 위 [표 안 row의 분류 컬럼 unique 값 list]에 *명시된 라벨*과 동일하게 적기.
91
+ - 예: 값이 "ITM_NM" list가 아니라 "C2_NM" list 안에 있으면 → field="C2_NM" (절대 ITM_NM이라 적지 말 것).
92
+ - field 잘못 적으면 row 필터링 0건이 되어 검증 실패. 정확한 컬럼명 사용 필수.
93
+
94
+ [응답 형식 — JSON only, 다른 텍스트 금지]
95
+ {{
96
+ "matches": [{{"field": "<컬럼명>", "value": "<매칭 값>"}}, ...],
97
+ "reasoning": "한 줄 이유"
98
+ }}
99
+
100
+ * matches는 0개 이상. 보통 1~3개. *상위 카테고리* 값은 절대 포함하지 마세요.
101
+ """
102
+
103
+
104
+ def _parse(raw: str) -> tuple[list[dict], str]:
105
+ """LLM 응답 파싱. (matches list, reasoning)."""
106
+ try:
107
+ m = re.search(r"\{.*\}", raw, re.DOTALL)
108
+ if not m:
109
+ return [], ""
110
+ data = json.loads(m.group(0))
111
+ matches = data.get("matches") or []
112
+ reasoning = str(data.get("reasoning") or "")
113
+ if not isinstance(matches, list):
114
+ return [], reasoning
115
+ out: list[dict] = []
116
+ for m in matches:
117
+ if isinstance(m, dict) and m.get("field") and m.get("value"):
118
+ out.append({"field": str(m["field"]), "value": str(m["value"])})
119
+ return out, reasoning
120
+ except Exception as e:
121
+ logger.debug(f"[row_matcher] 파싱 실패: {e}")
122
+ return [], ""
123
+
124
+
125
+ async def llm_select_rows(
126
+ *,
127
+ rows: list[dict],
128
+ indicator: str,
129
+ claim_text: str,
130
+ parent_path: str,
131
+ population: str,
132
+ config: dict | None,
133
+ ) -> list[dict]:
134
+ """LLM이 indicator와 의미적으로 매칭되는 row들을 식별해 반환.
135
+
136
+ Args:
137
+ rows: pop pre-filter 통과한 후보 row list.
138
+ indicator: schema.indicator.
139
+ claim_text: claim 원문 (앞 400자).
140
+ parent_path: schema.parent_path.
141
+ population: schema.population.
142
+ config: 전체 config dict.
143
+
144
+ Returns:
145
+ 매칭 row list. 빈 list면 LLM이 "이 표에 indicator 없다"고 판단.
146
+ """
147
+ if not rows or not indicator:
148
+ return []
149
+
150
+ # 1) unique 분류 값 추출 (ITM_NM/C1~C4_NM)
151
+ field_values: dict[str, list[str]] = {}
152
+ for f in _INDICATOR_FIELDS:
153
+ vals = _extract_unique_values(rows, f, max_n=80)
154
+ if vals:
155
+ field_values[f] = vals
156
+
157
+ if not field_values:
158
+ logger.info("[row_matcher] 분류 컬럼 값 없음 → LLM 호출 skip")
159
+ return []
160
+
161
+ # 2) LLM 호출
162
+ _rg_cfg = ((config or {}).get("kosis") or {}).get("relevance_guard") or {}
163
+ model_tier = str(_rg_cfg.get("row_match_model_tier") or _rg_cfg.get("model_tier") or "light").strip().lower()
164
+
165
+ prompt = _build_prompt(indicator, claim_text, parent_path, population, field_values)
166
+ from structverify.utils.llm_client import LLMClient
167
+ llm = LLMClient(config=(config or {}).get("llm") or {})
168
+ try:
169
+ raw = await llm.generate(
170
+ prompt=prompt,
171
+ system_prompt="KOSIS row 매칭 reviewer. JSON만 응답.",
172
+ model_tier=model_tier,
173
+ )
174
+ except Exception as e:
175
+ logger.warning(f"[row_matcher] LLM 호출 실패: {e}")
176
+ return []
177
+
178
+ matches, reasoning = _parse(raw)
179
+ if not matches:
180
+ logger.info(
181
+ f"[row_matcher] LLM: 매칭 없음 — indicator={indicator!r} "
182
+ f"reason={reasoning[:120]!r}"
183
+ )
184
+ return []
185
+
186
+ # field/value swap 휴리스틱: LLM이 field 자리에 실제 매칭 값을 넣고
187
+ # value 자리에 DT 같은 측정치(숫자/짧은 문자열)를 넣어 응답하는 케이스 회복.
188
+ # (예: field='체외충격파쇄석기', value='18' → field='ITM_NM', value='체외충격파쇄석기')
189
+ swapped: list[dict] = []
190
+ for m in matches:
191
+ f, v = m["field"], m["value"]
192
+ if f not in _INDICATOR_FIELDS and (v.isdigit() or len(v) <= 3):
193
+ logger.warning(
194
+ f"[row_matcher] LLM field/value swap 감지 — "
195
+ f"field={f!r} value={v!r} → value={f!r}로 재해석"
196
+ )
197
+ swapped.append({"field": _INDICATOR_FIELDS[0], "value": f})
198
+ else:
199
+ swapped.append(m)
200
+ matches = swapped
201
+
202
+ logger.info(
203
+ f"[row_matcher] LLM 매칭 {len(matches)}건: {matches} "
204
+ f"reason={reasoning[:80]!r}"
205
+ )
206
+
207
+ # 3) matches dict({field, value})로 rows 필터링
208
+ def _norm(s: str) -> str:
209
+ if not s:
210
+ return ""
211
+ s = str(s)
212
+ for ch in (" ", "·", "ㆍ", "・", "/", "-", "_", "(", ")", "[", "]"):
213
+ s = s.replace(ch, "")
214
+ return s.strip().lower()
215
+
216
+ norm_matches = [
217
+ {"field": m["field"], "value_norm": _norm(m["value"]), "value_raw": m["value"]}
218
+ for m in matches
219
+ ]
220
+ filtered: list[dict] = []
221
+ field_mismatches: list[tuple[str, str]] = [] # (llm_field, actual_field) 로깅용
222
+
223
+ for r in rows:
224
+ matched = False
225
+ for nm in norm_matches:
226
+ target_v_norm = nm["value_norm"]
227
+ if not target_v_norm:
228
+ continue
229
+ # 1차: LLM이 지정한 field 우선 검사
230
+ v = r.get(nm["field"])
231
+ if v is not None:
232
+ v_norm = _norm(str(v))
233
+ if target_v_norm in v_norm or v_norm in target_v_norm:
234
+ matched = True
235
+ break
236
+ # 2차: LLM이 field를 잘못 지정한 경우 대비 — 모든 INDICATOR_FIELDS에서 value 검색.
237
+ # LLM은 "체외충격파쇄석기"가 ITM_NM이라 우길 수 있는데 실제로는 C2_NM 같은 다른
238
+ # 컬럼에 있음. value가 unique 분류라 다른 컬럼에 우연히 같은 단어 들어있을 가능성 낮음.
239
+ for fname in _INDICATOR_FIELDS:
240
+ if fname == nm["field"]:
241
+ continue # 1차에서 이미 봄
242
+ v = r.get(fname)
243
+ if v is None:
244
+ continue
245
+ v_norm = _norm(str(v))
246
+ if target_v_norm in v_norm or v_norm in target_v_norm:
247
+ matched = True
248
+ field_mismatches.append((nm["field"], fname))
249
+ break
250
+ if matched:
251
+ break
252
+ if matched:
253
+ filtered.append(r)
254
+
255
+ if field_mismatches:
256
+ # LLM이 잘못 지정한 field → 실제 매칭된 field 로깅 (1회만)
257
+ _ex = field_mismatches[0]
258
+ logger.info(
259
+ f"[row_matcher] LLM이 field={_ex[0]!r}로 지정했으나 실제로는 "
260
+ f"{_ex[1]!r}에 매칭됨 — all-field fallback으로 회복 "
261
+ f"({len(field_mismatches)} rows)"
262
+ )
263
+
264
+ logger.info(
265
+ f"[row_matcher] LLM 매칭 적용: {len(rows)} rows → {len(filtered)} rows"
266
+ )
267
+ return filtered
File without changes