chaskiwasi 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- chaskiwasi/__init__.py +0 -0
- chaskiwasi/chunking/__init__.py +0 -0
- chaskiwasi/chunking/chunk_tokenizer.py +43 -0
- chaskiwasi/classification/__init__.py +0 -0
- chaskiwasi/classification/cascade_factory.py +68 -0
- chaskiwasi/classification/strategies/__init__.py +0 -0
- chaskiwasi/classification/strategies/base_strategy.py +43 -0
- chaskiwasi/classification/strategies/context_overlap_strategy.py +21 -0
- chaskiwasi/classification/strategies/cpu_regex_strategy.py +62 -0
- chaskiwasi/classification/strategies/gemini_streamer.py +99 -0
- chaskiwasi/classification/strategies/llm_router_strategy.py +71 -0
- chaskiwasi/config/__init__.py +0 -0
- chaskiwasi/config/settings.py +19 -0
- chaskiwasi/config/taxonomy_registry.py +55 -0
- chaskiwasi/consolidation/consolidator.py +158 -0
- chaskiwasi/ingestion/__init__.py +0 -0
- chaskiwasi/ingestion/docling_parser.py +64 -0
- chaskiwasi/query_engine/__init__.py +0 -0
- chaskiwasi/query_engine/cross_filter.py +74 -0
- chaskiwasi/query_engine/ollama_streamer.py +64 -0
- chaskiwasi/storage/__init__.py +0 -0
- chaskiwasi/storage/chroma_persistent.py +67 -0
- chaskiwasi-0.2.0.dist-info/METADATA +31 -0
- chaskiwasi-0.2.0.dist-info/RECORD +25 -0
- chaskiwasi-0.2.0.dist-info/WHEEL +4 -0
chaskiwasi/__init__.py
ADDED
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# dantesito/chasky/chunking/chunk_tokenizer.py
|
|
2
|
+
|
|
3
|
+
from typing import List
|
|
4
|
+
|
|
5
|
+
import tiktoken
|
|
6
|
+
|
|
7
|
+
from chaskiwasi.config.settings import Settings
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ChunkTokenizer:
|
|
11
|
+
"""Divide textos en fragmentos mediante ventanas de tokens de tiktoken."""
|
|
12
|
+
|
|
13
|
+
def __init__(self) -> None:
|
|
14
|
+
self._encoding = tiktoken.get_encoding(Settings.TIKTOKEN_ENCODING)
|
|
15
|
+
|
|
16
|
+
def split_text(self, text: str) -> List[str]:
|
|
17
|
+
"""Divide el texto en chunks determinados exclusivamente por tokens."""
|
|
18
|
+
if not isinstance(text, str) or not text:
|
|
19
|
+
return []
|
|
20
|
+
|
|
21
|
+
tokens: List[int] = self._encoding.encode(text)
|
|
22
|
+
|
|
23
|
+
if not tokens:
|
|
24
|
+
return []
|
|
25
|
+
|
|
26
|
+
chunk_size: int = Settings.CHUNK_SIZE
|
|
27
|
+
step: int = Settings.CHUNK_SIZE - Settings.CHUNK_OVERLAP
|
|
28
|
+
|
|
29
|
+
chunks: List[str] = []
|
|
30
|
+
start: int = 0
|
|
31
|
+
total_tokens: int = len(tokens)
|
|
32
|
+
|
|
33
|
+
while start < total_tokens:
|
|
34
|
+
end: int = min(start + chunk_size, total_tokens)
|
|
35
|
+
chunk_tokens: List[int] = tokens[start:end]
|
|
36
|
+
chunks.append(self._encoding.decode(chunk_tokens))
|
|
37
|
+
|
|
38
|
+
if end >= total_tokens:
|
|
39
|
+
break
|
|
40
|
+
|
|
41
|
+
start += step
|
|
42
|
+
|
|
43
|
+
return chunks
|
|
File without changes
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from typing import Any, List, Optional, Tuple
|
|
3
|
+
|
|
4
|
+
from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
class CascadeFactory:
|
|
9
|
+
"""
|
|
10
|
+
Fábrica y orquestador de clasificación en cascada agnóstico para chaskywasi.
|
|
11
|
+
Aplica una arquitectura de enrutamiento por estrategias ordenadas jerárquicamente.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
strategies: Optional[List[BaseStrategy]] = None,
|
|
17
|
+
default_source: Optional[Any] = None,
|
|
18
|
+
default_section: Optional[Any] = None,
|
|
19
|
+
) -> None:
|
|
20
|
+
self._strategies: List[BaseStrategy] = strategies or []
|
|
21
|
+
self.default_source = default_source
|
|
22
|
+
self.default_section = default_section
|
|
23
|
+
|
|
24
|
+
def process_chunk(
|
|
25
|
+
self,
|
|
26
|
+
text: Optional[str],
|
|
27
|
+
current_source: Optional[Any] = None,
|
|
28
|
+
) -> Tuple[Optional[Any], Optional[Any]]:
|
|
29
|
+
"""Procesa un fragmento individual de texto iterando sobre la lista de estrategias."""
|
|
30
|
+
effective_default_source = current_source or self.default_source
|
|
31
|
+
|
|
32
|
+
if not text or not text.strip():
|
|
33
|
+
return effective_default_source, self.default_section
|
|
34
|
+
|
|
35
|
+
for strategy in self._strategies:
|
|
36
|
+
try:
|
|
37
|
+
src_res, sec_res = strategy.classify(
|
|
38
|
+
text, current_source=effective_default_source
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
if sec_res is not None:
|
|
42
|
+
final_source = src_res or effective_default_source
|
|
43
|
+
return final_source, sec_res
|
|
44
|
+
except Exception as e:
|
|
45
|
+
logger.warning(
|
|
46
|
+
"Estrategia %s falló durante la clasificación: %s",
|
|
47
|
+
strategy.__class__.__name__,
|
|
48
|
+
str(e),
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
return effective_default_source, self.default_section
|
|
52
|
+
|
|
53
|
+
def process_chunks_batch(
|
|
54
|
+
self,
|
|
55
|
+
chunks: List[str],
|
|
56
|
+
current_source: Optional[Any] = None,
|
|
57
|
+
) -> List[Tuple[Optional[Any], Optional[Any]]]:
|
|
58
|
+
"""Procesa una lista de fragmentos secuencialmente manteniendo el contexto actual."""
|
|
59
|
+
results = []
|
|
60
|
+
active_source = current_source or self.default_source
|
|
61
|
+
|
|
62
|
+
for chunk in chunks:
|
|
63
|
+
src, sec = self.process_chunk(chunk, current_source=active_source)
|
|
64
|
+
if src is not None:
|
|
65
|
+
active_source = src
|
|
66
|
+
results.append((src, sec))
|
|
67
|
+
|
|
68
|
+
return results
|
|
File without changes
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# dantesito/chasky/classification/strategies/base_strategy.py
|
|
2
|
+
from abc import ABC, abstractmethod
|
|
3
|
+
from typing import Tuple, Optional, Any
|
|
4
|
+
|
|
5
|
+
# 🚀 CONTRATO ACTIVO: Importación obligatoria para control de tipos dinámicos
|
|
6
|
+
from chaskiwasi.config.taxonomy_registry import TaxonomyRegistry
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class BaseStrategy(ABC):
|
|
10
|
+
"""
|
|
11
|
+
Interfaz contractual agnóstica para todas las estrategias de inferencia semántica.
|
|
12
|
+
|
|
13
|
+
Expone de forma segura las clases de Enums inyectadas por el cliente mediante
|
|
14
|
+
propiedades dinámicas de solo lectura, eliminando la fragilidad de constructores.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
@property
|
|
18
|
+
def source_enum_class(self) -> Any:
|
|
19
|
+
"""
|
|
20
|
+
Devuelve de forma dinámica la clase SourceEnum del cliente.
|
|
21
|
+
🛡️ CORTAFUEGOS EN CALIENTE: Lanza RuntimeError si no fue inyectada al arrancar.
|
|
22
|
+
"""
|
|
23
|
+
return TaxonomyRegistry.get_source_enum()
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def section_enum_class(self) -> Any:
|
|
27
|
+
"""
|
|
28
|
+
Devuelve de forma dinámica la clase SectionEnum del cliente.
|
|
29
|
+
🛡️ CORTAFUEGOS EN CALIENTE: Lanza RuntimeError si no fue inyectada al arrancar.
|
|
30
|
+
"""
|
|
31
|
+
return TaxonomyRegistry.get_section_enum()
|
|
32
|
+
|
|
33
|
+
@abstractmethod
|
|
34
|
+
def classify(self, chunk: str, current_source: Optional[str] = None) -> Tuple[Optional[Any], Optional[Any]]:
|
|
35
|
+
"""
|
|
36
|
+
Analiza un fragmento y devuelve una tupla conteniendo las instancias
|
|
37
|
+
de los Enums dinámicos del cliente (SourceEnum, SectionEnum).
|
|
38
|
+
|
|
39
|
+
:param chunk: Fragmento de texto Markdown a evaluar.
|
|
40
|
+
:param current_source: Contexto de origen acumulado o arrastrado.
|
|
41
|
+
:return: Tupla conteniendo (Instancia de SourceEnum, Instancia de SectionEnum)
|
|
42
|
+
"""
|
|
43
|
+
pass
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# dantesito/chasky/classification/strategies/context_overlap_strategy.py
|
|
2
|
+
|
|
3
|
+
from typing import Optional, Tuple
|
|
4
|
+
|
|
5
|
+
from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
|
|
6
|
+
from chaskiwasi.config.taxonomy_registry import SectionEnum, SourceEnum
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ContextOverlapStrategy(BaseStrategy):
|
|
10
|
+
"""Arrastra la fuente institucional desde el contexto del chunk anterior."""
|
|
11
|
+
|
|
12
|
+
def classify(
|
|
13
|
+
self,
|
|
14
|
+
chunk_text: str,
|
|
15
|
+
current_source: Optional[SourceEnum] = None,
|
|
16
|
+
) -> Tuple[Optional[SourceEnum], Optional[SectionEnum]]:
|
|
17
|
+
"""Mantiene la fuente previa sin inspeccionar el contenido del chunk."""
|
|
18
|
+
if current_source is not None:
|
|
19
|
+
return current_source, None
|
|
20
|
+
|
|
21
|
+
return None, None
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from dataclasses import dataclass, field
|
|
3
|
+
from typing import Any, List, Optional, Pattern, Tuple
|
|
4
|
+
from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass
|
|
8
|
+
class RegexRule:
|
|
9
|
+
"""Estructura de datos para definir reglas de clasificación por Regex."""
|
|
10
|
+
source: Any
|
|
11
|
+
section: Any
|
|
12
|
+
patterns: List[str]
|
|
13
|
+
flags: int = re.IGNORECASE | re.DOTALL
|
|
14
|
+
compiled_patterns: List[Pattern[str]] = field(init=False, repr=False)
|
|
15
|
+
|
|
16
|
+
def __post_init__(self) -> None:
|
|
17
|
+
compiled: List[Pattern[str]] = []
|
|
18
|
+
for pattern in self.patterns:
|
|
19
|
+
try:
|
|
20
|
+
compiled.append(re.compile(pattern, self.flags))
|
|
21
|
+
except re.error as err:
|
|
22
|
+
raise ValueError(
|
|
23
|
+
f"Patrón Regex inválido '{pattern}' para la regla ({self.source}, {self.section}): {err}"
|
|
24
|
+
) from err
|
|
25
|
+
self.compiled_patterns = compiled
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class CPURegexStrategy(BaseStrategy):
|
|
29
|
+
"""Estrategia genérica que ejecuta reglas Regex inyectadas."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, rules: Optional[List[RegexRule]] = None) -> None:
|
|
32
|
+
self.rules: List[RegexRule] = list(rules) if rules else []
|
|
33
|
+
|
|
34
|
+
def add_rule(self, rule: RegexRule) -> None:
|
|
35
|
+
"""Permite registrar reglas dinámicamente."""
|
|
36
|
+
self.rules.append(rule)
|
|
37
|
+
|
|
38
|
+
def classify(
|
|
39
|
+
self,
|
|
40
|
+
chunk_text: str,
|
|
41
|
+
current_source: Optional[Any] = None,
|
|
42
|
+
) -> Tuple[Optional[Any], Optional[Any]]:
|
|
43
|
+
if not isinstance(chunk_text, str) or not chunk_text:
|
|
44
|
+
return None, None
|
|
45
|
+
|
|
46
|
+
# Evaluación con preferencia contextual por current_source si aplica
|
|
47
|
+
for rule in self.rules:
|
|
48
|
+
if current_source and rule.source != current_source:
|
|
49
|
+
continue
|
|
50
|
+
|
|
51
|
+
if any(pattern.search(chunk_text) for pattern in rule.compiled_patterns):
|
|
52
|
+
return rule.source, rule.section
|
|
53
|
+
|
|
54
|
+
# Segunda pasada para reglas generales si no coincidió en el contexto actual
|
|
55
|
+
if current_source:
|
|
56
|
+
for rule in self.rules:
|
|
57
|
+
if rule.source == current_source:
|
|
58
|
+
continue
|
|
59
|
+
if any(pattern.search(chunk_text) for pattern in rule.compiled_patterns):
|
|
60
|
+
return rule.source, rule.section
|
|
61
|
+
|
|
62
|
+
return None, None
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
from typing import Any, Dict, Optional, Tuple
|
|
4
|
+
|
|
5
|
+
from google import genai
|
|
6
|
+
from google.genai import types
|
|
7
|
+
|
|
8
|
+
from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
|
|
9
|
+
|
|
10
|
+
logging.getLogger("google.genai").setLevel(logging.ERROR)
|
|
11
|
+
logging.getLogger("google.genai._api_client").setLevel(logging.ERROR)
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class GeminiStreamer(BaseStrategy):
|
|
17
|
+
"""Estrategia de inferencia remota en la nube totalmente agnóstica al dominio."""
|
|
18
|
+
|
|
19
|
+
def __init__(
|
|
20
|
+
self,
|
|
21
|
+
system_instruction: str,
|
|
22
|
+
label_mapping: Dict[str, Any],
|
|
23
|
+
) -> None:
|
|
24
|
+
self.system_instruction = system_instruction
|
|
25
|
+
self.label_mapping = label_mapping
|
|
26
|
+
|
|
27
|
+
api_key = ""
|
|
28
|
+
env_key = os.environ.get("GEMINI_API_KEY", "").strip()
|
|
29
|
+
if env_key:
|
|
30
|
+
api_key = env_key
|
|
31
|
+
|
|
32
|
+
if not api_key:
|
|
33
|
+
env_path = ".env"
|
|
34
|
+
if os.path.exists(env_path):
|
|
35
|
+
with open(env_path, "r", encoding="utf-8") as f:
|
|
36
|
+
for line in f:
|
|
37
|
+
clean_line = line.strip()
|
|
38
|
+
if clean_line.startswith("GEMINI_API_KEY"):
|
|
39
|
+
try:
|
|
40
|
+
parsed_tokens = clean_line.split("=", 1)
|
|
41
|
+
if len(parsed_tokens) == 2:
|
|
42
|
+
parsed_key = parsed_tokens[1].strip().strip('"').strip("'")
|
|
43
|
+
if parsed_key:
|
|
44
|
+
api_key = parsed_key
|
|
45
|
+
break
|
|
46
|
+
except Exception:
|
|
47
|
+
pass
|
|
48
|
+
|
|
49
|
+
if not api_key or not (api_key.startswith("AIzaSy") or api_key.startswith("AQ.")):
|
|
50
|
+
raise ValueError(
|
|
51
|
+
"🚨 ERROR CRÍTICO DE CONFIGURACIÓN: La variable 'GEMINI_API_KEY' "
|
|
52
|
+
"no está definida en el entorno ni en el archivo .env raíz."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
self.client = genai.Client(api_key=api_key)
|
|
56
|
+
self.model_name = "gemini-3.5-flash-lite"
|
|
57
|
+
|
|
58
|
+
def classify(
|
|
59
|
+
self, chunk_text: str, current_source: Optional[Any] = None
|
|
60
|
+
) -> Tuple[Optional[Any], Optional[Any]]:
|
|
61
|
+
if not chunk_text or not chunk_text.strip():
|
|
62
|
+
return None, None
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
response = self.client.models.generate_content(
|
|
66
|
+
model=self.model_name,
|
|
67
|
+
contents=chunk_text,
|
|
68
|
+
config=types.GenerateContentConfig(
|
|
69
|
+
system_instruction=self.system_instruction,
|
|
70
|
+
temperature=0.1,
|
|
71
|
+
),
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
if not response or not hasattr(response, "text") or not response.text:
|
|
75
|
+
logger.warning("Respuesta vacía o no válida recibida de la API de Gemini.")
|
|
76
|
+
return None, None
|
|
77
|
+
|
|
78
|
+
label = response.text.strip().lower()
|
|
79
|
+
|
|
80
|
+
if label == "desconocido":
|
|
81
|
+
return None, None
|
|
82
|
+
|
|
83
|
+
section_enum = self.label_mapping.get(label)
|
|
84
|
+
if section_enum is None:
|
|
85
|
+
logger.warning(
|
|
86
|
+
"Etiqueta desconocida o fuera de taxonomía devuelta por Gemini: '%s'", label
|
|
87
|
+
)
|
|
88
|
+
return None, None
|
|
89
|
+
|
|
90
|
+
# Asumimos que Gemini por ahora solo devuelve la sección, manteniendo la firma original
|
|
91
|
+
return None, section_enum
|
|
92
|
+
|
|
93
|
+
except Exception as e:
|
|
94
|
+
logger.error(
|
|
95
|
+
"Error durante la inferencia remota con el cliente Google GenAI: %s",
|
|
96
|
+
str(e),
|
|
97
|
+
exc_info=True,
|
|
98
|
+
)
|
|
99
|
+
return None, None
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from typing import Any, Dict, Optional, Tuple
|
|
3
|
+
|
|
4
|
+
import requests
|
|
5
|
+
|
|
6
|
+
from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
|
|
7
|
+
from chaskiwasi.config.settings import settings
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class LLMRouterStrategy(BaseStrategy):
|
|
13
|
+
"""Estrategia agnóstica de clasificación de respaldo mediante Ollama local."""
|
|
14
|
+
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
system_prompt: str,
|
|
18
|
+
label_mapping: Dict[str, Any],
|
|
19
|
+
endpoint: str = "http://localhost:11434/api/generate",
|
|
20
|
+
model_name: Optional[str] = None,
|
|
21
|
+
keep_alive: Optional[int] = None,
|
|
22
|
+
timeout: int = 10,
|
|
23
|
+
) -> None:
|
|
24
|
+
self.system_prompt = system_prompt
|
|
25
|
+
self.label_mapping = label_mapping
|
|
26
|
+
self.endpoint = endpoint
|
|
27
|
+
self.model_name = model_name or settings.LLM_MODEL_NAME
|
|
28
|
+
self.keep_alive = keep_alive if keep_alive is not None else settings.OLLAMA_KEEP_ALIVE
|
|
29
|
+
self.timeout = timeout
|
|
30
|
+
|
|
31
|
+
def classify(
|
|
32
|
+
self,
|
|
33
|
+
chunk_text: str,
|
|
34
|
+
current_source: Optional[Any] = None,
|
|
35
|
+
) -> Tuple[Optional[Any], Optional[Any]]:
|
|
36
|
+
"""Clasifica el fragmento mediante el modelo local de Ollama."""
|
|
37
|
+
if not isinstance(chunk_text, str) or not chunk_text.strip():
|
|
38
|
+
return None, None
|
|
39
|
+
|
|
40
|
+
payload: Dict[str, Any] = {
|
|
41
|
+
"model": self.model_name,
|
|
42
|
+
"prompt": f"{self.system_prompt}\n\n{chunk_text}",
|
|
43
|
+
"stream": False,
|
|
44
|
+
"keep_alive": self.keep_alive,
|
|
45
|
+
"options": {
|
|
46
|
+
"num_predict": 10,
|
|
47
|
+
},
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
response = requests.post(
|
|
52
|
+
self.endpoint,
|
|
53
|
+
json=payload,
|
|
54
|
+
timeout=self.timeout,
|
|
55
|
+
)
|
|
56
|
+
response.raise_for_status()
|
|
57
|
+
response_data: Dict[str, Any] = response.json()
|
|
58
|
+
except (requests.RequestException, ValueError, TypeError) as err:
|
|
59
|
+
logger.error("Error en inferencia remota con Ollama: %s", err)
|
|
60
|
+
return None, None
|
|
61
|
+
|
|
62
|
+
raw_result = str(response_data.get("response", "")).strip().upper()
|
|
63
|
+
|
|
64
|
+
if "DESCONOCIDO" in raw_result or not raw_result:
|
|
65
|
+
return None, None
|
|
66
|
+
|
|
67
|
+
for label_key, mapped_enum in self.label_mapping.items():
|
|
68
|
+
if label_key.upper() in raw_result:
|
|
69
|
+
return None, mapped_enum
|
|
70
|
+
|
|
71
|
+
return None, None
|
|
File without changes
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# dantesito/chasky/config/settings.py
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
@dataclass(frozen=True)
|
|
6
|
+
class Settings:
|
|
7
|
+
"""
|
|
8
|
+
Configuración global inmutable para el motor dantesito-chasky.
|
|
9
|
+
Diseñada con restricciones estrictas de bajo consumo en memoria RAM.
|
|
10
|
+
"""
|
|
11
|
+
CHUNK_SIZE: int = 256
|
|
12
|
+
CHUNK_OVERLAP: int = 60
|
|
13
|
+
TIKTOKEN_ENCODING: str = "cl100k_base"
|
|
14
|
+
LLM_MODEL_NAME: str = "llama3.2:1b"
|
|
15
|
+
OLLAMA_KEEP_ALIVE: int = 0
|
|
16
|
+
CHROMA_PERSISTENT_PATH: str = "./data/chroma_db"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
settings = Settings()
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# dantesito/chasky/config/taxonomy_registry.py
|
|
2
|
+
import logging
|
|
3
|
+
from typing import Any, Optional
|
|
4
|
+
|
|
5
|
+
logger = logging.getLogger(__name__)
|
|
6
|
+
|
|
7
|
+
# Contenedores en RAM que guardarán las clases reales provistas por dantesito-quipu o los mocks de test
|
|
8
|
+
_SourceEnumClass: Optional[Any] = None
|
|
9
|
+
_SectionEnumClass: Optional[Any] = None
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class TaxonomyRegistry:
|
|
13
|
+
"""
|
|
14
|
+
Registro y validador agnóstico de taxonomías para PyPI.
|
|
15
|
+
Permite que la aplicación inyecte sus Enums en caliente al arrancar.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@classmethod
|
|
19
|
+
def inject_taxonomies(cls, source_enum_class: Any, section_enum_class: Any) -> None:
|
|
20
|
+
"""Inyecta de forma segura las clases de enums del cliente en el core."""
|
|
21
|
+
global _SourceEnumClass, _SectionEnumClass
|
|
22
|
+
_SourceEnumClass = source_enum_class
|
|
23
|
+
_SectionEnumClass = section_enum_class
|
|
24
|
+
logger.info("[CHASKY-REGISTRY] Taxonomías del cliente inyectadas con éxito.")
|
|
25
|
+
|
|
26
|
+
@classmethod
|
|
27
|
+
def get_source_enum(cls) -> Any:
|
|
28
|
+
"""Devuelve la clase SourceEnum activa o un Fallback si no fue inyectada."""
|
|
29
|
+
global _SourceEnumClass
|
|
30
|
+
if _SourceEnumClass is None:
|
|
31
|
+
raise RuntimeError("Falta inyectar la clase SourceEnum al arrancar la aplicación.")
|
|
32
|
+
return _SourceEnumClass
|
|
33
|
+
|
|
34
|
+
@classmethod
|
|
35
|
+
def get_section_enum(cls) -> Any:
|
|
36
|
+
"""Devuelve la clase SectionEnum activa o un Fallback si no fue inyectada."""
|
|
37
|
+
global _SectionEnumClass
|
|
38
|
+
if _SectionEnumClass is None:
|
|
39
|
+
raise RuntimeError("Falta inyectar la clase SectionEnum al arrancar la aplicación.")
|
|
40
|
+
return _SectionEnumClass
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# =====================================================================
|
|
44
|
+
# 🚀 INTERCEPTOR DE ATRIBUTOS A NIVEL DE MÓDULO (EL SECRETO DEL ÉXITO)
|
|
45
|
+
# =====================================================================
|
|
46
|
+
def __getattr__(name: str) -> Any:
|
|
47
|
+
"""
|
|
48
|
+
Simula de forma polimórfica la existencia de SourceEnum y SectionEnum.
|
|
49
|
+
Resuelve el ImportError interceptando los 'from taxonomy_registry import ...'
|
|
50
|
+
"""
|
|
51
|
+
if name == "SourceEnum":
|
|
52
|
+
return TaxonomyRegistry.get_source_enum()
|
|
53
|
+
if name == "SectionEnum":
|
|
54
|
+
return TaxonomyRegistry.get_section_enum()
|
|
55
|
+
raise AttributeError(f"El módulo '{__name__}' no posee el atributo '{name}'")
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
import gc
|
|
2
|
+
import io
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any, Dict, List, Optional, Union
|
|
7
|
+
|
|
8
|
+
from docling.datamodel.base_models import DocumentStream
|
|
9
|
+
from docling.document_converter import DocumentConverter
|
|
10
|
+
|
|
11
|
+
from chaskiwasi.classification.cascade_factory import CascadeFactory
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ChaskyConsolidator:
|
|
17
|
+
"""
|
|
18
|
+
Motor core agnóstico de consolidación y clasificación de legajos RAG.
|
|
19
|
+
|
|
20
|
+
No posee acoplamiento con taxonomías fijas. Procesa flujos dinámicos en RAM (bytes)
|
|
21
|
+
o archivos en disco, delegando la estrategia de clasificación al `CascadeFactory` inyectado.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
cascade_factory: Optional[CascadeFactory] = None,
|
|
27
|
+
converter: Optional[DocumentConverter] = None,
|
|
28
|
+
) -> None:
|
|
29
|
+
self._cascade_factory = cascade_factory or CascadeFactory()
|
|
30
|
+
self._converter = converter or DocumentConverter()
|
|
31
|
+
|
|
32
|
+
def build_master_expediente(
|
|
33
|
+
self,
|
|
34
|
+
global_id: str,
|
|
35
|
+
data_sources: Dict[str, Union[str, Path, bytes]],
|
|
36
|
+
output_json_path: Union[str, Path],
|
|
37
|
+
) -> None:
|
|
38
|
+
"""
|
|
39
|
+
Construye el reporte maestro consolidado procesando múltiples fuentes de datos.
|
|
40
|
+
"""
|
|
41
|
+
logger.info(f"🚀 [CHASKY-CORE] Inicializando pipeline agnóstico para el lote: {global_id}")
|
|
42
|
+
|
|
43
|
+
master_data: Dict[str, Any] = {
|
|
44
|
+
"global_id": global_id,
|
|
45
|
+
"metadatos_proceso": {"total_fuentes": len(data_sources)},
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
for key_fuente, source_data in data_sources.items():
|
|
49
|
+
logger.info(f"[CHASKY-CORE] Procesando canal de información: '{key_fuente}'")
|
|
50
|
+
|
|
51
|
+
is_json_bytes = isinstance(source_data, bytes) and source_data.lstrip().startswith((b"[", b"{"))
|
|
52
|
+
|
|
53
|
+
if "json" in key_fuente.lower() or is_json_bytes:
|
|
54
|
+
master_data[key_fuente] = self._ingest_json_generic(source_data)
|
|
55
|
+
else:
|
|
56
|
+
master_data[key_fuente] = self._ingest_document_batch(source_data, key_fuente)
|
|
57
|
+
|
|
58
|
+
output_path = Path(output_json_path)
|
|
59
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
60
|
+
|
|
61
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
62
|
+
json.dump(master_data, f, ensure_ascii=False, indent=2)
|
|
63
|
+
|
|
64
|
+
logger.info(f"💾 Reporte Maestro consolidado guardado con éxito en: {output_json_path}")
|
|
65
|
+
|
|
66
|
+
del master_data
|
|
67
|
+
gc.collect()
|
|
68
|
+
|
|
69
|
+
def _ingest_json_generic(self, json_source: Union[str, Path, bytes]) -> Dict[str, Any]:
|
|
70
|
+
"""Procesa flujos JSON estructurados mediante streaming con ijson."""
|
|
71
|
+
import ijson
|
|
72
|
+
|
|
73
|
+
items = []
|
|
74
|
+
file_name = "stream_ram.json" if isinstance(json_source, bytes) else Path(json_source).name
|
|
75
|
+
|
|
76
|
+
try:
|
|
77
|
+
if isinstance(json_source, bytes):
|
|
78
|
+
with io.BytesIO(json_source) as stream:
|
|
79
|
+
parser = ijson.items(stream, "item")
|
|
80
|
+
for item in parser:
|
|
81
|
+
items.append(item)
|
|
82
|
+
return {"file": file_name, "records": items}
|
|
83
|
+
|
|
84
|
+
json_path = Path(json_source)
|
|
85
|
+
if not json_path.exists():
|
|
86
|
+
return {"file": json_path.name, "records": []}
|
|
87
|
+
|
|
88
|
+
with open(json_path, "rb") as stream:
|
|
89
|
+
parser = ijson.items(stream, "item")
|
|
90
|
+
for item in parser:
|
|
91
|
+
items.append(item)
|
|
92
|
+
return {"file": json_path.name, "records": items}
|
|
93
|
+
|
|
94
|
+
except Exception as err:
|
|
95
|
+
logger.error(f"Error en extracción genérica JSON para {file_name}: {err}")
|
|
96
|
+
return {"file": file_name, "records": []}
|
|
97
|
+
|
|
98
|
+
def _ingest_document_batch(
|
|
99
|
+
self,
|
|
100
|
+
doc_source: Union[str, Path, bytes],
|
|
101
|
+
source_label: str,
|
|
102
|
+
) -> Dict[str, Any]:
|
|
103
|
+
"""Extrae y clasifica fragmentos de documentos por lotes (Batching)."""
|
|
104
|
+
file_name = "stream_ram.pdf" if isinstance(doc_source, bytes) else Path(doc_source).name
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
lista_completa_chunks = self._extract_pdf_chunks(doc_source)
|
|
108
|
+
if not lista_completa_chunks:
|
|
109
|
+
return {"file": file_name, "sections": []}
|
|
110
|
+
|
|
111
|
+
sections_found: List[Dict[str, Any]] = []
|
|
112
|
+
|
|
113
|
+
resultados_batch = self._cascade_factory.process_chunks_batch(
|
|
114
|
+
lista_completa_chunks,
|
|
115
|
+
current_source=source_label,
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
active_context = source_label
|
|
119
|
+
for i, chunk in enumerate(lista_completa_chunks):
|
|
120
|
+
detected_source, section = resultados_batch[i]
|
|
121
|
+
|
|
122
|
+
if detected_source is not None:
|
|
123
|
+
active_context = getattr(detected_source, "value", str(detected_source))
|
|
124
|
+
|
|
125
|
+
section_value = getattr(section, "value", str(section)) if section is not None else None
|
|
126
|
+
|
|
127
|
+
sections_found.append({
|
|
128
|
+
"chunk": chunk,
|
|
129
|
+
"source": active_context,
|
|
130
|
+
"section": section_value,
|
|
131
|
+
})
|
|
132
|
+
|
|
133
|
+
return {"file": file_name, "sections": sections_found}
|
|
134
|
+
|
|
135
|
+
except Exception as err:
|
|
136
|
+
logger.error(f"Error en procesamiento masivo del documento {file_name}: {err}")
|
|
137
|
+
return {"file": file_name, "sections": []}
|
|
138
|
+
|
|
139
|
+
def _extract_pdf_chunks(self, doc_source: Union[str, Path, bytes]) -> List[str]:
|
|
140
|
+
"""Extrae texto plano Markdown utilizando Docling alimentándose de memoria o disco."""
|
|
141
|
+
chunks: List[str] = []
|
|
142
|
+
try:
|
|
143
|
+
if isinstance(doc_source, bytes):
|
|
144
|
+
with io.BytesIO(doc_source) as stream:
|
|
145
|
+
doc_stream = DocumentStream(name="stream_in_memory.pdf", stream=stream)
|
|
146
|
+
doc_result = self._converter.convert(doc_stream)
|
|
147
|
+
else:
|
|
148
|
+
doc_result = self._converter.convert(str(doc_source))
|
|
149
|
+
|
|
150
|
+
markdown_text = doc_result.document.export_to_markdown()
|
|
151
|
+
if markdown_text:
|
|
152
|
+
chunks = [c.strip() for c in markdown_text.split("\n\n") if c.strip()]
|
|
153
|
+
|
|
154
|
+
del doc_result
|
|
155
|
+
except Exception as err:
|
|
156
|
+
logger.warning(f"No se pudo extraer el contenido vía Docling: {err}")
|
|
157
|
+
|
|
158
|
+
return chunks
|
|
File without changes
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# dantesito/chasky/ingestion/docling_parser.py
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from docling.document_converter import DocumentConverter, ConversionResult
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class DoclingParser:
|
|
9
|
+
"""
|
|
10
|
+
Orquestador de extracción de documentos para dantesito-chasky.
|
|
11
|
+
Procesa PDFs mediante el pipeline de Docling y persiste el resultado en Markdown
|
|
12
|
+
directamente a disco para minimizar el impacto en memoria volátil (RAM).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
def __init__(self, converter: Optional[DocumentConverter] = None) -> None:
|
|
16
|
+
"""
|
|
17
|
+
Inicializa el convertidor de Docling. Permite la inyección de una instancia
|
|
18
|
+
personalizada para facilitación de pruebas unitarias o configuraciones avanzadas.
|
|
19
|
+
"""
|
|
20
|
+
self._converter: DocumentConverter = converter or DocumentConverter()
|
|
21
|
+
|
|
22
|
+
def parse_to_markdown(self, pdf_path: str, output_md_path: str) -> str:
|
|
23
|
+
"""
|
|
24
|
+
Procesa un archivo PDF en `pdf_path`, extrae su estructura en formato Markdown
|
|
25
|
+
y la escribe inmediatamente en el archivo físico en `output_md_path`.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
pdf_path: Ruta del archivo PDF de entrada.
|
|
29
|
+
output_md_path: Ruta del archivo Markdown de salida.
|
|
30
|
+
|
|
31
|
+
Returns:
|
|
32
|
+
str: Contenido extraído en formato Markdown.
|
|
33
|
+
|
|
34
|
+
Raises:
|
|
35
|
+
FileNotFoundError: Si el archivo PDF especificado no existe.
|
|
36
|
+
ValueError: Si la ruta de entrada no es un PDF o el documento no se puede procesar.
|
|
37
|
+
"""
|
|
38
|
+
input_path = Path(pdf_path)
|
|
39
|
+
output_path = Path(output_md_path)
|
|
40
|
+
|
|
41
|
+
if not input_path.is_file():
|
|
42
|
+
raise FileNotFoundError(f"El archivo PDF no existe en la ruta: {pdf_path}")
|
|
43
|
+
|
|
44
|
+
if input_path.suffix.lower() != ".pdf":
|
|
45
|
+
raise ValueError(f"El archivo especificado no es un PDF válido: {pdf_path}")
|
|
46
|
+
|
|
47
|
+
try:
|
|
48
|
+
result: ConversionResult = self._converter.convert(str(input_path))
|
|
49
|
+
markdown_content: str = result.document.export_to_markdown()
|
|
50
|
+
|
|
51
|
+
# Asegurar la creación del directorio padre si no existe
|
|
52
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
|
|
54
|
+
# Volcado directo a disco para mantener un footprint mínimo de memoria
|
|
55
|
+
output_path.write_text(markdown_content, encoding="utf-8")
|
|
56
|
+
|
|
57
|
+
return markdown_content
|
|
58
|
+
|
|
59
|
+
except Exception as e:
|
|
60
|
+
if isinstance(e, (FileNotFoundError, ValueError)):
|
|
61
|
+
raise
|
|
62
|
+
raise ValueError(
|
|
63
|
+
f"Error al procesar y convertir el archivo PDF '{pdf_path}': {str(e)}"
|
|
64
|
+
) from e
|
|
File without changes
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
import importlib.metadata
|
|
2
|
+
import logging
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Any, Dict, List, Optional
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class IntentRule:
|
|
11
|
+
"""Define una regla de enrutamiento basada en palabras clave."""
|
|
12
|
+
keywords: List[str]
|
|
13
|
+
metadata_filter: Dict[str, Any]
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class CrossQueryFilter:
|
|
17
|
+
"""Generador dinámico de filtros cruzados para ChromaDB compatible con Entry Points."""
|
|
18
|
+
|
|
19
|
+
def __init__(
|
|
20
|
+
self,
|
|
21
|
+
load_plugins: bool = True,
|
|
22
|
+
custom_rules: Optional[List[IntentRule]] = None,
|
|
23
|
+
) -> None:
|
|
24
|
+
"""
|
|
25
|
+
Inicializa el filtro con reglas inyectadas manualmente o descubiertas vía plugins.
|
|
26
|
+
"""
|
|
27
|
+
self.rules: List[IntentRule] = custom_rules or []
|
|
28
|
+
|
|
29
|
+
if load_plugins:
|
|
30
|
+
self._load_rules_from_entry_points()
|
|
31
|
+
|
|
32
|
+
def _load_rules_from_entry_points(self) -> None:
|
|
33
|
+
"""
|
|
34
|
+
Escanea el entorno de Python buscando plugins registrados bajo 'chaskywasi.query_rules'.
|
|
35
|
+
"""
|
|
36
|
+
try:
|
|
37
|
+
# Compatibilidad nativa para Python 3.10+
|
|
38
|
+
entry_points = importlib.metadata.entry_points(group="chaskywasi.query_rules")
|
|
39
|
+
for ep in entry_points:
|
|
40
|
+
try:
|
|
41
|
+
plugin_callable = ep.load()
|
|
42
|
+
plugin_rules = plugin_callable()
|
|
43
|
+
|
|
44
|
+
if isinstance(plugin_rules, list):
|
|
45
|
+
self.rules.extend(plugin_rules)
|
|
46
|
+
logger.info("Plugin de reglas cargado exitosamente: '%s'", ep.name)
|
|
47
|
+
except Exception as e:
|
|
48
|
+
logger.warning("Fallo al cargar el plugin de reglas '%s': %s", ep.name, e)
|
|
49
|
+
except KeyError:
|
|
50
|
+
# No hay plugins instalados para este grupo
|
|
51
|
+
pass
|
|
52
|
+
|
|
53
|
+
def generate_where_clause(self, query_text: str, doc_id: str) -> Dict[str, Any]:
|
|
54
|
+
"""
|
|
55
|
+
Analiza el texto iterando sobre las reglas dinámicas y genera el diccionario
|
|
56
|
+
de metadatos bajo las restricciones de la API de ChromaDB.
|
|
57
|
+
"""
|
|
58
|
+
base_filter = {"id_documento": doc_id}
|
|
59
|
+
|
|
60
|
+
if not query_text or not isinstance(query_text, str) or not query_text.strip():
|
|
61
|
+
return base_filter
|
|
62
|
+
|
|
63
|
+
query_lower = query_text.lower()
|
|
64
|
+
|
|
65
|
+
for rule in self.rules:
|
|
66
|
+
if any(w in query_lower for w in rule.keywords):
|
|
67
|
+
return {
|
|
68
|
+
"$and": [
|
|
69
|
+
base_filter,
|
|
70
|
+
rule.metadata_filter
|
|
71
|
+
]
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
return base_filter
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# dantesito/chasky/query_engine/ollama_streamer.py
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from typing import Any, Dict, Iterator, Optional
|
|
5
|
+
|
|
6
|
+
import requests
|
|
7
|
+
|
|
8
|
+
from chaskiwasi.config.settings import Settings
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class OllamaStreamer:
|
|
12
|
+
"""Cliente HTTP para consumir respuestas de Ollama mediante streaming."""
|
|
13
|
+
|
|
14
|
+
_DEFAULT_ENDPOINT: str = "http://localhost:11434/api/generate"
|
|
15
|
+
_CONNECTION_ERROR_MESSAGE: str = "[ERROR] No se pudo conectar con Ollama."
|
|
16
|
+
|
|
17
|
+
def __init__(self, endpoint: str = _DEFAULT_ENDPOINT) -> None:
|
|
18
|
+
self._endpoint: str = endpoint
|
|
19
|
+
|
|
20
|
+
def stream_response(
|
|
21
|
+
self,
|
|
22
|
+
prompt: str,
|
|
23
|
+
system_prompt: Optional[str] = None,
|
|
24
|
+
) -> Iterator[str]:
|
|
25
|
+
"""Genera incrementalmente la respuesta del LLM mediante streaming."""
|
|
26
|
+
payload: Dict[str, Any] = {
|
|
27
|
+
"model": Settings.LLM_MODEL_NAME,
|
|
28
|
+
"prompt": prompt,
|
|
29
|
+
"stream": True,
|
|
30
|
+
"keep_alive": Settings.OLLAMA_KEEP_ALIVE,
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
if system_prompt is not None:
|
|
34
|
+
payload["system"] = system_prompt
|
|
35
|
+
|
|
36
|
+
try:
|
|
37
|
+
response: requests.Response = requests.post(
|
|
38
|
+
self._endpoint,
|
|
39
|
+
json=payload,
|
|
40
|
+
stream=True,
|
|
41
|
+
)
|
|
42
|
+
response.raise_for_status()
|
|
43
|
+
|
|
44
|
+
for line in response.iter_lines():
|
|
45
|
+
if not line:
|
|
46
|
+
continue
|
|
47
|
+
|
|
48
|
+
try:
|
|
49
|
+
decoded_line: str = (
|
|
50
|
+
line.decode("utf-8")
|
|
51
|
+
if isinstance(line, bytes)
|
|
52
|
+
else str(line)
|
|
53
|
+
)
|
|
54
|
+
data: Dict[str, Any] = json.loads(decoded_line)
|
|
55
|
+
except (UnicodeDecodeError, json.JSONDecodeError):
|
|
56
|
+
continue
|
|
57
|
+
|
|
58
|
+
fragment: Any = data.get("response")
|
|
59
|
+
|
|
60
|
+
if isinstance(fragment, str):
|
|
61
|
+
yield fragment
|
|
62
|
+
|
|
63
|
+
except requests.exceptions.ConnectionError:
|
|
64
|
+
yield self._CONNECTION_ERROR_MESSAGE
|
|
File without changes
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# dantesito/chasky/storage/chroma_persistent.py
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import List, Optional, Tuple
|
|
5
|
+
|
|
6
|
+
import chromadb
|
|
7
|
+
from chromadb.api.models.Collection import Collection
|
|
8
|
+
|
|
9
|
+
from chaskiwasi.config.settings import Settings
|
|
10
|
+
from chaskiwasi.config.taxonomy_registry import SectionEnum, SourceEnum
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ChromaPersistentManager:
|
|
14
|
+
"""Gestiona el almacenamiento persistente de fragmentos en ChromaDB."""
|
|
15
|
+
|
|
16
|
+
def __init__(
|
|
17
|
+
self,
|
|
18
|
+
persistence_path: Optional[str] = None,
|
|
19
|
+
) -> None:
|
|
20
|
+
path: str = (
|
|
21
|
+
persistence_path
|
|
22
|
+
if persistence_path is not None
|
|
23
|
+
else Settings.CHROMA_PERSISTENT_PATH
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
Path(path).mkdir(parents=True, exist_ok=True)
|
|
27
|
+
self._client = chromadb.PersistentClient(path=path)
|
|
28
|
+
|
|
29
|
+
def get_or_create_collection(self, collection_name: str) -> Collection:
|
|
30
|
+
"""Obtiene o crea una colección usando la configuración nativa de ChromaDB."""
|
|
31
|
+
return self._client.get_or_create_collection(
|
|
32
|
+
name=collection_name,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
def add_chunks(
|
|
36
|
+
self,
|
|
37
|
+
collection_name: str,
|
|
38
|
+
doc_id: str,
|
|
39
|
+
source: SourceEnum,
|
|
40
|
+
chunks: List[Tuple[SectionEnum, str]],
|
|
41
|
+
) -> None:
|
|
42
|
+
"""Agrega fragmentos y su taxonomía estricta a una colección."""
|
|
43
|
+
if not chunks:
|
|
44
|
+
return
|
|
45
|
+
|
|
46
|
+
collection: Collection = self.get_or_create_collection(collection_name)
|
|
47
|
+
|
|
48
|
+
documents: List[str] = []
|
|
49
|
+
metadatas: List[dict[str, str]] = []
|
|
50
|
+
ids: List[str] = []
|
|
51
|
+
|
|
52
|
+
for idx, (section, text) in enumerate(chunks):
|
|
53
|
+
documents.append(text)
|
|
54
|
+
metadatas.append(
|
|
55
|
+
{
|
|
56
|
+
"id_documento": doc_id,
|
|
57
|
+
"fuente": source.value,
|
|
58
|
+
"tipo_seccion": section.value,
|
|
59
|
+
}
|
|
60
|
+
)
|
|
61
|
+
ids.append(f"{doc_id}_{idx}")
|
|
62
|
+
|
|
63
|
+
collection.add(
|
|
64
|
+
documents=documents,
|
|
65
|
+
metadatas=metadatas,
|
|
66
|
+
ids=ids,
|
|
67
|
+
)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: chaskiwasi
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Sistema de consulta semántica para fuentes jurídicas.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Maorda/chaskywasi
|
|
6
|
+
Project-URL: Issues, https://github.com/Maorda/chaskywasi/issues
|
|
7
|
+
Author-email: Luis Maurtua <luis.maurtua@ejemplo.com>
|
|
8
|
+
License: MIT
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Requires-Dist: chromadb>=0.5.0
|
|
17
|
+
Requires-Dist: docling>=2.0.0
|
|
18
|
+
Requires-Dist: google-genai>=0.1.0
|
|
19
|
+
Requires-Dist: httpx>=0.27.0
|
|
20
|
+
Requires-Dist: ijson>=3.3.0
|
|
21
|
+
Requires-Dist: requests>=2.32.0
|
|
22
|
+
Requires-Dist: streamlit>=1.30.0
|
|
23
|
+
Requires-Dist: tiktoken>=0.7.0
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: anyio>=4.0.0; extra == 'dev'
|
|
26
|
+
Requires-Dist: faker>=20.0.0; extra == 'dev'
|
|
27
|
+
Requires-Dist: pytest-asyncio>=0.23.0; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
GOLA
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
chaskiwasi/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
chaskiwasi/chunking/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
3
|
+
chaskiwasi/chunking/chunk_tokenizer.py,sha256=xEJm1K5hqhiUIbHFI1gzYDUP0TwhJ1fYTe4vCBe_Qmg,1187
|
|
4
|
+
chaskiwasi/classification/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
5
|
+
chaskiwasi/classification/cascade_factory.py,sha256=-H3aCHHlRAaUwSBN1Li2CvyYO4NFwhsxo7USLPp7BOM,2440
|
|
6
|
+
chaskiwasi/classification/strategies/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
chaskiwasi/classification/strategies/base_strategy.py,sha256=xvFmcSVrVUuBoKbT5lBLAJrICSjXqopf2Obt3yI9qfE,1719
|
|
8
|
+
chaskiwasi/classification/strategies/context_overlap_strategy.py,sha256=Ac7NKilkW3xRYN0RzclSYkFeUmJGy539TFREegv6Kx0,743
|
|
9
|
+
chaskiwasi/classification/strategies/cpu_regex_strategy.py,sha256=eg8VeFTCoYCfjUIPVvXs1DLPv5P1Pwj8uv4mIhpywHQ,2295
|
|
10
|
+
chaskiwasi/classification/strategies/gemini_streamer.py,sha256=j5qLhJRmi4xNRsYtCth5mzn0t81ANemO8JGThUb252I,3626
|
|
11
|
+
chaskiwasi/classification/strategies/llm_router_strategy.py,sha256=VLJH6UnkpBNlO840fJH5SIsB2qu-bFyRMbhy9MK8eSE,2375
|
|
12
|
+
chaskiwasi/config/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
|
+
chaskiwasi/config/settings.py,sha256=0YOUJoiAkWuWVMoAmomKAOiYXfoxa50QRSKM5DWbIj0,516
|
|
14
|
+
chaskiwasi/config/taxonomy_registry.py,sha256=mAvjH1jqAf-yO_VTSb3HJwj5N582R1DGYhV5DjMlSxw,2271
|
|
15
|
+
chaskiwasi/consolidation/consolidator.py,sha256=QLPSnn5dxDJmFwQGAd7kG02QjGRM_3wtZwqPZXcf_BA,6129
|
|
16
|
+
chaskiwasi/ingestion/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
17
|
+
chaskiwasi/ingestion/docling_parser.py,sha256=UNEwot3LOYJrZcbhnnEI94UDm6DvM13w_jKjFUTII4U,2563
|
|
18
|
+
chaskiwasi/query_engine/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
19
|
+
chaskiwasi/query_engine/cross_filter.py,sha256=PvpAMnuyM8aiw-F_PUI8Mac775Yircgo6XOKY63xMDk,2551
|
|
20
|
+
chaskiwasi/query_engine/ollama_streamer.py,sha256=V2UsffssQ5Zy4EEsSxPQQQ4lrcN-ktcVe4xa-FXRqjw,1977
|
|
21
|
+
chaskiwasi/storage/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
22
|
+
chaskiwasi/storage/chroma_persistent.py,sha256=2cJlLZc0kmktVG65DYpYv1ZUCU3hsqQ3rXU5bTxWgt0,2000
|
|
23
|
+
chaskiwasi-0.2.0.dist-info/METADATA,sha256=-nGWk4TDNGbkhYcl-ajKACwUwruWvgKIuMrUiC9sgp0,1136
|
|
24
|
+
chaskiwasi-0.2.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
|
|
25
|
+
chaskiwasi-0.2.0.dist-info/RECORD,,
|