chaskiwasi 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
chaskiwasi/__init__.py ADDED
File without changes
File without changes
@@ -0,0 +1,43 @@
1
+ # dantesito/chasky/chunking/chunk_tokenizer.py
2
+
3
+ from typing import List
4
+
5
+ import tiktoken
6
+
7
+ from chaskiwasi.config.settings import Settings
8
+
9
+
10
+ class ChunkTokenizer:
11
+ """Divide textos en fragmentos mediante ventanas de tokens de tiktoken."""
12
+
13
+ def __init__(self) -> None:
14
+ self._encoding = tiktoken.get_encoding(Settings.TIKTOKEN_ENCODING)
15
+
16
+ def split_text(self, text: str) -> List[str]:
17
+ """Divide el texto en chunks determinados exclusivamente por tokens."""
18
+ if not isinstance(text, str) or not text:
19
+ return []
20
+
21
+ tokens: List[int] = self._encoding.encode(text)
22
+
23
+ if not tokens:
24
+ return []
25
+
26
+ chunk_size: int = Settings.CHUNK_SIZE
27
+ step: int = Settings.CHUNK_SIZE - Settings.CHUNK_OVERLAP
28
+
29
+ chunks: List[str] = []
30
+ start: int = 0
31
+ total_tokens: int = len(tokens)
32
+
33
+ while start < total_tokens:
34
+ end: int = min(start + chunk_size, total_tokens)
35
+ chunk_tokens: List[int] = tokens[start:end]
36
+ chunks.append(self._encoding.decode(chunk_tokens))
37
+
38
+ if end >= total_tokens:
39
+ break
40
+
41
+ start += step
42
+
43
+ return chunks
File without changes
@@ -0,0 +1,68 @@
1
+ import logging
2
+ from typing import Any, List, Optional, Tuple
3
+
4
+ from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ class CascadeFactory:
9
+ """
10
+ Fábrica y orquestador de clasificación en cascada agnóstico para chaskywasi.
11
+ Aplica una arquitectura de enrutamiento por estrategias ordenadas jerárquicamente.
12
+ """
13
+
14
+ def __init__(
15
+ self,
16
+ strategies: Optional[List[BaseStrategy]] = None,
17
+ default_source: Optional[Any] = None,
18
+ default_section: Optional[Any] = None,
19
+ ) -> None:
20
+ self._strategies: List[BaseStrategy] = strategies or []
21
+ self.default_source = default_source
22
+ self.default_section = default_section
23
+
24
+ def process_chunk(
25
+ self,
26
+ text: Optional[str],
27
+ current_source: Optional[Any] = None,
28
+ ) -> Tuple[Optional[Any], Optional[Any]]:
29
+ """Procesa un fragmento individual de texto iterando sobre la lista de estrategias."""
30
+ effective_default_source = current_source or self.default_source
31
+
32
+ if not text or not text.strip():
33
+ return effective_default_source, self.default_section
34
+
35
+ for strategy in self._strategies:
36
+ try:
37
+ src_res, sec_res = strategy.classify(
38
+ text, current_source=effective_default_source
39
+ )
40
+
41
+ if sec_res is not None:
42
+ final_source = src_res or effective_default_source
43
+ return final_source, sec_res
44
+ except Exception as e:
45
+ logger.warning(
46
+ "Estrategia %s falló durante la clasificación: %s",
47
+ strategy.__class__.__name__,
48
+ str(e),
49
+ )
50
+
51
+ return effective_default_source, self.default_section
52
+
53
+ def process_chunks_batch(
54
+ self,
55
+ chunks: List[str],
56
+ current_source: Optional[Any] = None,
57
+ ) -> List[Tuple[Optional[Any], Optional[Any]]]:
58
+ """Procesa una lista de fragmentos secuencialmente manteniendo el contexto actual."""
59
+ results = []
60
+ active_source = current_source or self.default_source
61
+
62
+ for chunk in chunks:
63
+ src, sec = self.process_chunk(chunk, current_source=active_source)
64
+ if src is not None:
65
+ active_source = src
66
+ results.append((src, sec))
67
+
68
+ return results
File without changes
@@ -0,0 +1,43 @@
1
+ # dantesito/chasky/classification/strategies/base_strategy.py
2
+ from abc import ABC, abstractmethod
3
+ from typing import Tuple, Optional, Any
4
+
5
+ # 🚀 CONTRATO ACTIVO: Importación obligatoria para control de tipos dinámicos
6
+ from chaskiwasi.config.taxonomy_registry import TaxonomyRegistry
7
+
8
+
9
+ class BaseStrategy(ABC):
10
+ """
11
+ Interfaz contractual agnóstica para todas las estrategias de inferencia semántica.
12
+
13
+ Expone de forma segura las clases de Enums inyectadas por el cliente mediante
14
+ propiedades dinámicas de solo lectura, eliminando la fragilidad de constructores.
15
+ """
16
+
17
+ @property
18
+ def source_enum_class(self) -> Any:
19
+ """
20
+ Devuelve de forma dinámica la clase SourceEnum del cliente.
21
+ 🛡️ CORTAFUEGOS EN CALIENTE: Lanza RuntimeError si no fue inyectada al arrancar.
22
+ """
23
+ return TaxonomyRegistry.get_source_enum()
24
+
25
+ @property
26
+ def section_enum_class(self) -> Any:
27
+ """
28
+ Devuelve de forma dinámica la clase SectionEnum del cliente.
29
+ 🛡️ CORTAFUEGOS EN CALIENTE: Lanza RuntimeError si no fue inyectada al arrancar.
30
+ """
31
+ return TaxonomyRegistry.get_section_enum()
32
+
33
+ @abstractmethod
34
+ def classify(self, chunk: str, current_source: Optional[str] = None) -> Tuple[Optional[Any], Optional[Any]]:
35
+ """
36
+ Analiza un fragmento y devuelve una tupla conteniendo las instancias
37
+ de los Enums dinámicos del cliente (SourceEnum, SectionEnum).
38
+
39
+ :param chunk: Fragmento de texto Markdown a evaluar.
40
+ :param current_source: Contexto de origen acumulado o arrastrado.
41
+ :return: Tupla conteniendo (Instancia de SourceEnum, Instancia de SectionEnum)
42
+ """
43
+ pass
@@ -0,0 +1,21 @@
1
+ # dantesito/chasky/classification/strategies/context_overlap_strategy.py
2
+
3
+ from typing import Optional, Tuple
4
+
5
+ from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
6
+ from chaskiwasi.config.taxonomy_registry import SectionEnum, SourceEnum
7
+
8
+
9
+ class ContextOverlapStrategy(BaseStrategy):
10
+ """Arrastra la fuente institucional desde el contexto del chunk anterior."""
11
+
12
+ def classify(
13
+ self,
14
+ chunk_text: str,
15
+ current_source: Optional[SourceEnum] = None,
16
+ ) -> Tuple[Optional[SourceEnum], Optional[SectionEnum]]:
17
+ """Mantiene la fuente previa sin inspeccionar el contenido del chunk."""
18
+ if current_source is not None:
19
+ return current_source, None
20
+
21
+ return None, None
@@ -0,0 +1,62 @@
1
+ import re
2
+ from dataclasses import dataclass, field
3
+ from typing import Any, List, Optional, Pattern, Tuple
4
+ from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
5
+
6
+
7
+ @dataclass
8
+ class RegexRule:
9
+ """Estructura de datos para definir reglas de clasificación por Regex."""
10
+ source: Any
11
+ section: Any
12
+ patterns: List[str]
13
+ flags: int = re.IGNORECASE | re.DOTALL
14
+ compiled_patterns: List[Pattern[str]] = field(init=False, repr=False)
15
+
16
+ def __post_init__(self) -> None:
17
+ compiled: List[Pattern[str]] = []
18
+ for pattern in self.patterns:
19
+ try:
20
+ compiled.append(re.compile(pattern, self.flags))
21
+ except re.error as err:
22
+ raise ValueError(
23
+ f"Patrón Regex inválido '{pattern}' para la regla ({self.source}, {self.section}): {err}"
24
+ ) from err
25
+ self.compiled_patterns = compiled
26
+
27
+
28
+ class CPURegexStrategy(BaseStrategy):
29
+ """Estrategia genérica que ejecuta reglas Regex inyectadas."""
30
+
31
+ def __init__(self, rules: Optional[List[RegexRule]] = None) -> None:
32
+ self.rules: List[RegexRule] = list(rules) if rules else []
33
+
34
+ def add_rule(self, rule: RegexRule) -> None:
35
+ """Permite registrar reglas dinámicamente."""
36
+ self.rules.append(rule)
37
+
38
+ def classify(
39
+ self,
40
+ chunk_text: str,
41
+ current_source: Optional[Any] = None,
42
+ ) -> Tuple[Optional[Any], Optional[Any]]:
43
+ if not isinstance(chunk_text, str) or not chunk_text:
44
+ return None, None
45
+
46
+ # Evaluación con preferencia contextual por current_source si aplica
47
+ for rule in self.rules:
48
+ if current_source and rule.source != current_source:
49
+ continue
50
+
51
+ if any(pattern.search(chunk_text) for pattern in rule.compiled_patterns):
52
+ return rule.source, rule.section
53
+
54
+ # Segunda pasada para reglas generales si no coincidió en el contexto actual
55
+ if current_source:
56
+ for rule in self.rules:
57
+ if rule.source == current_source:
58
+ continue
59
+ if any(pattern.search(chunk_text) for pattern in rule.compiled_patterns):
60
+ return rule.source, rule.section
61
+
62
+ return None, None
@@ -0,0 +1,99 @@
1
+ import logging
2
+ import os
3
+ from typing import Any, Dict, Optional, Tuple
4
+
5
+ from google import genai
6
+ from google.genai import types
7
+
8
+ from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
9
+
10
+ logging.getLogger("google.genai").setLevel(logging.ERROR)
11
+ logging.getLogger("google.genai._api_client").setLevel(logging.ERROR)
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ class GeminiStreamer(BaseStrategy):
17
+ """Estrategia de inferencia remota en la nube totalmente agnóstica al dominio."""
18
+
19
+ def __init__(
20
+ self,
21
+ system_instruction: str,
22
+ label_mapping: Dict[str, Any],
23
+ ) -> None:
24
+ self.system_instruction = system_instruction
25
+ self.label_mapping = label_mapping
26
+
27
+ api_key = ""
28
+ env_key = os.environ.get("GEMINI_API_KEY", "").strip()
29
+ if env_key:
30
+ api_key = env_key
31
+
32
+ if not api_key:
33
+ env_path = ".env"
34
+ if os.path.exists(env_path):
35
+ with open(env_path, "r", encoding="utf-8") as f:
36
+ for line in f:
37
+ clean_line = line.strip()
38
+ if clean_line.startswith("GEMINI_API_KEY"):
39
+ try:
40
+ parsed_tokens = clean_line.split("=", 1)
41
+ if len(parsed_tokens) == 2:
42
+ parsed_key = parsed_tokens[1].strip().strip('"').strip("'")
43
+ if parsed_key:
44
+ api_key = parsed_key
45
+ break
46
+ except Exception:
47
+ pass
48
+
49
+ if not api_key or not (api_key.startswith("AIzaSy") or api_key.startswith("AQ.")):
50
+ raise ValueError(
51
+ "🚨 ERROR CRÍTICO DE CONFIGURACIÓN: La variable 'GEMINI_API_KEY' "
52
+ "no está definida en el entorno ni en el archivo .env raíz."
53
+ )
54
+
55
+ self.client = genai.Client(api_key=api_key)
56
+ self.model_name = "gemini-3.5-flash-lite"
57
+
58
+ def classify(
59
+ self, chunk_text: str, current_source: Optional[Any] = None
60
+ ) -> Tuple[Optional[Any], Optional[Any]]:
61
+ if not chunk_text or not chunk_text.strip():
62
+ return None, None
63
+
64
+ try:
65
+ response = self.client.models.generate_content(
66
+ model=self.model_name,
67
+ contents=chunk_text,
68
+ config=types.GenerateContentConfig(
69
+ system_instruction=self.system_instruction,
70
+ temperature=0.1,
71
+ ),
72
+ )
73
+
74
+ if not response or not hasattr(response, "text") or not response.text:
75
+ logger.warning("Respuesta vacía o no válida recibida de la API de Gemini.")
76
+ return None, None
77
+
78
+ label = response.text.strip().lower()
79
+
80
+ if label == "desconocido":
81
+ return None, None
82
+
83
+ section_enum = self.label_mapping.get(label)
84
+ if section_enum is None:
85
+ logger.warning(
86
+ "Etiqueta desconocida o fuera de taxonomía devuelta por Gemini: '%s'", label
87
+ )
88
+ return None, None
89
+
90
+ # Asumimos que Gemini por ahora solo devuelve la sección, manteniendo la firma original
91
+ return None, section_enum
92
+
93
+ except Exception as e:
94
+ logger.error(
95
+ "Error durante la inferencia remota con el cliente Google GenAI: %s",
96
+ str(e),
97
+ exc_info=True,
98
+ )
99
+ return None, None
@@ -0,0 +1,71 @@
1
+ import logging
2
+ from typing import Any, Dict, Optional, Tuple
3
+
4
+ import requests
5
+
6
+ from chaskiwasi.classification.strategies.base_strategy import BaseStrategy
7
+ from chaskiwasi.config.settings import settings
8
+
9
+ logger = logging.getLogger(__name__)
10
+
11
+
12
+ class LLMRouterStrategy(BaseStrategy):
13
+ """Estrategia agnóstica de clasificación de respaldo mediante Ollama local."""
14
+
15
+ def __init__(
16
+ self,
17
+ system_prompt: str,
18
+ label_mapping: Dict[str, Any],
19
+ endpoint: str = "http://localhost:11434/api/generate",
20
+ model_name: Optional[str] = None,
21
+ keep_alive: Optional[int] = None,
22
+ timeout: int = 10,
23
+ ) -> None:
24
+ self.system_prompt = system_prompt
25
+ self.label_mapping = label_mapping
26
+ self.endpoint = endpoint
27
+ self.model_name = model_name or settings.LLM_MODEL_NAME
28
+ self.keep_alive = keep_alive if keep_alive is not None else settings.OLLAMA_KEEP_ALIVE
29
+ self.timeout = timeout
30
+
31
+ def classify(
32
+ self,
33
+ chunk_text: str,
34
+ current_source: Optional[Any] = None,
35
+ ) -> Tuple[Optional[Any], Optional[Any]]:
36
+ """Clasifica el fragmento mediante el modelo local de Ollama."""
37
+ if not isinstance(chunk_text, str) or not chunk_text.strip():
38
+ return None, None
39
+
40
+ payload: Dict[str, Any] = {
41
+ "model": self.model_name,
42
+ "prompt": f"{self.system_prompt}\n\n{chunk_text}",
43
+ "stream": False,
44
+ "keep_alive": self.keep_alive,
45
+ "options": {
46
+ "num_predict": 10,
47
+ },
48
+ }
49
+
50
+ try:
51
+ response = requests.post(
52
+ self.endpoint,
53
+ json=payload,
54
+ timeout=self.timeout,
55
+ )
56
+ response.raise_for_status()
57
+ response_data: Dict[str, Any] = response.json()
58
+ except (requests.RequestException, ValueError, TypeError) as err:
59
+ logger.error("Error en inferencia remota con Ollama: %s", err)
60
+ return None, None
61
+
62
+ raw_result = str(response_data.get("response", "")).strip().upper()
63
+
64
+ if "DESCONOCIDO" in raw_result or not raw_result:
65
+ return None, None
66
+
67
+ for label_key, mapped_enum in self.label_mapping.items():
68
+ if label_key.upper() in raw_result:
69
+ return None, mapped_enum
70
+
71
+ return None, None
File without changes
@@ -0,0 +1,19 @@
1
+ # dantesito/chasky/config/settings.py
2
+ from dataclasses import dataclass
3
+
4
+
5
+ @dataclass(frozen=True)
6
+ class Settings:
7
+ """
8
+ Configuración global inmutable para el motor dantesito-chasky.
9
+ Diseñada con restricciones estrictas de bajo consumo en memoria RAM.
10
+ """
11
+ CHUNK_SIZE: int = 256
12
+ CHUNK_OVERLAP: int = 60
13
+ TIKTOKEN_ENCODING: str = "cl100k_base"
14
+ LLM_MODEL_NAME: str = "llama3.2:1b"
15
+ OLLAMA_KEEP_ALIVE: int = 0
16
+ CHROMA_PERSISTENT_PATH: str = "./data/chroma_db"
17
+
18
+
19
+ settings = Settings()
@@ -0,0 +1,55 @@
1
+ # dantesito/chasky/config/taxonomy_registry.py
2
+ import logging
3
+ from typing import Any, Optional
4
+
5
+ logger = logging.getLogger(__name__)
6
+
7
+ # Contenedores en RAM que guardarán las clases reales provistas por dantesito-quipu o los mocks de test
8
+ _SourceEnumClass: Optional[Any] = None
9
+ _SectionEnumClass: Optional[Any] = None
10
+
11
+
12
+ class TaxonomyRegistry:
13
+ """
14
+ Registro y validador agnóstico de taxonomías para PyPI.
15
+ Permite que la aplicación inyecte sus Enums en caliente al arrancar.
16
+ """
17
+
18
+ @classmethod
19
+ def inject_taxonomies(cls, source_enum_class: Any, section_enum_class: Any) -> None:
20
+ """Inyecta de forma segura las clases de enums del cliente en el core."""
21
+ global _SourceEnumClass, _SectionEnumClass
22
+ _SourceEnumClass = source_enum_class
23
+ _SectionEnumClass = section_enum_class
24
+ logger.info("[CHASKY-REGISTRY] Taxonomías del cliente inyectadas con éxito.")
25
+
26
+ @classmethod
27
+ def get_source_enum(cls) -> Any:
28
+ """Devuelve la clase SourceEnum activa o un Fallback si no fue inyectada."""
29
+ global _SourceEnumClass
30
+ if _SourceEnumClass is None:
31
+ raise RuntimeError("Falta inyectar la clase SourceEnum al arrancar la aplicación.")
32
+ return _SourceEnumClass
33
+
34
+ @classmethod
35
+ def get_section_enum(cls) -> Any:
36
+ """Devuelve la clase SectionEnum activa o un Fallback si no fue inyectada."""
37
+ global _SectionEnumClass
38
+ if _SectionEnumClass is None:
39
+ raise RuntimeError("Falta inyectar la clase SectionEnum al arrancar la aplicación.")
40
+ return _SectionEnumClass
41
+
42
+
43
+ # =====================================================================
44
+ # 🚀 INTERCEPTOR DE ATRIBUTOS A NIVEL DE MÓDULO (EL SECRETO DEL ÉXITO)
45
+ # =====================================================================
46
+ def __getattr__(name: str) -> Any:
47
+ """
48
+ Simula de forma polimórfica la existencia de SourceEnum y SectionEnum.
49
+ Resuelve el ImportError interceptando los 'from taxonomy_registry import ...'
50
+ """
51
+ if name == "SourceEnum":
52
+ return TaxonomyRegistry.get_source_enum()
53
+ if name == "SectionEnum":
54
+ return TaxonomyRegistry.get_section_enum()
55
+ raise AttributeError(f"El módulo '{__name__}' no posee el atributo '{name}'")
@@ -0,0 +1,158 @@
1
+ import gc
2
+ import io
3
+ import json
4
+ import logging
5
+ from pathlib import Path
6
+ from typing import Any, Dict, List, Optional, Union
7
+
8
+ from docling.datamodel.base_models import DocumentStream
9
+ from docling.document_converter import DocumentConverter
10
+
11
+ from chaskiwasi.classification.cascade_factory import CascadeFactory
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ class ChaskyConsolidator:
17
+ """
18
+ Motor core agnóstico de consolidación y clasificación de legajos RAG.
19
+
20
+ No posee acoplamiento con taxonomías fijas. Procesa flujos dinámicos en RAM (bytes)
21
+ o archivos en disco, delegando la estrategia de clasificación al `CascadeFactory` inyectado.
22
+ """
23
+
24
+ def __init__(
25
+ self,
26
+ cascade_factory: Optional[CascadeFactory] = None,
27
+ converter: Optional[DocumentConverter] = None,
28
+ ) -> None:
29
+ self._cascade_factory = cascade_factory or CascadeFactory()
30
+ self._converter = converter or DocumentConverter()
31
+
32
+ def build_master_expediente(
33
+ self,
34
+ global_id: str,
35
+ data_sources: Dict[str, Union[str, Path, bytes]],
36
+ output_json_path: Union[str, Path],
37
+ ) -> None:
38
+ """
39
+ Construye el reporte maestro consolidado procesando múltiples fuentes de datos.
40
+ """
41
+ logger.info(f"🚀 [CHASKY-CORE] Inicializando pipeline agnóstico para el lote: {global_id}")
42
+
43
+ master_data: Dict[str, Any] = {
44
+ "global_id": global_id,
45
+ "metadatos_proceso": {"total_fuentes": len(data_sources)},
46
+ }
47
+
48
+ for key_fuente, source_data in data_sources.items():
49
+ logger.info(f"[CHASKY-CORE] Procesando canal de información: '{key_fuente}'")
50
+
51
+ is_json_bytes = isinstance(source_data, bytes) and source_data.lstrip().startswith((b"[", b"{"))
52
+
53
+ if "json" in key_fuente.lower() or is_json_bytes:
54
+ master_data[key_fuente] = self._ingest_json_generic(source_data)
55
+ else:
56
+ master_data[key_fuente] = self._ingest_document_batch(source_data, key_fuente)
57
+
58
+ output_path = Path(output_json_path)
59
+ output_path.parent.mkdir(parents=True, exist_ok=True)
60
+
61
+ with open(output_path, "w", encoding="utf-8") as f:
62
+ json.dump(master_data, f, ensure_ascii=False, indent=2)
63
+
64
+ logger.info(f"💾 Reporte Maestro consolidado guardado con éxito en: {output_json_path}")
65
+
66
+ del master_data
67
+ gc.collect()
68
+
69
+ def _ingest_json_generic(self, json_source: Union[str, Path, bytes]) -> Dict[str, Any]:
70
+ """Procesa flujos JSON estructurados mediante streaming con ijson."""
71
+ import ijson
72
+
73
+ items = []
74
+ file_name = "stream_ram.json" if isinstance(json_source, bytes) else Path(json_source).name
75
+
76
+ try:
77
+ if isinstance(json_source, bytes):
78
+ with io.BytesIO(json_source) as stream:
79
+ parser = ijson.items(stream, "item")
80
+ for item in parser:
81
+ items.append(item)
82
+ return {"file": file_name, "records": items}
83
+
84
+ json_path = Path(json_source)
85
+ if not json_path.exists():
86
+ return {"file": json_path.name, "records": []}
87
+
88
+ with open(json_path, "rb") as stream:
89
+ parser = ijson.items(stream, "item")
90
+ for item in parser:
91
+ items.append(item)
92
+ return {"file": json_path.name, "records": items}
93
+
94
+ except Exception as err:
95
+ logger.error(f"Error en extracción genérica JSON para {file_name}: {err}")
96
+ return {"file": file_name, "records": []}
97
+
98
+ def _ingest_document_batch(
99
+ self,
100
+ doc_source: Union[str, Path, bytes],
101
+ source_label: str,
102
+ ) -> Dict[str, Any]:
103
+ """Extrae y clasifica fragmentos de documentos por lotes (Batching)."""
104
+ file_name = "stream_ram.pdf" if isinstance(doc_source, bytes) else Path(doc_source).name
105
+
106
+ try:
107
+ lista_completa_chunks = self._extract_pdf_chunks(doc_source)
108
+ if not lista_completa_chunks:
109
+ return {"file": file_name, "sections": []}
110
+
111
+ sections_found: List[Dict[str, Any]] = []
112
+
113
+ resultados_batch = self._cascade_factory.process_chunks_batch(
114
+ lista_completa_chunks,
115
+ current_source=source_label,
116
+ )
117
+
118
+ active_context = source_label
119
+ for i, chunk in enumerate(lista_completa_chunks):
120
+ detected_source, section = resultados_batch[i]
121
+
122
+ if detected_source is not None:
123
+ active_context = getattr(detected_source, "value", str(detected_source))
124
+
125
+ section_value = getattr(section, "value", str(section)) if section is not None else None
126
+
127
+ sections_found.append({
128
+ "chunk": chunk,
129
+ "source": active_context,
130
+ "section": section_value,
131
+ })
132
+
133
+ return {"file": file_name, "sections": sections_found}
134
+
135
+ except Exception as err:
136
+ logger.error(f"Error en procesamiento masivo del documento {file_name}: {err}")
137
+ return {"file": file_name, "sections": []}
138
+
139
+ def _extract_pdf_chunks(self, doc_source: Union[str, Path, bytes]) -> List[str]:
140
+ """Extrae texto plano Markdown utilizando Docling alimentándose de memoria o disco."""
141
+ chunks: List[str] = []
142
+ try:
143
+ if isinstance(doc_source, bytes):
144
+ with io.BytesIO(doc_source) as stream:
145
+ doc_stream = DocumentStream(name="stream_in_memory.pdf", stream=stream)
146
+ doc_result = self._converter.convert(doc_stream)
147
+ else:
148
+ doc_result = self._converter.convert(str(doc_source))
149
+
150
+ markdown_text = doc_result.document.export_to_markdown()
151
+ if markdown_text:
152
+ chunks = [c.strip() for c in markdown_text.split("\n\n") if c.strip()]
153
+
154
+ del doc_result
155
+ except Exception as err:
156
+ logger.warning(f"No se pudo extraer el contenido vía Docling: {err}")
157
+
158
+ return chunks
File without changes
@@ -0,0 +1,64 @@
1
+ # dantesito/chasky/ingestion/docling_parser.py
2
+ from pathlib import Path
3
+ from typing import Optional
4
+
5
+ from docling.document_converter import DocumentConverter, ConversionResult
6
+
7
+
8
+ class DoclingParser:
9
+ """
10
+ Orquestador de extracción de documentos para dantesito-chasky.
11
+ Procesa PDFs mediante el pipeline de Docling y persiste el resultado en Markdown
12
+ directamente a disco para minimizar el impacto en memoria volátil (RAM).
13
+ """
14
+
15
+ def __init__(self, converter: Optional[DocumentConverter] = None) -> None:
16
+ """
17
+ Inicializa el convertidor de Docling. Permite la inyección de una instancia
18
+ personalizada para facilitación de pruebas unitarias o configuraciones avanzadas.
19
+ """
20
+ self._converter: DocumentConverter = converter or DocumentConverter()
21
+
22
+ def parse_to_markdown(self, pdf_path: str, output_md_path: str) -> str:
23
+ """
24
+ Procesa un archivo PDF en `pdf_path`, extrae su estructura en formato Markdown
25
+ y la escribe inmediatamente en el archivo físico en `output_md_path`.
26
+
27
+ Args:
28
+ pdf_path: Ruta del archivo PDF de entrada.
29
+ output_md_path: Ruta del archivo Markdown de salida.
30
+
31
+ Returns:
32
+ str: Contenido extraído en formato Markdown.
33
+
34
+ Raises:
35
+ FileNotFoundError: Si el archivo PDF especificado no existe.
36
+ ValueError: Si la ruta de entrada no es un PDF o el documento no se puede procesar.
37
+ """
38
+ input_path = Path(pdf_path)
39
+ output_path = Path(output_md_path)
40
+
41
+ if not input_path.is_file():
42
+ raise FileNotFoundError(f"El archivo PDF no existe en la ruta: {pdf_path}")
43
+
44
+ if input_path.suffix.lower() != ".pdf":
45
+ raise ValueError(f"El archivo especificado no es un PDF válido: {pdf_path}")
46
+
47
+ try:
48
+ result: ConversionResult = self._converter.convert(str(input_path))
49
+ markdown_content: str = result.document.export_to_markdown()
50
+
51
+ # Asegurar la creación del directorio padre si no existe
52
+ output_path.parent.mkdir(parents=True, exist_ok=True)
53
+
54
+ # Volcado directo a disco para mantener un footprint mínimo de memoria
55
+ output_path.write_text(markdown_content, encoding="utf-8")
56
+
57
+ return markdown_content
58
+
59
+ except Exception as e:
60
+ if isinstance(e, (FileNotFoundError, ValueError)):
61
+ raise
62
+ raise ValueError(
63
+ f"Error al procesar y convertir el archivo PDF '{pdf_path}': {str(e)}"
64
+ ) from e
File without changes
@@ -0,0 +1,74 @@
1
+ import importlib.metadata
2
+ import logging
3
+ from dataclasses import dataclass
4
+ from typing import Any, Dict, List, Optional
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+
9
+ @dataclass
10
+ class IntentRule:
11
+ """Define una regla de enrutamiento basada en palabras clave."""
12
+ keywords: List[str]
13
+ metadata_filter: Dict[str, Any]
14
+
15
+
16
+ class CrossQueryFilter:
17
+ """Generador dinámico de filtros cruzados para ChromaDB compatible con Entry Points."""
18
+
19
+ def __init__(
20
+ self,
21
+ load_plugins: bool = True,
22
+ custom_rules: Optional[List[IntentRule]] = None,
23
+ ) -> None:
24
+ """
25
+ Inicializa el filtro con reglas inyectadas manualmente o descubiertas vía plugins.
26
+ """
27
+ self.rules: List[IntentRule] = custom_rules or []
28
+
29
+ if load_plugins:
30
+ self._load_rules_from_entry_points()
31
+
32
+ def _load_rules_from_entry_points(self) -> None:
33
+ """
34
+ Escanea el entorno de Python buscando plugins registrados bajo 'chaskywasi.query_rules'.
35
+ """
36
+ try:
37
+ # Compatibilidad nativa para Python 3.10+
38
+ entry_points = importlib.metadata.entry_points(group="chaskywasi.query_rules")
39
+ for ep in entry_points:
40
+ try:
41
+ plugin_callable = ep.load()
42
+ plugin_rules = plugin_callable()
43
+
44
+ if isinstance(plugin_rules, list):
45
+ self.rules.extend(plugin_rules)
46
+ logger.info("Plugin de reglas cargado exitosamente: '%s'", ep.name)
47
+ except Exception as e:
48
+ logger.warning("Fallo al cargar el plugin de reglas '%s': %s", ep.name, e)
49
+ except KeyError:
50
+ # No hay plugins instalados para este grupo
51
+ pass
52
+
53
+ def generate_where_clause(self, query_text: str, doc_id: str) -> Dict[str, Any]:
54
+ """
55
+ Analiza el texto iterando sobre las reglas dinámicas y genera el diccionario
56
+ de metadatos bajo las restricciones de la API de ChromaDB.
57
+ """
58
+ base_filter = {"id_documento": doc_id}
59
+
60
+ if not query_text or not isinstance(query_text, str) or not query_text.strip():
61
+ return base_filter
62
+
63
+ query_lower = query_text.lower()
64
+
65
+ for rule in self.rules:
66
+ if any(w in query_lower for w in rule.keywords):
67
+ return {
68
+ "$and": [
69
+ base_filter,
70
+ rule.metadata_filter
71
+ ]
72
+ }
73
+
74
+ return base_filter
@@ -0,0 +1,64 @@
1
+ # dantesito/chasky/query_engine/ollama_streamer.py
2
+
3
+ import json
4
+ from typing import Any, Dict, Iterator, Optional
5
+
6
+ import requests
7
+
8
+ from chaskiwasi.config.settings import Settings
9
+
10
+
11
+ class OllamaStreamer:
12
+ """Cliente HTTP para consumir respuestas de Ollama mediante streaming."""
13
+
14
+ _DEFAULT_ENDPOINT: str = "http://localhost:11434/api/generate"
15
+ _CONNECTION_ERROR_MESSAGE: str = "[ERROR] No se pudo conectar con Ollama."
16
+
17
+ def __init__(self, endpoint: str = _DEFAULT_ENDPOINT) -> None:
18
+ self._endpoint: str = endpoint
19
+
20
+ def stream_response(
21
+ self,
22
+ prompt: str,
23
+ system_prompt: Optional[str] = None,
24
+ ) -> Iterator[str]:
25
+ """Genera incrementalmente la respuesta del LLM mediante streaming."""
26
+ payload: Dict[str, Any] = {
27
+ "model": Settings.LLM_MODEL_NAME,
28
+ "prompt": prompt,
29
+ "stream": True,
30
+ "keep_alive": Settings.OLLAMA_KEEP_ALIVE,
31
+ }
32
+
33
+ if system_prompt is not None:
34
+ payload["system"] = system_prompt
35
+
36
+ try:
37
+ response: requests.Response = requests.post(
38
+ self._endpoint,
39
+ json=payload,
40
+ stream=True,
41
+ )
42
+ response.raise_for_status()
43
+
44
+ for line in response.iter_lines():
45
+ if not line:
46
+ continue
47
+
48
+ try:
49
+ decoded_line: str = (
50
+ line.decode("utf-8")
51
+ if isinstance(line, bytes)
52
+ else str(line)
53
+ )
54
+ data: Dict[str, Any] = json.loads(decoded_line)
55
+ except (UnicodeDecodeError, json.JSONDecodeError):
56
+ continue
57
+
58
+ fragment: Any = data.get("response")
59
+
60
+ if isinstance(fragment, str):
61
+ yield fragment
62
+
63
+ except requests.exceptions.ConnectionError:
64
+ yield self._CONNECTION_ERROR_MESSAGE
File without changes
@@ -0,0 +1,67 @@
1
+ # dantesito/chasky/storage/chroma_persistent.py
2
+
3
+ from pathlib import Path
4
+ from typing import List, Optional, Tuple
5
+
6
+ import chromadb
7
+ from chromadb.api.models.Collection import Collection
8
+
9
+ from chaskiwasi.config.settings import Settings
10
+ from chaskiwasi.config.taxonomy_registry import SectionEnum, SourceEnum
11
+
12
+
13
+ class ChromaPersistentManager:
14
+ """Gestiona el almacenamiento persistente de fragmentos en ChromaDB."""
15
+
16
+ def __init__(
17
+ self,
18
+ persistence_path: Optional[str] = None,
19
+ ) -> None:
20
+ path: str = (
21
+ persistence_path
22
+ if persistence_path is not None
23
+ else Settings.CHROMA_PERSISTENT_PATH
24
+ )
25
+
26
+ Path(path).mkdir(parents=True, exist_ok=True)
27
+ self._client = chromadb.PersistentClient(path=path)
28
+
29
+ def get_or_create_collection(self, collection_name: str) -> Collection:
30
+ """Obtiene o crea una colección usando la configuración nativa de ChromaDB."""
31
+ return self._client.get_or_create_collection(
32
+ name=collection_name,
33
+ )
34
+
35
+ def add_chunks(
36
+ self,
37
+ collection_name: str,
38
+ doc_id: str,
39
+ source: SourceEnum,
40
+ chunks: List[Tuple[SectionEnum, str]],
41
+ ) -> None:
42
+ """Agrega fragmentos y su taxonomía estricta a una colección."""
43
+ if not chunks:
44
+ return
45
+
46
+ collection: Collection = self.get_or_create_collection(collection_name)
47
+
48
+ documents: List[str] = []
49
+ metadatas: List[dict[str, str]] = []
50
+ ids: List[str] = []
51
+
52
+ for idx, (section, text) in enumerate(chunks):
53
+ documents.append(text)
54
+ metadatas.append(
55
+ {
56
+ "id_documento": doc_id,
57
+ "fuente": source.value,
58
+ "tipo_seccion": section.value,
59
+ }
60
+ )
61
+ ids.append(f"{doc_id}_{idx}")
62
+
63
+ collection.add(
64
+ documents=documents,
65
+ metadatas=metadatas,
66
+ ids=ids,
67
+ )
@@ -0,0 +1,31 @@
1
+ Metadata-Version: 2.5
2
+ Name: chaskiwasi
3
+ Version: 0.2.0
4
+ Summary: Sistema de consulta semántica para fuentes jurídicas.
5
+ Project-URL: Homepage, https://github.com/Maorda/chaskywasi
6
+ Project-URL: Issues, https://github.com/Maorda/chaskywasi/issues
7
+ Author-email: Luis Maurtua <luis.maurtua@ejemplo.com>
8
+ License: MIT
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.10
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Requires-Python: >=3.10
16
+ Requires-Dist: chromadb>=0.5.0
17
+ Requires-Dist: docling>=2.0.0
18
+ Requires-Dist: google-genai>=0.1.0
19
+ Requires-Dist: httpx>=0.27.0
20
+ Requires-Dist: ijson>=3.3.0
21
+ Requires-Dist: requests>=2.32.0
22
+ Requires-Dist: streamlit>=1.30.0
23
+ Requires-Dist: tiktoken>=0.7.0
24
+ Provides-Extra: dev
25
+ Requires-Dist: anyio>=4.0.0; extra == 'dev'
26
+ Requires-Dist: faker>=20.0.0; extra == 'dev'
27
+ Requires-Dist: pytest-asyncio>=0.23.0; extra == 'dev'
28
+ Requires-Dist: pytest>=8.0.0; extra == 'dev'
29
+ Description-Content-Type: text/markdown
30
+
31
+ GOLA
@@ -0,0 +1,25 @@
1
+ chaskiwasi/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ chaskiwasi/chunking/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
3
+ chaskiwasi/chunking/chunk_tokenizer.py,sha256=xEJm1K5hqhiUIbHFI1gzYDUP0TwhJ1fYTe4vCBe_Qmg,1187
4
+ chaskiwasi/classification/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
5
+ chaskiwasi/classification/cascade_factory.py,sha256=-H3aCHHlRAaUwSBN1Li2CvyYO4NFwhsxo7USLPp7BOM,2440
6
+ chaskiwasi/classification/strategies/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ chaskiwasi/classification/strategies/base_strategy.py,sha256=xvFmcSVrVUuBoKbT5lBLAJrICSjXqopf2Obt3yI9qfE,1719
8
+ chaskiwasi/classification/strategies/context_overlap_strategy.py,sha256=Ac7NKilkW3xRYN0RzclSYkFeUmJGy539TFREegv6Kx0,743
9
+ chaskiwasi/classification/strategies/cpu_regex_strategy.py,sha256=eg8VeFTCoYCfjUIPVvXs1DLPv5P1Pwj8uv4mIhpywHQ,2295
10
+ chaskiwasi/classification/strategies/gemini_streamer.py,sha256=j5qLhJRmi4xNRsYtCth5mzn0t81ANemO8JGThUb252I,3626
11
+ chaskiwasi/classification/strategies/llm_router_strategy.py,sha256=VLJH6UnkpBNlO840fJH5SIsB2qu-bFyRMbhy9MK8eSE,2375
12
+ chaskiwasi/config/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
+ chaskiwasi/config/settings.py,sha256=0YOUJoiAkWuWVMoAmomKAOiYXfoxa50QRSKM5DWbIj0,516
14
+ chaskiwasi/config/taxonomy_registry.py,sha256=mAvjH1jqAf-yO_VTSb3HJwj5N582R1DGYhV5DjMlSxw,2271
15
+ chaskiwasi/consolidation/consolidator.py,sha256=QLPSnn5dxDJmFwQGAd7kG02QjGRM_3wtZwqPZXcf_BA,6129
16
+ chaskiwasi/ingestion/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
17
+ chaskiwasi/ingestion/docling_parser.py,sha256=UNEwot3LOYJrZcbhnnEI94UDm6DvM13w_jKjFUTII4U,2563
18
+ chaskiwasi/query_engine/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
19
+ chaskiwasi/query_engine/cross_filter.py,sha256=PvpAMnuyM8aiw-F_PUI8Mac775Yircgo6XOKY63xMDk,2551
20
+ chaskiwasi/query_engine/ollama_streamer.py,sha256=V2UsffssQ5Zy4EEsSxPQQQ4lrcN-ktcVe4xa-FXRqjw,1977
21
+ chaskiwasi/storage/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
22
+ chaskiwasi/storage/chroma_persistent.py,sha256=2cJlLZc0kmktVG65DYpYv1ZUCU3hsqQ3rXU5bTxWgt0,2000
23
+ chaskiwasi-0.2.0.dist-info/METADATA,sha256=-nGWk4TDNGbkhYcl-ajKACwUwruWvgKIuMrUiC9sgp0,1136
24
+ chaskiwasi-0.2.0.dist-info/WHEEL,sha256=THafob7ofN-NsuMN7Mg4qZyHaQI7KkD-QlcQatYhXPo,87
25
+ chaskiwasi-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.3
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any