betl 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. betl-0.1.0/.github/workflows/publish.yml +34 -0
  2. betl-0.1.0/.gitignore +151 -0
  3. betl-0.1.0/PKG-INFO +21 -0
  4. betl-0.1.0/README.md +1 -0
  5. betl-0.1.0/pyproject.toml +31 -0
  6. betl-0.1.0/src/core/__init__.py +0 -0
  7. betl-0.1.0/src/core/decorators/contract.py +54 -0
  8. betl-0.1.0/src/core/decorators/strategy.py +40 -0
  9. betl-0.1.0/src/core/extractors/__init__.py +0 -0
  10. betl-0.1.0/src/core/extractors/extraction_contract.py +126 -0
  11. betl-0.1.0/src/core/extractors/ocr/__init__.py +0 -0
  12. betl-0.1.0/src/core/extractors/ocr/captcha.py +120 -0
  13. betl-0.1.0/src/core/extractors/ocr/orquestator_ocr.py +133 -0
  14. betl-0.1.0/src/core/extractors/ocr/pdf_hybrid_extractor.py +162 -0
  15. betl-0.1.0/src/core/extractors/ocr/pdfnative.py +155 -0
  16. betl-0.1.0/src/core/extractors/ocr/pdfscan.py +455 -0
  17. betl-0.1.0/src/core/helpers/__init__.py +0 -0
  18. betl-0.1.0/src/core/helpers/get_or_create_meta.py +15 -0
  19. betl-0.1.0/src/core/manipulate/__init__.py +0 -0
  20. betl-0.1.0/src/core/manipulate/orquestador_manipulacion.py +174 -0
  21. betl-0.1.0/src/core/manipulate/strategy/__init__.py +0 -0
  22. betl-0.1.0/src/core/manipulate/strategy/llm.py +147 -0
  23. betl-0.1.0/src/core/manipulate/strategy/regex.py +204 -0
  24. betl-0.1.0/src/core/requirements.txt +8 -0
  25. betl-0.1.0/src/core/transformation/__init__.py +0 -0
  26. betl-0.1.0/src/core/transformation/factory/__init__.py +0 -0
  27. betl-0.1.0/src/core/transformation/factory/mapper_factory.py +106 -0
  28. betl-0.1.0/src/core/transformation/mergeContract.py +117 -0
  29. betl-0.1.0/src/core/transformation/merger.py +126 -0
  30. betl-0.1.0/src/core/utils/__init__.py +0 -0
  31. betl-0.1.0/src/core/utils/formatters.py +29 -0
@@ -0,0 +1,34 @@
1
+ name: Publish Python Package to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published] # Se activará solo cuando publiques una Release en GitHub
6
+
7
+ jobs:
8
+ deploy:
9
+ runs-on: ubuntu-latest
10
+
11
+ steps:
12
+ - name: Checkout code
13
+ uses: actions/checkout@v4
14
+
15
+ - name: Set up Python
16
+ uses: actions/setup-python@v5
17
+ with:
18
+ python-version: '3.10'
19
+
20
+ - name: Install build tools
21
+ run: |
22
+ python -m pip install --upgrade pip
23
+ pip install build twine
24
+
25
+ - name: Build package
26
+ run: |
27
+ python -m build
28
+
29
+ - name: Publish to PyPI
30
+ env:
31
+ TWINE_USERNAME: __token__
32
+ TWINE_PASSWORD: ${{ secrets.PYPI_API_TOKEN }} # Llama al secreto seguro que guardamos
33
+ run: |
34
+ python -m twine upload dist/*
betl-0.1.0/.gitignore ADDED
@@ -0,0 +1,151 @@
1
+ # Byte-compiled / optimized / DLL files
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+
6
+ # C extensions
7
+ *.so
8
+
9
+ # Distribution / packaging
10
+ .Python
11
+ build/
12
+ develop-eggs/
13
+ dist/
14
+ downloads/
15
+ eggs/
16
+ .eggs/
17
+ lib/
18
+ lib64/
19
+ parts/
20
+ sdist/
21
+ var/
22
+ wheels/
23
+ share/python-wheels/
24
+ *.egg-info/
25
+ .installed.cfg
26
+ *.egg
27
+ MANIFEST
28
+
29
+ # PyInstaller
30
+ # Usually these files are written by a python script from a template
31
+ # before launching pyinstaller
32
+ *.manifest
33
+ *.spec
34
+
35
+ # Installer logs
36
+ pip-log.txt
37
+ pip-delete-this-directory.txt
38
+
39
+ # Unit test / coverage reports
40
+ htmlcov/
41
+ .toml
42
+ .cache
43
+ .pytest_cache/
44
+ .nofap
45
+ .coverage
46
+ .coverage.*
47
+ .hypothesis/
48
+ .pytest_cache/
49
+ cover/
50
+
51
+ # Translations
52
+ *.mo
53
+ *.pot
54
+
55
+ # Django stuff:
56
+ *.log
57
+ local_settings.py
58
+ db.sqlite3
59
+ db.sqlite3-journal
60
+
61
+ # Flask stuff:
62
+ instance/
63
+ .webassets-cache
64
+
65
+ # Scrapy stuff:
66
+ .scrapy
67
+
68
+ # Sphinx documentation
69
+ docs/_build/
70
+
71
+ # PyBuilder
72
+ .pybuilder/
73
+ target/
74
+
75
+ # Jupyter Notebook
76
+ .ipynb_checkpoints
77
+
78
+ # IPython
79
+ profile_default/
80
+ ipython_config.py
81
+
82
+ # pyenv
83
+ # For a library or package, you might want to ignore these files since the setups is
84
+ # specific to the developer, not the project.
85
+ # .python-version
86
+
87
+ # pipenv
88
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
89
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
90
+ # with no cross-platform support, pipenv may install dependencies that don't work, or fail.
91
+ #Pipfile.lock
92
+
93
+ # poetry
94
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
95
+ #Poetry.lock
96
+
97
+ # pdm
98
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
99
+ #pdm.lock
100
+ # pdm build standard temporary directory
101
+ .pdm-build/
102
+
103
+ # PEP 582 project packages
104
+ __pypackages__/
105
+
106
+ # Celery stuff
107
+ celerybeat-schedule
108
+ celerybeat.pid
109
+
110
+ # SageMath parsed files
111
+ *.sage.py
112
+
113
+ # Environments
114
+ .env
115
+ .venv
116
+ env/
117
+ venv/
118
+ ENV/
119
+ env.bak/
120
+ venv.bak/
121
+
122
+ # Spyder project settings
123
+ .spyderproject
124
+ .spyproject
125
+
126
+ # Rope project settings
127
+ .ropeproject
128
+
129
+ # mkdocs documentation
130
+ /site
131
+
132
+ # mypy
133
+ .mypy_cache/
134
+ .dmypy.json
135
+ dmypy.json
136
+
137
+ # Pyre type checker
138
+ .pyre/
139
+
140
+ # pytype static type analyzer
141
+ .pytype/
142
+
143
+ # Cython debug symbols
144
+ cython_debug/
145
+
146
+ # IDEs and editors
147
+ .idea/
148
+ .vscode/
149
+ *.swp
150
+ *.swo
151
+ .DS_Store
betl-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,21 @@
1
+ Metadata-Version: 2.5
2
+ Name: betl
3
+ Version: 0.1.0
4
+ Summary: Una descripción corta de tu librería
5
+ Project-URL: Homepage, https://github.com
6
+ License: MIT
7
+ Classifier: License :: OSI Approved :: MIT License
8
+ Classifier: Operating System :: OS Independent
9
+ Classifier: Programming Language :: Python :: 3
10
+ Requires-Python: >=3.8
11
+ Requires-Dist: ddddocr==1.6.1
12
+ Requires-Dist: numpy==2.5.2
13
+ Requires-Dist: opencv-contrib-python==4.10.0.84
14
+ Requires-Dist: opencv-python==5.0.0.93
15
+ Requires-Dist: paddleocr==3.7.0
16
+ Requires-Dist: pydantic==2.13.5
17
+ Requires-Dist: pymupdf==1.28.2
18
+ Requires-Dist: requests==2.34.2
19
+ Description-Content-Type: text/markdown
20
+
21
+ holamundo
betl-0.1.0/README.md ADDED
@@ -0,0 +1 @@
1
+ holamundo
@@ -0,0 +1,31 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "betl"
7
+ version = "0.1.0"
8
+ description = "Una descripción corta de tu librería"
9
+ readme = "README.md"
10
+ requires-python = ">=3.8"
11
+ license = { text = "MIT" }
12
+ classifiers = [
13
+ "Programming Language :: Python :: 3",
14
+ "License :: OSI Approved :: MIT License",
15
+ "Operating System :: OS Independent",
16
+ ]
17
+ dependencies = [
18
+ "ddddocr==1.6.1",
19
+ "numpy==2.5.2",
20
+ "opencv_contrib_python==4.10.0.84",
21
+ "opencv_python==5.0.0.93",
22
+ "paddleocr==3.7.0",
23
+ "pydantic==2.13.5",
24
+ "pymupdf==1.28.2",
25
+ "Requests==2.34.2",
26
+ ]
27
+ [tool.hatch.build.targets.wheel]
28
+ packages = ["src/core"]
29
+
30
+ [project.urls]
31
+ Homepage = "https://github.com"
File without changes
@@ -0,0 +1,54 @@
1
+ import inspect
2
+ from typing import Optional, Dict, Any, Type
3
+ from core.extractors.extraction_contract import ExtractionContract, ExtractionField
4
+
5
+ # ==========================================
6
+ # 2. Decorador de Clase (Ensamblador)
7
+ # ==========================================
8
+ def contract(
9
+ mapper_type: Optional[str] = None,
10
+ version: str = "1.0",
11
+ metadata: Optional[Dict[str, Any]] = None,
12
+ ):
13
+ """Decorador de clase que inyecta la configuración del ExtractionContract en la clase misma."""
14
+ def decorator(cls: Type) -> Type:
15
+ # Inyectamos los metadatos directamente en la clase para que
16
+ # el método adaptar_a_dto pueda leer self._metadata
17
+ cls._metadata = metadata or {}
18
+ cls._metadata["mapper_type"] = mapper_type
19
+ cls._metadata["version"] = version
20
+
21
+ contract_inst = ExtractionContract(
22
+ mapper_type=mapper_type,
23
+ version=version,
24
+ metadata=cls._metadata
25
+ )
26
+ cls_fields = {}
27
+
28
+ for attr_name, attr in inspect.getmembers(cls):
29
+ if hasattr(attr, "__extraction_field__"):
30
+ f_data = attr.__extraction_field__
31
+ cls_fields[f_data["name"]] = f_data
32
+ contract_inst.add_field(
33
+ ExtractionField(
34
+ name=f_data["name"],
35
+ data_type=f_data["data_type"],
36
+ description=f_data["description"],
37
+ required=f_data["required"],
38
+ regex=f_data["regex"],
39
+ llm=f_data["llm"],
40
+ )
41
+ )
42
+
43
+ cls._metadata["__fields_raw__"] = cls_fields
44
+ cls._extraction_contract = contract_inst # Guardamos la instancia del contrato internamente
45
+
46
+ errores = contract_inst.validar()
47
+ if errores:
48
+ raise ValueError(f"Contrato declarativo '{cls.__name__}' inválido: {'; '.join(errores)}")
49
+
50
+ # Retornamos la CLASE original, ahora enriquecida.
51
+ # Así, cuando MergerService haga `instancia = contrato()`, será una instancia de MiContratoRemate.
52
+ return cls
53
+
54
+ return decorator
@@ -0,0 +1,40 @@
1
+ import re
2
+ from typing import Any, Callable, Dict, Optional
3
+ from core.helpers.get_or_create_meta import _get_or_create_meta
4
+ from core.extractors.extraction_contract import RegexObjective, LLMTarget
5
+
6
+
7
+ # ==========================================
8
+ # 3. Decoradores (Construcción del Contrato)
9
+ # ==========================================
10
+ def regex_strategy(pattern: str, flags: int = re.IGNORECASE, enabled: bool = True):
11
+ """Decorador para inyectar una regla de extracción basada en Regex."""
12
+ def decorator(func: Callable) -> Callable:
13
+ meta = _get_or_create_meta(func)
14
+ meta["regex"] = RegexObjective(enabled=enabled, pattern=pattern, flags=flags)
15
+ return func
16
+ return decorator
17
+
18
+ def llm_strategy(
19
+ instruction: Optional[str] = None,
20
+ structure: Optional[Dict[str, Any]] = None,
21
+ enabled: bool = True,
22
+ ):
23
+ """Decorador para inyectar una instrucción o estructura objetivo para un LLM."""
24
+ def decorator(func: Callable) -> Callable:
25
+ meta = _get_or_create_meta(func)
26
+ meta["llm"] = LLMTarget(enabled=enabled, instruction=instruction, structure=structure)
27
+ return func
28
+ return decorator
29
+
30
+ def campo(data_type: str = "string", description: Optional[str] = None, required: bool = False):
31
+ """Decorador para registrar propiedades base descriptivas del campo."""
32
+ def decorator(func: Callable) -> Callable:
33
+ meta = _get_or_create_meta(func)
34
+ meta.update({
35
+ "data_type": data_type,
36
+ "description": description,
37
+ "required": required,
38
+ })
39
+ return func
40
+ return decorator
File without changes
@@ -0,0 +1,126 @@
1
+ import re
2
+ from typing import Any, Dict, List, Optional
3
+ from pydantic import BaseModel, ConfigDict, Field
4
+
5
+
6
+ class RegexObjective(BaseModel):
7
+ """Objetivo runtime de extracción mediante Regex. Solo describe QUÉ ejecutar."""
8
+ model_config = ConfigDict(extra="forbid")
9
+
10
+ enabled: bool = False
11
+ pattern: Optional[str] = None
12
+ flags: int = 0
13
+
14
+
15
+ class LLMTarget(BaseModel):
16
+ """Objetivo runtime de extracción mediante LLM. Solo describe instrucción y estructura."""
17
+ model_config = ConfigDict(extra="forbid")
18
+
19
+ enabled: bool = False
20
+ instruction: Optional[str] = None
21
+ structure: Optional[Dict[str, Any]] = None
22
+
23
+
24
+ class ExtractionField(BaseModel):
25
+ """Representa un campo objetivo de información a extraer."""
26
+ model_config = ConfigDict(extra="forbid")
27
+
28
+ name: str
29
+ data_type: str = "string"
30
+ description: Optional[str] = None
31
+ required: bool = False
32
+ regex: Optional[RegexObjective] = None
33
+ llm: Optional[LLMTarget] = None
34
+
35
+
36
+ class ExtractionContract(BaseModel):
37
+ """Contrato runtime que consolida los campos y estrategias de extracción."""
38
+ model_config = ConfigDict(extra="forbid")
39
+
40
+ mapper_type: Optional[str] = None
41
+ version: str = "1.0"
42
+ fields: Dict[str, ExtractionField] = Field(default_factory=dict)
43
+ metadata: Dict[str, Any] = Field(default_factory=dict)
44
+
45
+ def add_field(self, field: ExtractionField) -> None:
46
+ """Agrega un campo al contrato normalizando su nombre."""
47
+ if not isinstance(field, ExtractionField):
48
+ raise TypeError("field debe ser una instancia de ExtractionField.")
49
+
50
+ nombre = field.name.strip()
51
+ if not nombre:
52
+ raise ValueError("El nombre del campo no puede estar vacío.")
53
+
54
+ if nombre in self.fields:
55
+ raise ValueError(f"El campo '{nombre}' ya existe en el ExtractionContract.")
56
+
57
+ if nombre != field.name:
58
+ field = field.model_copy(update={"name": nombre})
59
+
60
+ self.fields[nombre] = field
61
+
62
+ def get_field(self, nombre: str) -> Optional[ExtractionField]:
63
+ return self.fields.get(nombre)
64
+
65
+ def has_field(self, nombre: str) -> bool:
66
+ return nombre in self.fields
67
+
68
+ def count_fields(self) -> int:
69
+ return len(self.fields)
70
+
71
+ def get_regex_fields(self) -> Dict[str, ExtractionField]:
72
+ """Campos con estrategia Regex habilitada."""
73
+ return {
74
+ nombre: field
75
+ for nombre, field in self.fields.items()
76
+ if field.regex and field.regex.enabled
77
+ }
78
+
79
+ def get_llm_fields(self) -> Dict[str, ExtractionField]:
80
+ """Campos con estrategia LLM habilitada."""
81
+ return {
82
+ nombre: field
83
+ for nombre, field in self.fields.items()
84
+ if field.llm and field.llm.enabled
85
+ }
86
+
87
+ def validar(self) -> List[str]:
88
+ """Valida la consistencia interna y la validez sintáctica de las reglas."""
89
+ errores: List[str] = []
90
+
91
+ if not self.fields:
92
+ return ["El ExtractionContract no contiene campos."]
93
+
94
+ for nombre, field in self.fields.items():
95
+ nombre_limpio = nombre.strip()
96
+
97
+ if not nombre_limpio:
98
+ errores.append("Existe un campo con nombre vacío en las llaves del contrato.")
99
+
100
+ if not field.name or not field.name.strip():
101
+ errores.append(f"El campo '{nombre}' tiene un atributo 'name' vacío.")
102
+ elif field.name.strip() != nombre_limpio:
103
+ errores.append(f"El nombre interno '{field.name}' no coincide con su clave '{nombre}'.")
104
+
105
+ if not field.data_type or not field.data_type.strip():
106
+ errores.append(f"El campo '{nombre}' no tiene especificado 'data_type'.")
107
+
108
+ # Validaciones para Regex
109
+ if field.regex and field.regex.enabled:
110
+ if not field.regex.pattern or not field.regex.pattern.strip():
111
+ errores.append(f"El campo '{nombre}' tiene Regex habilitado pero 'pattern' está vacío.")
112
+ else:
113
+ # Intento de compilación real para asegurar sintaxis
114
+ try:
115
+ re.compile(field.regex.pattern, field.regex.flags)
116
+ except (re.error, ValueError, TypeError) as exc:
117
+ errores.append(f"El campo '{nombre}' posee una Regex inválida ({field.regex.pattern}): {exc}")
118
+
119
+ # Validaciones para LLM
120
+ if field.llm and field.llm.enabled:
121
+ if not field.llm.instruction and not field.llm.structure:
122
+ errores.append(
123
+ f"El campo '{nombre}' tiene LLM habilitado pero carece de 'instruction' o 'structure'."
124
+ )
125
+
126
+ return errores
File without changes
@@ -0,0 +1,120 @@
1
+ import logging
2
+ from pathlib import Path
3
+ from typing import Union
4
+
5
+ logger = logging.getLogger(__name__)
6
+
7
+
8
+ class CaptchaExtractor:
9
+ """
10
+ Extractor especializado para CAPTCHAs.
11
+
12
+ Utiliza ddddocr cuando está disponible.
13
+
14
+ RESPONSABILIDAD
15
+ ---------------
16
+ Recibir una imagen CAPTCHA (bytes o ruta) y devolver el texto reconocido.
17
+ """
18
+
19
+ def __init__(self):
20
+ """
21
+ Inicializa ddddocr de manera opcional.
22
+ """
23
+ self.engine = None
24
+ self._inicializar()
25
+
26
+ # =====================================================================
27
+ # INICIALIZACIÓN
28
+ # =====================================================================
29
+
30
+ def _inicializar(self) -> None:
31
+ """
32
+ Inicializa el motor ddddocr.
33
+ """
34
+ try:
35
+ import ddddocr
36
+
37
+ self.engine = ddddocr.DdddOcr(show_ad=False)
38
+ logger.info("[CAPTCHA] Motor ddddocr inicializado correctamente.")
39
+
40
+ except Exception as exc:
41
+ logger.warning(
42
+ f"[CAPTCHA] ddddocr no está disponible: {exc}"
43
+ )
44
+ self.engine = None
45
+
46
+ # =====================================================================
47
+ # DISPONIBILIDAD
48
+ # =====================================================================
49
+
50
+ @property
51
+ def disponible(self) -> bool:
52
+ """
53
+ Indica si el motor CAPTCHA está disponible.
54
+ """
55
+ return self.engine is not None
56
+
57
+ # =====================================================================
58
+ # PUNTOS DE ENTRADA / EXTRAER
59
+ # =====================================================================
60
+
61
+ def resolver(self, image_input: Union[str, Path, bytes]) -> str:
62
+ """
63
+ Método de entrada flexible esperado por el orquestador.
64
+ Acepta rutas de archivo (str/Path) o bytes directamente.
65
+ """
66
+ if not image_input:
67
+ logger.warning("[CAPTCHA] Entrada vacía enviada a resolver().")
68
+ return ""
69
+
70
+ # Si recibe una ruta de archivo (str o Path)
71
+ if isinstance(image_input, (str, Path)):
72
+ path = Path(image_input)
73
+ if not path.is_file():
74
+ logger.error(f"[CAPTCHA] Archivo no encontrado: {path}")
75
+ return ""
76
+ try:
77
+ with open(path, "rb") as f:
78
+ image_bytes = f.read()
79
+ return self.extraer(image_bytes)
80
+ except Exception as exc:
81
+ logger.error(f"[CAPTCHA] Error leyendo archivo de captcha: {exc}")
82
+ return ""
83
+
84
+ # Si ya son bytes directamente
85
+ elif isinstance(image_input, bytes):
86
+ return self.extraer(image_input)
87
+
88
+ else:
89
+ logger.error(f"[CAPTCHA] Tipo de entrada no soportado: {type(image_input)}")
90
+ return ""
91
+
92
+ def extraer(self, image_bytes: bytes) -> str:
93
+ """
94
+ Reconoce el texto de un CAPTCHA a partir de sus bytes.
95
+ """
96
+ if not image_bytes:
97
+ logger.warning("[CAPTCHA] Se recibió una imagen vacía.")
98
+ return ""
99
+
100
+ if not self.engine:
101
+ logger.warning("[CAPTCHA] El motor ddddocr no está disponible.")
102
+ return ""
103
+
104
+ try:
105
+ resultado = self.engine.classification(image_bytes)
106
+
107
+ if resultado is None:
108
+ return ""
109
+
110
+ texto = str(resultado).strip().upper()
111
+
112
+ logger.debug(
113
+ f"[CAPTCHA] CAPTCHA procesado exitosamente. Longitud={len(texto)}"
114
+ )
115
+
116
+ return texto
117
+
118
+ except Exception as exc:
119
+ logger.error(f"[CAPTCHA] Error procesando CAPTCHA: {exc}")
120
+ return ""