betl 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- betl-0.1.0/.github/workflows/publish.yml +34 -0
- betl-0.1.0/.gitignore +151 -0
- betl-0.1.0/PKG-INFO +21 -0
- betl-0.1.0/README.md +1 -0
- betl-0.1.0/pyproject.toml +31 -0
- betl-0.1.0/src/core/__init__.py +0 -0
- betl-0.1.0/src/core/decorators/contract.py +54 -0
- betl-0.1.0/src/core/decorators/strategy.py +40 -0
- betl-0.1.0/src/core/extractors/__init__.py +0 -0
- betl-0.1.0/src/core/extractors/extraction_contract.py +126 -0
- betl-0.1.0/src/core/extractors/ocr/__init__.py +0 -0
- betl-0.1.0/src/core/extractors/ocr/captcha.py +120 -0
- betl-0.1.0/src/core/extractors/ocr/orquestator_ocr.py +133 -0
- betl-0.1.0/src/core/extractors/ocr/pdf_hybrid_extractor.py +162 -0
- betl-0.1.0/src/core/extractors/ocr/pdfnative.py +155 -0
- betl-0.1.0/src/core/extractors/ocr/pdfscan.py +455 -0
- betl-0.1.0/src/core/helpers/__init__.py +0 -0
- betl-0.1.0/src/core/helpers/get_or_create_meta.py +15 -0
- betl-0.1.0/src/core/manipulate/__init__.py +0 -0
- betl-0.1.0/src/core/manipulate/orquestador_manipulacion.py +174 -0
- betl-0.1.0/src/core/manipulate/strategy/__init__.py +0 -0
- betl-0.1.0/src/core/manipulate/strategy/llm.py +147 -0
- betl-0.1.0/src/core/manipulate/strategy/regex.py +204 -0
- betl-0.1.0/src/core/requirements.txt +8 -0
- betl-0.1.0/src/core/transformation/__init__.py +0 -0
- betl-0.1.0/src/core/transformation/factory/__init__.py +0 -0
- betl-0.1.0/src/core/transformation/factory/mapper_factory.py +106 -0
- betl-0.1.0/src/core/transformation/mergeContract.py +117 -0
- betl-0.1.0/src/core/transformation/merger.py +126 -0
- betl-0.1.0/src/core/utils/__init__.py +0 -0
- betl-0.1.0/src/core/utils/formatters.py +29 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
name: Publish Python Package to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published] # Se activará solo cuando publiques una Release en GitHub
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
deploy:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
|
|
11
|
+
steps:
|
|
12
|
+
- name: Checkout code
|
|
13
|
+
uses: actions/checkout@v4
|
|
14
|
+
|
|
15
|
+
- name: Set up Python
|
|
16
|
+
uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: '3.10'
|
|
19
|
+
|
|
20
|
+
- name: Install build tools
|
|
21
|
+
run: |
|
|
22
|
+
python -m pip install --upgrade pip
|
|
23
|
+
pip install build twine
|
|
24
|
+
|
|
25
|
+
- name: Build package
|
|
26
|
+
run: |
|
|
27
|
+
python -m build
|
|
28
|
+
|
|
29
|
+
- name: Publish to PyPI
|
|
30
|
+
env:
|
|
31
|
+
TWINE_USERNAME: __token__
|
|
32
|
+
TWINE_PASSWORD: ${{ secrets.PYPI_API_TOKEN }} # Llama al secreto seguro que guardamos
|
|
33
|
+
run: |
|
|
34
|
+
python -m twine upload dist/*
|
betl-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
# Byte-compiled / optimized / DLL files
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
|
|
6
|
+
# C extensions
|
|
7
|
+
*.so
|
|
8
|
+
|
|
9
|
+
# Distribution / packaging
|
|
10
|
+
.Python
|
|
11
|
+
build/
|
|
12
|
+
develop-eggs/
|
|
13
|
+
dist/
|
|
14
|
+
downloads/
|
|
15
|
+
eggs/
|
|
16
|
+
.eggs/
|
|
17
|
+
lib/
|
|
18
|
+
lib64/
|
|
19
|
+
parts/
|
|
20
|
+
sdist/
|
|
21
|
+
var/
|
|
22
|
+
wheels/
|
|
23
|
+
share/python-wheels/
|
|
24
|
+
*.egg-info/
|
|
25
|
+
.installed.cfg
|
|
26
|
+
*.egg
|
|
27
|
+
MANIFEST
|
|
28
|
+
|
|
29
|
+
# PyInstaller
|
|
30
|
+
# Usually these files are written by a python script from a template
|
|
31
|
+
# before launching pyinstaller
|
|
32
|
+
*.manifest
|
|
33
|
+
*.spec
|
|
34
|
+
|
|
35
|
+
# Installer logs
|
|
36
|
+
pip-log.txt
|
|
37
|
+
pip-delete-this-directory.txt
|
|
38
|
+
|
|
39
|
+
# Unit test / coverage reports
|
|
40
|
+
htmlcov/
|
|
41
|
+
.toml
|
|
42
|
+
.cache
|
|
43
|
+
.pytest_cache/
|
|
44
|
+
.nofap
|
|
45
|
+
.coverage
|
|
46
|
+
.coverage.*
|
|
47
|
+
.hypothesis/
|
|
48
|
+
.pytest_cache/
|
|
49
|
+
cover/
|
|
50
|
+
|
|
51
|
+
# Translations
|
|
52
|
+
*.mo
|
|
53
|
+
*.pot
|
|
54
|
+
|
|
55
|
+
# Django stuff:
|
|
56
|
+
*.log
|
|
57
|
+
local_settings.py
|
|
58
|
+
db.sqlite3
|
|
59
|
+
db.sqlite3-journal
|
|
60
|
+
|
|
61
|
+
# Flask stuff:
|
|
62
|
+
instance/
|
|
63
|
+
.webassets-cache
|
|
64
|
+
|
|
65
|
+
# Scrapy stuff:
|
|
66
|
+
.scrapy
|
|
67
|
+
|
|
68
|
+
# Sphinx documentation
|
|
69
|
+
docs/_build/
|
|
70
|
+
|
|
71
|
+
# PyBuilder
|
|
72
|
+
.pybuilder/
|
|
73
|
+
target/
|
|
74
|
+
|
|
75
|
+
# Jupyter Notebook
|
|
76
|
+
.ipynb_checkpoints
|
|
77
|
+
|
|
78
|
+
# IPython
|
|
79
|
+
profile_default/
|
|
80
|
+
ipython_config.py
|
|
81
|
+
|
|
82
|
+
# pyenv
|
|
83
|
+
# For a library or package, you might want to ignore these files since the setups is
|
|
84
|
+
# specific to the developer, not the project.
|
|
85
|
+
# .python-version
|
|
86
|
+
|
|
87
|
+
# pipenv
|
|
88
|
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
|
89
|
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
|
90
|
+
# with no cross-platform support, pipenv may install dependencies that don't work, or fail.
|
|
91
|
+
#Pipfile.lock
|
|
92
|
+
|
|
93
|
+
# poetry
|
|
94
|
+
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
|
95
|
+
#Poetry.lock
|
|
96
|
+
|
|
97
|
+
# pdm
|
|
98
|
+
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
|
99
|
+
#pdm.lock
|
|
100
|
+
# pdm build standard temporary directory
|
|
101
|
+
.pdm-build/
|
|
102
|
+
|
|
103
|
+
# PEP 582 project packages
|
|
104
|
+
__pypackages__/
|
|
105
|
+
|
|
106
|
+
# Celery stuff
|
|
107
|
+
celerybeat-schedule
|
|
108
|
+
celerybeat.pid
|
|
109
|
+
|
|
110
|
+
# SageMath parsed files
|
|
111
|
+
*.sage.py
|
|
112
|
+
|
|
113
|
+
# Environments
|
|
114
|
+
.env
|
|
115
|
+
.venv
|
|
116
|
+
env/
|
|
117
|
+
venv/
|
|
118
|
+
ENV/
|
|
119
|
+
env.bak/
|
|
120
|
+
venv.bak/
|
|
121
|
+
|
|
122
|
+
# Spyder project settings
|
|
123
|
+
.spyderproject
|
|
124
|
+
.spyproject
|
|
125
|
+
|
|
126
|
+
# Rope project settings
|
|
127
|
+
.ropeproject
|
|
128
|
+
|
|
129
|
+
# mkdocs documentation
|
|
130
|
+
/site
|
|
131
|
+
|
|
132
|
+
# mypy
|
|
133
|
+
.mypy_cache/
|
|
134
|
+
.dmypy.json
|
|
135
|
+
dmypy.json
|
|
136
|
+
|
|
137
|
+
# Pyre type checker
|
|
138
|
+
.pyre/
|
|
139
|
+
|
|
140
|
+
# pytype static type analyzer
|
|
141
|
+
.pytype/
|
|
142
|
+
|
|
143
|
+
# Cython debug symbols
|
|
144
|
+
cython_debug/
|
|
145
|
+
|
|
146
|
+
# IDEs and editors
|
|
147
|
+
.idea/
|
|
148
|
+
.vscode/
|
|
149
|
+
*.swp
|
|
150
|
+
*.swo
|
|
151
|
+
.DS_Store
|
betl-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: betl
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Una descripción corta de tu librería
|
|
5
|
+
Project-URL: Homepage, https://github.com
|
|
6
|
+
License: MIT
|
|
7
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Requires-Python: >=3.8
|
|
11
|
+
Requires-Dist: ddddocr==1.6.1
|
|
12
|
+
Requires-Dist: numpy==2.5.2
|
|
13
|
+
Requires-Dist: opencv-contrib-python==4.10.0.84
|
|
14
|
+
Requires-Dist: opencv-python==5.0.0.93
|
|
15
|
+
Requires-Dist: paddleocr==3.7.0
|
|
16
|
+
Requires-Dist: pydantic==2.13.5
|
|
17
|
+
Requires-Dist: pymupdf==1.28.2
|
|
18
|
+
Requires-Dist: requests==2.34.2
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
holamundo
|
betl-0.1.0/README.md
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
holamundo
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "betl"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Una descripción corta de tu librería"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.8"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Programming Language :: Python :: 3",
|
|
14
|
+
"License :: OSI Approved :: MIT License",
|
|
15
|
+
"Operating System :: OS Independent",
|
|
16
|
+
]
|
|
17
|
+
dependencies = [
|
|
18
|
+
"ddddocr==1.6.1",
|
|
19
|
+
"numpy==2.5.2",
|
|
20
|
+
"opencv_contrib_python==4.10.0.84",
|
|
21
|
+
"opencv_python==5.0.0.93",
|
|
22
|
+
"paddleocr==3.7.0",
|
|
23
|
+
"pydantic==2.13.5",
|
|
24
|
+
"pymupdf==1.28.2",
|
|
25
|
+
"Requests==2.34.2",
|
|
26
|
+
]
|
|
27
|
+
[tool.hatch.build.targets.wheel]
|
|
28
|
+
packages = ["src/core"]
|
|
29
|
+
|
|
30
|
+
[project.urls]
|
|
31
|
+
Homepage = "https://github.com"
|
|
File without changes
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import inspect
|
|
2
|
+
from typing import Optional, Dict, Any, Type
|
|
3
|
+
from core.extractors.extraction_contract import ExtractionContract, ExtractionField
|
|
4
|
+
|
|
5
|
+
# ==========================================
|
|
6
|
+
# 2. Decorador de Clase (Ensamblador)
|
|
7
|
+
# ==========================================
|
|
8
|
+
def contract(
|
|
9
|
+
mapper_type: Optional[str] = None,
|
|
10
|
+
version: str = "1.0",
|
|
11
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
12
|
+
):
|
|
13
|
+
"""Decorador de clase que inyecta la configuración del ExtractionContract en la clase misma."""
|
|
14
|
+
def decorator(cls: Type) -> Type:
|
|
15
|
+
# Inyectamos los metadatos directamente en la clase para que
|
|
16
|
+
# el método adaptar_a_dto pueda leer self._metadata
|
|
17
|
+
cls._metadata = metadata or {}
|
|
18
|
+
cls._metadata["mapper_type"] = mapper_type
|
|
19
|
+
cls._metadata["version"] = version
|
|
20
|
+
|
|
21
|
+
contract_inst = ExtractionContract(
|
|
22
|
+
mapper_type=mapper_type,
|
|
23
|
+
version=version,
|
|
24
|
+
metadata=cls._metadata
|
|
25
|
+
)
|
|
26
|
+
cls_fields = {}
|
|
27
|
+
|
|
28
|
+
for attr_name, attr in inspect.getmembers(cls):
|
|
29
|
+
if hasattr(attr, "__extraction_field__"):
|
|
30
|
+
f_data = attr.__extraction_field__
|
|
31
|
+
cls_fields[f_data["name"]] = f_data
|
|
32
|
+
contract_inst.add_field(
|
|
33
|
+
ExtractionField(
|
|
34
|
+
name=f_data["name"],
|
|
35
|
+
data_type=f_data["data_type"],
|
|
36
|
+
description=f_data["description"],
|
|
37
|
+
required=f_data["required"],
|
|
38
|
+
regex=f_data["regex"],
|
|
39
|
+
llm=f_data["llm"],
|
|
40
|
+
)
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
cls._metadata["__fields_raw__"] = cls_fields
|
|
44
|
+
cls._extraction_contract = contract_inst # Guardamos la instancia del contrato internamente
|
|
45
|
+
|
|
46
|
+
errores = contract_inst.validar()
|
|
47
|
+
if errores:
|
|
48
|
+
raise ValueError(f"Contrato declarativo '{cls.__name__}' inválido: {'; '.join(errores)}")
|
|
49
|
+
|
|
50
|
+
# Retornamos la CLASE original, ahora enriquecida.
|
|
51
|
+
# Así, cuando MergerService haga `instancia = contrato()`, será una instancia de MiContratoRemate.
|
|
52
|
+
return cls
|
|
53
|
+
|
|
54
|
+
return decorator
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from typing import Any, Callable, Dict, Optional
|
|
3
|
+
from core.helpers.get_or_create_meta import _get_or_create_meta
|
|
4
|
+
from core.extractors.extraction_contract import RegexObjective, LLMTarget
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
# ==========================================
|
|
8
|
+
# 3. Decoradores (Construcción del Contrato)
|
|
9
|
+
# ==========================================
|
|
10
|
+
def regex_strategy(pattern: str, flags: int = re.IGNORECASE, enabled: bool = True):
|
|
11
|
+
"""Decorador para inyectar una regla de extracción basada en Regex."""
|
|
12
|
+
def decorator(func: Callable) -> Callable:
|
|
13
|
+
meta = _get_or_create_meta(func)
|
|
14
|
+
meta["regex"] = RegexObjective(enabled=enabled, pattern=pattern, flags=flags)
|
|
15
|
+
return func
|
|
16
|
+
return decorator
|
|
17
|
+
|
|
18
|
+
def llm_strategy(
|
|
19
|
+
instruction: Optional[str] = None,
|
|
20
|
+
structure: Optional[Dict[str, Any]] = None,
|
|
21
|
+
enabled: bool = True,
|
|
22
|
+
):
|
|
23
|
+
"""Decorador para inyectar una instrucción o estructura objetivo para un LLM."""
|
|
24
|
+
def decorator(func: Callable) -> Callable:
|
|
25
|
+
meta = _get_or_create_meta(func)
|
|
26
|
+
meta["llm"] = LLMTarget(enabled=enabled, instruction=instruction, structure=structure)
|
|
27
|
+
return func
|
|
28
|
+
return decorator
|
|
29
|
+
|
|
30
|
+
def campo(data_type: str = "string", description: Optional[str] = None, required: bool = False):
|
|
31
|
+
"""Decorador para registrar propiedades base descriptivas del campo."""
|
|
32
|
+
def decorator(func: Callable) -> Callable:
|
|
33
|
+
meta = _get_or_create_meta(func)
|
|
34
|
+
meta.update({
|
|
35
|
+
"data_type": data_type,
|
|
36
|
+
"description": description,
|
|
37
|
+
"required": required,
|
|
38
|
+
})
|
|
39
|
+
return func
|
|
40
|
+
return decorator
|
|
File without changes
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from typing import Any, Dict, List, Optional
|
|
3
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class RegexObjective(BaseModel):
|
|
7
|
+
"""Objetivo runtime de extracción mediante Regex. Solo describe QUÉ ejecutar."""
|
|
8
|
+
model_config = ConfigDict(extra="forbid")
|
|
9
|
+
|
|
10
|
+
enabled: bool = False
|
|
11
|
+
pattern: Optional[str] = None
|
|
12
|
+
flags: int = 0
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class LLMTarget(BaseModel):
|
|
16
|
+
"""Objetivo runtime de extracción mediante LLM. Solo describe instrucción y estructura."""
|
|
17
|
+
model_config = ConfigDict(extra="forbid")
|
|
18
|
+
|
|
19
|
+
enabled: bool = False
|
|
20
|
+
instruction: Optional[str] = None
|
|
21
|
+
structure: Optional[Dict[str, Any]] = None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ExtractionField(BaseModel):
|
|
25
|
+
"""Representa un campo objetivo de información a extraer."""
|
|
26
|
+
model_config = ConfigDict(extra="forbid")
|
|
27
|
+
|
|
28
|
+
name: str
|
|
29
|
+
data_type: str = "string"
|
|
30
|
+
description: Optional[str] = None
|
|
31
|
+
required: bool = False
|
|
32
|
+
regex: Optional[RegexObjective] = None
|
|
33
|
+
llm: Optional[LLMTarget] = None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class ExtractionContract(BaseModel):
|
|
37
|
+
"""Contrato runtime que consolida los campos y estrategias de extracción."""
|
|
38
|
+
model_config = ConfigDict(extra="forbid")
|
|
39
|
+
|
|
40
|
+
mapper_type: Optional[str] = None
|
|
41
|
+
version: str = "1.0"
|
|
42
|
+
fields: Dict[str, ExtractionField] = Field(default_factory=dict)
|
|
43
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
44
|
+
|
|
45
|
+
def add_field(self, field: ExtractionField) -> None:
|
|
46
|
+
"""Agrega un campo al contrato normalizando su nombre."""
|
|
47
|
+
if not isinstance(field, ExtractionField):
|
|
48
|
+
raise TypeError("field debe ser una instancia de ExtractionField.")
|
|
49
|
+
|
|
50
|
+
nombre = field.name.strip()
|
|
51
|
+
if not nombre:
|
|
52
|
+
raise ValueError("El nombre del campo no puede estar vacío.")
|
|
53
|
+
|
|
54
|
+
if nombre in self.fields:
|
|
55
|
+
raise ValueError(f"El campo '{nombre}' ya existe en el ExtractionContract.")
|
|
56
|
+
|
|
57
|
+
if nombre != field.name:
|
|
58
|
+
field = field.model_copy(update={"name": nombre})
|
|
59
|
+
|
|
60
|
+
self.fields[nombre] = field
|
|
61
|
+
|
|
62
|
+
def get_field(self, nombre: str) -> Optional[ExtractionField]:
|
|
63
|
+
return self.fields.get(nombre)
|
|
64
|
+
|
|
65
|
+
def has_field(self, nombre: str) -> bool:
|
|
66
|
+
return nombre in self.fields
|
|
67
|
+
|
|
68
|
+
def count_fields(self) -> int:
|
|
69
|
+
return len(self.fields)
|
|
70
|
+
|
|
71
|
+
def get_regex_fields(self) -> Dict[str, ExtractionField]:
|
|
72
|
+
"""Campos con estrategia Regex habilitada."""
|
|
73
|
+
return {
|
|
74
|
+
nombre: field
|
|
75
|
+
for nombre, field in self.fields.items()
|
|
76
|
+
if field.regex and field.regex.enabled
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
def get_llm_fields(self) -> Dict[str, ExtractionField]:
|
|
80
|
+
"""Campos con estrategia LLM habilitada."""
|
|
81
|
+
return {
|
|
82
|
+
nombre: field
|
|
83
|
+
for nombre, field in self.fields.items()
|
|
84
|
+
if field.llm and field.llm.enabled
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
def validar(self) -> List[str]:
|
|
88
|
+
"""Valida la consistencia interna y la validez sintáctica de las reglas."""
|
|
89
|
+
errores: List[str] = []
|
|
90
|
+
|
|
91
|
+
if not self.fields:
|
|
92
|
+
return ["El ExtractionContract no contiene campos."]
|
|
93
|
+
|
|
94
|
+
for nombre, field in self.fields.items():
|
|
95
|
+
nombre_limpio = nombre.strip()
|
|
96
|
+
|
|
97
|
+
if not nombre_limpio:
|
|
98
|
+
errores.append("Existe un campo con nombre vacío en las llaves del contrato.")
|
|
99
|
+
|
|
100
|
+
if not field.name or not field.name.strip():
|
|
101
|
+
errores.append(f"El campo '{nombre}' tiene un atributo 'name' vacío.")
|
|
102
|
+
elif field.name.strip() != nombre_limpio:
|
|
103
|
+
errores.append(f"El nombre interno '{field.name}' no coincide con su clave '{nombre}'.")
|
|
104
|
+
|
|
105
|
+
if not field.data_type or not field.data_type.strip():
|
|
106
|
+
errores.append(f"El campo '{nombre}' no tiene especificado 'data_type'.")
|
|
107
|
+
|
|
108
|
+
# Validaciones para Regex
|
|
109
|
+
if field.regex and field.regex.enabled:
|
|
110
|
+
if not field.regex.pattern or not field.regex.pattern.strip():
|
|
111
|
+
errores.append(f"El campo '{nombre}' tiene Regex habilitado pero 'pattern' está vacío.")
|
|
112
|
+
else:
|
|
113
|
+
# Intento de compilación real para asegurar sintaxis
|
|
114
|
+
try:
|
|
115
|
+
re.compile(field.regex.pattern, field.regex.flags)
|
|
116
|
+
except (re.error, ValueError, TypeError) as exc:
|
|
117
|
+
errores.append(f"El campo '{nombre}' posee una Regex inválida ({field.regex.pattern}): {exc}")
|
|
118
|
+
|
|
119
|
+
# Validaciones para LLM
|
|
120
|
+
if field.llm and field.llm.enabled:
|
|
121
|
+
if not field.llm.instruction and not field.llm.structure:
|
|
122
|
+
errores.append(
|
|
123
|
+
f"El campo '{nombre}' tiene LLM habilitado pero carece de 'instruction' o 'structure'."
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
return errores
|
|
File without changes
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Union
|
|
4
|
+
|
|
5
|
+
logger = logging.getLogger(__name__)
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class CaptchaExtractor:
|
|
9
|
+
"""
|
|
10
|
+
Extractor especializado para CAPTCHAs.
|
|
11
|
+
|
|
12
|
+
Utiliza ddddocr cuando está disponible.
|
|
13
|
+
|
|
14
|
+
RESPONSABILIDAD
|
|
15
|
+
---------------
|
|
16
|
+
Recibir una imagen CAPTCHA (bytes o ruta) y devolver el texto reconocido.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
def __init__(self):
|
|
20
|
+
"""
|
|
21
|
+
Inicializa ddddocr de manera opcional.
|
|
22
|
+
"""
|
|
23
|
+
self.engine = None
|
|
24
|
+
self._inicializar()
|
|
25
|
+
|
|
26
|
+
# =====================================================================
|
|
27
|
+
# INICIALIZACIÓN
|
|
28
|
+
# =====================================================================
|
|
29
|
+
|
|
30
|
+
def _inicializar(self) -> None:
|
|
31
|
+
"""
|
|
32
|
+
Inicializa el motor ddddocr.
|
|
33
|
+
"""
|
|
34
|
+
try:
|
|
35
|
+
import ddddocr
|
|
36
|
+
|
|
37
|
+
self.engine = ddddocr.DdddOcr(show_ad=False)
|
|
38
|
+
logger.info("[CAPTCHA] Motor ddddocr inicializado correctamente.")
|
|
39
|
+
|
|
40
|
+
except Exception as exc:
|
|
41
|
+
logger.warning(
|
|
42
|
+
f"[CAPTCHA] ddddocr no está disponible: {exc}"
|
|
43
|
+
)
|
|
44
|
+
self.engine = None
|
|
45
|
+
|
|
46
|
+
# =====================================================================
|
|
47
|
+
# DISPONIBILIDAD
|
|
48
|
+
# =====================================================================
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def disponible(self) -> bool:
|
|
52
|
+
"""
|
|
53
|
+
Indica si el motor CAPTCHA está disponible.
|
|
54
|
+
"""
|
|
55
|
+
return self.engine is not None
|
|
56
|
+
|
|
57
|
+
# =====================================================================
|
|
58
|
+
# PUNTOS DE ENTRADA / EXTRAER
|
|
59
|
+
# =====================================================================
|
|
60
|
+
|
|
61
|
+
def resolver(self, image_input: Union[str, Path, bytes]) -> str:
|
|
62
|
+
"""
|
|
63
|
+
Método de entrada flexible esperado por el orquestador.
|
|
64
|
+
Acepta rutas de archivo (str/Path) o bytes directamente.
|
|
65
|
+
"""
|
|
66
|
+
if not image_input:
|
|
67
|
+
logger.warning("[CAPTCHA] Entrada vacía enviada a resolver().")
|
|
68
|
+
return ""
|
|
69
|
+
|
|
70
|
+
# Si recibe una ruta de archivo (str o Path)
|
|
71
|
+
if isinstance(image_input, (str, Path)):
|
|
72
|
+
path = Path(image_input)
|
|
73
|
+
if not path.is_file():
|
|
74
|
+
logger.error(f"[CAPTCHA] Archivo no encontrado: {path}")
|
|
75
|
+
return ""
|
|
76
|
+
try:
|
|
77
|
+
with open(path, "rb") as f:
|
|
78
|
+
image_bytes = f.read()
|
|
79
|
+
return self.extraer(image_bytes)
|
|
80
|
+
except Exception as exc:
|
|
81
|
+
logger.error(f"[CAPTCHA] Error leyendo archivo de captcha: {exc}")
|
|
82
|
+
return ""
|
|
83
|
+
|
|
84
|
+
# Si ya son bytes directamente
|
|
85
|
+
elif isinstance(image_input, bytes):
|
|
86
|
+
return self.extraer(image_input)
|
|
87
|
+
|
|
88
|
+
else:
|
|
89
|
+
logger.error(f"[CAPTCHA] Tipo de entrada no soportado: {type(image_input)}")
|
|
90
|
+
return ""
|
|
91
|
+
|
|
92
|
+
def extraer(self, image_bytes: bytes) -> str:
|
|
93
|
+
"""
|
|
94
|
+
Reconoce el texto de un CAPTCHA a partir de sus bytes.
|
|
95
|
+
"""
|
|
96
|
+
if not image_bytes:
|
|
97
|
+
logger.warning("[CAPTCHA] Se recibió una imagen vacía.")
|
|
98
|
+
return ""
|
|
99
|
+
|
|
100
|
+
if not self.engine:
|
|
101
|
+
logger.warning("[CAPTCHA] El motor ddddocr no está disponible.")
|
|
102
|
+
return ""
|
|
103
|
+
|
|
104
|
+
try:
|
|
105
|
+
resultado = self.engine.classification(image_bytes)
|
|
106
|
+
|
|
107
|
+
if resultado is None:
|
|
108
|
+
return ""
|
|
109
|
+
|
|
110
|
+
texto = str(resultado).strip().upper()
|
|
111
|
+
|
|
112
|
+
logger.debug(
|
|
113
|
+
f"[CAPTCHA] CAPTCHA procesado exitosamente. Longitud={len(texto)}"
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
return texto
|
|
117
|
+
|
|
118
|
+
except Exception as exc:
|
|
119
|
+
logger.error(f"[CAPTCHA] Error procesando CAPTCHA: {exc}")
|
|
120
|
+
return ""
|