tarja 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tarja/__init__.py +65 -0
- tarja/__main__.py +5 -0
- tarja/cli.py +129 -0
- tarja/detect.py +187 -0
- tarja/entities.py +446 -0
- tarja/mask.py +121 -0
- tarja/normalise.py +77 -0
- tarja/py.typed +0 -0
- tarja/registry.py +87 -0
- tarja/validators/__init__.py +4 -0
- tarja/validators/cartao.py +46 -0
- tarja/validators/cep.py +28 -0
- tarja/validators/cib.py +61 -0
- tarja/validators/cnh.py +51 -0
- tarja/validators/cnj.py +67 -0
- tarja/validators/cnm.py +50 -0
- tarja/validators/cnpj.py +93 -0
- tarja/validators/cns.py +82 -0
- tarja/validators/cpf.py +78 -0
- tarja/validators/iptu.py +29 -0
- tarja/validators/matricula.py +29 -0
- tarja/validators/nis.py +66 -0
- tarja/validators/pix.py +19 -0
- tarja/validators/placa.py +44 -0
- tarja/validators/renavam.py +47 -0
- tarja/validators/telefone.py +43 -0
- tarja/validators/titulo.py +77 -0
- tarja/vault.py +201 -0
- tarja-0.5.0.dist-info/METADATA +189 -0
- tarja-0.5.0.dist-info/RECORD +34 -0
- tarja-0.5.0.dist-info/WHEEL +4 -0
- tarja-0.5.0.dist-info/entry_points.txt +2 -0
- tarja-0.5.0.dist-info/licenses/LICENSE +202 -0
- tarja-0.5.0.dist-info/licenses/NOTICE +7 -0
tarja/__init__.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# tarja/__init__.py
|
|
2
|
+
# Public API: find() (search), mask() (replace), validate() (check one value), Vault (reversible
|
|
3
|
+
# tokens), residual() (second-pass check), register_entity() (your own entities).
|
|
4
|
+
# author/autoria: https://github.com/macmaia
|
|
5
|
+
|
|
6
|
+
# import the validator modules (one file per document type in validators/)
|
|
7
|
+
from tarja.detect import Match, find, resolve_overlaps
|
|
8
|
+
from tarja.entities import ENTITIES
|
|
9
|
+
from tarja.mask import mask
|
|
10
|
+
from tarja.registry import register_entity, unregister_entity
|
|
11
|
+
from tarja.validators import cartao, cnj, cnpj, cns, cpf, nis
|
|
12
|
+
from tarja.vault import (
|
|
13
|
+
ProtectedText,
|
|
14
|
+
Vault,
|
|
15
|
+
VaultCollisionError,
|
|
16
|
+
VaultConsumedError,
|
|
17
|
+
VaultError,
|
|
18
|
+
VaultExpiredError,
|
|
19
|
+
VaultScopeError,
|
|
20
|
+
residual,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
# what "from tarja import *" exposes
|
|
24
|
+
__all__ = [
|
|
25
|
+
"ENTITIES",
|
|
26
|
+
"Match",
|
|
27
|
+
"ProtectedText",
|
|
28
|
+
"Vault",
|
|
29
|
+
"VaultCollisionError",
|
|
30
|
+
"VaultConsumedError",
|
|
31
|
+
"VaultError",
|
|
32
|
+
"VaultExpiredError",
|
|
33
|
+
"VaultScopeError",
|
|
34
|
+
"cartao",
|
|
35
|
+
"cnj",
|
|
36
|
+
"cnpj",
|
|
37
|
+
"cns",
|
|
38
|
+
"cpf",
|
|
39
|
+
"find",
|
|
40
|
+
"mask",
|
|
41
|
+
"nis",
|
|
42
|
+
"register_entity",
|
|
43
|
+
"residual",
|
|
44
|
+
"resolve_overlaps",
|
|
45
|
+
"unregister_entity",
|
|
46
|
+
"validate",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
# version read by hatch at build time (see pyproject.toml, [tool.hatch.version])
|
|
50
|
+
__version__ = "0.5.0"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def validate(entity: str, value: str) -> bool:
|
|
54
|
+
"""EN: Check value against the entity's rule. E.g. validate("BR_CPF", "529.982.247-25").
|
|
55
|
+
PT: Valida value pela regra da entidade. Ex: validate("BR_CPF", "529.982.247-25").
|
|
56
|
+
"""
|
|
57
|
+
# [VALIDATE] look the entity up in the registry (tarja/entities.py)
|
|
58
|
+
spec = ENTITIES.get(entity)
|
|
59
|
+
if spec is None:
|
|
60
|
+
# unknown entity -> explicit error, better than a silent False
|
|
61
|
+
raise ValueError(f"unknown entity / entidade desconhecida: {entity!r}")
|
|
62
|
+
# entities without a check digit can't be validated alone
|
|
63
|
+
if spec.validator is None:
|
|
64
|
+
raise ValueError(f"{entity} has no check digit / {entity} nao tem DV")
|
|
65
|
+
return spec.validator(value)
|
tarja/__main__.py
ADDED
tarja/cli.py
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# tarja/cli.py
|
|
2
|
+
# [CLI] command line. Examples:
|
|
3
|
+
# tarja scan contrato.txt -> JSON lines, one per finding (value hidden by default)
|
|
4
|
+
# tarja scan contrato.txt --format table -> readable table
|
|
5
|
+
# tarja scan contrato.txt --show-values -> include the raw identifier (careful)
|
|
6
|
+
# tarja mask contrato.txt > limpo.txt -> masked copy, strategy redact
|
|
7
|
+
# tarja mask - --strategy pseudonym_stable --salt "$TARJA_SALT" < in.txt
|
|
8
|
+
# cat log.txt | tarja scan - --entities BR_CPF,BR_CNPJ --min-score 0.9
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
|
|
17
|
+
from tarja import __version__
|
|
18
|
+
from tarja.detect import find
|
|
19
|
+
from tarja.entities import ENTITIES
|
|
20
|
+
from tarja.mask import STRATEGIES, STRATEGY_ALIASES, mask
|
|
21
|
+
|
|
22
|
+
# [CLI-LIMIT] input cap, so a huge file or endless pipe can't exhaust memory (override with --max-mb)
|
|
23
|
+
DEFAULT_MAX_MB = 50
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _read(path: str, max_mb: float = DEFAULT_MAX_MB) -> str:
|
|
27
|
+
# [CLI-READ] "-" reads stdin, both capped at max_mb (read one char over the cap to detect it)
|
|
28
|
+
limit = int(max_mb * 1024 * 1024)
|
|
29
|
+
if path == "-":
|
|
30
|
+
text = sys.stdin.read(limit + 1)
|
|
31
|
+
else:
|
|
32
|
+
with open(path, encoding="utf-8") as fh:
|
|
33
|
+
text = fh.read(limit + 1)
|
|
34
|
+
if len(text) > limit:
|
|
35
|
+
raise ValueError(f"input larger than {max_mb:g} MB, use --max-mb or split the file")
|
|
36
|
+
return text
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _entities(arg: str | None) -> list[str] | None:
|
|
40
|
+
# [CLI-ENTITIES] "BR_CPF,BR_CNPJ" -> list, None = all
|
|
41
|
+
if not arg:
|
|
42
|
+
return None
|
|
43
|
+
return [e.strip().upper() for e in arg.split(",") if e.strip()]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
47
|
+
# [CLI-PARSER]
|
|
48
|
+
p = argparse.ArgumentParser(
|
|
49
|
+
prog="tarja",
|
|
50
|
+
description="Find and mask Brazilian personal identifiers. / Acha e mascara identificadores brasileiros.",
|
|
51
|
+
)
|
|
52
|
+
p.add_argument("--version", action="version", version=f"tarja {__version__}")
|
|
53
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
54
|
+
|
|
55
|
+
# shared options
|
|
56
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
57
|
+
common.add_argument("path", help="file or - for stdin / arquivo ou - p/ stdin")
|
|
58
|
+
common.add_argument(
|
|
59
|
+
"--entities",
|
|
60
|
+
help=f"comma-separated / separadas por virgula. Default: all / todas ({','.join(ENTITIES)})",
|
|
61
|
+
)
|
|
62
|
+
common.add_argument("--min-score", type=float, default=0.0, help="drop below this / descarta abaixo disso")
|
|
63
|
+
common.add_argument(
|
|
64
|
+
"--max-mb",
|
|
65
|
+
type=float,
|
|
66
|
+
default=DEFAULT_MAX_MB,
|
|
67
|
+
help=f"input size cap in MB, default {DEFAULT_MAX_MB} / limite de entrada em MB",
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
# [CLI-SCAN]
|
|
71
|
+
scan = sub.add_parser("scan", parents=[common], help="list findings / lista o q achou")
|
|
72
|
+
scan.add_argument("--format", choices=("jsonl", "table"), default="jsonl")
|
|
73
|
+
scan.add_argument(
|
|
74
|
+
"--show-values",
|
|
75
|
+
action="store_true",
|
|
76
|
+
help="print raw identifiers (personal data!) / mostra o valor cru (dado pessoal!)",
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
scan.add_argument(
|
|
80
|
+
"--suspect",
|
|
81
|
+
action="store_true",
|
|
82
|
+
help="also report ID-shaped values with a wrong check digit (score 0) / reporta tb formato de ID c/ DV errado",
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
# [CLI-MASK]
|
|
86
|
+
mk = sub.add_parser("mask", parents=[common], help="print masked text / imprime o texto mascarado")
|
|
87
|
+
mk.add_argument("--strategy", choices=(*STRATEGIES, *STRATEGY_ALIASES), default="redact")
|
|
88
|
+
mk.add_argument(
|
|
89
|
+
"--salt",
|
|
90
|
+
default=os.environ.get("TARJA_SALT"),
|
|
91
|
+
help="secret key for pseudonym_stable, or set TARJA_SALT / chave secreta, ou use TARJA_SALT",
|
|
92
|
+
)
|
|
93
|
+
return p
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def main(argv: list[str] | None = None) -> int:
|
|
97
|
+
"""EN: Entry point (console script "tarja"). Returns an exit code.
|
|
98
|
+
PT: Ponto de entrada (script "tarja"). Devolve o exit code.
|
|
99
|
+
"""
|
|
100
|
+
args = _build_parser().parse_args(argv)
|
|
101
|
+
try:
|
|
102
|
+
text = _read(args.path, args.max_mb)
|
|
103
|
+
# [CLI-SUSPECT] only scan reports suspects, mask never touches them
|
|
104
|
+
suspect = args.command == "scan" and args.suspect
|
|
105
|
+
found = find(text, entities=_entities(args.entities), min_score=args.min_score, report_invalid=suspect)
|
|
106
|
+
if args.command == "mask":
|
|
107
|
+
# [CLI-MASK-RUN]
|
|
108
|
+
sys.stdout.write(mask(text, strategy=args.strategy, salt=args.salt, matches=found))
|
|
109
|
+
return 0
|
|
110
|
+
# [CLI-SCAN-RUN]
|
|
111
|
+
if args.format == "jsonl":
|
|
112
|
+
for m in found:
|
|
113
|
+
sys.stdout.write(json.dumps(m.to_dict(include_value=args.show_values), ensure_ascii=False) + "\n")
|
|
114
|
+
else:
|
|
115
|
+
# simple fixed-width table
|
|
116
|
+
sys.stdout.write(f"{'entity':<10} {'start':>7} {'end':>7} {'score':>5} value\n")
|
|
117
|
+
for m in found:
|
|
118
|
+
shown = m.value if args.show_values else "*" * len(m.value)
|
|
119
|
+
sys.stdout.write(f"{m.entity:<10} {m.start:>7} {m.end:>7} {m.score:>5.2f} {shown}\n")
|
|
120
|
+
# exit 1 when something was found, handy in CI
|
|
121
|
+
return 1 if found else 0
|
|
122
|
+
except (OSError, ValueError, UnicodeDecodeError) as exc:
|
|
123
|
+
# [CLI-ERROR] clean message instead of a traceback
|
|
124
|
+
sys.stderr.write(f"tarja: {exc}\n")
|
|
125
|
+
return 2
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
if __name__ == "__main__": # pragma: no cover
|
|
129
|
+
raise SystemExit(main())
|
tarja/detect.py
ADDED
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
# tarja/detect.py
|
|
2
|
+
# [DETECT] detection engine. Pipeline per entity:
|
|
3
|
+
# 1. normalise the text (same length, offsets kept)
|
|
4
|
+
# 2. run every candidate regex
|
|
5
|
+
# 3. validate the check digit (drop if wrong)
|
|
6
|
+
# 4. look for context words around the match -> final score
|
|
7
|
+
# 5. resolve overlaps between entities
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from collections.abc import Iterable
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from functools import lru_cache
|
|
15
|
+
|
|
16
|
+
from tarja.entities import ENTITIES, TIER_RANK, EntitySpec
|
|
17
|
+
from tarja.normalise import fold, normalise_text
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class Match:
|
|
22
|
+
"""EN: One detected identifier. start/end are offsets in the ORIGINAL text (text[start:end] == value).
|
|
23
|
+
PT: 1 identificador achado. start/end sao offsets no texto ORIGINAL (text[start:end] == value).
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
entity: str
|
|
27
|
+
start: int
|
|
28
|
+
end: int
|
|
29
|
+
value: str
|
|
30
|
+
score: float
|
|
31
|
+
tier: str
|
|
32
|
+
pattern: str
|
|
33
|
+
has_context: bool
|
|
34
|
+
# [DETECT-SUSPECT] False = right shape, WRONG check digit (only with find(report_invalid=True), score 0)
|
|
35
|
+
valid_dv: bool = True
|
|
36
|
+
|
|
37
|
+
def to_dict(self, include_value: bool = True) -> dict:
|
|
38
|
+
"""EN: Plain dict for JSON output. include_value=False drops the raw identifier.
|
|
39
|
+
PT: Dict simples p/ saida JSON. include_value=False tira o identificador cru.
|
|
40
|
+
"""
|
|
41
|
+
# [DETECT-DICT]
|
|
42
|
+
d = {
|
|
43
|
+
"entity": self.entity,
|
|
44
|
+
"start": self.start,
|
|
45
|
+
"end": self.end,
|
|
46
|
+
"score": self.score,
|
|
47
|
+
"tier": self.tier,
|
|
48
|
+
"pattern": self.pattern,
|
|
49
|
+
"has_context": self.has_context,
|
|
50
|
+
"valid_dv": self.valid_dv,
|
|
51
|
+
}
|
|
52
|
+
if include_value:
|
|
53
|
+
d["value"] = self.value
|
|
54
|
+
return d
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@lru_cache(maxsize=64)
|
|
58
|
+
def _context_regex(words: tuple[str, ...]) -> re.Pattern[str]:
|
|
59
|
+
# [DETECT-CONTEXT-RE] one regex per word list, whole words only ("sus" must not hit "suspenso")
|
|
60
|
+
alternatives = "|".join(re.escape(w) for w in sorted(words, key=len, reverse=True))
|
|
61
|
+
return re.compile(rf"(?<![0-9a-z])(?:{alternatives})(?![0-9a-z])")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@lru_cache(maxsize=64)
|
|
65
|
+
def _adjacent_regexes(before: tuple[str, ...], after: tuple[str, ...], gap: int) -> tuple[re.Pattern[str], re.Pattern[str]]:
|
|
66
|
+
# [DETECT-ADJACENT-RE] word + up to gap non-alphanumerics (+ an optional "o" from a folded "nº") at the END of the
|
|
67
|
+
# text before, and up to gap non-alphanumerics + one optional 1-3 letter linking word ("no", "do") + word at the
|
|
68
|
+
# START of the text after
|
|
69
|
+
alt = lambda ws: "|".join(re.escape(w) for w in sorted(ws, key=len, reverse=True)) or "(?!)" # noqa: E731
|
|
70
|
+
pre = re.compile(rf"(?<![0-9a-z])(?:{alt(before)})[^0-9a-z]{{0,{gap}}}(?:o[^0-9a-z]{{1,2}})?$")
|
|
71
|
+
post = re.compile(rf"^[^0-9a-z]{{0,{gap}}}(?:[a-z]{{1,3}}[^0-9a-z]{{1,2}})?(?:{alt(after)})(?![0-9a-z])")
|
|
72
|
+
return pre, post
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _has_context(folded: str, start: int, end: int, spec: EntitySpec) -> bool:
|
|
76
|
+
# [DETECT-CONTEXT] search window before and after the match in the folded (lowercase, no accent) text
|
|
77
|
+
if spec.context_before or spec.context_after:
|
|
78
|
+
# [DETECT-ADJACENT] tight mode, see EntitySpec.context_gap
|
|
79
|
+
pre, post = _adjacent_regexes(spec.context_before, spec.context_after, spec.context_gap)
|
|
80
|
+
reach = max((len(w) for w in spec.context_before), default=0) + spec.context_gap + 3
|
|
81
|
+
return bool(pre.search(folded[max(0, start - reach) : start])) or bool(post.search(folded[end : end + 40]))
|
|
82
|
+
lo = max(0, start - spec.context_window)
|
|
83
|
+
hi = min(len(folded), end + spec.context_window)
|
|
84
|
+
# blank out the match itself so the number can't count as its own context
|
|
85
|
+
window = folded[lo:start] + " " + folded[end:hi]
|
|
86
|
+
return _context_regex(spec.context_words).search(window) is not None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# [DETECT-SUSPECT-MIN] only "formatted" patterns (base score >= 0.5) can raise a suspect, so random digit
|
|
90
|
+
# runs don't flood the report
|
|
91
|
+
SUSPECT_MIN_PATTERN_SCORE = 0.5
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _candidates(norm: str, folded: str, text: str, spec: EntitySpec, report_invalid: bool = False) -> Iterable[Match]:
|
|
95
|
+
# [DETECT-CANDIDATES] every regex hit that passes the check digit, one Match per distinct span
|
|
96
|
+
seen: set[tuple[int, int]] = set()
|
|
97
|
+
for pat in spec.patterns:
|
|
98
|
+
for m in pat.regex.finditer(norm):
|
|
99
|
+
span = (m.start(), m.end())
|
|
100
|
+
if span in seen:
|
|
101
|
+
continue
|
|
102
|
+
seen.add(span)
|
|
103
|
+
# validate on the normalised slice (ASCII digits etc)
|
|
104
|
+
if spec.validator is not None and not spec.validator(m.group(0)):
|
|
105
|
+
# [DETECT-SUSPECT-EMIT] N1 look-alike with wrong DV, reported with score 0 when asked
|
|
106
|
+
if report_invalid and spec.tier == "N1" and pat.score >= SUSPECT_MIN_PATTERN_SCORE:
|
|
107
|
+
ctx = _has_context(folded, m.start(), m.end(), spec)
|
|
108
|
+
yield Match(spec.id, m.start(), m.end(), text[m.start() : m.end()], 0.0, spec.tier, pat.name, ctx, False)
|
|
109
|
+
continue
|
|
110
|
+
ctx = _has_context(folded, m.start(), m.end(), spec)
|
|
111
|
+
# entities that require context are dropped without it
|
|
112
|
+
if spec.context_required and not ctx:
|
|
113
|
+
continue
|
|
114
|
+
score = spec.score_with_context if ctx else spec.score_without_context
|
|
115
|
+
yield Match(
|
|
116
|
+
entity=spec.id,
|
|
117
|
+
start=m.start(),
|
|
118
|
+
end=m.end(),
|
|
119
|
+
value=text[m.start() : m.end()],
|
|
120
|
+
score=score,
|
|
121
|
+
tier=spec.tier,
|
|
122
|
+
pattern=pat.name,
|
|
123
|
+
has_context=ctx,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def resolve_overlaps(matches: Iterable[Match], order: list[str] | None = None) -> list[Match]:
|
|
128
|
+
"""EN: Keep the best match wherever spans overlap. Priority: stronger tier, then higher score,
|
|
129
|
+
then longer span, then registry order. Result is sorted by position.
|
|
130
|
+
PT: Fica c/ o melhor match onde os trechos se sobrepoem. Prioridade: nivel mais forte, dps score
|
|
131
|
+
maior, dps trecho mais longo, dps ordem do registro. Resultado ordenado por posicao.
|
|
132
|
+
"""
|
|
133
|
+
# [DETECT-OVERLAP]
|
|
134
|
+
order = order or list(ENTITIES)
|
|
135
|
+
rank = {e: i for i, e in enumerate(order)}
|
|
136
|
+
|
|
137
|
+
def key(m: Match) -> tuple:
|
|
138
|
+
# sort key, best first
|
|
139
|
+
return (TIER_RANK.get(m.tier, 9), -m.score, -(m.end - m.start), rank.get(m.entity, 99), m.start)
|
|
140
|
+
|
|
141
|
+
kept: list[Match] = []
|
|
142
|
+
for m in sorted(matches, key=key):
|
|
143
|
+
# keep only if it doesn't touch anything already kept
|
|
144
|
+
if all(m.end <= k.start or m.start >= k.end for k in kept):
|
|
145
|
+
kept.append(m)
|
|
146
|
+
return sorted(kept, key=lambda m: (m.start, m.end))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def find(
|
|
150
|
+
text: str,
|
|
151
|
+
entities: Iterable[str] | None = None,
|
|
152
|
+
min_score: float = 0.0,
|
|
153
|
+
resolve: bool = True,
|
|
154
|
+
report_invalid: bool = False,
|
|
155
|
+
) -> list[Match]:
|
|
156
|
+
"""EN: Find Brazilian identifiers in text.
|
|
157
|
+
entities: subset of ids (e.g. ["BR_CPF"]), default all. min_score: drop anything below.
|
|
158
|
+
resolve: resolve overlaps between entities (default True).
|
|
159
|
+
report_invalid: also return N1 look-alikes with a WRONG check digit (valid_dv=False, score 0), e.g. typos.
|
|
160
|
+
PT: Acha identificadores brasileiros no texto.
|
|
161
|
+
entities: subconjunto de ids (ex: ["BR_CPF"]), padrao todos. min_score: descarta abaixo disso.
|
|
162
|
+
resolve: resolve sobreposicao entre entidades (padrao True).
|
|
163
|
+
report_invalid: devolve tb parecidos N1 c/ DV ERRADO (valid_dv=False, score 0), ex: erro de digitacao.
|
|
164
|
+
"""
|
|
165
|
+
# [FIND] type check
|
|
166
|
+
if not isinstance(text, str):
|
|
167
|
+
raise TypeError("text must be str / text precisa ser str")
|
|
168
|
+
ids = list(entities) if entities is not None else list(ENTITIES)
|
|
169
|
+
unknown = [e for e in ids if e not in ENTITIES]
|
|
170
|
+
if unknown:
|
|
171
|
+
raise ValueError(f"unknown entities / entidades desconhecidas: {unknown}")
|
|
172
|
+
# [FIND-PREP] normalise once, fold once
|
|
173
|
+
norm = normalise_text(text)
|
|
174
|
+
folded = fold(norm)
|
|
175
|
+
found: list[Match] = []
|
|
176
|
+
for eid in ids:
|
|
177
|
+
found.extend(_candidates(norm, folded, text, ENTITIES[eid], report_invalid))
|
|
178
|
+
# [FIND-SPLIT] suspects never compete with valid matches
|
|
179
|
+
suspects = [m for m in found if not m.valid_dv]
|
|
180
|
+
found = [m for m in found if m.valid_dv and m.score >= min_score]
|
|
181
|
+
# [FIND-RESOLVE]
|
|
182
|
+
kept = resolve_overlaps(found, ids) if resolve else sorted(found, key=lambda m: (m.start, m.end))
|
|
183
|
+
if suspects:
|
|
184
|
+
# a suspect survives only where no valid match sits
|
|
185
|
+
free = [x for x in suspects if all(x.end <= k.start or x.start >= k.end for k in kept)]
|
|
186
|
+
kept = sorted(kept + resolve_overlaps(free, ids), key=lambda m: (m.start, m.end))
|
|
187
|
+
return kept
|