tarja 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tarja/__init__.py ADDED
@@ -0,0 +1,65 @@
1
+ # tarja/__init__.py
2
+ # Public API: find() (search), mask() (replace), validate() (check one value), Vault (reversible
3
+ # tokens), residual() (second-pass check), register_entity() (your own entities).
4
+ # author/autoria: https://github.com/macmaia
5
+
6
+ # import the validator modules (one file per document type in validators/)
7
+ from tarja.detect import Match, find, resolve_overlaps
8
+ from tarja.entities import ENTITIES
9
+ from tarja.mask import mask
10
+ from tarja.registry import register_entity, unregister_entity
11
+ from tarja.validators import cartao, cnj, cnpj, cns, cpf, nis
12
+ from tarja.vault import (
13
+ ProtectedText,
14
+ Vault,
15
+ VaultCollisionError,
16
+ VaultConsumedError,
17
+ VaultError,
18
+ VaultExpiredError,
19
+ VaultScopeError,
20
+ residual,
21
+ )
22
+
23
+ # what "from tarja import *" exposes
24
+ __all__ = [
25
+ "ENTITIES",
26
+ "Match",
27
+ "ProtectedText",
28
+ "Vault",
29
+ "VaultCollisionError",
30
+ "VaultConsumedError",
31
+ "VaultError",
32
+ "VaultExpiredError",
33
+ "VaultScopeError",
34
+ "cartao",
35
+ "cnj",
36
+ "cnpj",
37
+ "cns",
38
+ "cpf",
39
+ "find",
40
+ "mask",
41
+ "nis",
42
+ "register_entity",
43
+ "residual",
44
+ "resolve_overlaps",
45
+ "unregister_entity",
46
+ "validate",
47
+ ]
48
+
49
+ # version read by hatch at build time (see pyproject.toml, [tool.hatch.version])
50
+ __version__ = "0.5.0"
51
+
52
+
53
+ def validate(entity: str, value: str) -> bool:
54
+ """EN: Check value against the entity's rule. E.g. validate("BR_CPF", "529.982.247-25").
55
+ PT: Valida value pela regra da entidade. Ex: validate("BR_CPF", "529.982.247-25").
56
+ """
57
+ # [VALIDATE] look the entity up in the registry (tarja/entities.py)
58
+ spec = ENTITIES.get(entity)
59
+ if spec is None:
60
+ # unknown entity -> explicit error, better than a silent False
61
+ raise ValueError(f"unknown entity / entidade desconhecida: {entity!r}")
62
+ # entities without a check digit can't be validated alone
63
+ if spec.validator is None:
64
+ raise ValueError(f"{entity} has no check digit / {entity} nao tem DV")
65
+ return spec.validator(value)
tarja/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ # tarja/__main__.py
2
+ # lets you run "python -m tarja scan file.txt"
3
+ from tarja.cli import main
4
+
5
+ raise SystemExit(main())
tarja/cli.py ADDED
@@ -0,0 +1,129 @@
1
+ # tarja/cli.py
2
+ # [CLI] command line. Examples:
3
+ # tarja scan contrato.txt -> JSON lines, one per finding (value hidden by default)
4
+ # tarja scan contrato.txt --format table -> readable table
5
+ # tarja scan contrato.txt --show-values -> include the raw identifier (careful)
6
+ # tarja mask contrato.txt > limpo.txt -> masked copy, strategy redact
7
+ # tarja mask - --strategy pseudonym_stable --salt "$TARJA_SALT" < in.txt
8
+ # cat log.txt | tarja scan - --entities BR_CPF,BR_CNPJ --min-score 0.9
9
+
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import json
14
+ import os
15
+ import sys
16
+
17
+ from tarja import __version__
18
+ from tarja.detect import find
19
+ from tarja.entities import ENTITIES
20
+ from tarja.mask import STRATEGIES, STRATEGY_ALIASES, mask
21
+
22
+ # [CLI-LIMIT] input cap, so a huge file or endless pipe can't exhaust memory (override with --max-mb)
23
+ DEFAULT_MAX_MB = 50
24
+
25
+
26
+ def _read(path: str, max_mb: float = DEFAULT_MAX_MB) -> str:
27
+ # [CLI-READ] "-" reads stdin, both capped at max_mb (read one char over the cap to detect it)
28
+ limit = int(max_mb * 1024 * 1024)
29
+ if path == "-":
30
+ text = sys.stdin.read(limit + 1)
31
+ else:
32
+ with open(path, encoding="utf-8") as fh:
33
+ text = fh.read(limit + 1)
34
+ if len(text) > limit:
35
+ raise ValueError(f"input larger than {max_mb:g} MB, use --max-mb or split the file")
36
+ return text
37
+
38
+
39
+ def _entities(arg: str | None) -> list[str] | None:
40
+ # [CLI-ENTITIES] "BR_CPF,BR_CNPJ" -> list, None = all
41
+ if not arg:
42
+ return None
43
+ return [e.strip().upper() for e in arg.split(",") if e.strip()]
44
+
45
+
46
+ def _build_parser() -> argparse.ArgumentParser:
47
+ # [CLI-PARSER]
48
+ p = argparse.ArgumentParser(
49
+ prog="tarja",
50
+ description="Find and mask Brazilian personal identifiers. / Acha e mascara identificadores brasileiros.",
51
+ )
52
+ p.add_argument("--version", action="version", version=f"tarja {__version__}")
53
+ sub = p.add_subparsers(dest="command", required=True)
54
+
55
+ # shared options
56
+ common = argparse.ArgumentParser(add_help=False)
57
+ common.add_argument("path", help="file or - for stdin / arquivo ou - p/ stdin")
58
+ common.add_argument(
59
+ "--entities",
60
+ help=f"comma-separated / separadas por virgula. Default: all / todas ({','.join(ENTITIES)})",
61
+ )
62
+ common.add_argument("--min-score", type=float, default=0.0, help="drop below this / descarta abaixo disso")
63
+ common.add_argument(
64
+ "--max-mb",
65
+ type=float,
66
+ default=DEFAULT_MAX_MB,
67
+ help=f"input size cap in MB, default {DEFAULT_MAX_MB} / limite de entrada em MB",
68
+ )
69
+
70
+ # [CLI-SCAN]
71
+ scan = sub.add_parser("scan", parents=[common], help="list findings / lista o q achou")
72
+ scan.add_argument("--format", choices=("jsonl", "table"), default="jsonl")
73
+ scan.add_argument(
74
+ "--show-values",
75
+ action="store_true",
76
+ help="print raw identifiers (personal data!) / mostra o valor cru (dado pessoal!)",
77
+ )
78
+
79
+ scan.add_argument(
80
+ "--suspect",
81
+ action="store_true",
82
+ help="also report ID-shaped values with a wrong check digit (score 0) / reporta tb formato de ID c/ DV errado",
83
+ )
84
+
85
+ # [CLI-MASK]
86
+ mk = sub.add_parser("mask", parents=[common], help="print masked text / imprime o texto mascarado")
87
+ mk.add_argument("--strategy", choices=(*STRATEGIES, *STRATEGY_ALIASES), default="redact")
88
+ mk.add_argument(
89
+ "--salt",
90
+ default=os.environ.get("TARJA_SALT"),
91
+ help="secret key for pseudonym_stable, or set TARJA_SALT / chave secreta, ou use TARJA_SALT",
92
+ )
93
+ return p
94
+
95
+
96
+ def main(argv: list[str] | None = None) -> int:
97
+ """EN: Entry point (console script "tarja"). Returns an exit code.
98
+ PT: Ponto de entrada (script "tarja"). Devolve o exit code.
99
+ """
100
+ args = _build_parser().parse_args(argv)
101
+ try:
102
+ text = _read(args.path, args.max_mb)
103
+ # [CLI-SUSPECT] only scan reports suspects, mask never touches them
104
+ suspect = args.command == "scan" and args.suspect
105
+ found = find(text, entities=_entities(args.entities), min_score=args.min_score, report_invalid=suspect)
106
+ if args.command == "mask":
107
+ # [CLI-MASK-RUN]
108
+ sys.stdout.write(mask(text, strategy=args.strategy, salt=args.salt, matches=found))
109
+ return 0
110
+ # [CLI-SCAN-RUN]
111
+ if args.format == "jsonl":
112
+ for m in found:
113
+ sys.stdout.write(json.dumps(m.to_dict(include_value=args.show_values), ensure_ascii=False) + "\n")
114
+ else:
115
+ # simple fixed-width table
116
+ sys.stdout.write(f"{'entity':<10} {'start':>7} {'end':>7} {'score':>5} value\n")
117
+ for m in found:
118
+ shown = m.value if args.show_values else "*" * len(m.value)
119
+ sys.stdout.write(f"{m.entity:<10} {m.start:>7} {m.end:>7} {m.score:>5.2f} {shown}\n")
120
+ # exit 1 when something was found, handy in CI
121
+ return 1 if found else 0
122
+ except (OSError, ValueError, UnicodeDecodeError) as exc:
123
+ # [CLI-ERROR] clean message instead of a traceback
124
+ sys.stderr.write(f"tarja: {exc}\n")
125
+ return 2
126
+
127
+
128
+ if __name__ == "__main__": # pragma: no cover
129
+ raise SystemExit(main())
tarja/detect.py ADDED
@@ -0,0 +1,187 @@
1
+ # tarja/detect.py
2
+ # [DETECT] detection engine. Pipeline per entity:
3
+ # 1. normalise the text (same length, offsets kept)
4
+ # 2. run every candidate regex
5
+ # 3. validate the check digit (drop if wrong)
6
+ # 4. look for context words around the match -> final score
7
+ # 5. resolve overlaps between entities
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from collections.abc import Iterable
13
+ from dataclasses import dataclass
14
+ from functools import lru_cache
15
+
16
+ from tarja.entities import ENTITIES, TIER_RANK, EntitySpec
17
+ from tarja.normalise import fold, normalise_text
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class Match:
22
+ """EN: One detected identifier. start/end are offsets in the ORIGINAL text (text[start:end] == value).
23
+ PT: 1 identificador achado. start/end sao offsets no texto ORIGINAL (text[start:end] == value).
24
+ """
25
+
26
+ entity: str
27
+ start: int
28
+ end: int
29
+ value: str
30
+ score: float
31
+ tier: str
32
+ pattern: str
33
+ has_context: bool
34
+ # [DETECT-SUSPECT] False = right shape, WRONG check digit (only with find(report_invalid=True), score 0)
35
+ valid_dv: bool = True
36
+
37
+ def to_dict(self, include_value: bool = True) -> dict:
38
+ """EN: Plain dict for JSON output. include_value=False drops the raw identifier.
39
+ PT: Dict simples p/ saida JSON. include_value=False tira o identificador cru.
40
+ """
41
+ # [DETECT-DICT]
42
+ d = {
43
+ "entity": self.entity,
44
+ "start": self.start,
45
+ "end": self.end,
46
+ "score": self.score,
47
+ "tier": self.tier,
48
+ "pattern": self.pattern,
49
+ "has_context": self.has_context,
50
+ "valid_dv": self.valid_dv,
51
+ }
52
+ if include_value:
53
+ d["value"] = self.value
54
+ return d
55
+
56
+
57
+ @lru_cache(maxsize=64)
58
+ def _context_regex(words: tuple[str, ...]) -> re.Pattern[str]:
59
+ # [DETECT-CONTEXT-RE] one regex per word list, whole words only ("sus" must not hit "suspenso")
60
+ alternatives = "|".join(re.escape(w) for w in sorted(words, key=len, reverse=True))
61
+ return re.compile(rf"(?<![0-9a-z])(?:{alternatives})(?![0-9a-z])")
62
+
63
+
64
+ @lru_cache(maxsize=64)
65
+ def _adjacent_regexes(before: tuple[str, ...], after: tuple[str, ...], gap: int) -> tuple[re.Pattern[str], re.Pattern[str]]:
66
+ # [DETECT-ADJACENT-RE] word + up to gap non-alphanumerics (+ an optional "o" from a folded "nº") at the END of the
67
+ # text before, and up to gap non-alphanumerics + one optional 1-3 letter linking word ("no", "do") + word at the
68
+ # START of the text after
69
+ alt = lambda ws: "|".join(re.escape(w) for w in sorted(ws, key=len, reverse=True)) or "(?!)" # noqa: E731
70
+ pre = re.compile(rf"(?<![0-9a-z])(?:{alt(before)})[^0-9a-z]{{0,{gap}}}(?:o[^0-9a-z]{{1,2}})?$")
71
+ post = re.compile(rf"^[^0-9a-z]{{0,{gap}}}(?:[a-z]{{1,3}}[^0-9a-z]{{1,2}})?(?:{alt(after)})(?![0-9a-z])")
72
+ return pre, post
73
+
74
+
75
+ def _has_context(folded: str, start: int, end: int, spec: EntitySpec) -> bool:
76
+ # [DETECT-CONTEXT] search window before and after the match in the folded (lowercase, no accent) text
77
+ if spec.context_before or spec.context_after:
78
+ # [DETECT-ADJACENT] tight mode, see EntitySpec.context_gap
79
+ pre, post = _adjacent_regexes(spec.context_before, spec.context_after, spec.context_gap)
80
+ reach = max((len(w) for w in spec.context_before), default=0) + spec.context_gap + 3
81
+ return bool(pre.search(folded[max(0, start - reach) : start])) or bool(post.search(folded[end : end + 40]))
82
+ lo = max(0, start - spec.context_window)
83
+ hi = min(len(folded), end + spec.context_window)
84
+ # blank out the match itself so the number can't count as its own context
85
+ window = folded[lo:start] + " " + folded[end:hi]
86
+ return _context_regex(spec.context_words).search(window) is not None
87
+
88
+
89
+ # [DETECT-SUSPECT-MIN] only "formatted" patterns (base score >= 0.5) can raise a suspect, so random digit
90
+ # runs don't flood the report
91
+ SUSPECT_MIN_PATTERN_SCORE = 0.5
92
+
93
+
94
+ def _candidates(norm: str, folded: str, text: str, spec: EntitySpec, report_invalid: bool = False) -> Iterable[Match]:
95
+ # [DETECT-CANDIDATES] every regex hit that passes the check digit, one Match per distinct span
96
+ seen: set[tuple[int, int]] = set()
97
+ for pat in spec.patterns:
98
+ for m in pat.regex.finditer(norm):
99
+ span = (m.start(), m.end())
100
+ if span in seen:
101
+ continue
102
+ seen.add(span)
103
+ # validate on the normalised slice (ASCII digits etc)
104
+ if spec.validator is not None and not spec.validator(m.group(0)):
105
+ # [DETECT-SUSPECT-EMIT] N1 look-alike with wrong DV, reported with score 0 when asked
106
+ if report_invalid and spec.tier == "N1" and pat.score >= SUSPECT_MIN_PATTERN_SCORE:
107
+ ctx = _has_context(folded, m.start(), m.end(), spec)
108
+ yield Match(spec.id, m.start(), m.end(), text[m.start() : m.end()], 0.0, spec.tier, pat.name, ctx, False)
109
+ continue
110
+ ctx = _has_context(folded, m.start(), m.end(), spec)
111
+ # entities that require context are dropped without it
112
+ if spec.context_required and not ctx:
113
+ continue
114
+ score = spec.score_with_context if ctx else spec.score_without_context
115
+ yield Match(
116
+ entity=spec.id,
117
+ start=m.start(),
118
+ end=m.end(),
119
+ value=text[m.start() : m.end()],
120
+ score=score,
121
+ tier=spec.tier,
122
+ pattern=pat.name,
123
+ has_context=ctx,
124
+ )
125
+
126
+
127
+ def resolve_overlaps(matches: Iterable[Match], order: list[str] | None = None) -> list[Match]:
128
+ """EN: Keep the best match wherever spans overlap. Priority: stronger tier, then higher score,
129
+ then longer span, then registry order. Result is sorted by position.
130
+ PT: Fica c/ o melhor match onde os trechos se sobrepoem. Prioridade: nivel mais forte, dps score
131
+ maior, dps trecho mais longo, dps ordem do registro. Resultado ordenado por posicao.
132
+ """
133
+ # [DETECT-OVERLAP]
134
+ order = order or list(ENTITIES)
135
+ rank = {e: i for i, e in enumerate(order)}
136
+
137
+ def key(m: Match) -> tuple:
138
+ # sort key, best first
139
+ return (TIER_RANK.get(m.tier, 9), -m.score, -(m.end - m.start), rank.get(m.entity, 99), m.start)
140
+
141
+ kept: list[Match] = []
142
+ for m in sorted(matches, key=key):
143
+ # keep only if it doesn't touch anything already kept
144
+ if all(m.end <= k.start or m.start >= k.end for k in kept):
145
+ kept.append(m)
146
+ return sorted(kept, key=lambda m: (m.start, m.end))
147
+
148
+
149
+ def find(
150
+ text: str,
151
+ entities: Iterable[str] | None = None,
152
+ min_score: float = 0.0,
153
+ resolve: bool = True,
154
+ report_invalid: bool = False,
155
+ ) -> list[Match]:
156
+ """EN: Find Brazilian identifiers in text.
157
+ entities: subset of ids (e.g. ["BR_CPF"]), default all. min_score: drop anything below.
158
+ resolve: resolve overlaps between entities (default True).
159
+ report_invalid: also return N1 look-alikes with a WRONG check digit (valid_dv=False, score 0), e.g. typos.
160
+ PT: Acha identificadores brasileiros no texto.
161
+ entities: subconjunto de ids (ex: ["BR_CPF"]), padrao todos. min_score: descarta abaixo disso.
162
+ resolve: resolve sobreposicao entre entidades (padrao True).
163
+ report_invalid: devolve tb parecidos N1 c/ DV ERRADO (valid_dv=False, score 0), ex: erro de digitacao.
164
+ """
165
+ # [FIND] type check
166
+ if not isinstance(text, str):
167
+ raise TypeError("text must be str / text precisa ser str")
168
+ ids = list(entities) if entities is not None else list(ENTITIES)
169
+ unknown = [e for e in ids if e not in ENTITIES]
170
+ if unknown:
171
+ raise ValueError(f"unknown entities / entidades desconhecidas: {unknown}")
172
+ # [FIND-PREP] normalise once, fold once
173
+ norm = normalise_text(text)
174
+ folded = fold(norm)
175
+ found: list[Match] = []
176
+ for eid in ids:
177
+ found.extend(_candidates(norm, folded, text, ENTITIES[eid], report_invalid))
178
+ # [FIND-SPLIT] suspects never compete with valid matches
179
+ suspects = [m for m in found if not m.valid_dv]
180
+ found = [m for m in found if m.valid_dv and m.score >= min_score]
181
+ # [FIND-RESOLVE]
182
+ kept = resolve_overlaps(found, ids) if resolve else sorted(found, key=lambda m: (m.start, m.end))
183
+ if suspects:
184
+ # a suspect survives only where no valid match sits
185
+ free = [x for x in suspects if all(x.end <= k.start or x.start >= k.end for k in kept)]
186
+ kept = sorted(kept + resolve_overlaps(free, ids), key=lambda m: (m.start, m.end))
187
+ return kept