datajig 0.5.0.dev0__py3-none-macosx_11_0_arm64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- datajig/__init__.py +43 -0
- datajig/agent.py +270 -0
- datajig/analyzers/__init__.py +2 -0
- datajig/analyzers/changes.py +57 -0
- datajig/analyzers/distribution.py +114 -0
- datajig/analyzers/leakage.py +125 -0
- datajig/api.py +225 -0
- datajig/bin/datajig-core +0 -0
- datajig/cache.py +89 -0
- datajig/cli.py +359 -0
- datajig/hashing.py +23 -0
- datajig/inventory.py +210 -0
- datajig/layout.py +42 -0
- datajig/manifest.py +256 -0
- datajig/manifest_diff.py +182 -0
- datajig/matching.py +305 -0
- datajig/models.py +230 -0
- datajig/native.py +350 -0
- datajig/phash_index.py +79 -0
- datajig/policy.py +66 -0
- datajig/reporting/__init__.py +2 -0
- datajig/reporting/html.py +171 -0
- datajig/reporting/json.py +210 -0
- datajig/reporting/templates/review.html.j2 +53 -0
- datajig/snapshot.py +105 -0
- datajig/source.py +135 -0
- datajig-0.5.0.dev0.dist-info/METADATA +564 -0
- datajig-0.5.0.dev0.dist-info/RECORD +30 -0
- datajig-0.5.0.dev0.dist-info/WHEEL +6 -0
- datajig-0.5.0.dev0.dist-info/entry_points.txt +2 -0
datajig/__init__.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from importlib import import_module
|
|
4
|
+
from typing import TYPE_CHECKING, Any
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from datajig.api import compare
|
|
8
|
+
from datajig.models import (
|
|
9
|
+
DistributionDelta,
|
|
10
|
+
Finding,
|
|
11
|
+
PolicyResult,
|
|
12
|
+
ReviewReport,
|
|
13
|
+
ReviewStatus,
|
|
14
|
+
SampleMatch,
|
|
15
|
+
SampleRecord,
|
|
16
|
+
Severity,
|
|
17
|
+
)
|
|
18
|
+
from datajig.policy import PolicyConfig
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"DistributionDelta",
|
|
22
|
+
"Finding",
|
|
23
|
+
"PolicyConfig",
|
|
24
|
+
"PolicyResult",
|
|
25
|
+
"ReviewReport",
|
|
26
|
+
"ReviewStatus",
|
|
27
|
+
"SampleMatch",
|
|
28
|
+
"SampleRecord",
|
|
29
|
+
"Severity",
|
|
30
|
+
"compare",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
_MODEL_EXPORTS = frozenset(__all__) - {"PolicyConfig", "compare"}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def __getattr__(name: str) -> Any:
|
|
37
|
+
if name == "compare":
|
|
38
|
+
return import_module("datajig.api").compare
|
|
39
|
+
if name == "PolicyConfig":
|
|
40
|
+
return import_module("datajig.policy").PolicyConfig
|
|
41
|
+
if name in _MODEL_EXPORTS:
|
|
42
|
+
return getattr(import_module("datajig.models"), name)
|
|
43
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
datajig/agent.py
ADDED
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
6
|
+
|
|
7
|
+
from datajig.models import Finding, ReviewReport, Severity, indexed_findings
|
|
8
|
+
|
|
9
|
+
AGENT_API_VERSION = 1
|
|
10
|
+
DEFAULT_PAGE_SIZE = 50
|
|
11
|
+
MAX_PAGE_SIZE = 200
|
|
12
|
+
MAX_COMPACT_FINDINGS = 20
|
|
13
|
+
MAX_COMPACT_CODES = 50
|
|
14
|
+
MAX_COMPACT_SAMPLE_IDS = 5
|
|
15
|
+
MAX_COMPACT_EVIDENCE_ITEMS = 8
|
|
16
|
+
MAX_COMPACT_SEQUENCE_ITEMS = 5
|
|
17
|
+
MAX_COMPACT_TEXT_CHARS = 240
|
|
18
|
+
MAX_COMPACT_JSON_CHARS = 50_000
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class FindingNotFoundError(LookupError):
|
|
22
|
+
"""Raised when a report does not contain a requested finding ID."""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def capabilities() -> dict[str, object]:
|
|
26
|
+
return {
|
|
27
|
+
"agent_api_version": AGENT_API_VERSION,
|
|
28
|
+
"kind": "capabilities",
|
|
29
|
+
"tool": {"name": "datajig", "version": _tool_version()},
|
|
30
|
+
"report_schema_versions": [1],
|
|
31
|
+
"identity_namespace": "datajig-v1",
|
|
32
|
+
"commands": [
|
|
33
|
+
"capabilities",
|
|
34
|
+
"compare",
|
|
35
|
+
"explain",
|
|
36
|
+
"finding",
|
|
37
|
+
"findings",
|
|
38
|
+
"snapshot",
|
|
39
|
+
"snapshot-diff",
|
|
40
|
+
"snapshot-info",
|
|
41
|
+
],
|
|
42
|
+
"limits": {
|
|
43
|
+
"default_page_size": DEFAULT_PAGE_SIZE,
|
|
44
|
+
"max_page_size": MAX_PAGE_SIZE,
|
|
45
|
+
"default_compact_findings": 10,
|
|
46
|
+
"max_compact_findings": MAX_COMPACT_FINDINGS,
|
|
47
|
+
"max_compact_characters": MAX_COMPACT_JSON_CHARS,
|
|
48
|
+
},
|
|
49
|
+
"features": {
|
|
50
|
+
"compact_output": True,
|
|
51
|
+
"deterministic_finding_ids": True,
|
|
52
|
+
"filtered_pagination": True,
|
|
53
|
+
"native_manifest_backend": True,
|
|
54
|
+
"native_manifest_fallback": False,
|
|
55
|
+
"read_only_queries": True,
|
|
56
|
+
"snapshot_manifests": True,
|
|
57
|
+
},
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def list_findings(
|
|
62
|
+
report: ReviewReport,
|
|
63
|
+
*,
|
|
64
|
+
severities: tuple[Severity, ...] = (),
|
|
65
|
+
codes: tuple[str, ...] = (),
|
|
66
|
+
offset: int = 0,
|
|
67
|
+
limit: int = DEFAULT_PAGE_SIZE,
|
|
68
|
+
) -> dict[str, object]:
|
|
69
|
+
_validate_page(offset, limit)
|
|
70
|
+
severity_filter = set(severities)
|
|
71
|
+
code_filter = set(codes)
|
|
72
|
+
filtered = [
|
|
73
|
+
(finding_id, finding)
|
|
74
|
+
for finding_id, finding in indexed_findings(report)
|
|
75
|
+
if (not severity_filter or finding.severity in severity_filter)
|
|
76
|
+
and (not code_filter or finding.code in code_filter)
|
|
77
|
+
]
|
|
78
|
+
page = filtered[offset : offset + limit]
|
|
79
|
+
return {
|
|
80
|
+
"agent_api_version": AGENT_API_VERSION,
|
|
81
|
+
"kind": "finding_page",
|
|
82
|
+
"report": _report_descriptor(report),
|
|
83
|
+
"filters": {
|
|
84
|
+
"severities": sorted(item.value for item in severity_filter),
|
|
85
|
+
"codes": sorted(code_filter),
|
|
86
|
+
},
|
|
87
|
+
"page": {
|
|
88
|
+
"offset": offset,
|
|
89
|
+
"limit": limit,
|
|
90
|
+
"returned": len(page),
|
|
91
|
+
"total": len(filtered),
|
|
92
|
+
"has_more": offset + len(page) < len(filtered),
|
|
93
|
+
},
|
|
94
|
+
"findings": [
|
|
95
|
+
_finding_payload(report, finding_id, finding)
|
|
96
|
+
for finding_id, finding in page
|
|
97
|
+
],
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def get_finding(report: ReviewReport, finding_id: str) -> dict[str, object]:
|
|
102
|
+
for current_id, finding in indexed_findings(report):
|
|
103
|
+
if current_id == finding_id:
|
|
104
|
+
return {
|
|
105
|
+
"agent_api_version": AGENT_API_VERSION,
|
|
106
|
+
"kind": "finding",
|
|
107
|
+
"report": _report_descriptor(report),
|
|
108
|
+
"finding": _finding_payload(report, current_id, finding),
|
|
109
|
+
}
|
|
110
|
+
raise FindingNotFoundError(f"finding {finding_id!r} was not found")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def compact_summary(report: ReviewReport, *, limit: int = 10) -> dict[str, object]:
|
|
114
|
+
if limit < 0 or limit > MAX_COMPACT_FINDINGS:
|
|
115
|
+
raise ValueError(f"limit must be between 0 and {MAX_COMPACT_FINDINGS}")
|
|
116
|
+
indexed = indexed_findings(report)
|
|
117
|
+
severity_counts = Counter(finding.severity.value for _, finding in indexed)
|
|
118
|
+
code_counts = Counter(finding.code for _, finding in indexed)
|
|
119
|
+
selected = indexed[:limit]
|
|
120
|
+
compact_codes = _compact_code_counts(code_counts)
|
|
121
|
+
compact_findings = [
|
|
122
|
+
_compact_finding_payload(report, finding_id, finding)
|
|
123
|
+
for finding_id, finding in selected
|
|
124
|
+
]
|
|
125
|
+
result: dict[str, object] = {
|
|
126
|
+
"agent_api_version": AGENT_API_VERSION,
|
|
127
|
+
"kind": "compact_summary",
|
|
128
|
+
"report": _compact_report_descriptor(report),
|
|
129
|
+
"counts": {
|
|
130
|
+
"total": len(indexed),
|
|
131
|
+
"by_severity": {
|
|
132
|
+
severity.value: severity_counts.get(severity.value, 0)
|
|
133
|
+
for severity in sorted(Severity, key=lambda item: item.value)
|
|
134
|
+
},
|
|
135
|
+
"by_code": compact_codes,
|
|
136
|
+
},
|
|
137
|
+
"code_kinds_total": len(code_counts),
|
|
138
|
+
"code_kinds_truncated": len(code_counts) > len(compact_codes),
|
|
139
|
+
"findings": compact_findings,
|
|
140
|
+
"truncated": len(indexed) > len(compact_findings),
|
|
141
|
+
"budget_truncated": False,
|
|
142
|
+
}
|
|
143
|
+
while compact_findings and _json_size(result) > MAX_COMPACT_JSON_CHARS:
|
|
144
|
+
compact_findings.pop()
|
|
145
|
+
result["truncated"] = True
|
|
146
|
+
result["budget_truncated"] = True
|
|
147
|
+
return result
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _finding_payload(
|
|
151
|
+
report: ReviewReport,
|
|
152
|
+
finding_id: str,
|
|
153
|
+
finding: Finding,
|
|
154
|
+
) -> dict[str, object]:
|
|
155
|
+
configured = dict(report.policy.effective_policy).get(
|
|
156
|
+
finding.code, finding.severity.value
|
|
157
|
+
)
|
|
158
|
+
effective_severity = (
|
|
159
|
+
configured if isinstance(configured, str) else finding.severity.value
|
|
160
|
+
)
|
|
161
|
+
return {
|
|
162
|
+
"id": finding_id,
|
|
163
|
+
"code": finding.code,
|
|
164
|
+
"severity": finding.severity.value,
|
|
165
|
+
"effective_severity": effective_severity,
|
|
166
|
+
"message": finding.message,
|
|
167
|
+
"sample_ids": sorted(finding.sample_ids),
|
|
168
|
+
"evidence": dict(sorted(finding.evidence)),
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _report_descriptor(report: ReviewReport) -> dict[str, object]:
|
|
173
|
+
return {
|
|
174
|
+
"schema_version": report.schema_version,
|
|
175
|
+
"baseline": report.baseline,
|
|
176
|
+
"candidate": report.candidate,
|
|
177
|
+
"complete": report.complete,
|
|
178
|
+
"status": report.policy.status.value,
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _compact_report_descriptor(report: ReviewReport) -> dict[str, object]:
|
|
183
|
+
descriptor = _report_descriptor(report)
|
|
184
|
+
descriptor["baseline"] = _compact_text(report.baseline)
|
|
185
|
+
descriptor["candidate"] = _compact_text(report.candidate)
|
|
186
|
+
return descriptor
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _compact_finding_payload(
|
|
190
|
+
report: ReviewReport,
|
|
191
|
+
finding_id: str,
|
|
192
|
+
finding: Finding,
|
|
193
|
+
) -> dict[str, object]:
|
|
194
|
+
configured = dict(report.policy.effective_policy).get(
|
|
195
|
+
finding.code, finding.severity.value
|
|
196
|
+
)
|
|
197
|
+
effective_severity = (
|
|
198
|
+
configured if isinstance(configured, str) else finding.severity.value
|
|
199
|
+
)
|
|
200
|
+
sample_ids = sorted(finding.sample_ids)
|
|
201
|
+
evidence_items = sorted(finding.evidence)[:MAX_COMPACT_EVIDENCE_ITEMS]
|
|
202
|
+
return {
|
|
203
|
+
"id": finding_id,
|
|
204
|
+
"code": _compact_text(finding.code),
|
|
205
|
+
"severity": finding.severity.value,
|
|
206
|
+
"effective_severity": effective_severity,
|
|
207
|
+
"message": _compact_text(finding.message),
|
|
208
|
+
"message_truncated": len(finding.message) > MAX_COMPACT_TEXT_CHARS,
|
|
209
|
+
"sample_ids": [
|
|
210
|
+
_compact_text(item) for item in sample_ids[:MAX_COMPACT_SAMPLE_IDS]
|
|
211
|
+
],
|
|
212
|
+
"sample_ids_total": len(sample_ids),
|
|
213
|
+
"sample_ids_truncated": len(sample_ids) > MAX_COMPACT_SAMPLE_IDS,
|
|
214
|
+
"evidence": {
|
|
215
|
+
_compact_text(key): _compact_evidence_value(value)
|
|
216
|
+
for key, value in evidence_items
|
|
217
|
+
},
|
|
218
|
+
"evidence_total": len(finding.evidence),
|
|
219
|
+
"evidence_truncated": len(finding.evidence) > len(evidence_items),
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _compact_code_counts(counts: Counter[str]) -> dict[str, int]:
|
|
224
|
+
compact: dict[str, int] = {}
|
|
225
|
+
ordered = sorted(counts.items(), key=lambda item: (-item[1], item[0]))
|
|
226
|
+
for code, count in ordered[:MAX_COMPACT_CODES]:
|
|
227
|
+
key = _compact_text(code)
|
|
228
|
+
compact[key] = compact.get(key, 0) + count
|
|
229
|
+
return dict(sorted(compact.items()))
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _compact_evidence_value(value: object) -> object:
|
|
233
|
+
if isinstance(value, tuple):
|
|
234
|
+
return {
|
|
235
|
+
"items": [
|
|
236
|
+
_compact_evidence_value(item)
|
|
237
|
+
for item in value[:MAX_COMPACT_SEQUENCE_ITEMS]
|
|
238
|
+
],
|
|
239
|
+
"total": len(value),
|
|
240
|
+
"truncated": len(value) > MAX_COMPACT_SEQUENCE_ITEMS,
|
|
241
|
+
}
|
|
242
|
+
if isinstance(value, str):
|
|
243
|
+
return _compact_text(value)
|
|
244
|
+
return value
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _compact_text(value: str) -> str:
|
|
248
|
+
if len(value) <= MAX_COMPACT_TEXT_CHARS:
|
|
249
|
+
return value
|
|
250
|
+
return f"{value[: MAX_COMPACT_TEXT_CHARS - 1]}…"
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _json_size(value: object) -> int:
|
|
254
|
+
return len(
|
|
255
|
+
json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _validate_page(offset: int, limit: int) -> None:
|
|
260
|
+
if offset < 0:
|
|
261
|
+
raise ValueError("offset must be non-negative")
|
|
262
|
+
if limit < 1 or limit > MAX_PAGE_SIZE:
|
|
263
|
+
raise ValueError(f"limit must be between 1 and {MAX_PAGE_SIZE}")
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _tool_version() -> str:
|
|
267
|
+
try:
|
|
268
|
+
return version("datajig")
|
|
269
|
+
except PackageNotFoundError:
|
|
270
|
+
return "0+unknown"
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from datajig.matching import MatchResult
|
|
4
|
+
from datajig.models import Finding, Severity
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def analyze_changes(result: MatchResult) -> tuple[Finding, ...]:
|
|
8
|
+
before_by_path = {item.relative_path: item for item in result.baseline}
|
|
9
|
+
after_by_path = {item.relative_path: item for item in result.candidate}
|
|
10
|
+
findings: list[Finding] = list(result.findings)
|
|
11
|
+
|
|
12
|
+
for match in result.matches:
|
|
13
|
+
if match.confidence == "ambiguous":
|
|
14
|
+
continue
|
|
15
|
+
before = before_by_path[match.baseline_id]
|
|
16
|
+
after = after_by_path[match.candidate_id]
|
|
17
|
+
sample_ids = (before.relative_path, after.relative_path)
|
|
18
|
+
if before.split != after.split:
|
|
19
|
+
findings.append(_finding("SPLIT_CHANGED", "sample changed split", sample_ids))
|
|
20
|
+
if before.label != after.label:
|
|
21
|
+
findings.append(_finding("LABEL_CHANGED", "sample changed label", sample_ids))
|
|
22
|
+
if before.logical_path != after.logical_path:
|
|
23
|
+
findings.append(_finding("SAMPLE_MOVED", "sample path changed", sample_ids))
|
|
24
|
+
if before.content_hash != after.content_hash:
|
|
25
|
+
code = (
|
|
26
|
+
"PROBABLE_REENCODE"
|
|
27
|
+
if match.distance is not None and match.distance <= result.phash_threshold
|
|
28
|
+
else "CONTENT_CHANGED"
|
|
29
|
+
)
|
|
30
|
+
message = (
|
|
31
|
+
"sample bytes changed but perceptual content is similar"
|
|
32
|
+
if code == "PROBABLE_REENCODE"
|
|
33
|
+
else "sample content changed"
|
|
34
|
+
)
|
|
35
|
+
findings.append(_finding(code, message, sample_ids, distance=match.distance))
|
|
36
|
+
|
|
37
|
+
for item in result.unmatched_candidate:
|
|
38
|
+
findings.append(
|
|
39
|
+
_finding("SAMPLE_ADDED", "sample was added", (item.relative_path,))
|
|
40
|
+
)
|
|
41
|
+
for item in result.unmatched_baseline:
|
|
42
|
+
findings.append(
|
|
43
|
+
_finding("SAMPLE_REMOVED", "sample was removed", (item.relative_path,))
|
|
44
|
+
)
|
|
45
|
+
findings.sort(key=lambda item: (item.code, item.sample_ids))
|
|
46
|
+
return tuple(findings)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _finding(
|
|
50
|
+
code: str,
|
|
51
|
+
message: str,
|
|
52
|
+
sample_ids: tuple[str, ...],
|
|
53
|
+
*,
|
|
54
|
+
distance: int | None = None,
|
|
55
|
+
) -> Finding:
|
|
56
|
+
evidence = () if distance is None else (("perceptual_distance", distance),)
|
|
57
|
+
return Finding(code, Severity.INFO, message, sample_ids, evidence)
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
from datajig.inventory import InventoryResult
|
|
7
|
+
from datajig.models import DistributionDelta, SampleRecord
|
|
8
|
+
|
|
9
|
+
SIZE_BUCKET_BOUNDARIES = (256, 512, 1024)
|
|
10
|
+
ASPECT_RATIO_BOUNDARIES = (0.8, 1.25)
|
|
11
|
+
BUCKET_BOUNDARIES = (
|
|
12
|
+
"size: <256, 256-511, 512-1023, >=1024, unknown",
|
|
13
|
+
"aspect_ratio: portrait <0.8, square 0.8-1.25, landscape >1.25, unknown",
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class DistributionBucket:
|
|
19
|
+
dimension: str
|
|
20
|
+
key: str
|
|
21
|
+
count: int
|
|
22
|
+
proportion: float
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True, slots=True)
|
|
26
|
+
class DistributionSummary:
|
|
27
|
+
total: int
|
|
28
|
+
buckets: tuple[DistributionBucket, ...]
|
|
29
|
+
bucket_boundaries: tuple[str, ...] = BUCKET_BOUNDARIES
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def summarize_distribution(inventory: InventoryResult) -> DistributionSummary:
|
|
33
|
+
total = len(inventory.records)
|
|
34
|
+
counts: Counter[tuple[str, str]] = Counter()
|
|
35
|
+
for record in inventory.records:
|
|
36
|
+
for dimension, key in _dimensions(record):
|
|
37
|
+
counts[(dimension, key)] += 1
|
|
38
|
+
buckets = tuple(
|
|
39
|
+
DistributionBucket(
|
|
40
|
+
dimension,
|
|
41
|
+
key,
|
|
42
|
+
count,
|
|
43
|
+
count / total if total else 0.0,
|
|
44
|
+
)
|
|
45
|
+
for (dimension, key), count in sorted(counts.items())
|
|
46
|
+
)
|
|
47
|
+
return DistributionSummary(total, buckets)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def compare_distributions(
|
|
51
|
+
before: DistributionSummary, after: DistributionSummary
|
|
52
|
+
) -> tuple[DistributionDelta, ...]:
|
|
53
|
+
before_map = {(item.dimension, item.key): item for item in before.buckets}
|
|
54
|
+
after_map = {(item.dimension, item.key): item for item in after.buckets}
|
|
55
|
+
deltas: list[DistributionDelta] = []
|
|
56
|
+
for dimension, key in sorted(before_map.keys() | after_map.keys()):
|
|
57
|
+
before_item = before_map.get((dimension, key))
|
|
58
|
+
after_item = after_map.get((dimension, key))
|
|
59
|
+
before_count = before_item.count if before_item else 0
|
|
60
|
+
after_count = after_item.count if after_item else 0
|
|
61
|
+
before_proportion = before_item.proportion if before_item else 0.0
|
|
62
|
+
after_proportion = after_item.proportion if after_item else 0.0
|
|
63
|
+
percentage_delta = (
|
|
64
|
+
((after_proportion - before_proportion) / before_proportion) * 100
|
|
65
|
+
if before_proportion
|
|
66
|
+
else None
|
|
67
|
+
)
|
|
68
|
+
deltas.append(
|
|
69
|
+
DistributionDelta(
|
|
70
|
+
dimension,
|
|
71
|
+
key,
|
|
72
|
+
before_count,
|
|
73
|
+
after_count,
|
|
74
|
+
before_proportion,
|
|
75
|
+
after_proportion,
|
|
76
|
+
percentage_delta,
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
return tuple(deltas)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _dimensions(record: SampleRecord) -> tuple[tuple[str, str], ...]:
|
|
83
|
+
return (
|
|
84
|
+
("split", record.split),
|
|
85
|
+
("label", record.label),
|
|
86
|
+
("width", _size_bucket(record.width)),
|
|
87
|
+
("height", _size_bucket(record.height)),
|
|
88
|
+
("aspect_ratio", _aspect_bucket(record.width, record.height)),
|
|
89
|
+
("format", record.media_format or "unknown"),
|
|
90
|
+
("channels", str(record.channels) if record.channels is not None else "unknown"),
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _size_bucket(value: int | None) -> str:
|
|
95
|
+
if value is None:
|
|
96
|
+
return "unknown"
|
|
97
|
+
if value < SIZE_BUCKET_BOUNDARIES[0]:
|
|
98
|
+
return "<256"
|
|
99
|
+
if value < SIZE_BUCKET_BOUNDARIES[1]:
|
|
100
|
+
return "256-511"
|
|
101
|
+
if value < SIZE_BUCKET_BOUNDARIES[2]:
|
|
102
|
+
return "512-1023"
|
|
103
|
+
return ">=1024"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _aspect_bucket(width: int | None, height: int | None) -> str:
|
|
107
|
+
if width is None or height is None or height == 0:
|
|
108
|
+
return "unknown"
|
|
109
|
+
ratio = width / height
|
|
110
|
+
if ratio < ASPECT_RATIO_BOUNDARIES[0]:
|
|
111
|
+
return "portrait"
|
|
112
|
+
if ratio <= ASPECT_RATIO_BOUNDARIES[1]:
|
|
113
|
+
return "square"
|
|
114
|
+
return "landscape"
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections import defaultdict
|
|
4
|
+
|
|
5
|
+
from datajig.inventory import InventoryResult
|
|
6
|
+
from datajig.models import Finding, SampleRecord, Severity
|
|
7
|
+
from datajig.phash_index import PhashIndex
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def analyze_leakage(
|
|
11
|
+
inventory: InventoryResult, threshold: int
|
|
12
|
+
) -> tuple[Finding, ...]:
|
|
13
|
+
if threshold < 0:
|
|
14
|
+
raise ValueError("threshold must be non-negative")
|
|
15
|
+
findings: list[Finding] = []
|
|
16
|
+
by_digest: dict[str, list[SampleRecord]] = defaultdict(list)
|
|
17
|
+
for record in inventory.records:
|
|
18
|
+
by_digest[record.content_hash].append(record)
|
|
19
|
+
for digest in sorted(by_digest):
|
|
20
|
+
group = sorted(by_digest[digest], key=lambda item: item.relative_path)
|
|
21
|
+
if len(group) < 2:
|
|
22
|
+
continue
|
|
23
|
+
sample_ids = tuple(item.relative_path for item in group)
|
|
24
|
+
exact_evidence = (("group_size", len(group)), ("content_hash", digest))
|
|
25
|
+
findings.append(
|
|
26
|
+
Finding(
|
|
27
|
+
"EXACT_DUPLICATE_GROUP",
|
|
28
|
+
Severity.WARNING,
|
|
29
|
+
"identical image bytes appear more than once",
|
|
30
|
+
sample_ids,
|
|
31
|
+
exact_evidence,
|
|
32
|
+
)
|
|
33
|
+
)
|
|
34
|
+
if len({item.split for item in group}) > 1:
|
|
35
|
+
findings.append(
|
|
36
|
+
Finding(
|
|
37
|
+
"CROSS_SPLIT_EXACT_LEAKAGE",
|
|
38
|
+
Severity.ERROR,
|
|
39
|
+
"identical image bytes appear across splits",
|
|
40
|
+
sample_ids,
|
|
41
|
+
exact_evidence,
|
|
42
|
+
)
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
for group in _near_components(inventory.records, threshold):
|
|
46
|
+
sample_ids = tuple(item.relative_path for item in group)
|
|
47
|
+
near_evidence = (("group_size", len(group)), ("threshold", threshold))
|
|
48
|
+
findings.append(
|
|
49
|
+
Finding(
|
|
50
|
+
"NEAR_DUPLICATE_GROUP",
|
|
51
|
+
Severity.WARNING,
|
|
52
|
+
"perceptually similar images form a duplicate candidate group",
|
|
53
|
+
sample_ids,
|
|
54
|
+
near_evidence,
|
|
55
|
+
)
|
|
56
|
+
)
|
|
57
|
+
if len({item.split for item in group}) > 1:
|
|
58
|
+
findings.append(
|
|
59
|
+
Finding(
|
|
60
|
+
"CROSS_SPLIT_NEAR_LEAKAGE",
|
|
61
|
+
Severity.WARNING,
|
|
62
|
+
"perceptually similar images appear across splits",
|
|
63
|
+
sample_ids,
|
|
64
|
+
near_evidence,
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
findings.sort(key=lambda item: (item.code, item.sample_ids))
|
|
68
|
+
return tuple(findings)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _near_components(
|
|
72
|
+
records: tuple[SampleRecord, ...], threshold: int
|
|
73
|
+
) -> list[list[SampleRecord]]:
|
|
74
|
+
records = tuple(sorted(records, key=lambda item: item.relative_path))
|
|
75
|
+
groups: dict[str, dict[str, list[int]]] = defaultdict(lambda: defaultdict(list))
|
|
76
|
+
for index, record in enumerate(records):
|
|
77
|
+
if record.perceptual_hash is not None:
|
|
78
|
+
groups[record.perceptual_hash][record.content_hash].append(index)
|
|
79
|
+
|
|
80
|
+
parents = list(range(len(records)))
|
|
81
|
+
|
|
82
|
+
def find(index: int) -> int:
|
|
83
|
+
while parents[index] != index:
|
|
84
|
+
parents[index] = parents[parents[index]]
|
|
85
|
+
index = parents[index]
|
|
86
|
+
return index
|
|
87
|
+
|
|
88
|
+
def union(left: int, right: int) -> None:
|
|
89
|
+
left_root = find(left)
|
|
90
|
+
right_root = find(right)
|
|
91
|
+
if left_root != right_root:
|
|
92
|
+
parents[right_root] = left_root
|
|
93
|
+
|
|
94
|
+
def connect(left_group: list[int], right_group: list[int]) -> None:
|
|
95
|
+
for index in left_group[1:]:
|
|
96
|
+
union(left_group[0], index)
|
|
97
|
+
for index in right_group[1:]:
|
|
98
|
+
union(right_group[0], index)
|
|
99
|
+
union(left_group[0], right_group[0])
|
|
100
|
+
|
|
101
|
+
phash_index: PhashIndex[str] = PhashIndex()
|
|
102
|
+
for hash_value in sorted(groups):
|
|
103
|
+
digest_groups = groups[hash_value]
|
|
104
|
+
digest_values = sorted(digest_groups)
|
|
105
|
+
if len(digest_values) > 1:
|
|
106
|
+
anchor = digest_groups[digest_values[0]]
|
|
107
|
+
for other_digest in digest_values[1:]:
|
|
108
|
+
connect(anchor, digest_groups[other_digest])
|
|
109
|
+
for near_match in phash_index.query(hash_value, threshold):
|
|
110
|
+
for other_hash in near_match.items:
|
|
111
|
+
combined = [*digest_groups.items(), *groups[other_hash].items()]
|
|
112
|
+
anchor_digest, anchor_group = combined[0]
|
|
113
|
+
for other_digest, other_group in combined[1:]:
|
|
114
|
+
if anchor_digest != other_digest:
|
|
115
|
+
connect(anchor_group, other_group)
|
|
116
|
+
phash_index.add(hash_value, hash_value)
|
|
117
|
+
|
|
118
|
+
connected: dict[int, list[int]] = defaultdict(list)
|
|
119
|
+
for record_index in range(len(records)):
|
|
120
|
+
connected[find(record_index)].append(record_index)
|
|
121
|
+
return [
|
|
122
|
+
[records[index] for index in indexes]
|
|
123
|
+
for _, indexes in sorted(connected.items())
|
|
124
|
+
if len(indexes) > 1
|
|
125
|
+
]
|