datajig 0.5.0.dev0__py3-none-macosx_11_0_arm64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
datajig/__init__.py ADDED
@@ -0,0 +1,43 @@
1
+ from __future__ import annotations
2
+
3
+ from importlib import import_module
4
+ from typing import TYPE_CHECKING, Any
5
+
6
+ if TYPE_CHECKING:
7
+ from datajig.api import compare
8
+ from datajig.models import (
9
+ DistributionDelta,
10
+ Finding,
11
+ PolicyResult,
12
+ ReviewReport,
13
+ ReviewStatus,
14
+ SampleMatch,
15
+ SampleRecord,
16
+ Severity,
17
+ )
18
+ from datajig.policy import PolicyConfig
19
+
20
+ __all__ = [
21
+ "DistributionDelta",
22
+ "Finding",
23
+ "PolicyConfig",
24
+ "PolicyResult",
25
+ "ReviewReport",
26
+ "ReviewStatus",
27
+ "SampleMatch",
28
+ "SampleRecord",
29
+ "Severity",
30
+ "compare",
31
+ ]
32
+
33
+ _MODEL_EXPORTS = frozenset(__all__) - {"PolicyConfig", "compare"}
34
+
35
+
36
+ def __getattr__(name: str) -> Any:
37
+ if name == "compare":
38
+ return import_module("datajig.api").compare
39
+ if name == "PolicyConfig":
40
+ return import_module("datajig.policy").PolicyConfig
41
+ if name in _MODEL_EXPORTS:
42
+ return getattr(import_module("datajig.models"), name)
43
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
datajig/agent.py ADDED
@@ -0,0 +1,270 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections import Counter
5
+ from importlib.metadata import PackageNotFoundError, version
6
+
7
+ from datajig.models import Finding, ReviewReport, Severity, indexed_findings
8
+
9
+ AGENT_API_VERSION = 1
10
+ DEFAULT_PAGE_SIZE = 50
11
+ MAX_PAGE_SIZE = 200
12
+ MAX_COMPACT_FINDINGS = 20
13
+ MAX_COMPACT_CODES = 50
14
+ MAX_COMPACT_SAMPLE_IDS = 5
15
+ MAX_COMPACT_EVIDENCE_ITEMS = 8
16
+ MAX_COMPACT_SEQUENCE_ITEMS = 5
17
+ MAX_COMPACT_TEXT_CHARS = 240
18
+ MAX_COMPACT_JSON_CHARS = 50_000
19
+
20
+
21
+ class FindingNotFoundError(LookupError):
22
+ """Raised when a report does not contain a requested finding ID."""
23
+
24
+
25
+ def capabilities() -> dict[str, object]:
26
+ return {
27
+ "agent_api_version": AGENT_API_VERSION,
28
+ "kind": "capabilities",
29
+ "tool": {"name": "datajig", "version": _tool_version()},
30
+ "report_schema_versions": [1],
31
+ "identity_namespace": "datajig-v1",
32
+ "commands": [
33
+ "capabilities",
34
+ "compare",
35
+ "explain",
36
+ "finding",
37
+ "findings",
38
+ "snapshot",
39
+ "snapshot-diff",
40
+ "snapshot-info",
41
+ ],
42
+ "limits": {
43
+ "default_page_size": DEFAULT_PAGE_SIZE,
44
+ "max_page_size": MAX_PAGE_SIZE,
45
+ "default_compact_findings": 10,
46
+ "max_compact_findings": MAX_COMPACT_FINDINGS,
47
+ "max_compact_characters": MAX_COMPACT_JSON_CHARS,
48
+ },
49
+ "features": {
50
+ "compact_output": True,
51
+ "deterministic_finding_ids": True,
52
+ "filtered_pagination": True,
53
+ "native_manifest_backend": True,
54
+ "native_manifest_fallback": False,
55
+ "read_only_queries": True,
56
+ "snapshot_manifests": True,
57
+ },
58
+ }
59
+
60
+
61
+ def list_findings(
62
+ report: ReviewReport,
63
+ *,
64
+ severities: tuple[Severity, ...] = (),
65
+ codes: tuple[str, ...] = (),
66
+ offset: int = 0,
67
+ limit: int = DEFAULT_PAGE_SIZE,
68
+ ) -> dict[str, object]:
69
+ _validate_page(offset, limit)
70
+ severity_filter = set(severities)
71
+ code_filter = set(codes)
72
+ filtered = [
73
+ (finding_id, finding)
74
+ for finding_id, finding in indexed_findings(report)
75
+ if (not severity_filter or finding.severity in severity_filter)
76
+ and (not code_filter or finding.code in code_filter)
77
+ ]
78
+ page = filtered[offset : offset + limit]
79
+ return {
80
+ "agent_api_version": AGENT_API_VERSION,
81
+ "kind": "finding_page",
82
+ "report": _report_descriptor(report),
83
+ "filters": {
84
+ "severities": sorted(item.value for item in severity_filter),
85
+ "codes": sorted(code_filter),
86
+ },
87
+ "page": {
88
+ "offset": offset,
89
+ "limit": limit,
90
+ "returned": len(page),
91
+ "total": len(filtered),
92
+ "has_more": offset + len(page) < len(filtered),
93
+ },
94
+ "findings": [
95
+ _finding_payload(report, finding_id, finding)
96
+ for finding_id, finding in page
97
+ ],
98
+ }
99
+
100
+
101
+ def get_finding(report: ReviewReport, finding_id: str) -> dict[str, object]:
102
+ for current_id, finding in indexed_findings(report):
103
+ if current_id == finding_id:
104
+ return {
105
+ "agent_api_version": AGENT_API_VERSION,
106
+ "kind": "finding",
107
+ "report": _report_descriptor(report),
108
+ "finding": _finding_payload(report, current_id, finding),
109
+ }
110
+ raise FindingNotFoundError(f"finding {finding_id!r} was not found")
111
+
112
+
113
+ def compact_summary(report: ReviewReport, *, limit: int = 10) -> dict[str, object]:
114
+ if limit < 0 or limit > MAX_COMPACT_FINDINGS:
115
+ raise ValueError(f"limit must be between 0 and {MAX_COMPACT_FINDINGS}")
116
+ indexed = indexed_findings(report)
117
+ severity_counts = Counter(finding.severity.value for _, finding in indexed)
118
+ code_counts = Counter(finding.code for _, finding in indexed)
119
+ selected = indexed[:limit]
120
+ compact_codes = _compact_code_counts(code_counts)
121
+ compact_findings = [
122
+ _compact_finding_payload(report, finding_id, finding)
123
+ for finding_id, finding in selected
124
+ ]
125
+ result: dict[str, object] = {
126
+ "agent_api_version": AGENT_API_VERSION,
127
+ "kind": "compact_summary",
128
+ "report": _compact_report_descriptor(report),
129
+ "counts": {
130
+ "total": len(indexed),
131
+ "by_severity": {
132
+ severity.value: severity_counts.get(severity.value, 0)
133
+ for severity in sorted(Severity, key=lambda item: item.value)
134
+ },
135
+ "by_code": compact_codes,
136
+ },
137
+ "code_kinds_total": len(code_counts),
138
+ "code_kinds_truncated": len(code_counts) > len(compact_codes),
139
+ "findings": compact_findings,
140
+ "truncated": len(indexed) > len(compact_findings),
141
+ "budget_truncated": False,
142
+ }
143
+ while compact_findings and _json_size(result) > MAX_COMPACT_JSON_CHARS:
144
+ compact_findings.pop()
145
+ result["truncated"] = True
146
+ result["budget_truncated"] = True
147
+ return result
148
+
149
+
150
+ def _finding_payload(
151
+ report: ReviewReport,
152
+ finding_id: str,
153
+ finding: Finding,
154
+ ) -> dict[str, object]:
155
+ configured = dict(report.policy.effective_policy).get(
156
+ finding.code, finding.severity.value
157
+ )
158
+ effective_severity = (
159
+ configured if isinstance(configured, str) else finding.severity.value
160
+ )
161
+ return {
162
+ "id": finding_id,
163
+ "code": finding.code,
164
+ "severity": finding.severity.value,
165
+ "effective_severity": effective_severity,
166
+ "message": finding.message,
167
+ "sample_ids": sorted(finding.sample_ids),
168
+ "evidence": dict(sorted(finding.evidence)),
169
+ }
170
+
171
+
172
+ def _report_descriptor(report: ReviewReport) -> dict[str, object]:
173
+ return {
174
+ "schema_version": report.schema_version,
175
+ "baseline": report.baseline,
176
+ "candidate": report.candidate,
177
+ "complete": report.complete,
178
+ "status": report.policy.status.value,
179
+ }
180
+
181
+
182
+ def _compact_report_descriptor(report: ReviewReport) -> dict[str, object]:
183
+ descriptor = _report_descriptor(report)
184
+ descriptor["baseline"] = _compact_text(report.baseline)
185
+ descriptor["candidate"] = _compact_text(report.candidate)
186
+ return descriptor
187
+
188
+
189
+ def _compact_finding_payload(
190
+ report: ReviewReport,
191
+ finding_id: str,
192
+ finding: Finding,
193
+ ) -> dict[str, object]:
194
+ configured = dict(report.policy.effective_policy).get(
195
+ finding.code, finding.severity.value
196
+ )
197
+ effective_severity = (
198
+ configured if isinstance(configured, str) else finding.severity.value
199
+ )
200
+ sample_ids = sorted(finding.sample_ids)
201
+ evidence_items = sorted(finding.evidence)[:MAX_COMPACT_EVIDENCE_ITEMS]
202
+ return {
203
+ "id": finding_id,
204
+ "code": _compact_text(finding.code),
205
+ "severity": finding.severity.value,
206
+ "effective_severity": effective_severity,
207
+ "message": _compact_text(finding.message),
208
+ "message_truncated": len(finding.message) > MAX_COMPACT_TEXT_CHARS,
209
+ "sample_ids": [
210
+ _compact_text(item) for item in sample_ids[:MAX_COMPACT_SAMPLE_IDS]
211
+ ],
212
+ "sample_ids_total": len(sample_ids),
213
+ "sample_ids_truncated": len(sample_ids) > MAX_COMPACT_SAMPLE_IDS,
214
+ "evidence": {
215
+ _compact_text(key): _compact_evidence_value(value)
216
+ for key, value in evidence_items
217
+ },
218
+ "evidence_total": len(finding.evidence),
219
+ "evidence_truncated": len(finding.evidence) > len(evidence_items),
220
+ }
221
+
222
+
223
+ def _compact_code_counts(counts: Counter[str]) -> dict[str, int]:
224
+ compact: dict[str, int] = {}
225
+ ordered = sorted(counts.items(), key=lambda item: (-item[1], item[0]))
226
+ for code, count in ordered[:MAX_COMPACT_CODES]:
227
+ key = _compact_text(code)
228
+ compact[key] = compact.get(key, 0) + count
229
+ return dict(sorted(compact.items()))
230
+
231
+
232
+ def _compact_evidence_value(value: object) -> object:
233
+ if isinstance(value, tuple):
234
+ return {
235
+ "items": [
236
+ _compact_evidence_value(item)
237
+ for item in value[:MAX_COMPACT_SEQUENCE_ITEMS]
238
+ ],
239
+ "total": len(value),
240
+ "truncated": len(value) > MAX_COMPACT_SEQUENCE_ITEMS,
241
+ }
242
+ if isinstance(value, str):
243
+ return _compact_text(value)
244
+ return value
245
+
246
+
247
+ def _compact_text(value: str) -> str:
248
+ if len(value) <= MAX_COMPACT_TEXT_CHARS:
249
+ return value
250
+ return f"{value[: MAX_COMPACT_TEXT_CHARS - 1]}…"
251
+
252
+
253
+ def _json_size(value: object) -> int:
254
+ return len(
255
+ json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
256
+ )
257
+
258
+
259
+ def _validate_page(offset: int, limit: int) -> None:
260
+ if offset < 0:
261
+ raise ValueError("offset must be non-negative")
262
+ if limit < 1 or limit > MAX_PAGE_SIZE:
263
+ raise ValueError(f"limit must be between 1 and {MAX_PAGE_SIZE}")
264
+
265
+
266
+ def _tool_version() -> str:
267
+ try:
268
+ return version("datajig")
269
+ except PackageNotFoundError:
270
+ return "0+unknown"
@@ -0,0 +1,2 @@
1
+ """Semantic analyzers for dataset review findings."""
2
+
@@ -0,0 +1,57 @@
1
+ from __future__ import annotations
2
+
3
+ from datajig.matching import MatchResult
4
+ from datajig.models import Finding, Severity
5
+
6
+
7
+ def analyze_changes(result: MatchResult) -> tuple[Finding, ...]:
8
+ before_by_path = {item.relative_path: item for item in result.baseline}
9
+ after_by_path = {item.relative_path: item for item in result.candidate}
10
+ findings: list[Finding] = list(result.findings)
11
+
12
+ for match in result.matches:
13
+ if match.confidence == "ambiguous":
14
+ continue
15
+ before = before_by_path[match.baseline_id]
16
+ after = after_by_path[match.candidate_id]
17
+ sample_ids = (before.relative_path, after.relative_path)
18
+ if before.split != after.split:
19
+ findings.append(_finding("SPLIT_CHANGED", "sample changed split", sample_ids))
20
+ if before.label != after.label:
21
+ findings.append(_finding("LABEL_CHANGED", "sample changed label", sample_ids))
22
+ if before.logical_path != after.logical_path:
23
+ findings.append(_finding("SAMPLE_MOVED", "sample path changed", sample_ids))
24
+ if before.content_hash != after.content_hash:
25
+ code = (
26
+ "PROBABLE_REENCODE"
27
+ if match.distance is not None and match.distance <= result.phash_threshold
28
+ else "CONTENT_CHANGED"
29
+ )
30
+ message = (
31
+ "sample bytes changed but perceptual content is similar"
32
+ if code == "PROBABLE_REENCODE"
33
+ else "sample content changed"
34
+ )
35
+ findings.append(_finding(code, message, sample_ids, distance=match.distance))
36
+
37
+ for item in result.unmatched_candidate:
38
+ findings.append(
39
+ _finding("SAMPLE_ADDED", "sample was added", (item.relative_path,))
40
+ )
41
+ for item in result.unmatched_baseline:
42
+ findings.append(
43
+ _finding("SAMPLE_REMOVED", "sample was removed", (item.relative_path,))
44
+ )
45
+ findings.sort(key=lambda item: (item.code, item.sample_ids))
46
+ return tuple(findings)
47
+
48
+
49
+ def _finding(
50
+ code: str,
51
+ message: str,
52
+ sample_ids: tuple[str, ...],
53
+ *,
54
+ distance: int | None = None,
55
+ ) -> Finding:
56
+ evidence = () if distance is None else (("perceptual_distance", distance),)
57
+ return Finding(code, Severity.INFO, message, sample_ids, evidence)
@@ -0,0 +1,114 @@
1
+ from __future__ import annotations
2
+
3
+ from collections import Counter
4
+ from dataclasses import dataclass
5
+
6
+ from datajig.inventory import InventoryResult
7
+ from datajig.models import DistributionDelta, SampleRecord
8
+
9
+ SIZE_BUCKET_BOUNDARIES = (256, 512, 1024)
10
+ ASPECT_RATIO_BOUNDARIES = (0.8, 1.25)
11
+ BUCKET_BOUNDARIES = (
12
+ "size: <256, 256-511, 512-1023, >=1024, unknown",
13
+ "aspect_ratio: portrait <0.8, square 0.8-1.25, landscape >1.25, unknown",
14
+ )
15
+
16
+
17
+ @dataclass(frozen=True, slots=True)
18
+ class DistributionBucket:
19
+ dimension: str
20
+ key: str
21
+ count: int
22
+ proportion: float
23
+
24
+
25
+ @dataclass(frozen=True, slots=True)
26
+ class DistributionSummary:
27
+ total: int
28
+ buckets: tuple[DistributionBucket, ...]
29
+ bucket_boundaries: tuple[str, ...] = BUCKET_BOUNDARIES
30
+
31
+
32
+ def summarize_distribution(inventory: InventoryResult) -> DistributionSummary:
33
+ total = len(inventory.records)
34
+ counts: Counter[tuple[str, str]] = Counter()
35
+ for record in inventory.records:
36
+ for dimension, key in _dimensions(record):
37
+ counts[(dimension, key)] += 1
38
+ buckets = tuple(
39
+ DistributionBucket(
40
+ dimension,
41
+ key,
42
+ count,
43
+ count / total if total else 0.0,
44
+ )
45
+ for (dimension, key), count in sorted(counts.items())
46
+ )
47
+ return DistributionSummary(total, buckets)
48
+
49
+
50
+ def compare_distributions(
51
+ before: DistributionSummary, after: DistributionSummary
52
+ ) -> tuple[DistributionDelta, ...]:
53
+ before_map = {(item.dimension, item.key): item for item in before.buckets}
54
+ after_map = {(item.dimension, item.key): item for item in after.buckets}
55
+ deltas: list[DistributionDelta] = []
56
+ for dimension, key in sorted(before_map.keys() | after_map.keys()):
57
+ before_item = before_map.get((dimension, key))
58
+ after_item = after_map.get((dimension, key))
59
+ before_count = before_item.count if before_item else 0
60
+ after_count = after_item.count if after_item else 0
61
+ before_proportion = before_item.proportion if before_item else 0.0
62
+ after_proportion = after_item.proportion if after_item else 0.0
63
+ percentage_delta = (
64
+ ((after_proportion - before_proportion) / before_proportion) * 100
65
+ if before_proportion
66
+ else None
67
+ )
68
+ deltas.append(
69
+ DistributionDelta(
70
+ dimension,
71
+ key,
72
+ before_count,
73
+ after_count,
74
+ before_proportion,
75
+ after_proportion,
76
+ percentage_delta,
77
+ )
78
+ )
79
+ return tuple(deltas)
80
+
81
+
82
+ def _dimensions(record: SampleRecord) -> tuple[tuple[str, str], ...]:
83
+ return (
84
+ ("split", record.split),
85
+ ("label", record.label),
86
+ ("width", _size_bucket(record.width)),
87
+ ("height", _size_bucket(record.height)),
88
+ ("aspect_ratio", _aspect_bucket(record.width, record.height)),
89
+ ("format", record.media_format or "unknown"),
90
+ ("channels", str(record.channels) if record.channels is not None else "unknown"),
91
+ )
92
+
93
+
94
+ def _size_bucket(value: int | None) -> str:
95
+ if value is None:
96
+ return "unknown"
97
+ if value < SIZE_BUCKET_BOUNDARIES[0]:
98
+ return "<256"
99
+ if value < SIZE_BUCKET_BOUNDARIES[1]:
100
+ return "256-511"
101
+ if value < SIZE_BUCKET_BOUNDARIES[2]:
102
+ return "512-1023"
103
+ return ">=1024"
104
+
105
+
106
+ def _aspect_bucket(width: int | None, height: int | None) -> str:
107
+ if width is None or height is None or height == 0:
108
+ return "unknown"
109
+ ratio = width / height
110
+ if ratio < ASPECT_RATIO_BOUNDARIES[0]:
111
+ return "portrait"
112
+ if ratio <= ASPECT_RATIO_BOUNDARIES[1]:
113
+ return "square"
114
+ return "landscape"
@@ -0,0 +1,125 @@
1
+ from __future__ import annotations
2
+
3
+ from collections import defaultdict
4
+
5
+ from datajig.inventory import InventoryResult
6
+ from datajig.models import Finding, SampleRecord, Severity
7
+ from datajig.phash_index import PhashIndex
8
+
9
+
10
+ def analyze_leakage(
11
+ inventory: InventoryResult, threshold: int
12
+ ) -> tuple[Finding, ...]:
13
+ if threshold < 0:
14
+ raise ValueError("threshold must be non-negative")
15
+ findings: list[Finding] = []
16
+ by_digest: dict[str, list[SampleRecord]] = defaultdict(list)
17
+ for record in inventory.records:
18
+ by_digest[record.content_hash].append(record)
19
+ for digest in sorted(by_digest):
20
+ group = sorted(by_digest[digest], key=lambda item: item.relative_path)
21
+ if len(group) < 2:
22
+ continue
23
+ sample_ids = tuple(item.relative_path for item in group)
24
+ exact_evidence = (("group_size", len(group)), ("content_hash", digest))
25
+ findings.append(
26
+ Finding(
27
+ "EXACT_DUPLICATE_GROUP",
28
+ Severity.WARNING,
29
+ "identical image bytes appear more than once",
30
+ sample_ids,
31
+ exact_evidence,
32
+ )
33
+ )
34
+ if len({item.split for item in group}) > 1:
35
+ findings.append(
36
+ Finding(
37
+ "CROSS_SPLIT_EXACT_LEAKAGE",
38
+ Severity.ERROR,
39
+ "identical image bytes appear across splits",
40
+ sample_ids,
41
+ exact_evidence,
42
+ )
43
+ )
44
+
45
+ for group in _near_components(inventory.records, threshold):
46
+ sample_ids = tuple(item.relative_path for item in group)
47
+ near_evidence = (("group_size", len(group)), ("threshold", threshold))
48
+ findings.append(
49
+ Finding(
50
+ "NEAR_DUPLICATE_GROUP",
51
+ Severity.WARNING,
52
+ "perceptually similar images form a duplicate candidate group",
53
+ sample_ids,
54
+ near_evidence,
55
+ )
56
+ )
57
+ if len({item.split for item in group}) > 1:
58
+ findings.append(
59
+ Finding(
60
+ "CROSS_SPLIT_NEAR_LEAKAGE",
61
+ Severity.WARNING,
62
+ "perceptually similar images appear across splits",
63
+ sample_ids,
64
+ near_evidence,
65
+ )
66
+ )
67
+ findings.sort(key=lambda item: (item.code, item.sample_ids))
68
+ return tuple(findings)
69
+
70
+
71
+ def _near_components(
72
+ records: tuple[SampleRecord, ...], threshold: int
73
+ ) -> list[list[SampleRecord]]:
74
+ records = tuple(sorted(records, key=lambda item: item.relative_path))
75
+ groups: dict[str, dict[str, list[int]]] = defaultdict(lambda: defaultdict(list))
76
+ for index, record in enumerate(records):
77
+ if record.perceptual_hash is not None:
78
+ groups[record.perceptual_hash][record.content_hash].append(index)
79
+
80
+ parents = list(range(len(records)))
81
+
82
+ def find(index: int) -> int:
83
+ while parents[index] != index:
84
+ parents[index] = parents[parents[index]]
85
+ index = parents[index]
86
+ return index
87
+
88
+ def union(left: int, right: int) -> None:
89
+ left_root = find(left)
90
+ right_root = find(right)
91
+ if left_root != right_root:
92
+ parents[right_root] = left_root
93
+
94
+ def connect(left_group: list[int], right_group: list[int]) -> None:
95
+ for index in left_group[1:]:
96
+ union(left_group[0], index)
97
+ for index in right_group[1:]:
98
+ union(right_group[0], index)
99
+ union(left_group[0], right_group[0])
100
+
101
+ phash_index: PhashIndex[str] = PhashIndex()
102
+ for hash_value in sorted(groups):
103
+ digest_groups = groups[hash_value]
104
+ digest_values = sorted(digest_groups)
105
+ if len(digest_values) > 1:
106
+ anchor = digest_groups[digest_values[0]]
107
+ for other_digest in digest_values[1:]:
108
+ connect(anchor, digest_groups[other_digest])
109
+ for near_match in phash_index.query(hash_value, threshold):
110
+ for other_hash in near_match.items:
111
+ combined = [*digest_groups.items(), *groups[other_hash].items()]
112
+ anchor_digest, anchor_group = combined[0]
113
+ for other_digest, other_group in combined[1:]:
114
+ if anchor_digest != other_digest:
115
+ connect(anchor_group, other_group)
116
+ phash_index.add(hash_value, hash_value)
117
+
118
+ connected: dict[int, list[int]] = defaultdict(list)
119
+ for record_index in range(len(records)):
120
+ connected[find(record_index)].append(record_index)
121
+ return [
122
+ [records[index] for index in indexes]
123
+ for _, indexes in sorted(connected.items())
124
+ if len(indexes) > 1
125
+ ]