rubricloop 0.1.4__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rubricloop-0.1.4 → rubricloop-0.2.0}/PKG-INFO +55 -2
- {rubricloop-0.1.4 → rubricloop-0.2.0}/README.md +48 -1
- {rubricloop-0.1.4 → rubricloop-0.2.0}/pyproject.toml +9 -1
- rubricloop-0.2.0/src/rubricloop/__init__.py +101 -0
- rubricloop-0.2.0/src/rubricloop/artifacts.py +149 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/checks/__init__.py +26 -0
- rubricloop-0.2.0/src/rubricloop/checks/common.py +218 -0
- rubricloop-0.2.0/src/rubricloop/checks/media.py +340 -0
- rubricloop-0.2.0/src/rubricloop/compile.py +146 -0
- rubricloop-0.2.0/src/rubricloop/credentials.py +32 -0
- rubricloop-0.2.0/src/rubricloop/examples/llm_judge_example.py +101 -0
- rubricloop-0.2.0/src/rubricloop/examples/multimodal_example.py +59 -0
- rubricloop-0.2.0/src/rubricloop/examples/reference_feed_example.py +70 -0
- rubricloop-0.2.0/src/rubricloop/examples/streaming_example.py +52 -0
- rubricloop-0.2.0/src/rubricloop/judge.py +172 -0
- rubricloop-0.2.0/src/rubricloop/loop.py +649 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/vetting.py +44 -0
- rubricloop-0.2.0/src/rubricloop/references.py +133 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/rlpack.py +269 -14
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/rubric.py +46 -7
- rubricloop-0.2.0/src/rubricloop/safe_patterns.py +23 -0
- rubricloop-0.2.0/src/rubricloop/streaming.py +137 -0
- rubricloop-0.2.0/src/rubricloop/types.py +364 -0
- rubricloop-0.2.0/tests/test_compile.py +105 -0
- rubricloop-0.2.0/tests/test_credentials.py +13 -0
- rubricloop-0.2.0/tests/test_examples.py +33 -0
- rubricloop-0.2.0/tests/test_judge.py +176 -0
- rubricloop-0.2.0/tests/test_loop.py +175 -0
- rubricloop-0.2.0/tests/test_media.py +170 -0
- rubricloop-0.2.0/tests/test_references.py +117 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_rlpack.py +147 -10
- rubricloop-0.2.0/tests/test_streaming.py +179 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_vetting_scenarios.py +12 -1
- rubricloop-0.1.4/src/rubricloop/__init__.py +0 -59
- rubricloop-0.1.4/src/rubricloop/checks/common.py +0 -123
- rubricloop-0.1.4/src/rubricloop/loop.py +0 -199
- rubricloop-0.1.4/src/rubricloop/types.py +0 -166
- rubricloop-0.1.4/tests/test_loop.py +0 -76
- {rubricloop-0.1.4 → rubricloop-0.2.0}/.gitignore +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/LICENSE +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/RLPACK.md +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/VETTING.md +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/examples/sql_agent.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/examples/verify_reply.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/examples/verify_reply_registry/README.md +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/examples/verify_reply_registry/build_package.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/examples/verify_reply_registry/run-input.json +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/checks/sql.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/checks/structured.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/cli.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/examples/__init__.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/examples/verify_reply.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/__init__.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/base.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/catalog.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/commerce.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/finance.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/logistics.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/official.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/refund.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/security.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/sql.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/packs/support.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/py.typed +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/registry.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/sandboxes/__init__.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/src/rubricloop/sandboxes/sqlite.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_common_checks.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_invariant.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_refund_pack.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_registry_cli.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_sql_pack.py +0 -0
- {rubricloop-0.1.4 → rubricloop-0.2.0}/tests/test_wave_one_packs.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: rubricloop
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Bounded, auditable verification loops for AI agent output.
|
|
5
5
|
Project-URL: Homepage, https://rubricloop.com
|
|
6
6
|
Project-URL: Documentation, https://rubricloop.com/docs
|
|
@@ -25,6 +25,12 @@ Requires-Dist: build>=1.2; extra == 'dev'
|
|
|
25
25
|
Requires-Dist: pytest>=8.3; extra == 'dev'
|
|
26
26
|
Requires-Dist: ruff>=0.12; extra == 'dev'
|
|
27
27
|
Requires-Dist: twine<8,>=7; extra == 'dev'
|
|
28
|
+
Provides-Extra: judge
|
|
29
|
+
Requires-Dist: openai<4,>=2; extra == 'judge'
|
|
30
|
+
Provides-Extra: media
|
|
31
|
+
Requires-Dist: openpyxl<4,>=3.1; extra == 'media'
|
|
32
|
+
Requires-Dist: pillow<13,>=11; extra == 'media'
|
|
33
|
+
Requires-Dist: python-pptx<2,>=1.0; extra == 'media'
|
|
28
34
|
Description-Content-Type: text/markdown
|
|
29
35
|
|
|
30
36
|
# RubricLoop SDK
|
|
@@ -32,7 +38,10 @@ Description-Content-Type: text/markdown
|
|
|
32
38
|
RubricLoop checks AI agent output with ordinary code, returns exact failures to
|
|
33
39
|
the agent, and stops when the work passes or the run hits a clear limit.
|
|
34
40
|
|
|
35
|
-
|
|
41
|
+
Deterministic checks never call an LLM. An optional, explicitly configured
|
|
42
|
+
hybrid judge can score only the rule IDs you declare and cannot override a
|
|
43
|
+
deterministic failure. Your agent and judge can use any OpenAI-compatible
|
|
44
|
+
provider.
|
|
36
45
|
|
|
37
46
|
Native registry packages use the deterministic, signed
|
|
38
47
|
[`.rlpack` format](https://rubricloop.com/docs#package-lifecycle).
|
|
@@ -47,6 +56,14 @@ export OPENAI_API_KEY="sk-..."
|
|
|
47
56
|
python -m rubricloop.examples.verify_reply
|
|
48
57
|
```
|
|
49
58
|
|
|
59
|
+
Install only the extensions you use:
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install "rubricloop[judge]" # OpenAI-compatible LLM judge
|
|
63
|
+
pip install "rubricloop[media]" # image, workbook, and presentation extraction
|
|
64
|
+
pip install "rubricloop[judge,media]"
|
|
65
|
+
```
|
|
66
|
+
|
|
50
67
|
The example deliberately produces a short first draft containing a forbidden
|
|
51
68
|
promise. RubricLoop measures the failures and passes those diagnostics into the
|
|
52
69
|
next model call. The process exits successfully only when the reply satisfies
|
|
@@ -163,6 +180,42 @@ its failed rules to a reviewer.
|
|
|
163
180
|
- Local in-memory SQLite sandbox
|
|
164
181
|
- Ready-made `engineering/sql-safe` pack
|
|
165
182
|
- `support/refund-policy` action gate
|
|
183
|
+
- Signed, fresh, release-pinned reference feeds
|
|
184
|
+
- Scoped LLM-as-judge rules with token evidence and abstention
|
|
185
|
+
- Decisive streaming checks with cancellation or observation modes
|
|
186
|
+
- Digest-bound artifacts and deterministic media extraction
|
|
187
|
+
|
|
188
|
+
## Four extension examples
|
|
189
|
+
|
|
190
|
+
The wheel includes one offline, synthetic example for each extension study.
|
|
191
|
+
They require no API key or customer data and are exercised by the SDK test
|
|
192
|
+
suite:
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
python -m rubricloop.examples.reference_feed_example
|
|
196
|
+
python -m rubricloop.examples.llm_judge_example
|
|
197
|
+
python -m rubricloop.examples.streaming_example
|
|
198
|
+
python -m rubricloop.examples.multimodal_example
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
The main extension parameters preserve the original text-only behavior:
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
run = verify(
|
|
205
|
+
agent,
|
|
206
|
+
prompt,
|
|
207
|
+
rules,
|
|
208
|
+
judge=judge_config, # JudgeConfig; credentials are references
|
|
209
|
+
artifact=artifact, # Artifact or a sequence of artifacts
|
|
210
|
+
extract=extract_spec, # ExtractSpec or a sequence of extractors
|
|
211
|
+
streaming="cut", # False, True/"cut", or "observe"
|
|
212
|
+
constrain=True, # Pass compiled constraints when supported
|
|
213
|
+
)
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Extension evidence is emitted only when used through `run.references`,
|
|
217
|
+
`run.judge`, `run.artifacts`, `run.extractions`, `run.constraints`, and the
|
|
218
|
+
matching keys in `run.to_dict()`.
|
|
166
219
|
|
|
167
220
|
## Registry CLI
|
|
168
221
|
|
|
@@ -3,7 +3,10 @@
|
|
|
3
3
|
RubricLoop checks AI agent output with ordinary code, returns exact failures to
|
|
4
4
|
the agent, and stops when the work passes or the run hits a clear limit.
|
|
5
5
|
|
|
6
|
-
|
|
6
|
+
Deterministic checks never call an LLM. An optional, explicitly configured
|
|
7
|
+
hybrid judge can score only the rule IDs you declare and cannot override a
|
|
8
|
+
deterministic failure. Your agent and judge can use any OpenAI-compatible
|
|
9
|
+
provider.
|
|
7
10
|
|
|
8
11
|
Native registry packages use the deterministic, signed
|
|
9
12
|
[`.rlpack` format](https://rubricloop.com/docs#package-lifecycle).
|
|
@@ -18,6 +21,14 @@ export OPENAI_API_KEY="sk-..."
|
|
|
18
21
|
python -m rubricloop.examples.verify_reply
|
|
19
22
|
```
|
|
20
23
|
|
|
24
|
+
Install only the extensions you use:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install "rubricloop[judge]" # OpenAI-compatible LLM judge
|
|
28
|
+
pip install "rubricloop[media]" # image, workbook, and presentation extraction
|
|
29
|
+
pip install "rubricloop[judge,media]"
|
|
30
|
+
```
|
|
31
|
+
|
|
21
32
|
The example deliberately produces a short first draft containing a forbidden
|
|
22
33
|
promise. RubricLoop measures the failures and passes those diagnostics into the
|
|
23
34
|
next model call. The process exits successfully only when the reply satisfies
|
|
@@ -134,6 +145,42 @@ its failed rules to a reviewer.
|
|
|
134
145
|
- Local in-memory SQLite sandbox
|
|
135
146
|
- Ready-made `engineering/sql-safe` pack
|
|
136
147
|
- `support/refund-policy` action gate
|
|
148
|
+
- Signed, fresh, release-pinned reference feeds
|
|
149
|
+
- Scoped LLM-as-judge rules with token evidence and abstention
|
|
150
|
+
- Decisive streaming checks with cancellation or observation modes
|
|
151
|
+
- Digest-bound artifacts and deterministic media extraction
|
|
152
|
+
|
|
153
|
+
## Four extension examples
|
|
154
|
+
|
|
155
|
+
The wheel includes one offline, synthetic example for each extension study.
|
|
156
|
+
They require no API key or customer data and are exercised by the SDK test
|
|
157
|
+
suite:
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
python -m rubricloop.examples.reference_feed_example
|
|
161
|
+
python -m rubricloop.examples.llm_judge_example
|
|
162
|
+
python -m rubricloop.examples.streaming_example
|
|
163
|
+
python -m rubricloop.examples.multimodal_example
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
The main extension parameters preserve the original text-only behavior:
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
run = verify(
|
|
170
|
+
agent,
|
|
171
|
+
prompt,
|
|
172
|
+
rules,
|
|
173
|
+
judge=judge_config, # JudgeConfig; credentials are references
|
|
174
|
+
artifact=artifact, # Artifact or a sequence of artifacts
|
|
175
|
+
extract=extract_spec, # ExtractSpec or a sequence of extractors
|
|
176
|
+
streaming="cut", # False, True/"cut", or "observe"
|
|
177
|
+
constrain=True, # Pass compiled constraints when supported
|
|
178
|
+
)
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Extension evidence is emitted only when used through `run.references`,
|
|
182
|
+
`run.judge`, `run.artifacts`, `run.extractions`, `run.constraints`, and the
|
|
183
|
+
matching keys in `run.to_dict()`.
|
|
137
184
|
|
|
138
185
|
## Registry CLI
|
|
139
186
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "rubricloop"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Bounded, auditable verification loops for AI agent output."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -35,6 +35,14 @@ Documentation = "https://rubricloop.com/docs"
|
|
|
35
35
|
rubricloop = "rubricloop.cli:main"
|
|
36
36
|
|
|
37
37
|
[project.optional-dependencies]
|
|
38
|
+
judge = [
|
|
39
|
+
"openai>=2,<4",
|
|
40
|
+
]
|
|
41
|
+
media = [
|
|
42
|
+
"openpyxl>=3.1,<4",
|
|
43
|
+
"Pillow>=11,<13",
|
|
44
|
+
"python-pptx>=1.0,<2",
|
|
45
|
+
]
|
|
38
46
|
dev = [
|
|
39
47
|
"build>=1.2",
|
|
40
48
|
"pytest>=8.3",
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""RubricLoop public API."""
|
|
2
|
+
|
|
3
|
+
from .compile import (
|
|
4
|
+
BACKEND_SUPPORT,
|
|
5
|
+
ConstrainedAgentAdapter,
|
|
6
|
+
HostedConstrainedAgentAdapter,
|
|
7
|
+
VllmConstrainedAgentAdapter,
|
|
8
|
+
compile_constraints,
|
|
9
|
+
)
|
|
10
|
+
from .credentials import resolve_credential
|
|
11
|
+
from .loop import build_feedback, verify, verify_artifact
|
|
12
|
+
from .references import resolve_reference
|
|
13
|
+
from .rlpack import (
|
|
14
|
+
ArchiveLimits,
|
|
15
|
+
Ed25519Keyring,
|
|
16
|
+
Ed25519Signer,
|
|
17
|
+
Rlpack,
|
|
18
|
+
RlpackError,
|
|
19
|
+
SignatureError,
|
|
20
|
+
build_rlpack,
|
|
21
|
+
canonical_json,
|
|
22
|
+
create_manifest,
|
|
23
|
+
read_rlpack,
|
|
24
|
+
validate_manifest,
|
|
25
|
+
)
|
|
26
|
+
from .rubric import CheckContext, Rubric, Rule, rubric
|
|
27
|
+
from .streaming import StreamGuard, streamability_report
|
|
28
|
+
from .types import (
|
|
29
|
+
Abstention,
|
|
30
|
+
AgentRequest,
|
|
31
|
+
AgentResponse,
|
|
32
|
+
Artifact,
|
|
33
|
+
Budget,
|
|
34
|
+
CheckResult,
|
|
35
|
+
CompiledConstraints,
|
|
36
|
+
ExtractionRecord,
|
|
37
|
+
ExtractionRegion,
|
|
38
|
+
ExtractSpec,
|
|
39
|
+
Failure,
|
|
40
|
+
IterationRecord,
|
|
41
|
+
JudgeCallRecord,
|
|
42
|
+
JudgeConfig,
|
|
43
|
+
ReferenceFailure,
|
|
44
|
+
ReferenceSnapshot,
|
|
45
|
+
ReferenceSpec,
|
|
46
|
+
RunResult,
|
|
47
|
+
StreamHit,
|
|
48
|
+
Verdict,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
__all__ = [
|
|
52
|
+
"Abstention",
|
|
53
|
+
"AgentRequest",
|
|
54
|
+
"AgentResponse",
|
|
55
|
+
"Artifact",
|
|
56
|
+
"ArchiveLimits",
|
|
57
|
+
"Budget",
|
|
58
|
+
"BACKEND_SUPPORT",
|
|
59
|
+
"Ed25519Keyring",
|
|
60
|
+
"Ed25519Signer",
|
|
61
|
+
"Rlpack",
|
|
62
|
+
"RlpackError",
|
|
63
|
+
"SignatureError",
|
|
64
|
+
"CheckContext",
|
|
65
|
+
"CheckResult",
|
|
66
|
+
"CompiledConstraints",
|
|
67
|
+
"ConstrainedAgentAdapter",
|
|
68
|
+
"HostedConstrainedAgentAdapter",
|
|
69
|
+
"ExtractSpec",
|
|
70
|
+
"ExtractionRecord",
|
|
71
|
+
"ExtractionRegion",
|
|
72
|
+
"Failure",
|
|
73
|
+
"IterationRecord",
|
|
74
|
+
"JudgeCallRecord",
|
|
75
|
+
"JudgeConfig",
|
|
76
|
+
"ReferenceFailure",
|
|
77
|
+
"ReferenceSnapshot",
|
|
78
|
+
"ReferenceSpec",
|
|
79
|
+
"Rubric",
|
|
80
|
+
"Rule",
|
|
81
|
+
"RunResult",
|
|
82
|
+
"StreamHit",
|
|
83
|
+
"StreamGuard",
|
|
84
|
+
"streamability_report",
|
|
85
|
+
"Verdict",
|
|
86
|
+
"VllmConstrainedAgentAdapter",
|
|
87
|
+
"build_rlpack",
|
|
88
|
+
"build_feedback",
|
|
89
|
+
"canonical_json",
|
|
90
|
+
"compile_constraints",
|
|
91
|
+
"create_manifest",
|
|
92
|
+
"read_rlpack",
|
|
93
|
+
"resolve_credential",
|
|
94
|
+
"resolve_reference",
|
|
95
|
+
"rubric",
|
|
96
|
+
"validate_manifest",
|
|
97
|
+
"verify",
|
|
98
|
+
"verify_artifact",
|
|
99
|
+
]
|
|
100
|
+
|
|
101
|
+
__version__ = "0.2.0"
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Artifact integrity and bounded local extraction helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
from collections.abc import Iterable
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from .types import Artifact, ExtractionRecord, ExtractionRegion, ExtractSpec
|
|
13
|
+
|
|
14
|
+
MAX_ARTIFACT_BYTES = 20 * 1024 * 1024
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def prepare_artifacts(
|
|
18
|
+
value: Artifact | Iterable[Artifact] | None,
|
|
19
|
+
) -> tuple[tuple[Artifact, ...], dict[str, bytes]]:
|
|
20
|
+
if value is None:
|
|
21
|
+
return (), {}
|
|
22
|
+
artifacts = (value,) if isinstance(value, Artifact) else tuple(value)
|
|
23
|
+
names: set[str] = set()
|
|
24
|
+
payloads: dict[str, bytes] = {}
|
|
25
|
+
for artifact in artifacts:
|
|
26
|
+
if artifact.name in names:
|
|
27
|
+
raise ValueError(f"duplicate artifact name: {artifact.name}")
|
|
28
|
+
names.add(artifact.name)
|
|
29
|
+
payload = read_artifact(artifact)
|
|
30
|
+
payloads[artifact.name] = payload
|
|
31
|
+
return artifacts, payloads
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def read_artifact(artifact: Artifact) -> bytes:
|
|
35
|
+
source = artifact.source
|
|
36
|
+
if isinstance(source, bytes):
|
|
37
|
+
payload = source
|
|
38
|
+
elif isinstance(source, (str, Path)):
|
|
39
|
+
path = Path(source)
|
|
40
|
+
if path.stat().st_size > MAX_ARTIFACT_BYTES:
|
|
41
|
+
raise ValueError(f"artifact {artifact.name} exceeds the size limit")
|
|
42
|
+
payload = path.read_bytes()
|
|
43
|
+
elif hasattr(source, "read"):
|
|
44
|
+
position = source.tell() if hasattr(source, "tell") else None
|
|
45
|
+
payload = source.read(MAX_ARTIFACT_BYTES + 1)
|
|
46
|
+
if position is not None and hasattr(source, "seek"):
|
|
47
|
+
source.seek(position)
|
|
48
|
+
else:
|
|
49
|
+
raise TypeError(f"artifact {artifact.name} source must be a path or file-like")
|
|
50
|
+
if not isinstance(payload, bytes):
|
|
51
|
+
raise TypeError(f"artifact {artifact.name} source did not return bytes")
|
|
52
|
+
if len(payload) > MAX_ARTIFACT_BYTES:
|
|
53
|
+
raise ValueError(f"artifact {artifact.name} exceeds the size limit")
|
|
54
|
+
if len(payload) != artifact.size_bytes:
|
|
55
|
+
raise ValueError(f"artifact {artifact.name} size does not match")
|
|
56
|
+
digest = "sha256:" + hashlib.sha256(payload).hexdigest()
|
|
57
|
+
if digest != artifact.sha256:
|
|
58
|
+
raise ValueError(f"artifact {artifact.name} digest does not match")
|
|
59
|
+
return payload
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def run_extractions(
|
|
63
|
+
specs: Iterable[ExtractSpec],
|
|
64
|
+
artifacts: tuple[Artifact, ...],
|
|
65
|
+
payloads: dict[str, bytes],
|
|
66
|
+
) -> tuple[dict[str, ExtractionRecord], dict[str, Any]]:
|
|
67
|
+
records: dict[str, ExtractionRecord] = {}
|
|
68
|
+
evidence: dict[str, Any] = {}
|
|
69
|
+
by_name = {artifact.name: artifact for artifact in artifacts}
|
|
70
|
+
for spec in specs:
|
|
71
|
+
extraction_id = spec.id or spec.engine
|
|
72
|
+
try:
|
|
73
|
+
artifact = _select_artifact(spec, artifacts, by_name)
|
|
74
|
+
content, regions = _extract(spec, artifact, payloads[artifact.name])
|
|
75
|
+
record = ExtractionRecord(
|
|
76
|
+
engine=spec.engine,
|
|
77
|
+
version=spec.version,
|
|
78
|
+
config_sha256=_digest(dict(spec.config)),
|
|
79
|
+
content_sha256=_digest(content),
|
|
80
|
+
regions=regions,
|
|
81
|
+
content=content,
|
|
82
|
+
)
|
|
83
|
+
records[extraction_id] = record
|
|
84
|
+
evidence[extraction_id] = {
|
|
85
|
+
**record.evidence(),
|
|
86
|
+
"artifact": artifact.name,
|
|
87
|
+
}
|
|
88
|
+
except Exception as exc:
|
|
89
|
+
evidence[extraction_id] = {
|
|
90
|
+
"status": "unavailable",
|
|
91
|
+
"reason": f"{type(exc).__name__}: {str(exc)[:180]}",
|
|
92
|
+
}
|
|
93
|
+
return records, evidence
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _select_artifact(
|
|
97
|
+
spec: ExtractSpec,
|
|
98
|
+
artifacts: tuple[Artifact, ...],
|
|
99
|
+
by_name: dict[str, Artifact],
|
|
100
|
+
) -> Artifact:
|
|
101
|
+
if spec.artifact is not None:
|
|
102
|
+
if spec.artifact not in by_name:
|
|
103
|
+
raise ValueError(f"artifact {spec.artifact} is unavailable")
|
|
104
|
+
return by_name[spec.artifact]
|
|
105
|
+
if len(artifacts) != 1:
|
|
106
|
+
raise ValueError("extraction must name an artifact when the run has multiple")
|
|
107
|
+
return artifacts[0]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _extract(
|
|
111
|
+
spec: ExtractSpec, artifact: Artifact, payload: bytes
|
|
112
|
+
) -> tuple[Any, tuple[ExtractionRegion, ...]]:
|
|
113
|
+
if spec.engine == "provided":
|
|
114
|
+
if "content" not in spec.config:
|
|
115
|
+
raise ValueError("provided extraction requires config.content")
|
|
116
|
+
return spec.config["content"], ()
|
|
117
|
+
if spec.engine == "image.metadata":
|
|
118
|
+
try:
|
|
119
|
+
from PIL import Image
|
|
120
|
+
except ImportError as exc:
|
|
121
|
+
raise RuntimeError(
|
|
122
|
+
"install rubricloop[media] to extract image metadata"
|
|
123
|
+
) from exc
|
|
124
|
+
with Image.open(io.BytesIO(payload)) as image:
|
|
125
|
+
content = {
|
|
126
|
+
"width": image.width,
|
|
127
|
+
"height": image.height,
|
|
128
|
+
"mode": image.mode,
|
|
129
|
+
"format": image.format,
|
|
130
|
+
}
|
|
131
|
+
return content, (
|
|
132
|
+
ExtractionRegion(
|
|
133
|
+
id=f"image:{artifact.name}",
|
|
134
|
+
content=content,
|
|
135
|
+
confidence=1.0,
|
|
136
|
+
),
|
|
137
|
+
)
|
|
138
|
+
raise ValueError(f"unsupported local extraction engine: {spec.engine}")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _digest(value: Any) -> str:
|
|
142
|
+
encoded = json.dumps(
|
|
143
|
+
value,
|
|
144
|
+
sort_keys=True,
|
|
145
|
+
separators=(",", ":"),
|
|
146
|
+
ensure_ascii=False,
|
|
147
|
+
allow_nan=False,
|
|
148
|
+
).encode()
|
|
149
|
+
return "sha256:" + hashlib.sha256(encoded).hexdigest()
|
|
@@ -2,13 +2,27 @@
|
|
|
2
2
|
|
|
3
3
|
from .common import (
|
|
4
4
|
abstain,
|
|
5
|
+
citation_lookup,
|
|
5
6
|
forbidden_terms,
|
|
7
|
+
json_field_enum,
|
|
6
8
|
json_schema,
|
|
7
9
|
pii_free,
|
|
8
10
|
required_terms,
|
|
9
11
|
valid_json,
|
|
10
12
|
word_count,
|
|
11
13
|
)
|
|
14
|
+
from .media import (
|
|
15
|
+
EnvironmentResolver,
|
|
16
|
+
audio_script_adherence,
|
|
17
|
+
chart_data_match,
|
|
18
|
+
cross_modal_match,
|
|
19
|
+
layout_compliance,
|
|
20
|
+
media_metric,
|
|
21
|
+
media_pii_screen,
|
|
22
|
+
render_check,
|
|
23
|
+
synthetic_media_flag,
|
|
24
|
+
table_arithmetic,
|
|
25
|
+
)
|
|
12
26
|
from .sql import (
|
|
13
27
|
executes,
|
|
14
28
|
matches_reference,
|
|
@@ -21,10 +35,18 @@ from .sql import (
|
|
|
21
35
|
|
|
22
36
|
__all__ = [
|
|
23
37
|
"abstain",
|
|
38
|
+
"audio_script_adherence",
|
|
39
|
+
"chart_data_match",
|
|
40
|
+
"citation_lookup",
|
|
24
41
|
"executes",
|
|
25
42
|
"forbidden_terms",
|
|
43
|
+
"cross_modal_match",
|
|
44
|
+
"json_field_enum",
|
|
26
45
|
"json_schema",
|
|
27
46
|
"matches_reference",
|
|
47
|
+
"layout_compliance",
|
|
48
|
+
"media_metric",
|
|
49
|
+
"media_pii_screen",
|
|
28
50
|
"no_select_star",
|
|
29
51
|
"parses",
|
|
30
52
|
"pii_free",
|
|
@@ -32,6 +54,10 @@ __all__ = [
|
|
|
32
54
|
"required_columns",
|
|
33
55
|
"required_terms",
|
|
34
56
|
"schema_conformance",
|
|
57
|
+
"render_check",
|
|
58
|
+
"synthetic_media_flag",
|
|
59
|
+
"table_arithmetic",
|
|
35
60
|
"valid_json",
|
|
36
61
|
"word_count",
|
|
62
|
+
"EnvironmentResolver",
|
|
37
63
|
]
|