rubricloop 0.1.3__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rubricloop-0.1.3 → rubricloop-0.2.0}/PKG-INFO +60 -7
- {rubricloop-0.1.3 → rubricloop-0.2.0}/README.md +53 -6
- {rubricloop-0.1.3 → rubricloop-0.2.0}/examples/verify_reply_registry/README.md +24 -6
- {rubricloop-0.1.3 → rubricloop-0.2.0}/pyproject.toml +9 -1
- rubricloop-0.2.0/src/rubricloop/__init__.py +101 -0
- rubricloop-0.2.0/src/rubricloop/artifacts.py +149 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/checks/__init__.py +26 -0
- rubricloop-0.2.0/src/rubricloop/checks/common.py +218 -0
- rubricloop-0.2.0/src/rubricloop/checks/media.py +340 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/cli.py +1 -1
- rubricloop-0.2.0/src/rubricloop/compile.py +146 -0
- rubricloop-0.2.0/src/rubricloop/credentials.py +32 -0
- rubricloop-0.2.0/src/rubricloop/examples/llm_judge_example.py +101 -0
- rubricloop-0.2.0/src/rubricloop/examples/multimodal_example.py +59 -0
- rubricloop-0.2.0/src/rubricloop/examples/reference_feed_example.py +70 -0
- rubricloop-0.2.0/src/rubricloop/examples/streaming_example.py +52 -0
- rubricloop-0.2.0/src/rubricloop/judge.py +172 -0
- rubricloop-0.2.0/src/rubricloop/loop.py +649 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/__init__.py +2 -1
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/catalog.py +19 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/support.py +35 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/vetting.py +44 -0
- rubricloop-0.2.0/src/rubricloop/references.py +133 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/registry.py +1 -1
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/rlpack.py +269 -14
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/rubric.py +46 -7
- rubricloop-0.2.0/src/rubricloop/safe_patterns.py +23 -0
- rubricloop-0.2.0/src/rubricloop/streaming.py +137 -0
- rubricloop-0.2.0/src/rubricloop/types.py +364 -0
- rubricloop-0.2.0/tests/test_compile.py +105 -0
- rubricloop-0.2.0/tests/test_credentials.py +13 -0
- rubricloop-0.2.0/tests/test_examples.py +33 -0
- rubricloop-0.2.0/tests/test_judge.py +176 -0
- rubricloop-0.2.0/tests/test_loop.py +175 -0
- rubricloop-0.2.0/tests/test_media.py +170 -0
- rubricloop-0.2.0/tests/test_references.py +117 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_registry_cli.py +7 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_rlpack.py +147 -10
- rubricloop-0.2.0/tests/test_streaming.py +179 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_vetting_scenarios.py +12 -1
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_wave_one_packs.py +27 -0
- rubricloop-0.1.3/src/rubricloop/__init__.py +0 -59
- rubricloop-0.1.3/src/rubricloop/checks/common.py +0 -123
- rubricloop-0.1.3/src/rubricloop/loop.py +0 -199
- rubricloop-0.1.3/src/rubricloop/types.py +0 -166
- rubricloop-0.1.3/tests/test_loop.py +0 -76
- {rubricloop-0.1.3 → rubricloop-0.2.0}/.gitignore +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/LICENSE +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/RLPACK.md +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/VETTING.md +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/examples/sql_agent.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/examples/verify_reply.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/examples/verify_reply_registry/build_package.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/examples/verify_reply_registry/run-input.json +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/checks/sql.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/checks/structured.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/examples/__init__.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/examples/verify_reply.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/base.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/commerce.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/finance.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/logistics.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/official.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/refund.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/security.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/packs/sql.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/py.typed +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/sandboxes/__init__.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/src/rubricloop/sandboxes/sqlite.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_common_checks.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_invariant.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_refund_pack.py +0 -0
- {rubricloop-0.1.3 → rubricloop-0.2.0}/tests/test_sql_pack.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: rubricloop
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Bounded, auditable verification loops for AI agent output.
|
|
5
5
|
Project-URL: Homepage, https://rubricloop.com
|
|
6
6
|
Project-URL: Documentation, https://rubricloop.com/docs
|
|
@@ -25,6 +25,12 @@ Requires-Dist: build>=1.2; extra == 'dev'
|
|
|
25
25
|
Requires-Dist: pytest>=8.3; extra == 'dev'
|
|
26
26
|
Requires-Dist: ruff>=0.12; extra == 'dev'
|
|
27
27
|
Requires-Dist: twine<8,>=7; extra == 'dev'
|
|
28
|
+
Provides-Extra: judge
|
|
29
|
+
Requires-Dist: openai<4,>=2; extra == 'judge'
|
|
30
|
+
Provides-Extra: media
|
|
31
|
+
Requires-Dist: openpyxl<4,>=3.1; extra == 'media'
|
|
32
|
+
Requires-Dist: pillow<13,>=11; extra == 'media'
|
|
33
|
+
Requires-Dist: python-pptx<2,>=1.0; extra == 'media'
|
|
28
34
|
Description-Content-Type: text/markdown
|
|
29
35
|
|
|
30
36
|
# RubricLoop SDK
|
|
@@ -32,7 +38,10 @@ Description-Content-Type: text/markdown
|
|
|
32
38
|
RubricLoop checks AI agent output with ordinary code, returns exact failures to
|
|
33
39
|
the agent, and stops when the work passes or the run hits a clear limit.
|
|
34
40
|
|
|
35
|
-
|
|
41
|
+
Deterministic checks never call an LLM. An optional, explicitly configured
|
|
42
|
+
hybrid judge can score only the rule IDs you declare and cannot override a
|
|
43
|
+
deterministic failure. Your agent and judge can use any OpenAI-compatible
|
|
44
|
+
provider.
|
|
36
45
|
|
|
37
46
|
Native registry packages use the deterministic, signed
|
|
38
47
|
[`.rlpack` format](https://rubricloop.com/docs#package-lifecycle).
|
|
@@ -47,6 +56,14 @@ export OPENAI_API_KEY="sk-..."
|
|
|
47
56
|
python -m rubricloop.examples.verify_reply
|
|
48
57
|
```
|
|
49
58
|
|
|
59
|
+
Install only the extensions you use:
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install "rubricloop[judge]" # OpenAI-compatible LLM judge
|
|
63
|
+
pip install "rubricloop[media]" # image, workbook, and presentation extraction
|
|
64
|
+
pip install "rubricloop[judge,media]"
|
|
65
|
+
```
|
|
66
|
+
|
|
50
67
|
The example deliberately produces a short first draft containing a forbidden
|
|
51
68
|
promise. RubricLoop measures the failures and passes those diagnostics into the
|
|
52
69
|
next model call. The process exits successfully only when the reply satisfies
|
|
@@ -163,6 +180,42 @@ its failed rules to a reviewer.
|
|
|
163
180
|
- Local in-memory SQLite sandbox
|
|
164
181
|
- Ready-made `engineering/sql-safe` pack
|
|
165
182
|
- `support/refund-policy` action gate
|
|
183
|
+
- Signed, fresh, release-pinned reference feeds
|
|
184
|
+
- Scoped LLM-as-judge rules with token evidence and abstention
|
|
185
|
+
- Decisive streaming checks with cancellation or observation modes
|
|
186
|
+
- Digest-bound artifacts and deterministic media extraction
|
|
187
|
+
|
|
188
|
+
## Four extension examples
|
|
189
|
+
|
|
190
|
+
The wheel includes one offline, synthetic example for each extension study.
|
|
191
|
+
They require no API key or customer data and are exercised by the SDK test
|
|
192
|
+
suite:
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
python -m rubricloop.examples.reference_feed_example
|
|
196
|
+
python -m rubricloop.examples.llm_judge_example
|
|
197
|
+
python -m rubricloop.examples.streaming_example
|
|
198
|
+
python -m rubricloop.examples.multimodal_example
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
The main extension parameters preserve the original text-only behavior:
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
run = verify(
|
|
205
|
+
agent,
|
|
206
|
+
prompt,
|
|
207
|
+
rules,
|
|
208
|
+
judge=judge_config, # JudgeConfig; credentials are references
|
|
209
|
+
artifact=artifact, # Artifact or a sequence of artifacts
|
|
210
|
+
extract=extract_spec, # ExtractSpec or a sequence of extractors
|
|
211
|
+
streaming="cut", # False, True/"cut", or "observe"
|
|
212
|
+
constrain=True, # Pass compiled constraints when supported
|
|
213
|
+
)
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Extension evidence is emitted only when used through `run.references`,
|
|
217
|
+
`run.judge`, `run.artifacts`, `run.extractions`, `run.constraints`, and the
|
|
218
|
+
matching keys in `run.to_dict()`.
|
|
166
219
|
|
|
167
220
|
## Registry CLI
|
|
168
221
|
|
|
@@ -179,11 +232,11 @@ When the package is ready to share, create an API key in the
|
|
|
179
232
|
hosted registry:
|
|
180
233
|
|
|
181
234
|
```bash
|
|
182
|
-
rubricloop
|
|
183
|
-
rubricloop
|
|
184
|
-
rubricloop
|
|
185
|
-
rubricloop
|
|
186
|
-
rubricloop
|
|
235
|
+
rubricloop login --api-key "$RUBRICLOOP_API_KEY"
|
|
236
|
+
rubricloop create acme/check --visibility private
|
|
237
|
+
rubricloop push dist/acme-check.rlpack --tag latest
|
|
238
|
+
rubricloop vet acme/check@latest
|
|
239
|
+
rubricloop pull acme/check@latest
|
|
187
240
|
```
|
|
188
241
|
|
|
189
242
|
Public community packages are free. Private repositories are available only to
|
|
@@ -3,7 +3,10 @@
|
|
|
3
3
|
RubricLoop checks AI agent output with ordinary code, returns exact failures to
|
|
4
4
|
the agent, and stops when the work passes or the run hits a clear limit.
|
|
5
5
|
|
|
6
|
-
|
|
6
|
+
Deterministic checks never call an LLM. An optional, explicitly configured
|
|
7
|
+
hybrid judge can score only the rule IDs you declare and cannot override a
|
|
8
|
+
deterministic failure. Your agent and judge can use any OpenAI-compatible
|
|
9
|
+
provider.
|
|
7
10
|
|
|
8
11
|
Native registry packages use the deterministic, signed
|
|
9
12
|
[`.rlpack` format](https://rubricloop.com/docs#package-lifecycle).
|
|
@@ -18,6 +21,14 @@ export OPENAI_API_KEY="sk-..."
|
|
|
18
21
|
python -m rubricloop.examples.verify_reply
|
|
19
22
|
```
|
|
20
23
|
|
|
24
|
+
Install only the extensions you use:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
pip install "rubricloop[judge]" # OpenAI-compatible LLM judge
|
|
28
|
+
pip install "rubricloop[media]" # image, workbook, and presentation extraction
|
|
29
|
+
pip install "rubricloop[judge,media]"
|
|
30
|
+
```
|
|
31
|
+
|
|
21
32
|
The example deliberately produces a short first draft containing a forbidden
|
|
22
33
|
promise. RubricLoop measures the failures and passes those diagnostics into the
|
|
23
34
|
next model call. The process exits successfully only when the reply satisfies
|
|
@@ -134,6 +145,42 @@ its failed rules to a reviewer.
|
|
|
134
145
|
- Local in-memory SQLite sandbox
|
|
135
146
|
- Ready-made `engineering/sql-safe` pack
|
|
136
147
|
- `support/refund-policy` action gate
|
|
148
|
+
- Signed, fresh, release-pinned reference feeds
|
|
149
|
+
- Scoped LLM-as-judge rules with token evidence and abstention
|
|
150
|
+
- Decisive streaming checks with cancellation or observation modes
|
|
151
|
+
- Digest-bound artifacts and deterministic media extraction
|
|
152
|
+
|
|
153
|
+
## Four extension examples
|
|
154
|
+
|
|
155
|
+
The wheel includes one offline, synthetic example for each extension study.
|
|
156
|
+
They require no API key or customer data and are exercised by the SDK test
|
|
157
|
+
suite:
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
python -m rubricloop.examples.reference_feed_example
|
|
161
|
+
python -m rubricloop.examples.llm_judge_example
|
|
162
|
+
python -m rubricloop.examples.streaming_example
|
|
163
|
+
python -m rubricloop.examples.multimodal_example
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
The main extension parameters preserve the original text-only behavior:
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
run = verify(
|
|
170
|
+
agent,
|
|
171
|
+
prompt,
|
|
172
|
+
rules,
|
|
173
|
+
judge=judge_config, # JudgeConfig; credentials are references
|
|
174
|
+
artifact=artifact, # Artifact or a sequence of artifacts
|
|
175
|
+
extract=extract_spec, # ExtractSpec or a sequence of extractors
|
|
176
|
+
streaming="cut", # False, True/"cut", or "observe"
|
|
177
|
+
constrain=True, # Pass compiled constraints when supported
|
|
178
|
+
)
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Extension evidence is emitted only when used through `run.references`,
|
|
182
|
+
`run.judge`, `run.artifacts`, `run.extractions`, `run.constraints`, and the
|
|
183
|
+
matching keys in `run.to_dict()`.
|
|
137
184
|
|
|
138
185
|
## Registry CLI
|
|
139
186
|
|
|
@@ -150,11 +197,11 @@ When the package is ready to share, create an API key in the
|
|
|
150
197
|
hosted registry:
|
|
151
198
|
|
|
152
199
|
```bash
|
|
153
|
-
rubricloop
|
|
154
|
-
rubricloop
|
|
155
|
-
rubricloop
|
|
156
|
-
rubricloop
|
|
157
|
-
rubricloop
|
|
200
|
+
rubricloop login --api-key "$RUBRICLOOP_API_KEY"
|
|
201
|
+
rubricloop create acme/check --visibility private
|
|
202
|
+
rubricloop push dist/acme-check.rlpack --tag latest
|
|
203
|
+
rubricloop vet acme/check@latest
|
|
204
|
+
rubricloop pull acme/check@latest
|
|
158
205
|
```
|
|
159
206
|
|
|
160
207
|
Public community packages are free. Private repositories are available only to
|
|
@@ -4,6 +4,24 @@ This package uses the same three deterministic rules as
|
|
|
4
4
|
`sdk/examples/verify_reply.py`: 40 to 70 words, the word `timeline`, and no
|
|
5
5
|
`guaranteed refund` promise.
|
|
6
6
|
|
|
7
|
+
You can run the same checks through Hosted Verification before publishing and
|
|
8
|
+
vetting your own package:
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
curl https://api.rubricloop.dev/v1/verify \
|
|
12
|
+
-H "Content-Type: application/json" \
|
|
13
|
+
-H "X-API-Key: $RUBRICLOOP_API_KEY" \
|
|
14
|
+
-d '{
|
|
15
|
+
"pack_id": "support/reply-v1",
|
|
16
|
+
"output": {
|
|
17
|
+
"reply": "We apologize for the delay with your order. Our team is currently investigating the issue and working to resolve it as quickly as possible. You can expect an update on the timeline for your order shortly. Thank you for your patience."
|
|
18
|
+
},
|
|
19
|
+
"context": {}
|
|
20
|
+
}'
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
To publish the package itself:
|
|
24
|
+
|
|
7
25
|
```bash
|
|
8
26
|
pip install rubricloop openai
|
|
9
27
|
export OPENAI_API_KEY="sk-..."
|
|
@@ -12,24 +30,24 @@ python sdk/examples/verify_reply.py
|
|
|
12
30
|
python sdk/examples/verify_reply_registry/build_package.py \
|
|
13
31
|
--namespace YOUR_NAMESPACE
|
|
14
32
|
|
|
15
|
-
rubricloop
|
|
16
|
-
rubricloop
|
|
33
|
+
rubricloop login
|
|
34
|
+
rubricloop create \
|
|
17
35
|
YOUR_NAMESPACE/support-reply \
|
|
18
36
|
--visibility private \
|
|
19
37
|
--description "Checks support replies before they are sent."
|
|
20
38
|
rubricloop validate dist/support-reply.rlpack
|
|
21
39
|
rubricloop test dist/support-reply.rlpack
|
|
22
|
-
rubricloop
|
|
40
|
+
rubricloop push \
|
|
23
41
|
dist/support-reply.rlpack \
|
|
24
42
|
--tag latest
|
|
25
|
-
rubricloop
|
|
43
|
+
rubricloop vet \
|
|
26
44
|
YOUR_NAMESPACE/support-reply@latest
|
|
27
45
|
```
|
|
28
46
|
|
|
29
47
|
Pull the immutable bundle digest returned by `push`, not the mutable tag:
|
|
30
48
|
|
|
31
49
|
```bash
|
|
32
|
-
rubricloop
|
|
50
|
+
rubricloop pull \
|
|
33
51
|
YOUR_NAMESPACE/support-reply@sha256:BUNDLE_DIGEST \
|
|
34
52
|
--output dist/pulled-support-reply.rlpack
|
|
35
53
|
cmp dist/support-reply.rlpack dist/pulled-support-reply.rlpack
|
|
@@ -38,7 +56,7 @@ cmp dist/support-reply.rlpack dist/pulled-support-reply.rlpack
|
|
|
38
56
|
Once vetting passes and hosted execution is approved for that exact digest:
|
|
39
57
|
|
|
40
58
|
```bash
|
|
41
|
-
rubricloop
|
|
59
|
+
rubricloop run \
|
|
42
60
|
YOUR_NAMESPACE/support-reply@sha256:BUNDLE_DIGEST \
|
|
43
61
|
--input sdk/examples/verify_reply_registry/run-input.json
|
|
44
62
|
```
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "rubricloop"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Bounded, auditable verification loops for AI agent output."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -35,6 +35,14 @@ Documentation = "https://rubricloop.com/docs"
|
|
|
35
35
|
rubricloop = "rubricloop.cli:main"
|
|
36
36
|
|
|
37
37
|
[project.optional-dependencies]
|
|
38
|
+
judge = [
|
|
39
|
+
"openai>=2,<4",
|
|
40
|
+
]
|
|
41
|
+
media = [
|
|
42
|
+
"openpyxl>=3.1,<4",
|
|
43
|
+
"Pillow>=11,<13",
|
|
44
|
+
"python-pptx>=1.0,<2",
|
|
45
|
+
]
|
|
38
46
|
dev = [
|
|
39
47
|
"build>=1.2",
|
|
40
48
|
"pytest>=8.3",
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""RubricLoop public API."""
|
|
2
|
+
|
|
3
|
+
from .compile import (
|
|
4
|
+
BACKEND_SUPPORT,
|
|
5
|
+
ConstrainedAgentAdapter,
|
|
6
|
+
HostedConstrainedAgentAdapter,
|
|
7
|
+
VllmConstrainedAgentAdapter,
|
|
8
|
+
compile_constraints,
|
|
9
|
+
)
|
|
10
|
+
from .credentials import resolve_credential
|
|
11
|
+
from .loop import build_feedback, verify, verify_artifact
|
|
12
|
+
from .references import resolve_reference
|
|
13
|
+
from .rlpack import (
|
|
14
|
+
ArchiveLimits,
|
|
15
|
+
Ed25519Keyring,
|
|
16
|
+
Ed25519Signer,
|
|
17
|
+
Rlpack,
|
|
18
|
+
RlpackError,
|
|
19
|
+
SignatureError,
|
|
20
|
+
build_rlpack,
|
|
21
|
+
canonical_json,
|
|
22
|
+
create_manifest,
|
|
23
|
+
read_rlpack,
|
|
24
|
+
validate_manifest,
|
|
25
|
+
)
|
|
26
|
+
from .rubric import CheckContext, Rubric, Rule, rubric
|
|
27
|
+
from .streaming import StreamGuard, streamability_report
|
|
28
|
+
from .types import (
|
|
29
|
+
Abstention,
|
|
30
|
+
AgentRequest,
|
|
31
|
+
AgentResponse,
|
|
32
|
+
Artifact,
|
|
33
|
+
Budget,
|
|
34
|
+
CheckResult,
|
|
35
|
+
CompiledConstraints,
|
|
36
|
+
ExtractionRecord,
|
|
37
|
+
ExtractionRegion,
|
|
38
|
+
ExtractSpec,
|
|
39
|
+
Failure,
|
|
40
|
+
IterationRecord,
|
|
41
|
+
JudgeCallRecord,
|
|
42
|
+
JudgeConfig,
|
|
43
|
+
ReferenceFailure,
|
|
44
|
+
ReferenceSnapshot,
|
|
45
|
+
ReferenceSpec,
|
|
46
|
+
RunResult,
|
|
47
|
+
StreamHit,
|
|
48
|
+
Verdict,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
__all__ = [
|
|
52
|
+
"Abstention",
|
|
53
|
+
"AgentRequest",
|
|
54
|
+
"AgentResponse",
|
|
55
|
+
"Artifact",
|
|
56
|
+
"ArchiveLimits",
|
|
57
|
+
"Budget",
|
|
58
|
+
"BACKEND_SUPPORT",
|
|
59
|
+
"Ed25519Keyring",
|
|
60
|
+
"Ed25519Signer",
|
|
61
|
+
"Rlpack",
|
|
62
|
+
"RlpackError",
|
|
63
|
+
"SignatureError",
|
|
64
|
+
"CheckContext",
|
|
65
|
+
"CheckResult",
|
|
66
|
+
"CompiledConstraints",
|
|
67
|
+
"ConstrainedAgentAdapter",
|
|
68
|
+
"HostedConstrainedAgentAdapter",
|
|
69
|
+
"ExtractSpec",
|
|
70
|
+
"ExtractionRecord",
|
|
71
|
+
"ExtractionRegion",
|
|
72
|
+
"Failure",
|
|
73
|
+
"IterationRecord",
|
|
74
|
+
"JudgeCallRecord",
|
|
75
|
+
"JudgeConfig",
|
|
76
|
+
"ReferenceFailure",
|
|
77
|
+
"ReferenceSnapshot",
|
|
78
|
+
"ReferenceSpec",
|
|
79
|
+
"Rubric",
|
|
80
|
+
"Rule",
|
|
81
|
+
"RunResult",
|
|
82
|
+
"StreamHit",
|
|
83
|
+
"StreamGuard",
|
|
84
|
+
"streamability_report",
|
|
85
|
+
"Verdict",
|
|
86
|
+
"VllmConstrainedAgentAdapter",
|
|
87
|
+
"build_rlpack",
|
|
88
|
+
"build_feedback",
|
|
89
|
+
"canonical_json",
|
|
90
|
+
"compile_constraints",
|
|
91
|
+
"create_manifest",
|
|
92
|
+
"read_rlpack",
|
|
93
|
+
"resolve_credential",
|
|
94
|
+
"resolve_reference",
|
|
95
|
+
"rubric",
|
|
96
|
+
"validate_manifest",
|
|
97
|
+
"verify",
|
|
98
|
+
"verify_artifact",
|
|
99
|
+
]
|
|
100
|
+
|
|
101
|
+
__version__ = "0.2.0"
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Artifact integrity and bounded local extraction helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
from collections.abc import Iterable
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from .types import Artifact, ExtractionRecord, ExtractionRegion, ExtractSpec
|
|
13
|
+
|
|
14
|
+
MAX_ARTIFACT_BYTES = 20 * 1024 * 1024
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def prepare_artifacts(
|
|
18
|
+
value: Artifact | Iterable[Artifact] | None,
|
|
19
|
+
) -> tuple[tuple[Artifact, ...], dict[str, bytes]]:
|
|
20
|
+
if value is None:
|
|
21
|
+
return (), {}
|
|
22
|
+
artifacts = (value,) if isinstance(value, Artifact) else tuple(value)
|
|
23
|
+
names: set[str] = set()
|
|
24
|
+
payloads: dict[str, bytes] = {}
|
|
25
|
+
for artifact in artifacts:
|
|
26
|
+
if artifact.name in names:
|
|
27
|
+
raise ValueError(f"duplicate artifact name: {artifact.name}")
|
|
28
|
+
names.add(artifact.name)
|
|
29
|
+
payload = read_artifact(artifact)
|
|
30
|
+
payloads[artifact.name] = payload
|
|
31
|
+
return artifacts, payloads
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def read_artifact(artifact: Artifact) -> bytes:
|
|
35
|
+
source = artifact.source
|
|
36
|
+
if isinstance(source, bytes):
|
|
37
|
+
payload = source
|
|
38
|
+
elif isinstance(source, (str, Path)):
|
|
39
|
+
path = Path(source)
|
|
40
|
+
if path.stat().st_size > MAX_ARTIFACT_BYTES:
|
|
41
|
+
raise ValueError(f"artifact {artifact.name} exceeds the size limit")
|
|
42
|
+
payload = path.read_bytes()
|
|
43
|
+
elif hasattr(source, "read"):
|
|
44
|
+
position = source.tell() if hasattr(source, "tell") else None
|
|
45
|
+
payload = source.read(MAX_ARTIFACT_BYTES + 1)
|
|
46
|
+
if position is not None and hasattr(source, "seek"):
|
|
47
|
+
source.seek(position)
|
|
48
|
+
else:
|
|
49
|
+
raise TypeError(f"artifact {artifact.name} source must be a path or file-like")
|
|
50
|
+
if not isinstance(payload, bytes):
|
|
51
|
+
raise TypeError(f"artifact {artifact.name} source did not return bytes")
|
|
52
|
+
if len(payload) > MAX_ARTIFACT_BYTES:
|
|
53
|
+
raise ValueError(f"artifact {artifact.name} exceeds the size limit")
|
|
54
|
+
if len(payload) != artifact.size_bytes:
|
|
55
|
+
raise ValueError(f"artifact {artifact.name} size does not match")
|
|
56
|
+
digest = "sha256:" + hashlib.sha256(payload).hexdigest()
|
|
57
|
+
if digest != artifact.sha256:
|
|
58
|
+
raise ValueError(f"artifact {artifact.name} digest does not match")
|
|
59
|
+
return payload
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def run_extractions(
|
|
63
|
+
specs: Iterable[ExtractSpec],
|
|
64
|
+
artifacts: tuple[Artifact, ...],
|
|
65
|
+
payloads: dict[str, bytes],
|
|
66
|
+
) -> tuple[dict[str, ExtractionRecord], dict[str, Any]]:
|
|
67
|
+
records: dict[str, ExtractionRecord] = {}
|
|
68
|
+
evidence: dict[str, Any] = {}
|
|
69
|
+
by_name = {artifact.name: artifact for artifact in artifacts}
|
|
70
|
+
for spec in specs:
|
|
71
|
+
extraction_id = spec.id or spec.engine
|
|
72
|
+
try:
|
|
73
|
+
artifact = _select_artifact(spec, artifacts, by_name)
|
|
74
|
+
content, regions = _extract(spec, artifact, payloads[artifact.name])
|
|
75
|
+
record = ExtractionRecord(
|
|
76
|
+
engine=spec.engine,
|
|
77
|
+
version=spec.version,
|
|
78
|
+
config_sha256=_digest(dict(spec.config)),
|
|
79
|
+
content_sha256=_digest(content),
|
|
80
|
+
regions=regions,
|
|
81
|
+
content=content,
|
|
82
|
+
)
|
|
83
|
+
records[extraction_id] = record
|
|
84
|
+
evidence[extraction_id] = {
|
|
85
|
+
**record.evidence(),
|
|
86
|
+
"artifact": artifact.name,
|
|
87
|
+
}
|
|
88
|
+
except Exception as exc:
|
|
89
|
+
evidence[extraction_id] = {
|
|
90
|
+
"status": "unavailable",
|
|
91
|
+
"reason": f"{type(exc).__name__}: {str(exc)[:180]}",
|
|
92
|
+
}
|
|
93
|
+
return records, evidence
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _select_artifact(
|
|
97
|
+
spec: ExtractSpec,
|
|
98
|
+
artifacts: tuple[Artifact, ...],
|
|
99
|
+
by_name: dict[str, Artifact],
|
|
100
|
+
) -> Artifact:
|
|
101
|
+
if spec.artifact is not None:
|
|
102
|
+
if spec.artifact not in by_name:
|
|
103
|
+
raise ValueError(f"artifact {spec.artifact} is unavailable")
|
|
104
|
+
return by_name[spec.artifact]
|
|
105
|
+
if len(artifacts) != 1:
|
|
106
|
+
raise ValueError("extraction must name an artifact when the run has multiple")
|
|
107
|
+
return artifacts[0]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _extract(
|
|
111
|
+
spec: ExtractSpec, artifact: Artifact, payload: bytes
|
|
112
|
+
) -> tuple[Any, tuple[ExtractionRegion, ...]]:
|
|
113
|
+
if spec.engine == "provided":
|
|
114
|
+
if "content" not in spec.config:
|
|
115
|
+
raise ValueError("provided extraction requires config.content")
|
|
116
|
+
return spec.config["content"], ()
|
|
117
|
+
if spec.engine == "image.metadata":
|
|
118
|
+
try:
|
|
119
|
+
from PIL import Image
|
|
120
|
+
except ImportError as exc:
|
|
121
|
+
raise RuntimeError(
|
|
122
|
+
"install rubricloop[media] to extract image metadata"
|
|
123
|
+
) from exc
|
|
124
|
+
with Image.open(io.BytesIO(payload)) as image:
|
|
125
|
+
content = {
|
|
126
|
+
"width": image.width,
|
|
127
|
+
"height": image.height,
|
|
128
|
+
"mode": image.mode,
|
|
129
|
+
"format": image.format,
|
|
130
|
+
}
|
|
131
|
+
return content, (
|
|
132
|
+
ExtractionRegion(
|
|
133
|
+
id=f"image:{artifact.name}",
|
|
134
|
+
content=content,
|
|
135
|
+
confidence=1.0,
|
|
136
|
+
),
|
|
137
|
+
)
|
|
138
|
+
raise ValueError(f"unsupported local extraction engine: {spec.engine}")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _digest(value: Any) -> str:
|
|
142
|
+
encoded = json.dumps(
|
|
143
|
+
value,
|
|
144
|
+
sort_keys=True,
|
|
145
|
+
separators=(",", ":"),
|
|
146
|
+
ensure_ascii=False,
|
|
147
|
+
allow_nan=False,
|
|
148
|
+
).encode()
|
|
149
|
+
return "sha256:" + hashlib.sha256(encoded).hexdigest()
|
|
@@ -2,13 +2,27 @@
|
|
|
2
2
|
|
|
3
3
|
from .common import (
|
|
4
4
|
abstain,
|
|
5
|
+
citation_lookup,
|
|
5
6
|
forbidden_terms,
|
|
7
|
+
json_field_enum,
|
|
6
8
|
json_schema,
|
|
7
9
|
pii_free,
|
|
8
10
|
required_terms,
|
|
9
11
|
valid_json,
|
|
10
12
|
word_count,
|
|
11
13
|
)
|
|
14
|
+
from .media import (
|
|
15
|
+
EnvironmentResolver,
|
|
16
|
+
audio_script_adherence,
|
|
17
|
+
chart_data_match,
|
|
18
|
+
cross_modal_match,
|
|
19
|
+
layout_compliance,
|
|
20
|
+
media_metric,
|
|
21
|
+
media_pii_screen,
|
|
22
|
+
render_check,
|
|
23
|
+
synthetic_media_flag,
|
|
24
|
+
table_arithmetic,
|
|
25
|
+
)
|
|
12
26
|
from .sql import (
|
|
13
27
|
executes,
|
|
14
28
|
matches_reference,
|
|
@@ -21,10 +35,18 @@ from .sql import (
|
|
|
21
35
|
|
|
22
36
|
__all__ = [
|
|
23
37
|
"abstain",
|
|
38
|
+
"audio_script_adherence",
|
|
39
|
+
"chart_data_match",
|
|
40
|
+
"citation_lookup",
|
|
24
41
|
"executes",
|
|
25
42
|
"forbidden_terms",
|
|
43
|
+
"cross_modal_match",
|
|
44
|
+
"json_field_enum",
|
|
26
45
|
"json_schema",
|
|
27
46
|
"matches_reference",
|
|
47
|
+
"layout_compliance",
|
|
48
|
+
"media_metric",
|
|
49
|
+
"media_pii_screen",
|
|
28
50
|
"no_select_star",
|
|
29
51
|
"parses",
|
|
30
52
|
"pii_free",
|
|
@@ -32,6 +54,10 @@ __all__ = [
|
|
|
32
54
|
"required_columns",
|
|
33
55
|
"required_terms",
|
|
34
56
|
"schema_conformance",
|
|
57
|
+
"render_check",
|
|
58
|
+
"synthetic_media_flag",
|
|
59
|
+
"table_arithmetic",
|
|
35
60
|
"valid_json",
|
|
36
61
|
"word_count",
|
|
62
|
+
"EnvironmentResolver",
|
|
37
63
|
]
|