digline-bedrock 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- digline_bedrock-0.1.0/.gitignore +41 -0
- digline_bedrock-0.1.0/PKG-INFO +177 -0
- digline_bedrock-0.1.0/README.md +160 -0
- digline_bedrock-0.1.0/pyproject.toml +38 -0
- digline_bedrock-0.1.0/src/digline_bedrock/__init__.py +36 -0
- digline_bedrock-0.1.0/src/digline_bedrock/client.py +284 -0
- digline_bedrock-0.1.0/src/digline_bedrock/judge.py +117 -0
- digline_bedrock-0.1.0/src/digline_bedrock/pricing.py +161 -0
- digline_bedrock-0.1.0/src/digline_bedrock/py.typed +0 -0
- digline_bedrock-0.1.0/src/digline_bedrock/target.py +111 -0
- digline_bedrock-0.1.0/tests/_bedrock_fakes.py +81 -0
- digline_bedrock-0.1.0/tests/conftest.py +20 -0
- digline_bedrock-0.1.0/tests/test_bedrock_judge.py +385 -0
- digline_bedrock-0.1.0/tests/test_bedrock_readme.py +126 -0
- digline_bedrock-0.1.0/tests/test_bedrock_target.py +498 -0
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# Python bytecode and build output
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
build/
|
|
5
|
+
dist/
|
|
6
|
+
*.egg-info/
|
|
7
|
+
|
|
8
|
+
# Environment. Recreated by `uv sync`; it pins absolute paths, so it must
|
|
9
|
+
# never be committed.
|
|
10
|
+
.venv/
|
|
11
|
+
.env
|
|
12
|
+
|
|
13
|
+
# Tool caches. Each already drops its own `.gitignore`; listed here so a
|
|
14
|
+
# fresh clone is clean before the tools have run once.
|
|
15
|
+
.pytest_cache/
|
|
16
|
+
.ruff_cache/
|
|
17
|
+
.mypy_cache/
|
|
18
|
+
|
|
19
|
+
# IDE. Excluded because the project files carry machine-specific SDK paths
|
|
20
|
+
# (`digline.iml` names the interpreter by absolute path). Drop these two lines
|
|
21
|
+
# to version the shared part, and keep ignoring `.idea/workspace.xml`.
|
|
22
|
+
.idea/
|
|
23
|
+
*.iml
|
|
24
|
+
|
|
25
|
+
# Local Claude Code settings. `.claude/settings.json`, if it appears, is shared
|
|
26
|
+
# and stays versioned. `CLAUDE.local.md` is the personal working agreement —
|
|
27
|
+
# how I want to be worked with — as against `CLAUDE.md`, which is the project.
|
|
28
|
+
.claude/settings.local.json
|
|
29
|
+
CLAUDE.local.md
|
|
30
|
+
|
|
31
|
+
# macOS
|
|
32
|
+
.DS_Store
|
|
33
|
+
|
|
34
|
+
# Working material that stays local and is not part of the package.
|
|
35
|
+
private/
|
|
36
|
+
|
|
37
|
+
# NOT ignored: `.digline/`. Decision 2 — baselines are versioned, run
|
|
38
|
+
# artifacts are not — and the split is enforced one level down, by the
|
|
39
|
+
# `.gitignore` the store itself writes into `.digline/` (`*/runs/`).
|
|
40
|
+
# Ignoring `.digline/` here would take the baselines out of git with it.
|
|
41
|
+
to-publish/
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: digline-bedrock
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Amazon Bedrock target and judges for digline, on the Converse API.
|
|
5
|
+
Author-email: Alessandro Prandini <alessandro.prandini@ict-group.it>
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Classifier: Development Status :: 3 - Alpha
|
|
8
|
+
Classifier: Intended Audience :: Developers
|
|
9
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Classifier: Topic :: Software Development :: Testing
|
|
12
|
+
Classifier: Typing :: Typed
|
|
13
|
+
Requires-Python: >=3.12
|
|
14
|
+
Requires-Dist: boto3>=1.35
|
|
15
|
+
Requires-Dist: digline>=0.1.3
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
|
|
18
|
+
# digline-bedrock
|
|
19
|
+
|
|
20
|
+
An [Amazon Bedrock](https://aws.amazon.com/bedrock/) target **and judges** for
|
|
21
|
+
[digline](https://pypi.org/project/digline/), on the Converse API: a prompt file
|
|
22
|
+
goes in, a priced `Response` comes out.
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
pip install digline-bedrock
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
**Requires Python 3.12+**, like digline itself.
|
|
29
|
+
|
|
30
|
+
## Quickstart
|
|
31
|
+
|
|
32
|
+
No credential argument exists: the AWS chain — environment, profile, IAM role,
|
|
33
|
+
instance metadata — is boto3's job, and this package never reads it.
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from digline_bedrock import BedrockTarget
|
|
37
|
+
|
|
38
|
+
target = BedrockTarget(
|
|
39
|
+
"prompts/answer.md",
|
|
40
|
+
model="eu.anthropic.claude-sonnet-4-20250514-v1:0",
|
|
41
|
+
max_tokens=1024,
|
|
42
|
+
)
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`model` takes a model id or an inference profile id. The **region** is resolved
|
|
46
|
+
when the target is built — from `region=` if you pass one, otherwise from the
|
|
47
|
+
client the chain produced — and the price list follows from it:
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
from digline_bedrock import BedrockTarget
|
|
51
|
+
|
|
52
|
+
target = BedrockTarget(
|
|
53
|
+
"prompts/answer.md",
|
|
54
|
+
model="us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
55
|
+
max_tokens=1024,
|
|
56
|
+
region="us-east-1",
|
|
57
|
+
)
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
`target.region` is read-only, and that is the point: what was priced is what was
|
|
61
|
+
called. A missing region fails **there**, when the target is built, not on case
|
|
62
|
+
thirty-seven with thirty-six paid calls behind it.
|
|
63
|
+
|
|
64
|
+
## The judges run in the same account
|
|
65
|
+
|
|
66
|
+
A plugin is a target **and** a judge ([ADR
|
|
67
|
+
0004](https://github.com/digline/digline/blob/main/docs/adr/0004-every-plugin-is-a-target-and-a-judge.md)),
|
|
68
|
+
which on Bedrock is usually the whole reason the model is there: what a judge is
|
|
69
|
+
sent is the model's *output*, and it stays inside the same account, the same
|
|
70
|
+
region and the same IAM role.
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from digline.core import LlmRubric
|
|
74
|
+
from digline_bedrock import BedrockJudge
|
|
75
|
+
|
|
76
|
+
judge = BedrockJudge(model="eu.anthropic.claude-haiku-4-5-20251001-v1:0")
|
|
77
|
+
rubric = LlmRubric(
|
|
78
|
+
rubric="The answer is one sentence and cites the passage it came from.",
|
|
79
|
+
judge=judge,
|
|
80
|
+
threshold=0.8,
|
|
81
|
+
tolerance=0.05,
|
|
82
|
+
)
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
```python
|
|
86
|
+
from digline.core import Faithfulness
|
|
87
|
+
from digline_bedrock import BedrockClaimJudge
|
|
88
|
+
|
|
89
|
+
faithful = Faithfulness(
|
|
90
|
+
judge=BedrockClaimJudge(model="eu.anthropic.claude-haiku-4-5-20251001-v1:0"),
|
|
91
|
+
threshold=0.9,
|
|
92
|
+
tolerance=0.05,
|
|
93
|
+
)
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Converse has no structured-output mode, so the reply shape is asked for in the
|
|
97
|
+
system prompt and read back leniently — a fenced block or a sentence in front of
|
|
98
|
+
the object both parse. What judging cost is counted on the judge and never
|
|
99
|
+
reset:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
from digline_bedrock import BedrockJudge
|
|
103
|
+
|
|
104
|
+
judge = BedrockJudge(model="eu.anthropic.claude-haiku-4-5-20251001-v1:0")
|
|
105
|
+
print(f"{judge.calls} judgements, {judge.spent_usd:.4f} USD, {judge.latency_ms:.0f} ms")
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## Prices, and what is not priced
|
|
109
|
+
|
|
110
|
+
`bedrock_pricing(region)` is seeded for **us-east-1, us-west-2, eu-west-1,
|
|
111
|
+
eu-central-1 and eu-west-3**, with the Anthropic models. Bedrock prices by model
|
|
112
|
+
*and* by region, and a figure invented for a region nobody checked would be
|
|
113
|
+
wrong in the direction nobody notices — so everything else raises at `preflight`
|
|
114
|
+
and is served with one argument:
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
from digline.targets import ModelPrice
|
|
118
|
+
from digline_bedrock import BedrockTarget, bedrock_pricing
|
|
119
|
+
|
|
120
|
+
target = BedrockTarget(
|
|
121
|
+
"prompts/answer.md",
|
|
122
|
+
model="amazon.nova-pro-v1:0",
|
|
123
|
+
max_tokens=1024,
|
|
124
|
+
region="us-east-1",
|
|
125
|
+
pricing=bedrock_pricing("us-east-1").override(
|
|
126
|
+
"amazon.nova-pro-v1:0", ModelPrice(input_per_mtok=0.80, output_per_mtok=3.20)
|
|
127
|
+
),
|
|
128
|
+
)
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
An **application inference profile** is an ARN and is opaque: it is never in the
|
|
132
|
+
list, fails `preflight`, and is served the same way. That is intended, not a
|
|
133
|
+
gap — a run that cannot say what it cost must not run.
|
|
134
|
+
|
|
135
|
+
A model you brought in through **Custom Model Import**, or one behind
|
|
136
|
+
**Provisioned Throughput**, has no per-token bill at all: it is billed by
|
|
137
|
+
model-copy-hour and model-unit-hour. Say so out loud rather than leaving it
|
|
138
|
+
unpriced:
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
from digline_bedrock import BedrockTarget, free
|
|
142
|
+
|
|
143
|
+
target = BedrockTarget(
|
|
144
|
+
"prompts/answer.md",
|
|
145
|
+
model="my-imported-model",
|
|
146
|
+
max_tokens=1024,
|
|
147
|
+
region="eu-west-1",
|
|
148
|
+
pricing=free("my-imported-model"),
|
|
149
|
+
)
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## The details that bite
|
|
153
|
+
|
|
154
|
+
**Your account never leaves the machine.** A botocore failure names the assumed
|
|
155
|
+
role — `arn:aws:sts::<account>:assumed-role/…` — and digline quotes a target's
|
|
156
|
+
exception into the `reason` of every verdict of that case, which lands in a
|
|
157
|
+
committed run file. So every AWS error is re-raised as `BedrockCallFailed` with
|
|
158
|
+
ARNs and account ids removed; the original stays on `__cause__`, in memory, for
|
|
159
|
+
a debugger.
|
|
160
|
+
|
|
161
|
+
**Cached tokens.** Converse counts cached reads **beside** `inputTokens`, not
|
|
162
|
+
inside them — measured against the API on 2026-08-28, not inferred from the
|
|
163
|
+
field names: a warm call reported `inputTokens=10`, `cacheReadInputTokens=12002`
|
|
164
|
+
and `totalTokens=12016`. So they are added rather than subtracted, which is the
|
|
165
|
+
opposite of OpenAI's convention. It is one constant in `digline_bedrock.client`,
|
|
166
|
+
re-measured by a live test, because getting it wrong is invisible in the
|
|
167
|
+
direction of good news.
|
|
168
|
+
|
|
169
|
+
**`additional_request_fields`** reaches Converse's
|
|
170
|
+
`additionalModelRequestFields` and changes what the model does. It is **not**
|
|
171
|
+
part of `config_hash` — no more than `temperature`, `model` or `max_tokens` are:
|
|
172
|
+
the fingerprint covers the rules that judge a run, not the system being judged.
|
|
173
|
+
Two runs differing only in these fields will read as "same configuration as the
|
|
174
|
+
reference". If you need the difference to show, keep the fields in a file and
|
|
175
|
+
declare it in `Suite.artifacts`, and the report carries the diff.
|
|
176
|
+
|
|
177
|
+
Apache-2.0. Docs: [digline/digline](https://github.com/digline/digline).
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
# digline-bedrock
|
|
2
|
+
|
|
3
|
+
An [Amazon Bedrock](https://aws.amazon.com/bedrock/) target **and judges** for
|
|
4
|
+
[digline](https://pypi.org/project/digline/), on the Converse API: a prompt file
|
|
5
|
+
goes in, a priced `Response` comes out.
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
pip install digline-bedrock
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
**Requires Python 3.12+**, like digline itself.
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
No credential argument exists: the AWS chain — environment, profile, IAM role,
|
|
16
|
+
instance metadata — is boto3's job, and this package never reads it.
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
from digline_bedrock import BedrockTarget
|
|
20
|
+
|
|
21
|
+
target = BedrockTarget(
|
|
22
|
+
"prompts/answer.md",
|
|
23
|
+
model="eu.anthropic.claude-sonnet-4-20250514-v1:0",
|
|
24
|
+
max_tokens=1024,
|
|
25
|
+
)
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
`model` takes a model id or an inference profile id. The **region** is resolved
|
|
29
|
+
when the target is built — from `region=` if you pass one, otherwise from the
|
|
30
|
+
client the chain produced — and the price list follows from it:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from digline_bedrock import BedrockTarget
|
|
34
|
+
|
|
35
|
+
target = BedrockTarget(
|
|
36
|
+
"prompts/answer.md",
|
|
37
|
+
model="us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
|
38
|
+
max_tokens=1024,
|
|
39
|
+
region="us-east-1",
|
|
40
|
+
)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`target.region` is read-only, and that is the point: what was priced is what was
|
|
44
|
+
called. A missing region fails **there**, when the target is built, not on case
|
|
45
|
+
thirty-seven with thirty-six paid calls behind it.
|
|
46
|
+
|
|
47
|
+
## The judges run in the same account
|
|
48
|
+
|
|
49
|
+
A plugin is a target **and** a judge ([ADR
|
|
50
|
+
0004](https://github.com/digline/digline/blob/main/docs/adr/0004-every-plugin-is-a-target-and-a-judge.md)),
|
|
51
|
+
which on Bedrock is usually the whole reason the model is there: what a judge is
|
|
52
|
+
sent is the model's *output*, and it stays inside the same account, the same
|
|
53
|
+
region and the same IAM role.
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from digline.core import LlmRubric
|
|
57
|
+
from digline_bedrock import BedrockJudge
|
|
58
|
+
|
|
59
|
+
judge = BedrockJudge(model="eu.anthropic.claude-haiku-4-5-20251001-v1:0")
|
|
60
|
+
rubric = LlmRubric(
|
|
61
|
+
rubric="The answer is one sentence and cites the passage it came from.",
|
|
62
|
+
judge=judge,
|
|
63
|
+
threshold=0.8,
|
|
64
|
+
tolerance=0.05,
|
|
65
|
+
)
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from digline.core import Faithfulness
|
|
70
|
+
from digline_bedrock import BedrockClaimJudge
|
|
71
|
+
|
|
72
|
+
faithful = Faithfulness(
|
|
73
|
+
judge=BedrockClaimJudge(model="eu.anthropic.claude-haiku-4-5-20251001-v1:0"),
|
|
74
|
+
threshold=0.9,
|
|
75
|
+
tolerance=0.05,
|
|
76
|
+
)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Converse has no structured-output mode, so the reply shape is asked for in the
|
|
80
|
+
system prompt and read back leniently — a fenced block or a sentence in front of
|
|
81
|
+
the object both parse. What judging cost is counted on the judge and never
|
|
82
|
+
reset:
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from digline_bedrock import BedrockJudge
|
|
86
|
+
|
|
87
|
+
judge = BedrockJudge(model="eu.anthropic.claude-haiku-4-5-20251001-v1:0")
|
|
88
|
+
print(f"{judge.calls} judgements, {judge.spent_usd:.4f} USD, {judge.latency_ms:.0f} ms")
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
## Prices, and what is not priced
|
|
92
|
+
|
|
93
|
+
`bedrock_pricing(region)` is seeded for **us-east-1, us-west-2, eu-west-1,
|
|
94
|
+
eu-central-1 and eu-west-3**, with the Anthropic models. Bedrock prices by model
|
|
95
|
+
*and* by region, and a figure invented for a region nobody checked would be
|
|
96
|
+
wrong in the direction nobody notices — so everything else raises at `preflight`
|
|
97
|
+
and is served with one argument:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
from digline.targets import ModelPrice
|
|
101
|
+
from digline_bedrock import BedrockTarget, bedrock_pricing
|
|
102
|
+
|
|
103
|
+
target = BedrockTarget(
|
|
104
|
+
"prompts/answer.md",
|
|
105
|
+
model="amazon.nova-pro-v1:0",
|
|
106
|
+
max_tokens=1024,
|
|
107
|
+
region="us-east-1",
|
|
108
|
+
pricing=bedrock_pricing("us-east-1").override(
|
|
109
|
+
"amazon.nova-pro-v1:0", ModelPrice(input_per_mtok=0.80, output_per_mtok=3.20)
|
|
110
|
+
),
|
|
111
|
+
)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
An **application inference profile** is an ARN and is opaque: it is never in the
|
|
115
|
+
list, fails `preflight`, and is served the same way. That is intended, not a
|
|
116
|
+
gap — a run that cannot say what it cost must not run.
|
|
117
|
+
|
|
118
|
+
A model you brought in through **Custom Model Import**, or one behind
|
|
119
|
+
**Provisioned Throughput**, has no per-token bill at all: it is billed by
|
|
120
|
+
model-copy-hour and model-unit-hour. Say so out loud rather than leaving it
|
|
121
|
+
unpriced:
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
from digline_bedrock import BedrockTarget, free
|
|
125
|
+
|
|
126
|
+
target = BedrockTarget(
|
|
127
|
+
"prompts/answer.md",
|
|
128
|
+
model="my-imported-model",
|
|
129
|
+
max_tokens=1024,
|
|
130
|
+
region="eu-west-1",
|
|
131
|
+
pricing=free("my-imported-model"),
|
|
132
|
+
)
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
## The details that bite
|
|
136
|
+
|
|
137
|
+
**Your account never leaves the machine.** A botocore failure names the assumed
|
|
138
|
+
role — `arn:aws:sts::<account>:assumed-role/…` — and digline quotes a target's
|
|
139
|
+
exception into the `reason` of every verdict of that case, which lands in a
|
|
140
|
+
committed run file. So every AWS error is re-raised as `BedrockCallFailed` with
|
|
141
|
+
ARNs and account ids removed; the original stays on `__cause__`, in memory, for
|
|
142
|
+
a debugger.
|
|
143
|
+
|
|
144
|
+
**Cached tokens.** Converse counts cached reads **beside** `inputTokens`, not
|
|
145
|
+
inside them — measured against the API on 2026-08-28, not inferred from the
|
|
146
|
+
field names: a warm call reported `inputTokens=10`, `cacheReadInputTokens=12002`
|
|
147
|
+
and `totalTokens=12016`. So they are added rather than subtracted, which is the
|
|
148
|
+
opposite of OpenAI's convention. It is one constant in `digline_bedrock.client`,
|
|
149
|
+
re-measured by a live test, because getting it wrong is invisible in the
|
|
150
|
+
direction of good news.
|
|
151
|
+
|
|
152
|
+
**`additional_request_fields`** reaches Converse's
|
|
153
|
+
`additionalModelRequestFields` and changes what the model does. It is **not**
|
|
154
|
+
part of `config_hash` — no more than `temperature`, `model` or `max_tokens` are:
|
|
155
|
+
the fingerprint covers the rules that judge a run, not the system being judged.
|
|
156
|
+
Two runs differing only in these fields will read as "same configuration as the
|
|
157
|
+
reference". If you need the difference to show, keep the fields in a file and
|
|
158
|
+
declare it in `Suite.artifacts`, and the report carries the diff.
|
|
159
|
+
|
|
160
|
+
Apache-2.0. Docs: [digline/digline](https://github.com/digline/digline).
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "digline-bedrock"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Amazon Bedrock target and judges for digline, on the Converse API."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.12"
|
|
7
|
+
license = "Apache-2.0"
|
|
8
|
+
authors = [
|
|
9
|
+
{ name = "Alessandro Prandini", email = "alessandro.prandini@ict-group.it" },
|
|
10
|
+
]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 3 - Alpha",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
15
|
+
"Programming Language :: Python :: 3.12",
|
|
16
|
+
"Topic :: Software Development :: Testing",
|
|
17
|
+
"Typing :: Typed",
|
|
18
|
+
]
|
|
19
|
+
dependencies = [
|
|
20
|
+
# Pinned to a version that exists, like the other two plugins: published,
|
|
21
|
+
# an unpinned `digline` would let a resolver pick something older than the
|
|
22
|
+
# API this plugin is written against — `JudgeBase` and its two subclasses
|
|
23
|
+
# are 0.1.3.
|
|
24
|
+
"digline>=0.1.3",
|
|
25
|
+
# boto3 and nothing else. No wrapper, no gateway: the Converse API is one
|
|
26
|
+
# call, and a layer on top would only add a second thing to be wrong.
|
|
27
|
+
"boto3>=1.35",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[build-system]
|
|
31
|
+
requires = ["hatchling>=1.27"]
|
|
32
|
+
build-backend = "hatchling.build"
|
|
33
|
+
|
|
34
|
+
[tool.hatch.build.targets.wheel]
|
|
35
|
+
packages = ["src/digline_bedrock"]
|
|
36
|
+
|
|
37
|
+
[tool.uv.sources]
|
|
38
|
+
digline = { workspace = true }
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Amazon Bedrock target and judges for digline, on the Converse API.
|
|
2
|
+
|
|
3
|
+
Installed beside digline, never inside it: `pip install digline` must not pull
|
|
4
|
+
somebody's HTTP client along with it.
|
|
5
|
+
|
|
6
|
+
A plugin is a target **and** a judge (ADR 0004), so a suite generates and judges
|
|
7
|
+
inside one AWS account, in one region, under one IAM role — which for Bedrock is
|
|
8
|
+
usually the whole reason the model is there.
|
|
9
|
+
|
|
10
|
+
No credential ever reaches this package: the chain is boto3's job.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from digline_bedrock.client import BedrockCallFailed, BedrockChat, scrub
|
|
14
|
+
from digline_bedrock.judge import BedrockClaimJudge, BedrockJudge
|
|
15
|
+
from digline_bedrock.pricing import (
|
|
16
|
+
BASE_PRICES,
|
|
17
|
+
PRICES_READ_ON,
|
|
18
|
+
SEEDED_REGIONS,
|
|
19
|
+
bedrock_pricing,
|
|
20
|
+
free,
|
|
21
|
+
)
|
|
22
|
+
from digline_bedrock.target import BedrockTarget
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"BASE_PRICES",
|
|
26
|
+
"PRICES_READ_ON",
|
|
27
|
+
"SEEDED_REGIONS",
|
|
28
|
+
"BedrockCallFailed",
|
|
29
|
+
"BedrockChat",
|
|
30
|
+
"BedrockClaimJudge",
|
|
31
|
+
"BedrockJudge",
|
|
32
|
+
"BedrockTarget",
|
|
33
|
+
"bedrock_pricing",
|
|
34
|
+
"free",
|
|
35
|
+
"scrub",
|
|
36
|
+
]
|