annotide-training 0.1.0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- annotide_training-0.1.2/LICENSE.md +110 -0
- annotide_training-0.1.2/PKG-INFO +217 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/README.md +1 -1
- annotide_training-0.1.2/annotide_training.egg-info/PKG-INFO +217 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/SOURCES.txt +1 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/pyproject.toml +29 -5
- annotide_training-0.1.0/PKG-INFO +0 -19
- annotide_training-0.1.0/annotide_training.egg-info/PKG-INFO +0 -19
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/__init__.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/cli.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/dataset.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/detector.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/evaluate.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/pipeline.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/tracking.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/trainers.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training/webhook.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/dependency_links.txt +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/entry_points.txt +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/requires.txt +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/top_level.txt +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/setup.cfg +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/tests/test_dataset.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/tests/test_detector.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/tests/test_pipeline.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/tests/test_trainers.py +0 -0
- {annotide_training-0.1.0 → annotide_training-0.1.2}/tests/test_webhook.py +0 -0
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
Copyright (c) 2026 R Squared Data Solutions (https://annotide.com)
|
|
2
|
+
|
|
3
|
+
Annotide is licensed under the Elastic License 2.0, reproduced below.
|
|
4
|
+
|
|
5
|
+
In plain words (the terms below govern, not this summary): you may use,
|
|
6
|
+
copy, modify and run Annotide, including for commercial work. Without a
|
|
7
|
+
licence key it runs as the Community edition, for up to three active users.
|
|
8
|
+
A licence key (https://annotide.com/pricing) adds users and the Business
|
|
9
|
+
features. You may not get around the licence key or remove what it
|
|
10
|
+
protects, and you may not offer Annotide to others as a hosted or managed
|
|
11
|
+
service.
|
|
12
|
+
|
|
13
|
+
The Python SDK, CLI and MCP server in `sdk/` (`pip install annotide`) are
|
|
14
|
+
licensed separately under the Apache License 2.0; see `sdk/LICENSE`.
|
|
15
|
+
|
|
16
|
+
---
|
|
17
|
+
|
|
18
|
+
Elastic License 2.0
|
|
19
|
+
|
|
20
|
+
URL: https://www.elastic.co/licensing/elastic-license
|
|
21
|
+
|
|
22
|
+
Acceptance
|
|
23
|
+
|
|
24
|
+
By using the software, you agree to all of the terms and conditions below.
|
|
25
|
+
|
|
26
|
+
Copyright License
|
|
27
|
+
|
|
28
|
+
The licensor grants you a non-exclusive, royalty-free, worldwide,
|
|
29
|
+
non-sublicensable, non-transferable license to use, copy, distribute, make
|
|
30
|
+
available, and prepare derivative works of the software, in each case subject to
|
|
31
|
+
the limitations and conditions below.
|
|
32
|
+
|
|
33
|
+
Limitations
|
|
34
|
+
|
|
35
|
+
You may not provide the software to third parties as a hosted or managed
|
|
36
|
+
service, where the service provides users with access to any substantial set of
|
|
37
|
+
the features or functionality of the software.
|
|
38
|
+
|
|
39
|
+
You may not move, change, disable, or circumvent the license key functionality
|
|
40
|
+
in the software, and you may not remove or obscure any functionality in the
|
|
41
|
+
software that is protected by the license key.
|
|
42
|
+
|
|
43
|
+
You may not alter, remove, or obscure any licensing, copyright, or other notices
|
|
44
|
+
of the licensor in the software. Any use of the licensor’s trademarks is subject
|
|
45
|
+
to applicable law.
|
|
46
|
+
|
|
47
|
+
Patents
|
|
48
|
+
|
|
49
|
+
The licensor grants you a license, under any patent claims the licensor can
|
|
50
|
+
license, or becomes able to license, to make, have made, use, sell, offer for
|
|
51
|
+
sale, import and have imported the software, in each case subject to the
|
|
52
|
+
limitations and conditions in this license. This license does not cover any
|
|
53
|
+
patent claims that you cause to be infringed by modifications or additions to
|
|
54
|
+
the software. If you or your company make any written claim that the software
|
|
55
|
+
infringes or contributes to infringement of any patent, your patent license for
|
|
56
|
+
the software granted under these terms ends immediately. If your company makes
|
|
57
|
+
such a claim, your patent license ends immediately for work on behalf of your
|
|
58
|
+
company.
|
|
59
|
+
|
|
60
|
+
Notices
|
|
61
|
+
|
|
62
|
+
You must ensure that anyone who gets a copy of any part of the software from you
|
|
63
|
+
also gets a copy of these terms.
|
|
64
|
+
|
|
65
|
+
If you modify the software, you must include in any modified copies of the
|
|
66
|
+
software prominent notices stating that you have modified the software.
|
|
67
|
+
|
|
68
|
+
## No Other Rights
|
|
69
|
+
|
|
70
|
+
These terms do not imply any licenses other than those expressly granted in
|
|
71
|
+
these terms.
|
|
72
|
+
|
|
73
|
+
Termination
|
|
74
|
+
|
|
75
|
+
If you use the software in violation of these terms, such use is not licensed,
|
|
76
|
+
and your licenses will automatically terminate. If the licensor provides you
|
|
77
|
+
with a notice of your violation, and you cease all violation of this license no
|
|
78
|
+
later than 30 days after you receive that notice, your licenses will be
|
|
79
|
+
reinstated retroactively. However, if you violate these terms after such
|
|
80
|
+
reinstatement, any additional violation of these terms will cause your licenses
|
|
81
|
+
to terminate automatically and permanently.
|
|
82
|
+
|
|
83
|
+
No Liability
|
|
84
|
+
|
|
85
|
+
As far as the law allows, the software comes as is, without any warranty or
|
|
86
|
+
condition, and the licensor will not be liable to you for any damages arising
|
|
87
|
+
out of these terms or the use or nature of the software, under any kind of
|
|
88
|
+
legal claim.
|
|
89
|
+
|
|
90
|
+
Definitions
|
|
91
|
+
|
|
92
|
+
The licensor is the entity offering these terms, and the software is the
|
|
93
|
+
software the licensor makes available under these terms, including any portion
|
|
94
|
+
of it.
|
|
95
|
+
|
|
96
|
+
you refers to the individual or entity agreeing to these terms.
|
|
97
|
+
|
|
98
|
+
your company is any legal entity, sole proprietorship, or other kind of
|
|
99
|
+
organization that you work for, plus all organizations that have control over,
|
|
100
|
+
are under the control of, or are under common control with that
|
|
101
|
+
organization. control means ownership of substantially all the assets of an
|
|
102
|
+
entity, or the power to direct its management and policies by vote, contract, or
|
|
103
|
+
otherwise. Control can be direct or indirect.
|
|
104
|
+
|
|
105
|
+
your licenses are all the licenses granted to you for the software under
|
|
106
|
+
these terms.
|
|
107
|
+
|
|
108
|
+
use means anything you do with the software requiring one of your licenses.
|
|
109
|
+
|
|
110
|
+
trademark means trademarks, service marks, and similar rights.
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: annotide-training
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: Reference training pipeline for the Annotide annotation platform, run on your own compute
|
|
5
|
+
Author: Annotide
|
|
6
|
+
License-Expression: Elastic-2.0
|
|
7
|
+
Project-URL: Homepage, https://annotide.com
|
|
8
|
+
Project-URL: Documentation, https://github.com/annotide/annotide/tree/main/training#readme
|
|
9
|
+
Project-URL: Source, https://github.com/annotide/annotide/tree/main/training
|
|
10
|
+
Project-URL: Issues, https://github.com/annotide/annotide/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/annotide/annotide/releases
|
|
12
|
+
Keywords: annotation,training,computer-vision,mlops,onnx,mlflow
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
22
|
+
Requires-Python: >=3.12
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE.md
|
|
25
|
+
Requires-Dist: annotide>=0.1
|
|
26
|
+
Provides-Extra: torch
|
|
27
|
+
Requires-Dist: torch>=2.5; extra == "torch"
|
|
28
|
+
Requires-Dist: torchvision>=0.20; extra == "torch"
|
|
29
|
+
Requires-Dist: onnx>=1.16; extra == "torch"
|
|
30
|
+
Requires-Dist: onnxruntime>=1.20; extra == "torch"
|
|
31
|
+
Requires-Dist: numpy>=2.1; extra == "torch"
|
|
32
|
+
Requires-Dist: pillow>=11.0; extra == "torch"
|
|
33
|
+
Provides-Extra: mlflow
|
|
34
|
+
Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
|
|
35
|
+
Provides-Extra: dev
|
|
36
|
+
Requires-Dist: pytest>=8.3; extra == "dev"
|
|
37
|
+
Requires-Dist: ruff>=0.8; extra == "dev"
|
|
38
|
+
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# Reference training pipeline
|
|
42
|
+
|
|
43
|
+
The platform never trains a model itself (ML-9). It freezes a snapshot,
|
|
44
|
+
emits `retrain.requested` to the project's webhooks, and records whatever
|
|
45
|
+
version comes back with its lineage (EXP-8). This package is the other end:
|
|
46
|
+
a pipeline a customer runs next to their own compute.
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
retrain.requested ─▶ prepare ─▶ split ─▶ train ─▶ evaluate ─▶ register
|
|
50
|
+
(webhook) export EXP-3 Trainer box P/R/F1 POST /models/{id}/versions
|
|
51
|
+
+ digest snapshot_id + digest + training_run
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
1. **prepare** — reads the snapshot row, refuses a digest that is not the
|
|
55
|
+
snapshot's, queues a `native` export of it, downloads the archive through
|
|
56
|
+
its signed URL (the API key never goes to the storage host) and checks
|
|
57
|
+
the archive's manifest names the same snapshot and digest.
|
|
58
|
+
2. **split** — uses the snapshot's own train / val / test partition when it
|
|
59
|
+
has one; otherwise splits with the platform's rule
|
|
60
|
+
(`sha256("{seed}:item:{id}")`), so the same seed gives the same sets.
|
|
61
|
+
3. **train** — a pluggable `Trainer` (`annotide_training/trainers.py`).
|
|
62
|
+
4. **evaluate** — box precision / recall / F1 at IoU 0.5, per class and
|
|
63
|
+
overall, on `test` (or `val` when `test` is empty). The same scorer for
|
|
64
|
+
every trainer.
|
|
65
|
+
5. **register** — adds a model version with `snapshot_id`,
|
|
66
|
+
`snapshot_digest`, the metrics and a `training_run` record (run id,
|
|
67
|
+
trainer, params, timings, export job, split counts, artifact location,
|
|
68
|
+
webhook delivery id). The platform re-checks the digest (409 otherwise).
|
|
69
|
+
|
|
70
|
+
Artifacts and a `run.json` land in `runs/{run_id}/`. Where the artifact goes
|
|
71
|
+
next — an ONNX file for `model-service`'s `onnx` backend, a model registry —
|
|
72
|
+
is the trainer's business; the platform only stores the lineage.
|
|
73
|
+
|
|
74
|
+
## Run it
|
|
75
|
+
|
|
76
|
+
```sh
|
|
77
|
+
cd training
|
|
78
|
+
python3.12 -m venv .venv
|
|
79
|
+
.venv/bin/pip install -e ../sdk # first: the local SDK, not the PyPI release
|
|
80
|
+
.venv/bin/pip install -e '.[dev]'
|
|
81
|
+
|
|
82
|
+
export ANNOTIDE_API_URL=http://localhost:8000 # site root, not /api/v1
|
|
83
|
+
export ANNOTIDE_API_KEY=... # an API key with the `write` scope (AUTH-4)
|
|
84
|
+
|
|
85
|
+
# One snapshot, now. --no-register trains and evaluates without adding a version.
|
|
86
|
+
.venv/bin/annotide-train --model <model-id> run --project <project-id> --snapshot <snapshot-id>
|
|
87
|
+
|
|
88
|
+
# Or receive webhooks: subscribe http://<host>:8088/ to retrain.requested
|
|
89
|
+
export ANNOTIDE_WEBHOOK_SECRET=... # the webhook's signing secret
|
|
90
|
+
.venv/bin/annotide-train --model <fallback-model-id> serve --host 0.0.0.0 --port 8088
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
`serve` verifies `X-Annotation-Signature` (5 min tolerance, constant-time),
|
|
94
|
+
answers `202` at once and trains on a thread (deliveries time out), skips a
|
|
95
|
+
repeated `X-Annotation-Delivery`, and acknowledges signed events it does not
|
|
96
|
+
act on (another event, no snapshot) with `200` so they are not retried.
|
|
97
|
+
The event's `model_id` wins over `--model`.
|
|
98
|
+
|
|
99
|
+
Options: `--trainer` (`baseline`, `fasterrcnn` or `package.module:ClassName`),
|
|
100
|
+
`--param key=value` (JSON values; repeatable, passed to the trainer),
|
|
101
|
+
`--split 0.8,0.1,0.1` (unsplit snapshots only), `--out runs`,
|
|
102
|
+
`--quantize int8` and `--teacher VERSION_ID` (below).
|
|
103
|
+
|
|
104
|
+
Every registered version carries the metrics the Models page compares
|
|
105
|
+
versions on: `precision` / `recall` / `f1` and `per_class` on the held-out
|
|
106
|
+
split, `size_bytes`, `latency_ms_p50` (median `predict` time per item on this
|
|
107
|
+
machine, `latency_device`) and `dtype`.
|
|
108
|
+
|
|
109
|
+
### Distillation and quantization
|
|
110
|
+
|
|
111
|
+
Both are recorded as the version's `derivation` and `parent_version_id`
|
|
112
|
+
(`docs/CONTRACTS.md`), which the Models page draws as the model's family.
|
|
113
|
+
|
|
114
|
+
**Distillation** here is the annotation loop with a big model as the
|
|
115
|
+
teacher: pre-label with the teacher's version (`POST /projects/{id}/prelabel`),
|
|
116
|
+
let people correct the drafts, take a snapshot, and train a smaller model on
|
|
117
|
+
it with `--teacher <the teacher's version id>`. The new version is registered
|
|
118
|
+
as `distilled` from the teacher, so its F1, size and latency sit next to the
|
|
119
|
+
teacher's. The teacher can be any model of the organisation, including one
|
|
120
|
+
the platform only calls through an endpoint.
|
|
121
|
+
|
|
122
|
+
**Quantization**: `--quantize int8` stores the trained model again at 8 bits,
|
|
123
|
+
scores it on the same held-out split and registers it as a `quantized` child
|
|
124
|
+
of the version this run registered, in `runs/{run_id}/int8/`. The trainer
|
|
125
|
+
implements `quantize(model, calibration, dtype)`; the calibration records are
|
|
126
|
+
a fixed-seed sample of the train split. `fasterrcnn` uses onnxruntime's static
|
|
127
|
+
QDQ quantization (Conv, Gemm, MatMul, per-channel weights); in a smoke test
|
|
128
|
+
the 76 MB model became 20 MB. How much faster it runs depends on the CPU:
|
|
129
|
+
x86 with VNNI / AMX gains most, Apple silicon little. Read the int8
|
|
130
|
+
version's F1 before switching prelabelling to it. `baseline` stores its
|
|
131
|
+
priors as 8-bit fractions, so the path runs without torch.
|
|
132
|
+
|
|
133
|
+
### Recording runs in MLflow (API-6)
|
|
134
|
+
|
|
135
|
+
With the `mlflow` extra (`pip install -e '.[dev,mlflow]'`), `--mlflow-experiment
|
|
136
|
+
NAME` also records each run in MLflow — a local server (`make mlflow` at the
|
|
137
|
+
repo root, `--mlflow-uri http://localhost:5001`), Databricks or Azure ML,
|
|
138
|
+
wherever `MLFLOW_TRACKING_URI` points and however that environment signs in.
|
|
139
|
+
The run gets the params, the numeric metrics, the artifacts under `model/`,
|
|
140
|
+
the tags `annotation.snapshot_id` / `annotation.snapshot_digest` and the
|
|
141
|
+
snapshot as its dataset input (`source_type` `annotation-snapshot`).
|
|
142
|
+
`--mlflow-register NAME` also registers it as a version of that registered
|
|
143
|
+
model. The platform's *Import from MLflow* (`POST /models/{id}/versions/import`)
|
|
144
|
+
reads the lineage back from the run or the registered version, so a version
|
|
145
|
+
imported that way is linked to its snapshot as one registered by this
|
|
146
|
+
pipeline is. `training_run.mlflow` names the run either way.
|
|
147
|
+
|
|
148
|
+
## The Faster R-CNN trainer
|
|
149
|
+
|
|
150
|
+
`--trainer fasterrcnn` fine-tunes torchvision's Faster R-CNN
|
|
151
|
+
(MobileNetV3-Large FPN) on the snapshot's boxes and exports it to ONNX in the
|
|
152
|
+
format `model-service`'s `onnx` backend loads:
|
|
153
|
+
|
|
154
|
+
```sh
|
|
155
|
+
pip install -e '.[torch]' # torch, torchvision, onnx, onnxruntime (CPU wheels:
|
|
156
|
+
# --index-url https://download.pytorch.org/whl/cpu)
|
|
157
|
+
annotide-train run … --trainer fasterrcnn \
|
|
158
|
+
--param image_root=/mnt/images --param epochs=20
|
|
159
|
+
|
|
160
|
+
# runs/{run_id}/fasterrcnn.onnx + fasterrcnn.names → the model service:
|
|
161
|
+
MODEL_BACKEND=onnx MODEL_PATH=runs/{run_id}/fasterrcnn.onnx uvicorn app.main:app
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
- **Images** come from `image_root`, a directory where the customer's own
|
|
165
|
+
storage is mounted or synced (blobfuse2, gcsfuse, `aws s3 sync`): export
|
|
166
|
+
records carry only each item's `path`, and the platform never serves media
|
|
167
|
+
to the trainer. Missing files fail the run before training starts; a path
|
|
168
|
+
that resolves outside `image_root` is refused.
|
|
169
|
+
- **Params**: `epochs` (10), `batch_size` (4), `lr` (0.01, SGD + cosine),
|
|
170
|
+
`weights` (`coco` | `imagenet` | `none`), `hflip` (true), `seed` (0),
|
|
171
|
+
`device` (`cuda` if available, else `cpu`), `score_threshold` (0.5).
|
|
172
|
+
- **Input** is letterboxed to 640 × 640 with grey padding, the same transform
|
|
173
|
+
the model service applies, for training, export and evaluation alike.
|
|
174
|
+
- **Evaluation runs the exported ONNX file** through onnxruntime, not the
|
|
175
|
+
torch model, so the registered metrics are for the artifact that ships.
|
|
176
|
+
The export is also checked against torch on a few training images and a
|
|
177
|
+
blank frame before it is accepted.
|
|
178
|
+
- **Licences**: torchvision, onnx and onnxruntime are BSD / Apache / MIT.
|
|
179
|
+
`weights=coco` starts from torchvision's COCO checkpoint; check its terms
|
|
180
|
+
for your use, or start from `imagenet` / `none`.
|
|
181
|
+
- **Cost**: the artifact is ~76 MB. On an Apple M-series CPU, 48 images ×
|
|
182
|
+
6 epochs take about 90 s; use `device=cuda` for real datasets.
|
|
183
|
+
|
|
184
|
+
## Bring your own trainer
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
from annotide_training.trainers import Prediction, TrainedModel
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
class MyDetector:
|
|
191
|
+
name = "my-detector"
|
|
192
|
+
|
|
193
|
+
def train(self, train, val, params) -> TrainedModel:
|
|
194
|
+
# train: native export records — item_id, path, width, height, shapes[…]
|
|
195
|
+
# Media is the customer's own storage: read it from there.
|
|
196
|
+
...
|
|
197
|
+
return TrainedModel(artifact=onnx_bytes, filename="model.onnx", classes=[...])
|
|
198
|
+
|
|
199
|
+
def predict(self, model, record) -> list[Prediction]: ...
|
|
200
|
+
|
|
201
|
+
# Optional, for --quantize: the same model at lower precision, loadable
|
|
202
|
+
# by predict above.
|
|
203
|
+
def quantize(self, model, calibration, dtype) -> TrainedModel: ...
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
`annotide-train --trainer my_package.detector:MyDetector …`. The framework
|
|
207
|
+
it needs is its own dependency, never the platform's.
|
|
208
|
+
|
|
209
|
+
`BaselineTrainer` needs nothing: it learns how often each class appears and
|
|
210
|
+
where it usually sits, and predicts that. It exists to exercise every stage
|
|
211
|
+
end to end and to give a floor a real model has to beat.
|
|
212
|
+
|
|
213
|
+
## Develop
|
|
214
|
+
|
|
215
|
+
```sh
|
|
216
|
+
.venv/bin/ruff check . && .venv/bin/ruff format --check . && .venv/bin/mypy annotide_training tests && .venv/bin/pytest
|
|
217
|
+
```
|
|
@@ -36,7 +36,7 @@ is the trainer's business; the platform only stores the lineage.
|
|
|
36
36
|
```sh
|
|
37
37
|
cd training
|
|
38
38
|
python3.12 -m venv .venv
|
|
39
|
-
.venv/bin/pip install -e ../sdk # first: the SDK
|
|
39
|
+
.venv/bin/pip install -e ../sdk # first: the local SDK, not the PyPI release
|
|
40
40
|
.venv/bin/pip install -e '.[dev]'
|
|
41
41
|
|
|
42
42
|
export ANNOTIDE_API_URL=http://localhost:8000 # site root, not /api/v1
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: annotide-training
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: Reference training pipeline for the Annotide annotation platform, run on your own compute
|
|
5
|
+
Author: Annotide
|
|
6
|
+
License-Expression: Elastic-2.0
|
|
7
|
+
Project-URL: Homepage, https://annotide.com
|
|
8
|
+
Project-URL: Documentation, https://github.com/annotide/annotide/tree/main/training#readme
|
|
9
|
+
Project-URL: Source, https://github.com/annotide/annotide/tree/main/training
|
|
10
|
+
Project-URL: Issues, https://github.com/annotide/annotide/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/annotide/annotide/releases
|
|
12
|
+
Keywords: annotation,training,computer-vision,mlops,onnx,mlflow
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
22
|
+
Requires-Python: >=3.12
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE.md
|
|
25
|
+
Requires-Dist: annotide>=0.1
|
|
26
|
+
Provides-Extra: torch
|
|
27
|
+
Requires-Dist: torch>=2.5; extra == "torch"
|
|
28
|
+
Requires-Dist: torchvision>=0.20; extra == "torch"
|
|
29
|
+
Requires-Dist: onnx>=1.16; extra == "torch"
|
|
30
|
+
Requires-Dist: onnxruntime>=1.20; extra == "torch"
|
|
31
|
+
Requires-Dist: numpy>=2.1; extra == "torch"
|
|
32
|
+
Requires-Dist: pillow>=11.0; extra == "torch"
|
|
33
|
+
Provides-Extra: mlflow
|
|
34
|
+
Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
|
|
35
|
+
Provides-Extra: dev
|
|
36
|
+
Requires-Dist: pytest>=8.3; extra == "dev"
|
|
37
|
+
Requires-Dist: ruff>=0.8; extra == "dev"
|
|
38
|
+
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# Reference training pipeline
|
|
42
|
+
|
|
43
|
+
The platform never trains a model itself (ML-9). It freezes a snapshot,
|
|
44
|
+
emits `retrain.requested` to the project's webhooks, and records whatever
|
|
45
|
+
version comes back with its lineage (EXP-8). This package is the other end:
|
|
46
|
+
a pipeline a customer runs next to their own compute.
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
retrain.requested ─▶ prepare ─▶ split ─▶ train ─▶ evaluate ─▶ register
|
|
50
|
+
(webhook) export EXP-3 Trainer box P/R/F1 POST /models/{id}/versions
|
|
51
|
+
+ digest snapshot_id + digest + training_run
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
1. **prepare** — reads the snapshot row, refuses a digest that is not the
|
|
55
|
+
snapshot's, queues a `native` export of it, downloads the archive through
|
|
56
|
+
its signed URL (the API key never goes to the storage host) and checks
|
|
57
|
+
the archive's manifest names the same snapshot and digest.
|
|
58
|
+
2. **split** — uses the snapshot's own train / val / test partition when it
|
|
59
|
+
has one; otherwise splits with the platform's rule
|
|
60
|
+
(`sha256("{seed}:item:{id}")`), so the same seed gives the same sets.
|
|
61
|
+
3. **train** — a pluggable `Trainer` (`annotide_training/trainers.py`).
|
|
62
|
+
4. **evaluate** — box precision / recall / F1 at IoU 0.5, per class and
|
|
63
|
+
overall, on `test` (or `val` when `test` is empty). The same scorer for
|
|
64
|
+
every trainer.
|
|
65
|
+
5. **register** — adds a model version with `snapshot_id`,
|
|
66
|
+
`snapshot_digest`, the metrics and a `training_run` record (run id,
|
|
67
|
+
trainer, params, timings, export job, split counts, artifact location,
|
|
68
|
+
webhook delivery id). The platform re-checks the digest (409 otherwise).
|
|
69
|
+
|
|
70
|
+
Artifacts and a `run.json` land in `runs/{run_id}/`. Where the artifact goes
|
|
71
|
+
next — an ONNX file for `model-service`'s `onnx` backend, a model registry —
|
|
72
|
+
is the trainer's business; the platform only stores the lineage.
|
|
73
|
+
|
|
74
|
+
## Run it
|
|
75
|
+
|
|
76
|
+
```sh
|
|
77
|
+
cd training
|
|
78
|
+
python3.12 -m venv .venv
|
|
79
|
+
.venv/bin/pip install -e ../sdk # first: the local SDK, not the PyPI release
|
|
80
|
+
.venv/bin/pip install -e '.[dev]'
|
|
81
|
+
|
|
82
|
+
export ANNOTIDE_API_URL=http://localhost:8000 # site root, not /api/v1
|
|
83
|
+
export ANNOTIDE_API_KEY=... # an API key with the `write` scope (AUTH-4)
|
|
84
|
+
|
|
85
|
+
# One snapshot, now. --no-register trains and evaluates without adding a version.
|
|
86
|
+
.venv/bin/annotide-train --model <model-id> run --project <project-id> --snapshot <snapshot-id>
|
|
87
|
+
|
|
88
|
+
# Or receive webhooks: subscribe http://<host>:8088/ to retrain.requested
|
|
89
|
+
export ANNOTIDE_WEBHOOK_SECRET=... # the webhook's signing secret
|
|
90
|
+
.venv/bin/annotide-train --model <fallback-model-id> serve --host 0.0.0.0 --port 8088
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
`serve` verifies `X-Annotation-Signature` (5 min tolerance, constant-time),
|
|
94
|
+
answers `202` at once and trains on a thread (deliveries time out), skips a
|
|
95
|
+
repeated `X-Annotation-Delivery`, and acknowledges signed events it does not
|
|
96
|
+
act on (another event, no snapshot) with `200` so they are not retried.
|
|
97
|
+
The event's `model_id` wins over `--model`.
|
|
98
|
+
|
|
99
|
+
Options: `--trainer` (`baseline`, `fasterrcnn` or `package.module:ClassName`),
|
|
100
|
+
`--param key=value` (JSON values; repeatable, passed to the trainer),
|
|
101
|
+
`--split 0.8,0.1,0.1` (unsplit snapshots only), `--out runs`,
|
|
102
|
+
`--quantize int8` and `--teacher VERSION_ID` (below).
|
|
103
|
+
|
|
104
|
+
Every registered version carries the metrics the Models page compares
|
|
105
|
+
versions on: `precision` / `recall` / `f1` and `per_class` on the held-out
|
|
106
|
+
split, `size_bytes`, `latency_ms_p50` (median `predict` time per item on this
|
|
107
|
+
machine, `latency_device`) and `dtype`.
|
|
108
|
+
|
|
109
|
+
### Distillation and quantization
|
|
110
|
+
|
|
111
|
+
Both are recorded as the version's `derivation` and `parent_version_id`
|
|
112
|
+
(`docs/CONTRACTS.md`), which the Models page draws as the model's family.
|
|
113
|
+
|
|
114
|
+
**Distillation** here is the annotation loop with a big model as the
|
|
115
|
+
teacher: pre-label with the teacher's version (`POST /projects/{id}/prelabel`),
|
|
116
|
+
let people correct the drafts, take a snapshot, and train a smaller model on
|
|
117
|
+
it with `--teacher <the teacher's version id>`. The new version is registered
|
|
118
|
+
as `distilled` from the teacher, so its F1, size and latency sit next to the
|
|
119
|
+
teacher's. The teacher can be any model of the organisation, including one
|
|
120
|
+
the platform only calls through an endpoint.
|
|
121
|
+
|
|
122
|
+
**Quantization**: `--quantize int8` stores the trained model again at 8 bits,
|
|
123
|
+
scores it on the same held-out split and registers it as a `quantized` child
|
|
124
|
+
of the version this run registered, in `runs/{run_id}/int8/`. The trainer
|
|
125
|
+
implements `quantize(model, calibration, dtype)`; the calibration records are
|
|
126
|
+
a fixed-seed sample of the train split. `fasterrcnn` uses onnxruntime's static
|
|
127
|
+
QDQ quantization (Conv, Gemm, MatMul, per-channel weights); in a smoke test
|
|
128
|
+
the 76 MB model became 20 MB. How much faster it runs depends on the CPU:
|
|
129
|
+
x86 with VNNI / AMX gains most, Apple silicon little. Read the int8
|
|
130
|
+
version's F1 before switching prelabelling to it. `baseline` stores its
|
|
131
|
+
priors as 8-bit fractions, so the path runs without torch.
|
|
132
|
+
|
|
133
|
+
### Recording runs in MLflow (API-6)
|
|
134
|
+
|
|
135
|
+
With the `mlflow` extra (`pip install -e '.[dev,mlflow]'`), `--mlflow-experiment
|
|
136
|
+
NAME` also records each run in MLflow — a local server (`make mlflow` at the
|
|
137
|
+
repo root, `--mlflow-uri http://localhost:5001`), Databricks or Azure ML,
|
|
138
|
+
wherever `MLFLOW_TRACKING_URI` points and however that environment signs in.
|
|
139
|
+
The run gets the params, the numeric metrics, the artifacts under `model/`,
|
|
140
|
+
the tags `annotation.snapshot_id` / `annotation.snapshot_digest` and the
|
|
141
|
+
snapshot as its dataset input (`source_type` `annotation-snapshot`).
|
|
142
|
+
`--mlflow-register NAME` also registers it as a version of that registered
|
|
143
|
+
model. The platform's *Import from MLflow* (`POST /models/{id}/versions/import`)
|
|
144
|
+
reads the lineage back from the run or the registered version, so a version
|
|
145
|
+
imported that way is linked to its snapshot as one registered by this
|
|
146
|
+
pipeline is. `training_run.mlflow` names the run either way.
|
|
147
|
+
|
|
148
|
+
## The Faster R-CNN trainer
|
|
149
|
+
|
|
150
|
+
`--trainer fasterrcnn` fine-tunes torchvision's Faster R-CNN
|
|
151
|
+
(MobileNetV3-Large FPN) on the snapshot's boxes and exports it to ONNX in the
|
|
152
|
+
format `model-service`'s `onnx` backend loads:
|
|
153
|
+
|
|
154
|
+
```sh
|
|
155
|
+
pip install -e '.[torch]' # torch, torchvision, onnx, onnxruntime (CPU wheels:
|
|
156
|
+
# --index-url https://download.pytorch.org/whl/cpu)
|
|
157
|
+
annotide-train run … --trainer fasterrcnn \
|
|
158
|
+
--param image_root=/mnt/images --param epochs=20
|
|
159
|
+
|
|
160
|
+
# runs/{run_id}/fasterrcnn.onnx + fasterrcnn.names → the model service:
|
|
161
|
+
MODEL_BACKEND=onnx MODEL_PATH=runs/{run_id}/fasterrcnn.onnx uvicorn app.main:app
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
- **Images** come from `image_root`, a directory where the customer's own
|
|
165
|
+
storage is mounted or synced (blobfuse2, gcsfuse, `aws s3 sync`): export
|
|
166
|
+
records carry only each item's `path`, and the platform never serves media
|
|
167
|
+
to the trainer. Missing files fail the run before training starts; a path
|
|
168
|
+
that resolves outside `image_root` is refused.
|
|
169
|
+
- **Params**: `epochs` (10), `batch_size` (4), `lr` (0.01, SGD + cosine),
|
|
170
|
+
`weights` (`coco` | `imagenet` | `none`), `hflip` (true), `seed` (0),
|
|
171
|
+
`device` (`cuda` if available, else `cpu`), `score_threshold` (0.5).
|
|
172
|
+
- **Input** is letterboxed to 640 × 640 with grey padding, the same transform
|
|
173
|
+
the model service applies, for training, export and evaluation alike.
|
|
174
|
+
- **Evaluation runs the exported ONNX file** through onnxruntime, not the
|
|
175
|
+
torch model, so the registered metrics are for the artifact that ships.
|
|
176
|
+
The export is also checked against torch on a few training images and a
|
|
177
|
+
blank frame before it is accepted.
|
|
178
|
+
- **Licences**: torchvision, onnx and onnxruntime are BSD / Apache / MIT.
|
|
179
|
+
`weights=coco` starts from torchvision's COCO checkpoint; check its terms
|
|
180
|
+
for your use, or start from `imagenet` / `none`.
|
|
181
|
+
- **Cost**: the artifact is ~76 MB. On an Apple M-series CPU, 48 images ×
|
|
182
|
+
6 epochs take about 90 s; use `device=cuda` for real datasets.
|
|
183
|
+
|
|
184
|
+
## Bring your own trainer
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
from annotide_training.trainers import Prediction, TrainedModel
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
class MyDetector:
|
|
191
|
+
name = "my-detector"
|
|
192
|
+
|
|
193
|
+
def train(self, train, val, params) -> TrainedModel:
|
|
194
|
+
# train: native export records — item_id, path, width, height, shapes[…]
|
|
195
|
+
# Media is the customer's own storage: read it from there.
|
|
196
|
+
...
|
|
197
|
+
return TrainedModel(artifact=onnx_bytes, filename="model.onnx", classes=[...])
|
|
198
|
+
|
|
199
|
+
def predict(self, model, record) -> list[Prediction]: ...
|
|
200
|
+
|
|
201
|
+
# Optional, for --quantize: the same model at lower precision, loadable
|
|
202
|
+
# by predict above.
|
|
203
|
+
def quantize(self, model, calibration, dtype) -> TrainedModel: ...
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
`annotide-train --trainer my_package.detector:MyDetector …`. The framework
|
|
207
|
+
it needs is its own dependency, never the platform's.
|
|
208
|
+
|
|
209
|
+
`BaselineTrainer` needs nothing: it learns how often each class appears and
|
|
210
|
+
where it usually sits, and predicts that. It exists to exercise every stage
|
|
211
|
+
end to end and to give a floor a real model has to beat.
|
|
212
|
+
|
|
213
|
+
## Develop
|
|
214
|
+
|
|
215
|
+
```sh
|
|
216
|
+
.venv/bin/ruff check . && .venv/bin/ruff format --check . && .venv/bin/mypy annotide_training tests && .venv/bin/pytest
|
|
217
|
+
```
|
|
@@ -1,13 +1,30 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "annotide-training"
|
|
3
|
-
version = "0.1.
|
|
4
|
-
description = "
|
|
3
|
+
version = "0.1.2"
|
|
4
|
+
description = "Reference training pipeline for the Annotide annotation platform, run on your own compute"
|
|
5
|
+
readme = "README.md"
|
|
5
6
|
requires-python = ">=3.12"
|
|
7
|
+
# ELv2 like the rest of the repo; only sdk/ is Apache-2.0 (LICENSE.md at the root).
|
|
8
|
+
license = "Elastic-2.0"
|
|
9
|
+
license-files = ["LICENSE.md"]
|
|
10
|
+
authors = [{ name = "Annotide" }]
|
|
11
|
+
keywords = ["annotation", "training", "computer-vision", "mlops", "onnx", "mlflow"]
|
|
12
|
+
# No `License ::` classifiers: PEP 639 `license` expressions replace them.
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 3 - Alpha",
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
20
|
+
"Programming Language :: Python :: 3.12",
|
|
21
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
23
|
+
]
|
|
6
24
|
# The platform never trains (ML-9); this runs on the customer's side. The
|
|
7
25
|
# pipeline talks to the platform through the SDK (../sdk, API-3) — a real
|
|
8
26
|
# trainer brings its own framework as an extra, never the platform image.
|
|
9
|
-
#
|
|
10
|
-
# will look the name up on PyPI.
|
|
27
|
+
# In a checkout, install ../sdk first so pip uses it instead of PyPI's.
|
|
11
28
|
dependencies = [
|
|
12
29
|
"annotide>=0.1",
|
|
13
30
|
]
|
|
@@ -31,11 +48,18 @@ dev = [
|
|
|
31
48
|
"mypy>=1.13",
|
|
32
49
|
]
|
|
33
50
|
|
|
51
|
+
[project.urls]
|
|
52
|
+
Homepage = "https://annotide.com"
|
|
53
|
+
Documentation = "https://github.com/annotide/annotide/tree/main/training#readme"
|
|
54
|
+
Source = "https://github.com/annotide/annotide/tree/main/training"
|
|
55
|
+
Issues = "https://github.com/annotide/annotide/issues"
|
|
56
|
+
Changelog = "https://github.com/annotide/annotide/releases"
|
|
57
|
+
|
|
34
58
|
[project.scripts]
|
|
35
59
|
annotide-train = "annotide_training.cli:main"
|
|
36
60
|
|
|
37
61
|
[build-system]
|
|
38
|
-
requires = ["setuptools>=
|
|
62
|
+
requires = ["setuptools>=77"] # PEP 639 `license` expressions
|
|
39
63
|
build-backend = "setuptools.build_meta"
|
|
40
64
|
|
|
41
65
|
[tool.setuptools.packages.find]
|
annotide_training-0.1.0/PKG-INFO
DELETED
|
@@ -1,19 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: annotide-training
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: Cloud-agnostic annotation platform — reference customer-side training pipeline (ML-9, EXP-8)
|
|
5
|
-
Requires-Python: >=3.12
|
|
6
|
-
Requires-Dist: annotide>=0.1
|
|
7
|
-
Provides-Extra: torch
|
|
8
|
-
Requires-Dist: torch>=2.5; extra == "torch"
|
|
9
|
-
Requires-Dist: torchvision>=0.20; extra == "torch"
|
|
10
|
-
Requires-Dist: onnx>=1.16; extra == "torch"
|
|
11
|
-
Requires-Dist: onnxruntime>=1.20; extra == "torch"
|
|
12
|
-
Requires-Dist: numpy>=2.1; extra == "torch"
|
|
13
|
-
Requires-Dist: pillow>=11.0; extra == "torch"
|
|
14
|
-
Provides-Extra: mlflow
|
|
15
|
-
Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
|
|
16
|
-
Provides-Extra: dev
|
|
17
|
-
Requires-Dist: pytest>=8.3; extra == "dev"
|
|
18
|
-
Requires-Dist: ruff>=0.8; extra == "dev"
|
|
19
|
-
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
@@ -1,19 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: annotide-training
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: Cloud-agnostic annotation platform — reference customer-side training pipeline (ML-9, EXP-8)
|
|
5
|
-
Requires-Python: >=3.12
|
|
6
|
-
Requires-Dist: annotide>=0.1
|
|
7
|
-
Provides-Extra: torch
|
|
8
|
-
Requires-Dist: torch>=2.5; extra == "torch"
|
|
9
|
-
Requires-Dist: torchvision>=0.20; extra == "torch"
|
|
10
|
-
Requires-Dist: onnx>=1.16; extra == "torch"
|
|
11
|
-
Requires-Dist: onnxruntime>=1.20; extra == "torch"
|
|
12
|
-
Requires-Dist: numpy>=2.1; extra == "torch"
|
|
13
|
-
Requires-Dist: pillow>=11.0; extra == "torch"
|
|
14
|
-
Provides-Extra: mlflow
|
|
15
|
-
Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
|
|
16
|
-
Provides-Extra: dev
|
|
17
|
-
Requires-Dist: pytest>=8.3; extra == "dev"
|
|
18
|
-
Requires-Dist: ruff>=0.8; extra == "dev"
|
|
19
|
-
Requires-Dist: mypy>=1.13; extra == "dev"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
{annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/entry_points.txt
RENAMED
|
File without changes
|
|
File without changes
|
{annotide_training-0.1.0 → annotide_training-0.1.2}/annotide_training.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|