annotide-training 0.1.0__tar.gz → 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. annotide_training-0.1.1/LICENSE.md +110 -0
  2. annotide_training-0.1.1/PKG-INFO +217 -0
  3. {annotide_training-0.1.0 → annotide_training-0.1.1}/README.md +1 -1
  4. annotide_training-0.1.1/annotide_training.egg-info/PKG-INFO +217 -0
  5. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training.egg-info/SOURCES.txt +1 -0
  6. {annotide_training-0.1.0 → annotide_training-0.1.1}/pyproject.toml +29 -5
  7. annotide_training-0.1.0/PKG-INFO +0 -19
  8. annotide_training-0.1.0/annotide_training.egg-info/PKG-INFO +0 -19
  9. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/__init__.py +0 -0
  10. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/cli.py +0 -0
  11. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/dataset.py +0 -0
  12. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/detector.py +0 -0
  13. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/evaluate.py +0 -0
  14. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/pipeline.py +0 -0
  15. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/tracking.py +0 -0
  16. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/trainers.py +0 -0
  17. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training/webhook.py +0 -0
  18. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training.egg-info/dependency_links.txt +0 -0
  19. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training.egg-info/entry_points.txt +0 -0
  20. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training.egg-info/requires.txt +0 -0
  21. {annotide_training-0.1.0 → annotide_training-0.1.1}/annotide_training.egg-info/top_level.txt +0 -0
  22. {annotide_training-0.1.0 → annotide_training-0.1.1}/setup.cfg +0 -0
  23. {annotide_training-0.1.0 → annotide_training-0.1.1}/tests/test_dataset.py +0 -0
  24. {annotide_training-0.1.0 → annotide_training-0.1.1}/tests/test_detector.py +0 -0
  25. {annotide_training-0.1.0 → annotide_training-0.1.1}/tests/test_pipeline.py +0 -0
  26. {annotide_training-0.1.0 → annotide_training-0.1.1}/tests/test_trainers.py +0 -0
  27. {annotide_training-0.1.0 → annotide_training-0.1.1}/tests/test_webhook.py +0 -0
@@ -0,0 +1,110 @@
1
+ Copyright (c) 2026 R Squared Data Solutions (https://annotide.com)
2
+
3
+ Annotide is licensed under the Elastic License 2.0, reproduced below.
4
+
5
+ In plain words (the terms below govern, not this summary): you may use,
6
+ copy, modify and run Annotide, including for commercial work. Without a
7
+ licence key it runs as the Community edition, for up to three active users.
8
+ A licence key (https://annotide.com/pricing) adds users and the Business
9
+ features. You may not get around the licence key or remove what it
10
+ protects, and you may not offer Annotide to others as a hosted or managed
11
+ service.
12
+
13
+ The Python SDK, CLI and MCP server in `sdk/` (`pip install annotide`) are
14
+ licensed separately under the Apache License 2.0; see `sdk/LICENSE`.
15
+
16
+ ---
17
+
18
+ Elastic License 2.0
19
+
20
+ URL: https://www.elastic.co/licensing/elastic-license
21
+
22
+ Acceptance
23
+
24
+ By using the software, you agree to all of the terms and conditions below.
25
+
26
+ Copyright License
27
+
28
+ The licensor grants you a non-exclusive, royalty-free, worldwide,
29
+ non-sublicensable, non-transferable license to use, copy, distribute, make
30
+ available, and prepare derivative works of the software, in each case subject to
31
+ the limitations and conditions below.
32
+
33
+ Limitations
34
+
35
+ You may not provide the software to third parties as a hosted or managed
36
+ service, where the service provides users with access to any substantial set of
37
+ the features or functionality of the software.
38
+
39
+ You may not move, change, disable, or circumvent the license key functionality
40
+ in the software, and you may not remove or obscure any functionality in the
41
+ software that is protected by the license key.
42
+
43
+ You may not alter, remove, or obscure any licensing, copyright, or other notices
44
+ of the licensor in the software. Any use of the licensor’s trademarks is subject
45
+ to applicable law.
46
+
47
+ Patents
48
+
49
+ The licensor grants you a license, under any patent claims the licensor can
50
+ license, or becomes able to license, to make, have made, use, sell, offer for
51
+ sale, import and have imported the software, in each case subject to the
52
+ limitations and conditions in this license. This license does not cover any
53
+ patent claims that you cause to be infringed by modifications or additions to
54
+ the software. If you or your company make any written claim that the software
55
+ infringes or contributes to infringement of any patent, your patent license for
56
+ the software granted under these terms ends immediately. If your company makes
57
+ such a claim, your patent license ends immediately for work on behalf of your
58
+ company.
59
+
60
+ Notices
61
+
62
+ You must ensure that anyone who gets a copy of any part of the software from you
63
+ also gets a copy of these terms.
64
+
65
+ If you modify the software, you must include in any modified copies of the
66
+ software prominent notices stating that you have modified the software.
67
+
68
+ ## No Other Rights
69
+
70
+ These terms do not imply any licenses other than those expressly granted in
71
+ these terms.
72
+
73
+ Termination
74
+
75
+ If you use the software in violation of these terms, such use is not licensed,
76
+ and your licenses will automatically terminate. If the licensor provides you
77
+ with a notice of your violation, and you cease all violation of this license no
78
+ later than 30 days after you receive that notice, your licenses will be
79
+ reinstated retroactively. However, if you violate these terms after such
80
+ reinstatement, any additional violation of these terms will cause your licenses
81
+ to terminate automatically and permanently.
82
+
83
+ No Liability
84
+
85
+ As far as the law allows, the software comes as is, without any warranty or
86
+ condition, and the licensor will not be liable to you for any damages arising
87
+ out of these terms or the use or nature of the software, under any kind of
88
+ legal claim.
89
+
90
+ Definitions
91
+
92
+ The licensor is the entity offering these terms, and the software is the
93
+ software the licensor makes available under these terms, including any portion
94
+ of it.
95
+
96
+ you refers to the individual or entity agreeing to these terms.
97
+
98
+ your company is any legal entity, sole proprietorship, or other kind of
99
+ organization that you work for, plus all organizations that have control over,
100
+ are under the control of, or are under common control with that
101
+ organization. control means ownership of substantially all the assets of an
102
+ entity, or the power to direct its management and policies by vote, contract, or
103
+ otherwise. Control can be direct or indirect.
104
+
105
+ your licenses are all the licenses granted to you for the software under
106
+ these terms.
107
+
108
+ use means anything you do with the software requiring one of your licenses.
109
+
110
+ trademark means trademarks, service marks, and similar rights.
@@ -0,0 +1,217 @@
1
+ Metadata-Version: 2.4
2
+ Name: annotide-training
3
+ Version: 0.1.1
4
+ Summary: Reference training pipeline for the Annotide annotation platform, run on your own compute
5
+ Author: Annotide
6
+ License-Expression: Elastic-2.0
7
+ Project-URL: Homepage, https://annotide.com
8
+ Project-URL: Documentation, https://github.com/annotide/annotide/tree/main/training#readme
9
+ Project-URL: Source, https://github.com/annotide/annotide/tree/main/training
10
+ Project-URL: Issues, https://github.com/annotide/annotide/issues
11
+ Project-URL: Changelog, https://github.com/annotide/annotide/releases
12
+ Keywords: annotation,training,computer-vision,mlops,onnx,mlflow
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
22
+ Requires-Python: >=3.12
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE.md
25
+ Requires-Dist: annotide>=0.1
26
+ Provides-Extra: torch
27
+ Requires-Dist: torch>=2.5; extra == "torch"
28
+ Requires-Dist: torchvision>=0.20; extra == "torch"
29
+ Requires-Dist: onnx>=1.16; extra == "torch"
30
+ Requires-Dist: onnxruntime>=1.20; extra == "torch"
31
+ Requires-Dist: numpy>=2.1; extra == "torch"
32
+ Requires-Dist: pillow>=11.0; extra == "torch"
33
+ Provides-Extra: mlflow
34
+ Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
35
+ Provides-Extra: dev
36
+ Requires-Dist: pytest>=8.3; extra == "dev"
37
+ Requires-Dist: ruff>=0.8; extra == "dev"
38
+ Requires-Dist: mypy>=1.13; extra == "dev"
39
+ Dynamic: license-file
40
+
41
+ # Reference training pipeline
42
+
43
+ The platform never trains a model itself (ML-9). It freezes a snapshot,
44
+ emits `retrain.requested` to the project's webhooks, and records whatever
45
+ version comes back with its lineage (EXP-8). This package is the other end:
46
+ a pipeline a customer runs next to their own compute.
47
+
48
+ ```
49
+ retrain.requested ─▶ prepare ─▶ split ─▶ train ─▶ evaluate ─▶ register
50
+ (webhook) export EXP-3 Trainer box P/R/F1 POST /models/{id}/versions
51
+ + digest snapshot_id + digest + training_run
52
+ ```
53
+
54
+ 1. **prepare** — reads the snapshot row, refuses a digest that is not the
55
+ snapshot's, queues a `native` export of it, downloads the archive through
56
+ its signed URL (the API key never goes to the storage host) and checks
57
+ the archive's manifest names the same snapshot and digest.
58
+ 2. **split** — uses the snapshot's own train / val / test partition when it
59
+ has one; otherwise splits with the platform's rule
60
+ (`sha256("{seed}:item:{id}")`), so the same seed gives the same sets.
61
+ 3. **train** — a pluggable `Trainer` (`annotide_training/trainers.py`).
62
+ 4. **evaluate** — box precision / recall / F1 at IoU 0.5, per class and
63
+ overall, on `test` (or `val` when `test` is empty). The same scorer for
64
+ every trainer.
65
+ 5. **register** — adds a model version with `snapshot_id`,
66
+ `snapshot_digest`, the metrics and a `training_run` record (run id,
67
+ trainer, params, timings, export job, split counts, artifact location,
68
+ webhook delivery id). The platform re-checks the digest (409 otherwise).
69
+
70
+ Artifacts and a `run.json` land in `runs/{run_id}/`. Where the artifact goes
71
+ next — an ONNX file for `model-service`'s `onnx` backend, a model registry —
72
+ is the trainer's business; the platform only stores the lineage.
73
+
74
+ ## Run it
75
+
76
+ ```sh
77
+ cd training
78
+ python3.12 -m venv .venv
79
+ .venv/bin/pip install -e ../sdk # first: the local SDK, not the PyPI release
80
+ .venv/bin/pip install -e '.[dev]'
81
+
82
+ export ANNOTIDE_API_URL=http://localhost:8000 # site root, not /api/v1
83
+ export ANNOTIDE_API_KEY=... # an API key with the `write` scope (AUTH-4)
84
+
85
+ # One snapshot, now. --no-register trains and evaluates without adding a version.
86
+ .venv/bin/annotide-train --model <model-id> run --project <project-id> --snapshot <snapshot-id>
87
+
88
+ # Or receive webhooks: subscribe http://<host>:8088/ to retrain.requested
89
+ export ANNOTIDE_WEBHOOK_SECRET=... # the webhook's signing secret
90
+ .venv/bin/annotide-train --model <fallback-model-id> serve --host 0.0.0.0 --port 8088
91
+ ```
92
+
93
+ `serve` verifies `X-Annotation-Signature` (5 min tolerance, constant-time),
94
+ answers `202` at once and trains on a thread (deliveries time out), skips a
95
+ repeated `X-Annotation-Delivery`, and acknowledges signed events it does not
96
+ act on (another event, no snapshot) with `200` so they are not retried.
97
+ The event's `model_id` wins over `--model`.
98
+
99
+ Options: `--trainer` (`baseline`, `fasterrcnn` or `package.module:ClassName`),
100
+ `--param key=value` (JSON values; repeatable, passed to the trainer),
101
+ `--split 0.8,0.1,0.1` (unsplit snapshots only), `--out runs`,
102
+ `--quantize int8` and `--teacher VERSION_ID` (below).
103
+
104
+ Every registered version carries the metrics the Models page compares
105
+ versions on: `precision` / `recall` / `f1` and `per_class` on the held-out
106
+ split, `size_bytes`, `latency_ms_p50` (median `predict` time per item on this
107
+ machine, `latency_device`) and `dtype`.
108
+
109
+ ### Distillation and quantization
110
+
111
+ Both are recorded as the version's `derivation` and `parent_version_id`
112
+ (`docs/CONTRACTS.md`), which the Models page draws as the model's family.
113
+
114
+ **Distillation** here is the annotation loop with a big model as the
115
+ teacher: pre-label with the teacher's version (`POST /projects/{id}/prelabel`),
116
+ let people correct the drafts, take a snapshot, and train a smaller model on
117
+ it with `--teacher <the teacher's version id>`. The new version is registered
118
+ as `distilled` from the teacher, so its F1, size and latency sit next to the
119
+ teacher's. The teacher can be any model of the organisation, including one
120
+ the platform only calls through an endpoint.
121
+
122
+ **Quantization**: `--quantize int8` stores the trained model again at 8 bits,
123
+ scores it on the same held-out split and registers it as a `quantized` child
124
+ of the version this run registered, in `runs/{run_id}/int8/`. The trainer
125
+ implements `quantize(model, calibration, dtype)`; the calibration records are
126
+ a fixed-seed sample of the train split. `fasterrcnn` uses onnxruntime's static
127
+ QDQ quantization (Conv, Gemm, MatMul, per-channel weights); in a smoke test
128
+ the 76 MB model became 20 MB. How much faster it runs depends on the CPU:
129
+ x86 with VNNI / AMX gains most, Apple silicon little. Read the int8
130
+ version's F1 before switching prelabelling to it. `baseline` stores its
131
+ priors as 8-bit fractions, so the path runs without torch.
132
+
133
+ ### Recording runs in MLflow (API-6)
134
+
135
+ With the `mlflow` extra (`pip install -e '.[dev,mlflow]'`), `--mlflow-experiment
136
+ NAME` also records each run in MLflow — a local server (`make mlflow` at the
137
+ repo root, `--mlflow-uri http://localhost:5001`), Databricks or Azure ML,
138
+ wherever `MLFLOW_TRACKING_URI` points and however that environment signs in.
139
+ The run gets the params, the numeric metrics, the artifacts under `model/`,
140
+ the tags `annotation.snapshot_id` / `annotation.snapshot_digest` and the
141
+ snapshot as its dataset input (`source_type` `annotation-snapshot`).
142
+ `--mlflow-register NAME` also registers it as a version of that registered
143
+ model. The platform's *Import from MLflow* (`POST /models/{id}/versions/import`)
144
+ reads the lineage back from the run or the registered version, so a version
145
+ imported that way is linked to its snapshot as one registered by this
146
+ pipeline is. `training_run.mlflow` names the run either way.
147
+
148
+ ## The Faster R-CNN trainer
149
+
150
+ `--trainer fasterrcnn` fine-tunes torchvision's Faster R-CNN
151
+ (MobileNetV3-Large FPN) on the snapshot's boxes and exports it to ONNX in the
152
+ format `model-service`'s `onnx` backend loads:
153
+
154
+ ```sh
155
+ pip install -e '.[torch]' # torch, torchvision, onnx, onnxruntime (CPU wheels:
156
+ # --index-url https://download.pytorch.org/whl/cpu)
157
+ annotide-train run … --trainer fasterrcnn \
158
+ --param image_root=/mnt/images --param epochs=20
159
+
160
+ # runs/{run_id}/fasterrcnn.onnx + fasterrcnn.names → the model service:
161
+ MODEL_BACKEND=onnx MODEL_PATH=runs/{run_id}/fasterrcnn.onnx uvicorn app.main:app
162
+ ```
163
+
164
+ - **Images** come from `image_root`, a directory where the customer's own
165
+ storage is mounted or synced (blobfuse2, gcsfuse, `aws s3 sync`): export
166
+ records carry only each item's `path`, and the platform never serves media
167
+ to the trainer. Missing files fail the run before training starts; a path
168
+ that resolves outside `image_root` is refused.
169
+ - **Params**: `epochs` (10), `batch_size` (4), `lr` (0.01, SGD + cosine),
170
+ `weights` (`coco` | `imagenet` | `none`), `hflip` (true), `seed` (0),
171
+ `device` (`cuda` if available, else `cpu`), `score_threshold` (0.5).
172
+ - **Input** is letterboxed to 640 × 640 with grey padding, the same transform
173
+ the model service applies, for training, export and evaluation alike.
174
+ - **Evaluation runs the exported ONNX file** through onnxruntime, not the
175
+ torch model, so the registered metrics are for the artifact that ships.
176
+ The export is also checked against torch on a few training images and a
177
+ blank frame before it is accepted.
178
+ - **Licences**: torchvision, onnx and onnxruntime are BSD / Apache / MIT.
179
+ `weights=coco` starts from torchvision's COCO checkpoint; check its terms
180
+ for your use, or start from `imagenet` / `none`.
181
+ - **Cost**: the artifact is ~76 MB. On an Apple M-series CPU, 48 images ×
182
+ 6 epochs take about 90 s; use `device=cuda` for real datasets.
183
+
184
+ ## Bring your own trainer
185
+
186
+ ```python
187
+ from annotide_training.trainers import Prediction, TrainedModel
188
+
189
+
190
+ class MyDetector:
191
+ name = "my-detector"
192
+
193
+ def train(self, train, val, params) -> TrainedModel:
194
+ # train: native export records — item_id, path, width, height, shapes[…]
195
+ # Media is the customer's own storage: read it from there.
196
+ ...
197
+ return TrainedModel(artifact=onnx_bytes, filename="model.onnx", classes=[...])
198
+
199
+ def predict(self, model, record) -> list[Prediction]: ...
200
+
201
+ # Optional, for --quantize: the same model at lower precision, loadable
202
+ # by predict above.
203
+ def quantize(self, model, calibration, dtype) -> TrainedModel: ...
204
+ ```
205
+
206
+ `annotide-train --trainer my_package.detector:MyDetector …`. The framework
207
+ it needs is its own dependency, never the platform's.
208
+
209
+ `BaselineTrainer` needs nothing: it learns how often each class appears and
210
+ where it usually sits, and predicts that. It exists to exercise every stage
211
+ end to end and to give a floor a real model has to beat.
212
+
213
+ ## Develop
214
+
215
+ ```sh
216
+ .venv/bin/ruff check . && .venv/bin/ruff format --check . && .venv/bin/mypy annotide_training tests && .venv/bin/pytest
217
+ ```
@@ -36,7 +36,7 @@ is the trainer's business; the platform only stores the lineage.
36
36
  ```sh
37
37
  cd training
38
38
  python3.12 -m venv .venv
39
- .venv/bin/pip install -e ../sdk # first: the SDK is not on a package index
39
+ .venv/bin/pip install -e ../sdk # first: the local SDK, not the PyPI release
40
40
  .venv/bin/pip install -e '.[dev]'
41
41
 
42
42
  export ANNOTIDE_API_URL=http://localhost:8000 # site root, not /api/v1
@@ -0,0 +1,217 @@
1
+ Metadata-Version: 2.4
2
+ Name: annotide-training
3
+ Version: 0.1.1
4
+ Summary: Reference training pipeline for the Annotide annotation platform, run on your own compute
5
+ Author: Annotide
6
+ License-Expression: Elastic-2.0
7
+ Project-URL: Homepage, https://annotide.com
8
+ Project-URL: Documentation, https://github.com/annotide/annotide/tree/main/training#readme
9
+ Project-URL: Source, https://github.com/annotide/annotide/tree/main/training
10
+ Project-URL: Issues, https://github.com/annotide/annotide/issues
11
+ Project-URL: Changelog, https://github.com/annotide/annotide/releases
12
+ Keywords: annotation,training,computer-vision,mlops,onnx,mlflow
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Classifier: Topic :: Scientific/Engineering :: Image Recognition
22
+ Requires-Python: >=3.12
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE.md
25
+ Requires-Dist: annotide>=0.1
26
+ Provides-Extra: torch
27
+ Requires-Dist: torch>=2.5; extra == "torch"
28
+ Requires-Dist: torchvision>=0.20; extra == "torch"
29
+ Requires-Dist: onnx>=1.16; extra == "torch"
30
+ Requires-Dist: onnxruntime>=1.20; extra == "torch"
31
+ Requires-Dist: numpy>=2.1; extra == "torch"
32
+ Requires-Dist: pillow>=11.0; extra == "torch"
33
+ Provides-Extra: mlflow
34
+ Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
35
+ Provides-Extra: dev
36
+ Requires-Dist: pytest>=8.3; extra == "dev"
37
+ Requires-Dist: ruff>=0.8; extra == "dev"
38
+ Requires-Dist: mypy>=1.13; extra == "dev"
39
+ Dynamic: license-file
40
+
41
+ # Reference training pipeline
42
+
43
+ The platform never trains a model itself (ML-9). It freezes a snapshot,
44
+ emits `retrain.requested` to the project's webhooks, and records whatever
45
+ version comes back with its lineage (EXP-8). This package is the other end:
46
+ a pipeline a customer runs next to their own compute.
47
+
48
+ ```
49
+ retrain.requested ─▶ prepare ─▶ split ─▶ train ─▶ evaluate ─▶ register
50
+ (webhook) export EXP-3 Trainer box P/R/F1 POST /models/{id}/versions
51
+ + digest snapshot_id + digest + training_run
52
+ ```
53
+
54
+ 1. **prepare** — reads the snapshot row, refuses a digest that is not the
55
+ snapshot's, queues a `native` export of it, downloads the archive through
56
+ its signed URL (the API key never goes to the storage host) and checks
57
+ the archive's manifest names the same snapshot and digest.
58
+ 2. **split** — uses the snapshot's own train / val / test partition when it
59
+ has one; otherwise splits with the platform's rule
60
+ (`sha256("{seed}:item:{id}")`), so the same seed gives the same sets.
61
+ 3. **train** — a pluggable `Trainer` (`annotide_training/trainers.py`).
62
+ 4. **evaluate** — box precision / recall / F1 at IoU 0.5, per class and
63
+ overall, on `test` (or `val` when `test` is empty). The same scorer for
64
+ every trainer.
65
+ 5. **register** — adds a model version with `snapshot_id`,
66
+ `snapshot_digest`, the metrics and a `training_run` record (run id,
67
+ trainer, params, timings, export job, split counts, artifact location,
68
+ webhook delivery id). The platform re-checks the digest (409 otherwise).
69
+
70
+ Artifacts and a `run.json` land in `runs/{run_id}/`. Where the artifact goes
71
+ next — an ONNX file for `model-service`'s `onnx` backend, a model registry —
72
+ is the trainer's business; the platform only stores the lineage.
73
+
74
+ ## Run it
75
+
76
+ ```sh
77
+ cd training
78
+ python3.12 -m venv .venv
79
+ .venv/bin/pip install -e ../sdk # first: the local SDK, not the PyPI release
80
+ .venv/bin/pip install -e '.[dev]'
81
+
82
+ export ANNOTIDE_API_URL=http://localhost:8000 # site root, not /api/v1
83
+ export ANNOTIDE_API_KEY=... # an API key with the `write` scope (AUTH-4)
84
+
85
+ # One snapshot, now. --no-register trains and evaluates without adding a version.
86
+ .venv/bin/annotide-train --model <model-id> run --project <project-id> --snapshot <snapshot-id>
87
+
88
+ # Or receive webhooks: subscribe http://<host>:8088/ to retrain.requested
89
+ export ANNOTIDE_WEBHOOK_SECRET=... # the webhook's signing secret
90
+ .venv/bin/annotide-train --model <fallback-model-id> serve --host 0.0.0.0 --port 8088
91
+ ```
92
+
93
+ `serve` verifies `X-Annotation-Signature` (5 min tolerance, constant-time),
94
+ answers `202` at once and trains on a thread (deliveries time out), skips a
95
+ repeated `X-Annotation-Delivery`, and acknowledges signed events it does not
96
+ act on (another event, no snapshot) with `200` so they are not retried.
97
+ The event's `model_id` wins over `--model`.
98
+
99
+ Options: `--trainer` (`baseline`, `fasterrcnn` or `package.module:ClassName`),
100
+ `--param key=value` (JSON values; repeatable, passed to the trainer),
101
+ `--split 0.8,0.1,0.1` (unsplit snapshots only), `--out runs`,
102
+ `--quantize int8` and `--teacher VERSION_ID` (below).
103
+
104
+ Every registered version carries the metrics the Models page compares
105
+ versions on: `precision` / `recall` / `f1` and `per_class` on the held-out
106
+ split, `size_bytes`, `latency_ms_p50` (median `predict` time per item on this
107
+ machine, `latency_device`) and `dtype`.
108
+
109
+ ### Distillation and quantization
110
+
111
+ Both are recorded as the version's `derivation` and `parent_version_id`
112
+ (`docs/CONTRACTS.md`), which the Models page draws as the model's family.
113
+
114
+ **Distillation** here is the annotation loop with a big model as the
115
+ teacher: pre-label with the teacher's version (`POST /projects/{id}/prelabel`),
116
+ let people correct the drafts, take a snapshot, and train a smaller model on
117
+ it with `--teacher <the teacher's version id>`. The new version is registered
118
+ as `distilled` from the teacher, so its F1, size and latency sit next to the
119
+ teacher's. The teacher can be any model of the organisation, including one
120
+ the platform only calls through an endpoint.
121
+
122
+ **Quantization**: `--quantize int8` stores the trained model again at 8 bits,
123
+ scores it on the same held-out split and registers it as a `quantized` child
124
+ of the version this run registered, in `runs/{run_id}/int8/`. The trainer
125
+ implements `quantize(model, calibration, dtype)`; the calibration records are
126
+ a fixed-seed sample of the train split. `fasterrcnn` uses onnxruntime's static
127
+ QDQ quantization (Conv, Gemm, MatMul, per-channel weights); in a smoke test
128
+ the 76 MB model became 20 MB. How much faster it runs depends on the CPU:
129
+ x86 with VNNI / AMX gains most, Apple silicon little. Read the int8
130
+ version's F1 before switching prelabelling to it. `baseline` stores its
131
+ priors as 8-bit fractions, so the path runs without torch.
132
+
133
+ ### Recording runs in MLflow (API-6)
134
+
135
+ With the `mlflow` extra (`pip install -e '.[dev,mlflow]'`), `--mlflow-experiment
136
+ NAME` also records each run in MLflow — a local server (`make mlflow` at the
137
+ repo root, `--mlflow-uri http://localhost:5001`), Databricks or Azure ML,
138
+ wherever `MLFLOW_TRACKING_URI` points and however that environment signs in.
139
+ The run gets the params, the numeric metrics, the artifacts under `model/`,
140
+ the tags `annotation.snapshot_id` / `annotation.snapshot_digest` and the
141
+ snapshot as its dataset input (`source_type` `annotation-snapshot`).
142
+ `--mlflow-register NAME` also registers it as a version of that registered
143
+ model. The platform's *Import from MLflow* (`POST /models/{id}/versions/import`)
144
+ reads the lineage back from the run or the registered version, so a version
145
+ imported that way is linked to its snapshot as one registered by this
146
+ pipeline is. `training_run.mlflow` names the run either way.
147
+
148
+ ## The Faster R-CNN trainer
149
+
150
+ `--trainer fasterrcnn` fine-tunes torchvision's Faster R-CNN
151
+ (MobileNetV3-Large FPN) on the snapshot's boxes and exports it to ONNX in the
152
+ format `model-service`'s `onnx` backend loads:
153
+
154
+ ```sh
155
+ pip install -e '.[torch]' # torch, torchvision, onnx, onnxruntime (CPU wheels:
156
+ # --index-url https://download.pytorch.org/whl/cpu)
157
+ annotide-train run … --trainer fasterrcnn \
158
+ --param image_root=/mnt/images --param epochs=20
159
+
160
+ # runs/{run_id}/fasterrcnn.onnx + fasterrcnn.names → the model service:
161
+ MODEL_BACKEND=onnx MODEL_PATH=runs/{run_id}/fasterrcnn.onnx uvicorn app.main:app
162
+ ```
163
+
164
+ - **Images** come from `image_root`, a directory where the customer's own
165
+ storage is mounted or synced (blobfuse2, gcsfuse, `aws s3 sync`): export
166
+ records carry only each item's `path`, and the platform never serves media
167
+ to the trainer. Missing files fail the run before training starts; a path
168
+ that resolves outside `image_root` is refused.
169
+ - **Params**: `epochs` (10), `batch_size` (4), `lr` (0.01, SGD + cosine),
170
+ `weights` (`coco` | `imagenet` | `none`), `hflip` (true), `seed` (0),
171
+ `device` (`cuda` if available, else `cpu`), `score_threshold` (0.5).
172
+ - **Input** is letterboxed to 640 × 640 with grey padding, the same transform
173
+ the model service applies, for training, export and evaluation alike.
174
+ - **Evaluation runs the exported ONNX file** through onnxruntime, not the
175
+ torch model, so the registered metrics are for the artifact that ships.
176
+ The export is also checked against torch on a few training images and a
177
+ blank frame before it is accepted.
178
+ - **Licences**: torchvision, onnx and onnxruntime are BSD / Apache / MIT.
179
+ `weights=coco` starts from torchvision's COCO checkpoint; check its terms
180
+ for your use, or start from `imagenet` / `none`.
181
+ - **Cost**: the artifact is ~76 MB. On an Apple M-series CPU, 48 images ×
182
+ 6 epochs take about 90 s; use `device=cuda` for real datasets.
183
+
184
+ ## Bring your own trainer
185
+
186
+ ```python
187
+ from annotide_training.trainers import Prediction, TrainedModel
188
+
189
+
190
+ class MyDetector:
191
+ name = "my-detector"
192
+
193
+ def train(self, train, val, params) -> TrainedModel:
194
+ # train: native export records — item_id, path, width, height, shapes[…]
195
+ # Media is the customer's own storage: read it from there.
196
+ ...
197
+ return TrainedModel(artifact=onnx_bytes, filename="model.onnx", classes=[...])
198
+
199
+ def predict(self, model, record) -> list[Prediction]: ...
200
+
201
+ # Optional, for --quantize: the same model at lower precision, loadable
202
+ # by predict above.
203
+ def quantize(self, model, calibration, dtype) -> TrainedModel: ...
204
+ ```
205
+
206
+ `annotide-train --trainer my_package.detector:MyDetector …`. The framework
207
+ it needs is its own dependency, never the platform's.
208
+
209
+ `BaselineTrainer` needs nothing: it learns how often each class appears and
210
+ where it usually sits, and predicts that. It exists to exercise every stage
211
+ end to end and to give a floor a real model has to beat.
212
+
213
+ ## Develop
214
+
215
+ ```sh
216
+ .venv/bin/ruff check . && .venv/bin/ruff format --check . && .venv/bin/mypy annotide_training tests && .venv/bin/pytest
217
+ ```
@@ -1,3 +1,4 @@
1
+ LICENSE.md
1
2
  README.md
2
3
  pyproject.toml
3
4
  annotide_training/__init__.py
@@ -1,13 +1,30 @@
1
1
  [project]
2
2
  name = "annotide-training"
3
- version = "0.1.0"
4
- description = "Cloud-agnostic annotation platform — reference customer-side training pipeline (ML-9, EXP-8)"
3
+ version = "0.1.1"
4
+ description = "Reference training pipeline for the Annotide annotation platform, run on your own compute"
5
+ readme = "README.md"
5
6
  requires-python = ">=3.12"
7
+ # ELv2 like the rest of the repo; only sdk/ is Apache-2.0 (LICENSE.md at the root).
8
+ license = "Elastic-2.0"
9
+ license-files = ["LICENSE.md"]
10
+ authors = [{ name = "Annotide" }]
11
+ keywords = ["annotation", "training", "computer-vision", "mlops", "onnx", "mlflow"]
12
+ # No `License ::` classifiers: PEP 639 `license` expressions replace them.
13
+ classifiers = [
14
+ "Development Status :: 3 - Alpha",
15
+ "Intended Audience :: Developers",
16
+ "Intended Audience :: Science/Research",
17
+ "Operating System :: OS Independent",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3 :: Only",
20
+ "Programming Language :: Python :: 3.12",
21
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
22
+ "Topic :: Scientific/Engineering :: Image Recognition",
23
+ ]
6
24
  # The platform never trains (ML-9); this runs on the customer's side. The
7
25
  # pipeline talks to the platform through the SDK (../sdk, API-3) — a real
8
26
  # trainer brings its own framework as an extra, never the platform image.
9
- # annotide is not on a package index yet: install ../sdk first, or pip
10
- # will look the name up on PyPI.
27
+ # In a checkout, install ../sdk first so pip uses it instead of PyPI's.
11
28
  dependencies = [
12
29
  "annotide>=0.1",
13
30
  ]
@@ -31,11 +48,18 @@ dev = [
31
48
  "mypy>=1.13",
32
49
  ]
33
50
 
51
+ [project.urls]
52
+ Homepage = "https://annotide.com"
53
+ Documentation = "https://github.com/annotide/annotide/tree/main/training#readme"
54
+ Source = "https://github.com/annotide/annotide/tree/main/training"
55
+ Issues = "https://github.com/annotide/annotide/issues"
56
+ Changelog = "https://github.com/annotide/annotide/releases"
57
+
34
58
  [project.scripts]
35
59
  annotide-train = "annotide_training.cli:main"
36
60
 
37
61
  [build-system]
38
- requires = ["setuptools>=75"]
62
+ requires = ["setuptools>=77"] # PEP 639 `license` expressions
39
63
  build-backend = "setuptools.build_meta"
40
64
 
41
65
  [tool.setuptools.packages.find]
@@ -1,19 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: annotide-training
3
- Version: 0.1.0
4
- Summary: Cloud-agnostic annotation platform — reference customer-side training pipeline (ML-9, EXP-8)
5
- Requires-Python: >=3.12
6
- Requires-Dist: annotide>=0.1
7
- Provides-Extra: torch
8
- Requires-Dist: torch>=2.5; extra == "torch"
9
- Requires-Dist: torchvision>=0.20; extra == "torch"
10
- Requires-Dist: onnx>=1.16; extra == "torch"
11
- Requires-Dist: onnxruntime>=1.20; extra == "torch"
12
- Requires-Dist: numpy>=2.1; extra == "torch"
13
- Requires-Dist: pillow>=11.0; extra == "torch"
14
- Provides-Extra: mlflow
15
- Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
16
- Provides-Extra: dev
17
- Requires-Dist: pytest>=8.3; extra == "dev"
18
- Requires-Dist: ruff>=0.8; extra == "dev"
19
- Requires-Dist: mypy>=1.13; extra == "dev"
@@ -1,19 +0,0 @@
1
- Metadata-Version: 2.4
2
- Name: annotide-training
3
- Version: 0.1.0
4
- Summary: Cloud-agnostic annotation platform — reference customer-side training pipeline (ML-9, EXP-8)
5
- Requires-Python: >=3.12
6
- Requires-Dist: annotide>=0.1
7
- Provides-Extra: torch
8
- Requires-Dist: torch>=2.5; extra == "torch"
9
- Requires-Dist: torchvision>=0.20; extra == "torch"
10
- Requires-Dist: onnx>=1.16; extra == "torch"
11
- Requires-Dist: onnxruntime>=1.20; extra == "torch"
12
- Requires-Dist: numpy>=2.1; extra == "torch"
13
- Requires-Dist: pillow>=11.0; extra == "torch"
14
- Provides-Extra: mlflow
15
- Requires-Dist: mlflow-skinny>=3.1; extra == "mlflow"
16
- Provides-Extra: dev
17
- Requires-Dist: pytest>=8.3; extra == "dev"
18
- Requires-Dist: ruff>=0.8; extra == "dev"
19
- Requires-Dist: mypy>=1.13; extra == "dev"