visiontrack-mot 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- visiontrack_mot-0.1.0/LICENSE +21 -0
- visiontrack_mot-0.1.0/PKG-INFO +287 -0
- visiontrack_mot-0.1.0/README.md +235 -0
- visiontrack_mot-0.1.0/pyproject.toml +68 -0
- visiontrack_mot-0.1.0/setup.cfg +4 -0
- visiontrack_mot-0.1.0/src/visiontrack/__init__.py +48 -0
- visiontrack_mot-0.1.0/src/visiontrack/appearance/__init__.py +21 -0
- visiontrack_mot-0.1.0/src/visiontrack/appearance/embedder.py +169 -0
- visiontrack_mot-0.1.0/src/visiontrack/appearance/gallery.py +37 -0
- visiontrack_mot-0.1.0/src/visiontrack/appearance/reid_onnx.py +149 -0
- visiontrack_mot-0.1.0/src/visiontrack/cli.py +316 -0
- visiontrack_mot-0.1.0/src/visiontrack/core/__init__.py +6 -0
- visiontrack_mot-0.1.0/src/visiontrack/core/assignment.py +181 -0
- visiontrack_mot-0.1.0/src/visiontrack/core/geometry.py +177 -0
- visiontrack_mot-0.1.0/src/visiontrack/core/kalman.py +245 -0
- visiontrack_mot-0.1.0/src/visiontrack/datasets/__init__.py +13 -0
- visiontrack_mot-0.1.0/src/visiontrack/datasets/cache.py +177 -0
- visiontrack_mot-0.1.0/src/visiontrack/datasets/splits.py +84 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/__init__.py +14 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/base.py +79 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/cached.py +57 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/dancetrack_loader.py +128 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/mot_loader.py +268 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/noise.py +118 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/onnx_yolo.py +138 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/synthetic.py +254 -0
- visiontrack_mot-0.1.0/src/visiontrack/detection/yolox_onnx.py +170 -0
- visiontrack_mot-0.1.0/src/visiontrack/eval/__init__.py +24 -0
- visiontrack_mot-0.1.0/src/visiontrack/eval/calibration.py +98 -0
- visiontrack_mot-0.1.0/src/visiontrack/eval/hota.py +258 -0
- visiontrack_mot-0.1.0/src/visiontrack/eval/mot.py +225 -0
- visiontrack_mot-0.1.0/src/visiontrack/eval/mot17.py +179 -0
- visiontrack_mot-0.1.0/src/visiontrack/eval/stats.py +159 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/__init__.py +8 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/config.py +111 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/cost.py +165 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/motion/__init__.py +4 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/motion/gmc.py +70 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/motion/oc.py +116 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/motion/residual.py +149 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/presets.py +106 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/track.py +177 -0
- visiontrack_mot-0.1.0/src/visiontrack/tracking/tracker.py +339 -0
- visiontrack_mot-0.1.0/src/visiontrack/video.py +180 -0
- visiontrack_mot-0.1.0/src/visiontrack/viz/__init__.py +6 -0
- visiontrack_mot-0.1.0/src/visiontrack/viz/draw.py +156 -0
- visiontrack_mot-0.1.0/src/visiontrack_mot.egg-info/PKG-INFO +287 -0
- visiontrack_mot-0.1.0/src/visiontrack_mot.egg-info/SOURCES.txt +76 -0
- visiontrack_mot-0.1.0/src/visiontrack_mot.egg-info/dependency_links.txt +1 -0
- visiontrack_mot-0.1.0/src/visiontrack_mot.egg-info/entry_points.txt +2 -0
- visiontrack_mot-0.1.0/src/visiontrack_mot.egg-info/requires.txt +34 -0
- visiontrack_mot-0.1.0/src/visiontrack_mot.egg-info/top_level.txt +1 -0
- visiontrack_mot-0.1.0/tests/test_appearance.py +183 -0
- visiontrack_mot-0.1.0/tests/test_assignment.py +70 -0
- visiontrack_mot-0.1.0/tests/test_benchmark.py +81 -0
- visiontrack_mot-0.1.0/tests/test_cost.py +122 -0
- visiontrack_mot-0.1.0/tests/test_dancetrack_loader.py +80 -0
- visiontrack_mot-0.1.0/tests/test_error_taxonomy.py +77 -0
- visiontrack_mot-0.1.0/tests/test_experiments.py +122 -0
- visiontrack_mot-0.1.0/tests/test_geometry.py +63 -0
- visiontrack_mot-0.1.0/tests/test_gmc.py +73 -0
- visiontrack_mot-0.1.0/tests/test_hota.py +122 -0
- visiontrack_mot-0.1.0/tests/test_hota_vs_trackeval.py +142 -0
- visiontrack_mot-0.1.0/tests/test_kalman.py +82 -0
- visiontrack_mot-0.1.0/tests/test_mot.py +83 -0
- visiontrack_mot-0.1.0/tests/test_mot17_runner.py +103 -0
- visiontrack_mot-0.1.0/tests/test_mot_loader.py +168 -0
- visiontrack_mot-0.1.0/tests/test_oc_sort.py +142 -0
- visiontrack_mot-0.1.0/tests/test_presets.py +94 -0
- visiontrack_mot-0.1.0/tests/test_profile_fps.py +24 -0
- visiontrack_mot-0.1.0/tests/test_public_api.py +33 -0
- visiontrack_mot-0.1.0/tests/test_residual.py +105 -0
- visiontrack_mot-0.1.0/tests/test_spatial_embedder.py +48 -0
- visiontrack_mot-0.1.0/tests/test_stats.py +80 -0
- visiontrack_mot-0.1.0/tests/test_synthetic_appearance.py +106 -0
- visiontrack_mot-0.1.0/tests/test_tracker.py +105 -0
- visiontrack_mot-0.1.0/tests/test_uncertainty.py +121 -0
- visiontrack_mot-0.1.0/tests/test_video.py +161 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Rushikesh Hulage
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: visiontrack-mot
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Online multi-object tracking (Kalman + Hungarian + ByteTrack) built from scratch on NumPy
|
|
5
|
+
Author: Rushikesh Hulage
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://visiontrack.hulage.in
|
|
8
|
+
Project-URL: Repository, https://github.com/hulagerushikesh/visiontrack
|
|
9
|
+
Project-URL: Issues, https://github.com/hulagerushikesh/visiontrack/issues
|
|
10
|
+
Keywords: computer-vision,multi-object-tracking,kalman-filter,bytetrack,sort
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Image Recognition
|
|
19
|
+
Classifier: Intended Audience :: Science/Research
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Requires-Dist: numpy>=1.23
|
|
24
|
+
Provides-Extra: viz
|
|
25
|
+
Requires-Dist: matplotlib>=3.6; extra == "viz"
|
|
26
|
+
Requires-Dist: pillow>=9.0; extra == "viz"
|
|
27
|
+
Provides-Extra: onnx
|
|
28
|
+
Requires-Dist: onnxruntime>=1.15; extra == "onnx"
|
|
29
|
+
Provides-Extra: video
|
|
30
|
+
Requires-Dist: onnxruntime>=1.15; extra == "video"
|
|
31
|
+
Requires-Dist: imageio>=2.31; extra == "video"
|
|
32
|
+
Requires-Dist: imageio-ffmpeg>=0.4; extra == "video"
|
|
33
|
+
Requires-Dist: pillow>=9.0; extra == "video"
|
|
34
|
+
Provides-Extra: appearance
|
|
35
|
+
Requires-Dist: matplotlib>=3.6; extra == "appearance"
|
|
36
|
+
Requires-Dist: pillow>=9.0; extra == "appearance"
|
|
37
|
+
Provides-Extra: experiments
|
|
38
|
+
Requires-Dist: pandas>=1.5; extra == "experiments"
|
|
39
|
+
Requires-Dist: pyarrow>=10; extra == "experiments"
|
|
40
|
+
Requires-Dist: scipy>=1.9; extra == "experiments"
|
|
41
|
+
Requires-Dist: pyyaml>=6; extra == "experiments"
|
|
42
|
+
Requires-Dist: matplotlib>=3.6; extra == "experiments"
|
|
43
|
+
Provides-Extra: dev
|
|
44
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
45
|
+
Requires-Dist: scipy>=1.9; extra == "dev"
|
|
46
|
+
Requires-Dist: matplotlib>=3.6; extra == "dev"
|
|
47
|
+
Requires-Dist: pillow>=9.0; extra == "dev"
|
|
48
|
+
Requires-Dist: pandas>=1.5; extra == "dev"
|
|
49
|
+
Requires-Dist: pyarrow>=10; extra == "dev"
|
|
50
|
+
Requires-Dist: pyyaml>=6; extra == "dev"
|
|
51
|
+
Dynamic: license-file
|
|
52
|
+
|
|
53
|
+
# VisionTrack
|
|
54
|
+
|
|
55
|
+
[](https://github.com/hulagerushikesh/visiontrack/actions/workflows/ci.yml)
|
|
56
|
+
[](https://www.python.org/)
|
|
57
|
+
[](LICENSE)
|
|
58
|
+
[](https://visiontrack.hulage.in)
|
|
59
|
+
[](https://colab.research.google.com/github/hulagerushikesh/visiontrack/blob/main/notebooks/reproduce.ipynb)
|
|
60
|
+
|
|
61
|
+
**A from-scratch multi-object tracker, used as a controlled study of *when* the field's standard tricks actually help.**
|
|
62
|
+
|
|
63
|
+
The tracker — an 8-state **Kalman filter**, an O(n³) **Hungarian** solver, and **ByteTrack** two-stage association — is implemented from first principles on NumPy, with no ML framework in the core. On top of it sits a reproducible experiment harness that measures, on **real MOT17** with seed variance and paired significance tests, whether **appearance** and **uncertainty-aware association** actually improve tracking. Several of the answers are honest negatives — which is the point.
|
|
64
|
+
|
|
65
|
+

|
|
66
|
+
|
|
67
|
+
*The tracker on a synthetic scene — boxes in, stable per-object IDs out. A **real-MOT17** version can be rendered locally on your own copy of the dataset with `python scripts/render_mot17_demo.py` (the frames aren't redistributed here — MOT17 is under a non-commercial share-alike licence).*
|
|
68
|
+
|
|
69
|
+
---
|
|
70
|
+
|
|
71
|
+
## Abstract
|
|
72
|
+
|
|
73
|
+
Most tracking repositories are a thin wrapper over a detector plus a vendored tracker; every reported number is a single run on one configuration. VisionTrack inverts that: the estimation and association math *is* the deliverable (independently tested against SciPy and `trackeval`), and it is used to run a **falsifiable study**. We reproduce a from-scratch ByteTrack baseline on MOT17 public detections (HOTA/IDF1 verified against `trackeval` to within `1.4e-3`), then ablate two common enhancements under seed variance and Wilcoxon significance:
|
|
74
|
+
|
|
75
|
+
- **Appearance (RQ1)** — a per-track re-ID cost helps *association* on MOT17 (ID switches 188 → 170) but the effect is small and not significant with a cheap descriptor.
|
|
76
|
+
- **Uncertainty (RQ3)** — folding calibrated Kalman uncertainty into the cost is **null**, and *calibrating* the filter to real motion is **actively harmful** under detector noise (ID switches 40 → 400+). The filter's apparent under-confidence turns out to be a robustness feature, not a bug.
|
|
77
|
+
|
|
78
|
+

|
|
79
|
+
|
|
80
|
+
*Synthetic sanity check: seven objects over 120 frames — tracks **#3** and **#7** cross in the centre and keep their identities.*
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## Research questions
|
|
85
|
+
|
|
86
|
+
| | Question | Answer on MOT17 |
|
|
87
|
+
|---|---|---|
|
|
88
|
+
| **RQ1** | Does an appearance (re-ID) association cost improve tracking? | Yes — deep re-ID **significantly** cuts ID switches on **both MOT17 and DanceTrack** (p<0.05). The predicted "appearance hurts on near-identical dancers" **sign-flip does not occur**: a gated cost that only *ranks* within the motion gate is robustly beneficial-or-neutral |
|
|
89
|
+
| **RQ2** | Does a learned motion residual on top of constant-velocity help? | **It helps *prediction* only where motion is non-linear** (DanceTrack next-centre error −12.5%; MOT17 +21.5% worse, since CV is already ~1px there) — but that gain **hurts *tracking*** on both (train-on-GT/infer-on-noisy skew + it perturbs the association gate). A from-scratch NumPy MLP |
|
|
90
|
+
| **RQ3** | Does folding *calibrated* Kalman uncertainty into the cost reduce switches under noise? | **No** — null as a soft cost; calibrating the filter *hurts* badly under detector noise |
|
|
91
|
+
|
|
92
|
+
The value is the same whichever way each result falls: a controlled study that shows a trick *doesn't* help is a contribution, not a failure.
|
|
93
|
+
|
|
94
|
+
---
|
|
95
|
+
|
|
96
|
+
## Method
|
|
97
|
+
|
|
98
|
+
**The from-scratch core** (`core/`, NumPy only):
|
|
99
|
+
|
|
100
|
+
- **Kalman filter** — 8-state constant-velocity model in DeepSORT's `xyah` parametrization, height-scaled noise for scale-invariance, Joseph-form covariance update, Mahalanobis gating. Convergence-tested.
|
|
101
|
+
- **Hungarian assignment** — rectangular O(n³) Kuhn–Munkres with dual potentials; validated against `scipy.optimize.linear_sum_assignment` on 150 random matrices.
|
|
102
|
+
- **ByteTrack two-stage association** + a `Tentative → Confirmed → Deleted` lifecycle FSM.
|
|
103
|
+
|
|
104
|
+
**The ablation surface** (`tracking/cost.py`): the association cost is factored so each hypothesis is one weighted term with a hard gate deciding feasibility and the terms only *ranking* feasible pairs (DeepSORT-style fusion):
|
|
105
|
+
|
|
106
|
+
```
|
|
107
|
+
cost = w_iou·motion ⊕ w_app·appearance ⊕ w_unc·uncertainty (gate: IoU + class + Mahalanobis)
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
At `w_app = w_unc = 0` this is **bit-identical** to the plain `1 − IoU` baseline, so every ablation toggles exactly one variable.
|
|
111
|
+
|
|
112
|
+
**Rigor** (`eval/`, `experiments/`):
|
|
113
|
+
|
|
114
|
+
- **Metrics**: from-scratch **CLEAR-MOT + IDF1 + HOTA**, cross-checked against `trackeval` end-to-end on real MOT17 (MOTA/IDF1 exact, HOTA within `1.4e-3`).
|
|
115
|
+
- **Seed variance + paired significance**: every configuration is run over seeds; variants are compared paired (same sequences/seeds) with a **paired bootstrap + Wilcoxon** test and Cohen's d.
|
|
116
|
+
- **Compute once**: detections and appearance embeddings are cached to disk (5.2 MB detections; 6.3 MB colour-histogram or 67 MB deep-re-ID features for all of MOT17-train), so the raw ~5 GB of frames can be deleted and every experiment runs CPU-only in seconds.
|
|
117
|
+
|
|
118
|
+
---
|
|
119
|
+
|
|
120
|
+
## Results
|
|
121
|
+
|
|
122
|
+
### Baseline — from-scratch ByteTrack on real MOT17 (val-half, public detections)
|
|
123
|
+
|
|
124
|
+
| detector | MOTA | IDF1 | HOTA | DetA | AssA |
|
|
125
|
+
|----------|-----:|-----:|-----:|-----:|-----:|
|
|
126
|
+
| **SDP** | 0.624 | 0.673 | 0.565 | 0.542 | 0.589 |
|
|
127
|
+
| **FRCNN** | 0.469 | 0.570 | 0.497 | 0.423 | 0.587 |
|
|
128
|
+
| **DPM** | 0.115 | 0.182 | 0.193 | 0.093 | 0.404 |
|
|
129
|
+
|
|
130
|
+
Squarely in the public-detection neighbourhood (the famous ~76 MOTA ByteTrack uses a private YOLOX detector), with the expected detector ordering. `scripts/xcheck_mot17_trackeval.py` confirms our whole pipeline (preprocessing + metrics) matches `trackeval` on a real sequence.
|
|
131
|
+
|
|
132
|
+
### RQ1 — appearance (MOT17 FRCNN), from-scratch histogram vs deep re-ID
|
|
133
|
+
|
|
134
|
+
Δ vs the `w_app=0` motion-only baseline (HOTA 0.497 / IDF1 0.570 / AssA 0.587 / IDSW 188), same 7-sequence pairing:
|
|
135
|
+
|
|
136
|
+
| w_app=0.6 embedder | HOTA | IDF1 | AssA | IDSW |
|
|
137
|
+
|--------------------|------|------|------|-----:|
|
|
138
|
+
| from-scratch colour histogram | +0.001 | +0.002 | +0.003 | 170 (−18) |
|
|
139
|
+
| **deep re-ID** (OSNet-x0.25, MSMT17, ONNX) | **+0.004** | **+0.004** | **+0.008** | **163 (−25)** |
|
|
140
|
+
|
|
141
|
+

|
|
142
|
+
|
|
143
|
+
Appearance's clearest effect is on **ID switches**. A deep re-ID embedder — a pretrained OSNet run through `appearance/reid_onnx.py` behind the same interface — **roughly doubles** the association gain over the hand-crafted histogram and cuts ID switches to **163 (−13%)**, earning that weight *early* (−14 IDSW already at `w_app=0.15`).
|
|
144
|
+
|
|
145
|
+
Pooling all **three public detectors** (DPM/FRCNN/SDP → 21 seq×detector units) for statistical power, the **ID-switch and IDF1 reductions become significant** (paired Wilcoxon p<0.05); association-quality (AssA/HOTA) stays marginal (p≈0.06). The instructive twist: appearance is **completely inert on the weak DPM detector** — its poorly-localized boxes yield mis-framed crops, so the re-ID embeddings carry no identity signal. **Detection/crop quality gates whether appearance helps at all**, the opposite of the "weak detector needs it most" intuition, and it doesn't stratify by crowd density or occlusion. [Details →](docs/PHASE3.md)
|
|
146
|
+
|
|
147
|
+

|
|
148
|
+
|
|
149
|
+
### RQ2 — learned motion residual (helps prediction, hurts tracking)
|
|
150
|
+
|
|
151
|
+
A from-scratch NumPy MLP (hand-written back-prop + Adam — no framework) predicts a correction to the constant-velocity Kalman mean, trained on GT trajectories.
|
|
152
|
+
|
|
153
|
+
| | CV next-centre error | + residual |
|
|
154
|
+
|---|---|---|
|
|
155
|
+
| MOT17 (near-linear) | 1.05 px | **1.27 (−21% worse)** |
|
|
156
|
+
| DanceTrack (non-linear) | 6.89 px | **6.03 (+12% better)** |
|
|
157
|
+
|
|
158
|
+
Open-loop, the residual helps *exactly* where motion is non-linear — but wired into the tracker it **hurts identity on both** (DanceTrack HOTA −0.043, +29 IDSW; MOT17 HOTA −0.013, +12 IDSW). Trained on clean GT but fed the tracker's noisy estimates, its correction mis-fires and perturbs the association gate. A better predictor that makes a worse tracker. [Details →](docs/PHASE4.md)
|
|
159
|
+
|
|
160
|
+
### RQ3 — calibrated uncertainty (the interesting negative)
|
|
161
|
+
|
|
162
|
+
Stepping the Kalman filter along real MOT17 GT, the innovation χ² is **0.15** where a calibrated filter would give **4.0** — it is ~25× *under-confident*, so its 95% gate captures **100%** of innovations and never rejects.
|
|
163
|
+
|
|
164
|
+

|
|
165
|
+
|
|
166
|
+
Consequences, measured:
|
|
167
|
+
|
|
168
|
+
- Folding uncertainty into the cost (`w_unc`) is **null** (the gate is inert, the signal near-constant).
|
|
169
|
+
- *Calibrating* the filter (`kf_noise_scale=0.19`) is **catastrophic under detector noise**: HOTA −0.205, **ID switches 40 → 400+** (both p<0.05). A tight gate tuned to clean motion rejects noisy-but-correct detections and tracks fragment.
|
|
170
|
+
|
|
171
|
+
**The loose gate is a robustness feature, not a bug.** This unifies with the ablation finding that disabling the gate barely changes clean-data results. [Details →](docs/PHASE5.md)
|
|
172
|
+
|
|
173
|
+
### Throughput
|
|
174
|
+
|
|
175
|
+
Real-time on CPU (single-threaded): **2334 FPS** at 4 objects down to **214 FPS** at 64 — comfortably ≥30 FPS across the range (`benchmarks/bench_tracker.py`).
|
|
176
|
+
|
|
177
|
+
---
|
|
178
|
+
|
|
179
|
+
## Install & use
|
|
180
|
+
|
|
181
|
+
```bash
|
|
182
|
+
pip install visiontrack-mot # core: NumPy only
|
|
183
|
+
pip install 'visiontrack-mot[video]' # + run on your own video files
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
from visiontrack import ByteTracker, TrackerConfig
|
|
188
|
+
tracker = ByteTracker(TrackerConfig())
|
|
189
|
+
for frame_detections in stream: # list[Detection]
|
|
190
|
+
observations = tracker.update(frame_detections)
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
Track a real video from the command line (needs a YOLOX ONNX model, see [docs/VIDEO.md](docs/VIDEO.md)):
|
|
194
|
+
|
|
195
|
+
```bash
|
|
196
|
+
visiontrack track input.mp4 out.mp4 --model models/yolox_nano.onnx
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
Full public API: [docs/API.md](docs/API.md) · packaging/release: [docs/RELEASE.md](docs/RELEASE.md).
|
|
200
|
+
|
|
201
|
+
**Benchmark a set of trackers** into one report (leaderboard + paired significance + ID-switch error taxonomy) — `make benchmark`, or see the live report at **[visiontrack.hulage.in/benchmark](https://visiontrack.hulage.in/benchmark)**.
|
|
202
|
+
|
|
203
|
+
## Reproduce
|
|
204
|
+
|
|
205
|
+
**No install? Run the synthetic study in your browser:**
|
|
206
|
+
[](https://colab.research.google.com/github/hulagerushikesh/visiontrack/blob/main/notebooks/reproduce.ipynb)
|
|
207
|
+
— clones, installs the harness extra, and reproduces the seed-varied, significance-tested synthetic tables (a couple of minutes, CPU-only). Notebook: [`notebooks/reproduce.ipynb`](notebooks/reproduce.ipynb).
|
|
208
|
+
|
|
209
|
+
```bash
|
|
210
|
+
make install # editable install with all extras
|
|
211
|
+
make test # full test suite
|
|
212
|
+
make reproduce-synth # synthetic harness + RQ3 probe — NO dataset needed
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
Full real-data reproduction (after a one-time cache build — see [docs/PHASE0.md](docs/PHASE0.md)):
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
python data/cache/precompute.py --data-root ~/MOT17 --detector FRCNN --out data/cache/mot17
|
|
219
|
+
python data/cache/precompute_embeddings.py --data-root ~/MOT17 --detector FRCNN --cache-dir data/cache/mot17
|
|
220
|
+
make reproduce # regenerates every table and figure above
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
Every run is pinned by a config hash; bootstrap resampling and synthetic scenes are seeded. Each phase has a runbook in [`docs/`](docs/) (`PHASE0.md` … `PHASE5.md`).
|
|
224
|
+
|
|
225
|
+
---
|
|
226
|
+
|
|
227
|
+
## Engineering
|
|
228
|
+
|
|
229
|
+
- **Zero ML-framework dependency in the core** — just NumPy. Heavy/optional deps (`scipy`, `pandas`, `matplotlib`, `pillow`, `onnxruntime`, `torch`) are isolated in extras (`[experiments]`, `[appearance]`, `[onnx]`) and lazily imported; nothing in `core/` imports them.
|
|
230
|
+
- **253 tests**: unit, property (Hungarian vs SciPy), convergence (Kalman), metric cross-checks (HOTA/IDF1 vs `trackeval`), and end-to-end integration with a MOTA floor.
|
|
231
|
+
- **CI** on Python 3.10/3.11/3.12 + ruff.
|
|
232
|
+
|
|
233
|
+
```
|
|
234
|
+
src/visiontrack/
|
|
235
|
+
core/ geometry · kalman · assignment ← from-scratch math
|
|
236
|
+
detection/ base · synthetic · onnx_yolo · mot_loader · noise
|
|
237
|
+
appearance/ embedder (colour-hist) · reid_onnx · gallery (RQ1)
|
|
238
|
+
tracking/ tracker · track (FSM) · cost (ablation surface) · config
|
|
239
|
+
eval/ mot (CLEAR-MOT) · hota (HOTA/IDF1) · stats · calibration (RQ3)
|
|
240
|
+
datasets/ splits (frozen) · cache (detections + embeddings)
|
|
241
|
+
experiments/ run_matrix · analyze · appearance_study · uncertainty_study · configs/
|
|
242
|
+
data/cache/ precompute · precompute_embeddings
|
|
243
|
+
scripts/ xcheck_mot17_trackeval.py
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
---
|
|
247
|
+
|
|
248
|
+
## Limitations & honest negatives
|
|
249
|
+
|
|
250
|
+
- **Public-detection, train/val split** — reproducible and self-contained, not test-server leaderboard numbers (deliberate).
|
|
251
|
+
- **RQ1's effect is real but small on MOT17** — a deep re-ID embedder (OSNet-x0.25, MSMT17) roughly doubles the association gain over the from-scratch histogram; pooling all three detectors (21 units) makes the **ID-switch/IDF1 reduction significant**, but association quality (AssA/HOTA) stays marginal and the benefit is capped by public-detection crop quality. A controlled **synthetic probe** (dialing inter-object appearance similarity) confirms the benefit *grows with object distinctness* and is significant on AssA/IDF1 — but never flips to *harmful*. And the decisive test — real **DanceTrack** (near-identical dancers, non-linear motion, the predicted "appearance hurts" case) — **refutes the hypothesis**: deep re-ID *still* significantly cuts ID switches there (217→202, p<0.05). Across all three probes a gated appearance cost is robustly beneficial-or-neutral. The "appearance hurts on DanceTrack" failure is a property of *appearance-vetoing* designs, not of uniform appearance itself — our cost only lets appearance **rank within the motion gate, never veto** a feasible match. Re-ID weights carry a non-commercial dataset licence and are **not committed** (regenerate from your own download).
|
|
252
|
+
- **RQ2 (learned motion residual) is answered — an honest negative.** A from-scratch NumPy MLP (hand-written back-prop, no framework) lowers open-loop next-centre error on DanceTrack (−12.5%) but not MOT17 (near-linear, ~1px CV error → +21.5% worse); crucially, inside the tracker it **hurts** on both (train-on-GT / infer-on-noisy-estimates distribution shift + it perturbs the association gate). Trained on GT trajectories; weights gitignored. [Details → docs/PHASE4.md](docs/PHASE4.md)
|
|
253
|
+
- **The cross-dataset RQ1 test is done on DanceTrack, with a scope caveat** — detections are oracle-perturbed GT (DanceTrack ships none), which gives clean crops that *favour* appearance; a poor real detector would weaken it (cf. DPM). The cost is weighted-and-gated, not pure-appearance. So the refutation is scoped: *a gated ranking appearance cost does not hurt, even on DanceTrack.* SportsMOT (RQ2 maneuver data) is still deferred.
|
|
254
|
+
## Interactive demo
|
|
255
|
+
|
|
256
|
+
A self-contained, pre-baked demo that makes the RQ1 result tangible: the same
|
|
257
|
+
synthetic scene tracked two ways — motion-only vs motion + appearance — with the
|
|
258
|
+
running ID-switch count side by side. Boxes are coloured by track ID, so a colour
|
|
259
|
+
flip on a moving object *is* a switch.
|
|
260
|
+
|
|
261
|
+
- **Try it live:** **[visiontrack.hulage.in/demo](https://visiontrack.hulage.in/demo)** — no install, runs in the browser (deployed from `viz/webdemo/` via Vercel; project home at [visiontrack.hulage.in](https://visiontrack.hulage.in)).
|
|
262
|
+
- **Build it locally:** `make demo` → open [`viz/webdemo/index.html`](viz/webdemo/index.html) in any browser (no server, no dataset, no toolchain — inference is pre-baked from the NumPy tracker on a synthetic scene).
|
|
263
|
+
- On the selected scene, appearance cuts ID switches **37 → 29 (−22%)** and lifts IDF1 — the study result, watchable frame by frame.
|
|
264
|
+
|
|
265
|
+
## Write-up
|
|
266
|
+
|
|
267
|
+
A narrative walk-through of the study — why appearance *refuses* to hurt (even on
|
|
268
|
+
DanceTrack), how a better motion predictor made a *worse* tracker, and why the
|
|
269
|
+
filter's loose gate is a feature: **[visiontrack.hulage.in/writeup](https://visiontrack.hulage.in/writeup)**.
|
|
270
|
+
The honest negatives, explained in plain prose rather than tables.
|
|
271
|
+
|
|
272
|
+
## Study guide
|
|
273
|
+
|
|
274
|
+
New to the concepts? Two self-paced, interactive learning files (single self-contained
|
|
275
|
+
HTML, progress checkboxes saved in your browser — open in any browser):
|
|
276
|
+
|
|
277
|
+
- [`docs/LEARNING_PATH.html`](docs/LEARNING_PATH.html) — **this project, topic by topic**:
|
|
278
|
+
Kalman → Hungarian → ByteTrack → metrics → the three research questions → where to take
|
|
279
|
+
it next, with every concept linked to the file it lives in.
|
|
280
|
+
- [`docs/CV_ROADMAP.html`](docs/CV_ROADMAP.html) — **the whole field, in order**: a
|
|
281
|
+
junior → mid → senior → research computer-vision roadmap (foundations → deep learning →
|
|
282
|
+
detection/segmentation/tracking → transformers/generative/3D → production → doing research),
|
|
283
|
+
with what-to-build and canonical resources at each stage.
|
|
284
|
+
|
|
285
|
+
## License
|
|
286
|
+
|
|
287
|
+
MIT
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
# VisionTrack
|
|
2
|
+
|
|
3
|
+
[](https://github.com/hulagerushikesh/visiontrack/actions/workflows/ci.yml)
|
|
4
|
+
[](https://www.python.org/)
|
|
5
|
+
[](LICENSE)
|
|
6
|
+
[](https://visiontrack.hulage.in)
|
|
7
|
+
[](https://colab.research.google.com/github/hulagerushikesh/visiontrack/blob/main/notebooks/reproduce.ipynb)
|
|
8
|
+
|
|
9
|
+
**A from-scratch multi-object tracker, used as a controlled study of *when* the field's standard tricks actually help.**
|
|
10
|
+
|
|
11
|
+
The tracker — an 8-state **Kalman filter**, an O(n³) **Hungarian** solver, and **ByteTrack** two-stage association — is implemented from first principles on NumPy, with no ML framework in the core. On top of it sits a reproducible experiment harness that measures, on **real MOT17** with seed variance and paired significance tests, whether **appearance** and **uncertainty-aware association** actually improve tracking. Several of the answers are honest negatives — which is the point.
|
|
12
|
+
|
|
13
|
+

|
|
14
|
+
|
|
15
|
+
*The tracker on a synthetic scene — boxes in, stable per-object IDs out. A **real-MOT17** version can be rendered locally on your own copy of the dataset with `python scripts/render_mot17_demo.py` (the frames aren't redistributed here — MOT17 is under a non-commercial share-alike licence).*
|
|
16
|
+
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
## Abstract
|
|
20
|
+
|
|
21
|
+
Most tracking repositories are a thin wrapper over a detector plus a vendored tracker; every reported number is a single run on one configuration. VisionTrack inverts that: the estimation and association math *is* the deliverable (independently tested against SciPy and `trackeval`), and it is used to run a **falsifiable study**. We reproduce a from-scratch ByteTrack baseline on MOT17 public detections (HOTA/IDF1 verified against `trackeval` to within `1.4e-3`), then ablate two common enhancements under seed variance and Wilcoxon significance:
|
|
22
|
+
|
|
23
|
+
- **Appearance (RQ1)** — a per-track re-ID cost helps *association* on MOT17 (ID switches 188 → 170) but the effect is small and not significant with a cheap descriptor.
|
|
24
|
+
- **Uncertainty (RQ3)** — folding calibrated Kalman uncertainty into the cost is **null**, and *calibrating* the filter to real motion is **actively harmful** under detector noise (ID switches 40 → 400+). The filter's apparent under-confidence turns out to be a robustness feature, not a bug.
|
|
25
|
+
|
|
26
|
+

|
|
27
|
+
|
|
28
|
+
*Synthetic sanity check: seven objects over 120 frames — tracks **#3** and **#7** cross in the centre and keep their identities.*
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Research questions
|
|
33
|
+
|
|
34
|
+
| | Question | Answer on MOT17 |
|
|
35
|
+
|---|---|---|
|
|
36
|
+
| **RQ1** | Does an appearance (re-ID) association cost improve tracking? | Yes — deep re-ID **significantly** cuts ID switches on **both MOT17 and DanceTrack** (p<0.05). The predicted "appearance hurts on near-identical dancers" **sign-flip does not occur**: a gated cost that only *ranks* within the motion gate is robustly beneficial-or-neutral |
|
|
37
|
+
| **RQ2** | Does a learned motion residual on top of constant-velocity help? | **It helps *prediction* only where motion is non-linear** (DanceTrack next-centre error −12.5%; MOT17 +21.5% worse, since CV is already ~1px there) — but that gain **hurts *tracking*** on both (train-on-GT/infer-on-noisy skew + it perturbs the association gate). A from-scratch NumPy MLP |
|
|
38
|
+
| **RQ3** | Does folding *calibrated* Kalman uncertainty into the cost reduce switches under noise? | **No** — null as a soft cost; calibrating the filter *hurts* badly under detector noise |
|
|
39
|
+
|
|
40
|
+
The value is the same whichever way each result falls: a controlled study that shows a trick *doesn't* help is a contribution, not a failure.
|
|
41
|
+
|
|
42
|
+
---
|
|
43
|
+
|
|
44
|
+
## Method
|
|
45
|
+
|
|
46
|
+
**The from-scratch core** (`core/`, NumPy only):
|
|
47
|
+
|
|
48
|
+
- **Kalman filter** — 8-state constant-velocity model in DeepSORT's `xyah` parametrization, height-scaled noise for scale-invariance, Joseph-form covariance update, Mahalanobis gating. Convergence-tested.
|
|
49
|
+
- **Hungarian assignment** — rectangular O(n³) Kuhn–Munkres with dual potentials; validated against `scipy.optimize.linear_sum_assignment` on 150 random matrices.
|
|
50
|
+
- **ByteTrack two-stage association** + a `Tentative → Confirmed → Deleted` lifecycle FSM.
|
|
51
|
+
|
|
52
|
+
**The ablation surface** (`tracking/cost.py`): the association cost is factored so each hypothesis is one weighted term with a hard gate deciding feasibility and the terms only *ranking* feasible pairs (DeepSORT-style fusion):
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
cost = w_iou·motion ⊕ w_app·appearance ⊕ w_unc·uncertainty (gate: IoU + class + Mahalanobis)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
At `w_app = w_unc = 0` this is **bit-identical** to the plain `1 − IoU` baseline, so every ablation toggles exactly one variable.
|
|
59
|
+
|
|
60
|
+
**Rigor** (`eval/`, `experiments/`):
|
|
61
|
+
|
|
62
|
+
- **Metrics**: from-scratch **CLEAR-MOT + IDF1 + HOTA**, cross-checked against `trackeval` end-to-end on real MOT17 (MOTA/IDF1 exact, HOTA within `1.4e-3`).
|
|
63
|
+
- **Seed variance + paired significance**: every configuration is run over seeds; variants are compared paired (same sequences/seeds) with a **paired bootstrap + Wilcoxon** test and Cohen's d.
|
|
64
|
+
- **Compute once**: detections and appearance embeddings are cached to disk (5.2 MB detections; 6.3 MB colour-histogram or 67 MB deep-re-ID features for all of MOT17-train), so the raw ~5 GB of frames can be deleted and every experiment runs CPU-only in seconds.
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
## Results
|
|
69
|
+
|
|
70
|
+
### Baseline — from-scratch ByteTrack on real MOT17 (val-half, public detections)
|
|
71
|
+
|
|
72
|
+
| detector | MOTA | IDF1 | HOTA | DetA | AssA |
|
|
73
|
+
|----------|-----:|-----:|-----:|-----:|-----:|
|
|
74
|
+
| **SDP** | 0.624 | 0.673 | 0.565 | 0.542 | 0.589 |
|
|
75
|
+
| **FRCNN** | 0.469 | 0.570 | 0.497 | 0.423 | 0.587 |
|
|
76
|
+
| **DPM** | 0.115 | 0.182 | 0.193 | 0.093 | 0.404 |
|
|
77
|
+
|
|
78
|
+
Squarely in the public-detection neighbourhood (the famous ~76 MOTA ByteTrack uses a private YOLOX detector), with the expected detector ordering. `scripts/xcheck_mot17_trackeval.py` confirms our whole pipeline (preprocessing + metrics) matches `trackeval` on a real sequence.
|
|
79
|
+
|
|
80
|
+
### RQ1 — appearance (MOT17 FRCNN), from-scratch histogram vs deep re-ID
|
|
81
|
+
|
|
82
|
+
Δ vs the `w_app=0` motion-only baseline (HOTA 0.497 / IDF1 0.570 / AssA 0.587 / IDSW 188), same 7-sequence pairing:
|
|
83
|
+
|
|
84
|
+
| w_app=0.6 embedder | HOTA | IDF1 | AssA | IDSW |
|
|
85
|
+
|--------------------|------|------|------|-----:|
|
|
86
|
+
| from-scratch colour histogram | +0.001 | +0.002 | +0.003 | 170 (−18) |
|
|
87
|
+
| **deep re-ID** (OSNet-x0.25, MSMT17, ONNX) | **+0.004** | **+0.004** | **+0.008** | **163 (−25)** |
|
|
88
|
+
|
|
89
|
+

|
|
90
|
+
|
|
91
|
+
Appearance's clearest effect is on **ID switches**. A deep re-ID embedder — a pretrained OSNet run through `appearance/reid_onnx.py` behind the same interface — **roughly doubles** the association gain over the hand-crafted histogram and cuts ID switches to **163 (−13%)**, earning that weight *early* (−14 IDSW already at `w_app=0.15`).
|
|
92
|
+
|
|
93
|
+
Pooling all **three public detectors** (DPM/FRCNN/SDP → 21 seq×detector units) for statistical power, the **ID-switch and IDF1 reductions become significant** (paired Wilcoxon p<0.05); association-quality (AssA/HOTA) stays marginal (p≈0.06). The instructive twist: appearance is **completely inert on the weak DPM detector** — its poorly-localized boxes yield mis-framed crops, so the re-ID embeddings carry no identity signal. **Detection/crop quality gates whether appearance helps at all**, the opposite of the "weak detector needs it most" intuition, and it doesn't stratify by crowd density or occlusion. [Details →](docs/PHASE3.md)
|
|
94
|
+
|
|
95
|
+

|
|
96
|
+
|
|
97
|
+
### RQ2 — learned motion residual (helps prediction, hurts tracking)
|
|
98
|
+
|
|
99
|
+
A from-scratch NumPy MLP (hand-written back-prop + Adam — no framework) predicts a correction to the constant-velocity Kalman mean, trained on GT trajectories.
|
|
100
|
+
|
|
101
|
+
| | CV next-centre error | + residual |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| MOT17 (near-linear) | 1.05 px | **1.27 (−21% worse)** |
|
|
104
|
+
| DanceTrack (non-linear) | 6.89 px | **6.03 (+12% better)** |
|
|
105
|
+
|
|
106
|
+
Open-loop, the residual helps *exactly* where motion is non-linear — but wired into the tracker it **hurts identity on both** (DanceTrack HOTA −0.043, +29 IDSW; MOT17 HOTA −0.013, +12 IDSW). Trained on clean GT but fed the tracker's noisy estimates, its correction mis-fires and perturbs the association gate. A better predictor that makes a worse tracker. [Details →](docs/PHASE4.md)
|
|
107
|
+
|
|
108
|
+
### RQ3 — calibrated uncertainty (the interesting negative)
|
|
109
|
+
|
|
110
|
+
Stepping the Kalman filter along real MOT17 GT, the innovation χ² is **0.15** where a calibrated filter would give **4.0** — it is ~25× *under-confident*, so its 95% gate captures **100%** of innovations and never rejects.
|
|
111
|
+
|
|
112
|
+

|
|
113
|
+
|
|
114
|
+
Consequences, measured:
|
|
115
|
+
|
|
116
|
+
- Folding uncertainty into the cost (`w_unc`) is **null** (the gate is inert, the signal near-constant).
|
|
117
|
+
- *Calibrating* the filter (`kf_noise_scale=0.19`) is **catastrophic under detector noise**: HOTA −0.205, **ID switches 40 → 400+** (both p<0.05). A tight gate tuned to clean motion rejects noisy-but-correct detections and tracks fragment.
|
|
118
|
+
|
|
119
|
+
**The loose gate is a robustness feature, not a bug.** This unifies with the ablation finding that disabling the gate barely changes clean-data results. [Details →](docs/PHASE5.md)
|
|
120
|
+
|
|
121
|
+
### Throughput
|
|
122
|
+
|
|
123
|
+
Real-time on CPU (single-threaded): **2334 FPS** at 4 objects down to **214 FPS** at 64 — comfortably ≥30 FPS across the range (`benchmarks/bench_tracker.py`).
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
## Install & use
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
pip install visiontrack-mot # core: NumPy only
|
|
131
|
+
pip install 'visiontrack-mot[video]' # + run on your own video files
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
from visiontrack import ByteTracker, TrackerConfig
|
|
136
|
+
tracker = ByteTracker(TrackerConfig())
|
|
137
|
+
for frame_detections in stream: # list[Detection]
|
|
138
|
+
observations = tracker.update(frame_detections)
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
Track a real video from the command line (needs a YOLOX ONNX model, see [docs/VIDEO.md](docs/VIDEO.md)):
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
visiontrack track input.mp4 out.mp4 --model models/yolox_nano.onnx
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Full public API: [docs/API.md](docs/API.md) · packaging/release: [docs/RELEASE.md](docs/RELEASE.md).
|
|
148
|
+
|
|
149
|
+
**Benchmark a set of trackers** into one report (leaderboard + paired significance + ID-switch error taxonomy) — `make benchmark`, or see the live report at **[visiontrack.hulage.in/benchmark](https://visiontrack.hulage.in/benchmark)**.
|
|
150
|
+
|
|
151
|
+
## Reproduce
|
|
152
|
+
|
|
153
|
+
**No install? Run the synthetic study in your browser:**
|
|
154
|
+
[](https://colab.research.google.com/github/hulagerushikesh/visiontrack/blob/main/notebooks/reproduce.ipynb)
|
|
155
|
+
— clones, installs the harness extra, and reproduces the seed-varied, significance-tested synthetic tables (a couple of minutes, CPU-only). Notebook: [`notebooks/reproduce.ipynb`](notebooks/reproduce.ipynb).
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
make install # editable install with all extras
|
|
159
|
+
make test # full test suite
|
|
160
|
+
make reproduce-synth # synthetic harness + RQ3 probe — NO dataset needed
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Full real-data reproduction (after a one-time cache build — see [docs/PHASE0.md](docs/PHASE0.md)):
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
python data/cache/precompute.py --data-root ~/MOT17 --detector FRCNN --out data/cache/mot17
|
|
167
|
+
python data/cache/precompute_embeddings.py --data-root ~/MOT17 --detector FRCNN --cache-dir data/cache/mot17
|
|
168
|
+
make reproduce # regenerates every table and figure above
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
Every run is pinned by a config hash; bootstrap resampling and synthetic scenes are seeded. Each phase has a runbook in [`docs/`](docs/) (`PHASE0.md` … `PHASE5.md`).
|
|
172
|
+
|
|
173
|
+
---
|
|
174
|
+
|
|
175
|
+
## Engineering
|
|
176
|
+
|
|
177
|
+
- **Zero ML-framework dependency in the core** — just NumPy. Heavy/optional deps (`scipy`, `pandas`, `matplotlib`, `pillow`, `onnxruntime`, `torch`) are isolated in extras (`[experiments]`, `[appearance]`, `[onnx]`) and lazily imported; nothing in `core/` imports them.
|
|
178
|
+
- **253 tests**: unit, property (Hungarian vs SciPy), convergence (Kalman), metric cross-checks (HOTA/IDF1 vs `trackeval`), and end-to-end integration with a MOTA floor.
|
|
179
|
+
- **CI** on Python 3.10/3.11/3.12 + ruff.
|
|
180
|
+
|
|
181
|
+
```
|
|
182
|
+
src/visiontrack/
|
|
183
|
+
core/ geometry · kalman · assignment ← from-scratch math
|
|
184
|
+
detection/ base · synthetic · onnx_yolo · mot_loader · noise
|
|
185
|
+
appearance/ embedder (colour-hist) · reid_onnx · gallery (RQ1)
|
|
186
|
+
tracking/ tracker · track (FSM) · cost (ablation surface) · config
|
|
187
|
+
eval/ mot (CLEAR-MOT) · hota (HOTA/IDF1) · stats · calibration (RQ3)
|
|
188
|
+
datasets/ splits (frozen) · cache (detections + embeddings)
|
|
189
|
+
experiments/ run_matrix · analyze · appearance_study · uncertainty_study · configs/
|
|
190
|
+
data/cache/ precompute · precompute_embeddings
|
|
191
|
+
scripts/ xcheck_mot17_trackeval.py
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
---
|
|
195
|
+
|
|
196
|
+
## Limitations & honest negatives
|
|
197
|
+
|
|
198
|
+
- **Public-detection, train/val split** — reproducible and self-contained, not test-server leaderboard numbers (deliberate).
|
|
199
|
+
- **RQ1's effect is real but small on MOT17** — a deep re-ID embedder (OSNet-x0.25, MSMT17) roughly doubles the association gain over the from-scratch histogram; pooling all three detectors (21 units) makes the **ID-switch/IDF1 reduction significant**, but association quality (AssA/HOTA) stays marginal and the benefit is capped by public-detection crop quality. A controlled **synthetic probe** (dialing inter-object appearance similarity) confirms the benefit *grows with object distinctness* and is significant on AssA/IDF1 — but never flips to *harmful*. And the decisive test — real **DanceTrack** (near-identical dancers, non-linear motion, the predicted "appearance hurts" case) — **refutes the hypothesis**: deep re-ID *still* significantly cuts ID switches there (217→202, p<0.05). Across all three probes a gated appearance cost is robustly beneficial-or-neutral. The "appearance hurts on DanceTrack" failure is a property of *appearance-vetoing* designs, not of uniform appearance itself — our cost only lets appearance **rank within the motion gate, never veto** a feasible match. Re-ID weights carry a non-commercial dataset licence and are **not committed** (regenerate from your own download).
|
|
200
|
+
- **RQ2 (learned motion residual) is answered — an honest negative.** A from-scratch NumPy MLP (hand-written back-prop, no framework) lowers open-loop next-centre error on DanceTrack (−12.5%) but not MOT17 (near-linear, ~1px CV error → +21.5% worse); crucially, inside the tracker it **hurts** on both (train-on-GT / infer-on-noisy-estimates distribution shift + it perturbs the association gate). Trained on GT trajectories; weights gitignored. [Details → docs/PHASE4.md](docs/PHASE4.md)
|
|
201
|
+
- **The cross-dataset RQ1 test is done on DanceTrack, with a scope caveat** — detections are oracle-perturbed GT (DanceTrack ships none), which gives clean crops that *favour* appearance; a poor real detector would weaken it (cf. DPM). The cost is weighted-and-gated, not pure-appearance. So the refutation is scoped: *a gated ranking appearance cost does not hurt, even on DanceTrack.* SportsMOT (RQ2 maneuver data) is still deferred.
|
|
202
|
+
## Interactive demo
|
|
203
|
+
|
|
204
|
+
A self-contained, pre-baked demo that makes the RQ1 result tangible: the same
|
|
205
|
+
synthetic scene tracked two ways — motion-only vs motion + appearance — with the
|
|
206
|
+
running ID-switch count side by side. Boxes are coloured by track ID, so a colour
|
|
207
|
+
flip on a moving object *is* a switch.
|
|
208
|
+
|
|
209
|
+
- **Try it live:** **[visiontrack.hulage.in/demo](https://visiontrack.hulage.in/demo)** — no install, runs in the browser (deployed from `viz/webdemo/` via Vercel; project home at [visiontrack.hulage.in](https://visiontrack.hulage.in)).
|
|
210
|
+
- **Build it locally:** `make demo` → open [`viz/webdemo/index.html`](viz/webdemo/index.html) in any browser (no server, no dataset, no toolchain — inference is pre-baked from the NumPy tracker on a synthetic scene).
|
|
211
|
+
- On the selected scene, appearance cuts ID switches **37 → 29 (−22%)** and lifts IDF1 — the study result, watchable frame by frame.
|
|
212
|
+
|
|
213
|
+
## Write-up
|
|
214
|
+
|
|
215
|
+
A narrative walk-through of the study — why appearance *refuses* to hurt (even on
|
|
216
|
+
DanceTrack), how a better motion predictor made a *worse* tracker, and why the
|
|
217
|
+
filter's loose gate is a feature: **[visiontrack.hulage.in/writeup](https://visiontrack.hulage.in/writeup)**.
|
|
218
|
+
The honest negatives, explained in plain prose rather than tables.
|
|
219
|
+
|
|
220
|
+
## Study guide
|
|
221
|
+
|
|
222
|
+
New to the concepts? Two self-paced, interactive learning files (single self-contained
|
|
223
|
+
HTML, progress checkboxes saved in your browser — open in any browser):
|
|
224
|
+
|
|
225
|
+
- [`docs/LEARNING_PATH.html`](docs/LEARNING_PATH.html) — **this project, topic by topic**:
|
|
226
|
+
Kalman → Hungarian → ByteTrack → metrics → the three research questions → where to take
|
|
227
|
+
it next, with every concept linked to the file it lives in.
|
|
228
|
+
- [`docs/CV_ROADMAP.html`](docs/CV_ROADMAP.html) — **the whole field, in order**: a
|
|
229
|
+
junior → mid → senior → research computer-vision roadmap (foundations → deep learning →
|
|
230
|
+
detection/segmentation/tracking → transformers/generative/3D → production → doing research),
|
|
231
|
+
with what-to-build and canonical resources at each stage.
|
|
232
|
+
|
|
233
|
+
## License
|
|
234
|
+
|
|
235
|
+
MIT
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "visiontrack-mot"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Online multi-object tracking (Kalman + Hungarian + ByteTrack) built from scratch on NumPy"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Rushikesh Hulage" }]
|
|
13
|
+
keywords = ["computer-vision", "multi-object-tracking", "kalman-filter", "bytetrack", "sort"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
23
|
+
"Intended Audience :: Science/Research",
|
|
24
|
+
]
|
|
25
|
+
dependencies = ["numpy>=1.23"]
|
|
26
|
+
|
|
27
|
+
[project.urls]
|
|
28
|
+
Homepage = "https://visiontrack.hulage.in"
|
|
29
|
+
Repository = "https://github.com/hulagerushikesh/visiontrack"
|
|
30
|
+
Issues = "https://github.com/hulagerushikesh/visiontrack/issues"
|
|
31
|
+
|
|
32
|
+
[project.optional-dependencies]
|
|
33
|
+
viz = ["matplotlib>=3.6", "pillow>=9.0"]
|
|
34
|
+
onnx = ["onnxruntime>=1.15"]
|
|
35
|
+
# Run the tracker on real video files: decode/encode via imageio + ffmpeg,
|
|
36
|
+
# detect with an ONNX model. Lazily imported; never touched by the core.
|
|
37
|
+
video = ["onnxruntime>=1.15", "imageio>=2.31", "imageio-ffmpeg>=0.4", "pillow>=9.0"]
|
|
38
|
+
# Appearance / re-ID: the from-scratch colour-histogram embedder uses
|
|
39
|
+
# matplotlib's rgb_to_hsv; embedding precompute reads frames with Pillow.
|
|
40
|
+
appearance = ["matplotlib>=3.6", "pillow>=9.0"]
|
|
41
|
+
# Statistical-rigor harness under experiments/ (not needed by the core package).
|
|
42
|
+
experiments = ["pandas>=1.5", "pyarrow>=10", "scipy>=1.9", "pyyaml>=6", "matplotlib>=3.6"]
|
|
43
|
+
dev = ["pytest>=7.0", "scipy>=1.9", "matplotlib>=3.6", "pillow>=9.0", "pandas>=1.5", "pyarrow>=10", "pyyaml>=6"]
|
|
44
|
+
|
|
45
|
+
[project.scripts]
|
|
46
|
+
visiontrack = "visiontrack.cli:main"
|
|
47
|
+
|
|
48
|
+
[tool.setuptools.packages.find]
|
|
49
|
+
where = ["src"]
|
|
50
|
+
|
|
51
|
+
[tool.pytest.ini_options]
|
|
52
|
+
testpaths = ["tests"]
|
|
53
|
+
addopts = "-q -m 'not slow'" # real-data/long tests are opt-in: pytest -m slow
|
|
54
|
+
markers = ["slow: real-data or long-running tests (need dataset caches); opt in with -m slow"]
|
|
55
|
+
pythonpath = ["src", "."] # "." makes the experiments/ harness importable in tests
|
|
56
|
+
|
|
57
|
+
[tool.ruff]
|
|
58
|
+
line-length = 100
|
|
59
|
+
target-version = "py310"
|
|
60
|
+
|
|
61
|
+
[tool.ruff.lint]
|
|
62
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
63
|
+
# E741: single-char names (i, j, u, v) are idiomatic in the linear-algebra code.
|
|
64
|
+
ignore = ["E741"]
|
|
65
|
+
|
|
66
|
+
[tool.ruff.lint.per-file-ignores]
|
|
67
|
+
# The benchmark HTML renderer is a CSS/HTML template string — long lines are content.
|
|
68
|
+
"experiments/_benchmark_html.py" = ["E501"]
|