opensportslib 0.2.0.dev2__py3-none-any.whl → 0.2.0.dev4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,37 @@
1
+ from opensportslib.apis import VQAModel
2
+
3
+
4
+ def main():
5
+ """
6
+ Minimal VQA example.
7
+ Update config, question, and dataset paths before running.
8
+ """
9
+
10
+ my_model = VQAModel(
11
+ config="examples/configs/vqa_qwen.yaml",
12
+ weights=None, # optional: path or Hugging Face model ID
13
+ )
14
+
15
+ predictions = my_model.infer(
16
+ test_set="/path/to/test_annotations.json",
17
+ )
18
+
19
+ print(predictions)
20
+
21
+ single_prediction = my_model.infer(
22
+ video_path="/path/to/video.mp4",
23
+ question="What card would you give? Why?",
24
+ )
25
+
26
+ print(single_prediction)
27
+
28
+ metrics = my_model.evaluate(
29
+ test_set="/path/to/test_annotations.json",
30
+ predictions=predictions,
31
+ )
32
+
33
+ print(metrics)
34
+
35
+
36
+ if __name__ == "__main__":
37
+ main()
opensportslib/cli.py CHANGED
@@ -11,8 +11,8 @@ def main(argv: Optional[list[str]] = None) -> int:
11
11
  parser.add_argument("command", choices=["setup"])
12
12
  parser.add_argument("--pyg", action="store_true")
13
13
  parser.add_argument("--dali", action="store_true")
14
- parser.add_argument("--xvars", action="store_true")
15
- parser.add_argument("--qwen", action="store_true")
14
+ parser.add_argument("--vqa_xvars", action="store_true")
15
+ parser.add_argument("--vqa_qwen", action="store_true")
16
16
 
17
17
  args = parser.parse_args(argv)
18
18
 
@@ -20,8 +20,8 @@ def main(argv: Optional[list[str]] = None) -> int:
20
20
  setup(
21
21
  pyg=args.pyg,
22
22
  dali=args.dali,
23
- xvars=args.xvars,
24
- qwen=args.qwen
23
+ vqa_xvars=args.vqa_xvars,
24
+ vqa_qwen=args.vqa_qwen
25
25
  )
26
26
  return 0
27
27
 
@@ -1,14 +1,13 @@
1
- TASK: VQA
1
+ TASK: vqa
2
2
  VERSION: 2
3
3
 
4
4
  SYSTEM:
5
5
  paths:
6
- log_dir: ./logs
7
- save_dir: ./checkpoints_vqa_lora
8
- work_dir: ./checkpoints_vqa_lora
6
+ save_dir: ./checkpoints_vqa
7
+ work_dir: ${SYSTEM.paths.save_dir}
9
8
  device: cuda
10
9
  gpu:
11
- count: 4
10
+ count: 1
12
11
  id: 0
13
12
  reproducibility:
14
13
  use_seed: true
@@ -90,41 +89,11 @@ MODEL:
90
89
  strict: true
91
90
  map_location: null
92
91
  format: auto
93
- components:
94
- video_encoder:
95
- kind: encoder
96
- source:
97
- provider: opensportslib
98
- # XVARS-trained classifier weights to load into that architecture
99
- load:
100
- weights_path: /home/vorajv/X-VARS/weights/14_model.pth.tar
101
- params:
102
- # CLIP architecture and image processor to instantiate.
103
- feature_source: indexed_or_raw_clip
104
- vision_tower: openai/clip-vit-large-patch14
105
- feature_dim: 1024
106
- overrides: {}
107
- mm_projector:
108
- kind: projector
109
- source:
110
- provider: opensportslib
111
- params:
112
- input_dim: 1024
113
- overrides: {}
114
- llm_decoder:
115
- kind: decoder
116
- source:
117
- provider: opensportslib
118
- params:
119
- repo_id: /home/vorajv/X-VARS/weights/base_model_videoChatGPT
120
- overrides: {}
121
92
  topology:
122
93
  - from: video_encoder
123
94
  to: mm_projector
124
95
  - from: mm_projector
125
96
  to: llm_decoder
126
- metadata:
127
- backend: xvars_videochatgpt
128
97
 
129
98
  IO:
130
99
  inputs:
@@ -137,20 +106,15 @@ IO:
137
106
  TRAIN:
138
107
  trainer:
139
108
  type: vqa
140
-
141
109
  epochs: 3
142
-
143
110
  criterion:
144
111
  type: CrossEntropyLoss
145
-
146
112
  optimizer:
147
113
  type: AdamW
148
114
  lr: 0.0002
149
115
  weight_decay: 0.001
150
-
151
116
  scheduler:
152
117
  type: constant
153
-
154
118
  execution:
155
119
  enabled: true
156
120
  training_backend: xvars_videochatgpt_lora
@@ -159,45 +123,22 @@ TRAIN:
159
123
  acc_grad_iter: 8
160
124
  log_interval: 1
161
125
  dry_run: false
162
-
163
126
  xvars:
164
127
  feature_mode: strict_xvars
165
- # Optional: path to a separate mm_projector checkpoint.
166
- # null means use the projector already embedded in base_model_videoChatGPT.
167
128
  projection_path: null
168
-
169
129
  prompt:
170
130
  style: detailed
171
- system_prompt: You are Video-ChatGPT, a large vision-language assistant. You are able to understand the video content that the user provides, and assist the user with a variety of tasks using natural language.Follow the instructions carefully and explain your answers in detail based on the provided video.
172
131
  include_priors: true
173
132
  prediction_prior_adapter: xvars_referee
174
133
  prior_fields: [action, offence, contact, bodypart]
175
134
  video_token_len: 300
176
-
177
135
  generation:
178
136
  max_new_tokens: 128
179
137
  temperature: 0.0
180
-
181
- # Optional XFoul-only generated-answer smoke test.
182
- # During LoRA training, this runs generation on one known training sample and
183
- # checks that the answer still contains referee-domain terms and avoids known
184
- # code-like failure strings. Disable or replace these values for non-XFoul data.
185
- generated_validation:
186
- enabled: true
187
- sample_id: action_0
188
- every_steps: 25
189
- max_new_tokens: 128
190
- require_relevance: true
191
- required_terms: [foul, card, challenge, spa, dogso, advantage]
192
- forbidden_terms: [get_children, django, httpclient, "```python", "```php"]
193
-
194
- # OpenSportsLib-native VQA evaluation config. This does not correspond to
195
- # an upstream X-VARS benchmark scorer.
196
138
  eval_profile:
197
139
  metric_set: [exact_match, contains_match, token_f1, referee_semantic]
198
140
  aggregation: mean
199
141
  exclusions: []
200
-
201
142
  sft:
202
143
  max_seq_length: 480
203
144
  include_video_tokens: true
@@ -205,11 +146,7 @@ TRAIN:
205
146
  reference_mode: all
206
147
  append_eos_token: true
207
148
  gradient_checkpointing: true
208
- save_strategy: "epoch"
209
-
210
- hf:
211
- tokenizer_id: ${MODEL.components.llm_decoder.params.repo_id}
212
-
149
+ save_strategy: epoch
213
150
  lora:
214
151
  r: 16
215
152
  alpha: 32
@@ -218,26 +155,18 @@ TRAIN:
218
155
  prepare_kbit: true
219
156
  target_modules: [mm_projector, upsample_features, up_proj, down_proj, gate_proj, k_proj, q_proj, v_proj, o_proj]
220
157
  exclude_modules: '^base_lm\.model\.mm_projector$'
221
-
222
158
  quantization:
223
159
  enabled: false
224
160
  load_in_4bit: true
225
161
  bnb_4bit_quant_type: nf4
226
162
  compute_dtype: float16
227
163
  bnb_4bit_use_double_quant: true
228
-
229
- # LoRA adapter save policy:
230
- # save_adapter: true saves the LoRA adapter artifacts.
231
- # merge_and_save: false keeps the base model and adapter separate.
232
- # If you set merge_and_save: true, it would export a merged model for standalone inference.
233
164
  checkpoint:
234
165
  save_adapter: true
235
166
  merge_and_save: false
236
-
237
167
  selection:
238
168
  monitor: loss
239
169
  mode: min
240
-
241
170
  checkpoint:
242
171
  save_every: 1
243
172
  save_best: true
@@ -1,95 +1,8 @@
1
- TASK: VQA
2
- VERSION: 2
3
-
4
1
  SYSTEM:
5
2
  paths:
6
- log_dir: ./logs
7
3
  save_dir: ./checkpoints_vqa_qwen
8
- work_dir: ./checkpoints_vqa_qwen
9
- device: cuda
10
- gpu:
11
- count: 1
12
- id: 0
13
- reproducibility:
14
- use_seed: true
15
- seed: 42
16
-
17
- DATA:
18
- common:
19
- dataset_name: OSL-XFoul
20
- data_root: /home/vorajv/dataset/OSL-XFoul
21
- feature_index: /home/vorajv/dataset/OSL-XFoul/feature_index.json
22
- prediction_index: /home/vorajv/dataset/OSL-XFoul/prediction_index.json
23
- runtime:
24
- loader_backend: opencv
25
- splits:
26
- train:
27
- annotation_path: /home/vorajv/dataset/OSL-XFoul/train.json
28
- source_path: /home/vorajv/dataset/OSL-XFoul
29
- dataloader:
30
- batch_size: 1
31
- shuffle: true
32
- num_workers: 0
33
- pin_memory: false
34
- mp_context: spawn
35
- persistent_workers: false
36
- valid:
37
- annotation_path: /home/vorajv/dataset/OSL-XFoul/valid.json
38
- source_path: /home/vorajv/dataset/OSL-XFoul
39
- dataloader:
40
- batch_size: 1
41
- shuffle: false
42
- num_workers: 0
43
- pin_memory: false
44
- mp_context: spawn
45
- persistent_workers: false
46
- test:
47
- annotation_path: /home/vorajv/dataset/OSL-XFoul/test.json
48
- source_path: /home/vorajv/dataset/OSL-XFoul
49
- dataloader:
50
- batch_size: 1
51
- shuffle: false
52
- num_workers: 0
53
- pin_memory: false
54
- mp_context: spawn
55
- persistent_workers: false
56
- inputs:
57
- video:
58
- modality: video
59
- representation: raw
60
- source:
61
- format: mp4
62
- sampling:
63
- num_frames: 100
64
- input_fps: 25
65
- target_fps: 17
66
- start_frame: 63
67
- end_frame: 87
68
- transform: {}
69
- augmentations: {}
70
- params: {}
71
- question:
72
- modality: text
73
- representation: raw
74
- source:
75
- format: json
76
- sampling: {}
77
- transform: {}
78
- augmentations: {}
79
- params: {}
80
4
 
81
5
  MODEL:
82
- runtime:
83
- dtype: fp16
84
- device: auto
85
- compile: false
86
- freeze: false
87
- load:
88
- checkpoint_path: null
89
- pretrained: false
90
- strict: true
91
- map_location: null
92
- format: auto
93
6
  components:
94
7
  video_encoder:
95
8
  kind: encoder
@@ -114,83 +27,19 @@ MODEL:
114
27
  kind: decoder
115
28
  source:
116
29
  provider: huggingface
117
- # Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
30
+ # Supported models: Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
118
31
  name: Qwen/Qwen3.5-9B-Base
119
32
  params:
120
- # Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
33
+ # Supported models: Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
121
34
  repo_id: Qwen/Qwen3.5-9B-Base
122
35
  overrides: {}
123
- topology:
124
- - from: video_encoder
125
- to: mm_projector
126
- - from: mm_projector
127
- to: llm_decoder
128
36
  metadata:
129
37
  backend: qwen_xvars_infer
130
38
 
131
- IO:
132
- inputs:
133
- video: video_encoder
134
- question: llm_decoder
135
- outputs:
136
- answer_text: llm_decoder
137
- explanation_text: llm_decoder
138
-
139
39
  TRAIN:
140
- trainer:
141
- type: vqa
142
-
143
- epochs: 3
144
-
145
- criterion:
146
- type: CrossEntropyLoss
147
-
148
- optimizer:
149
- type: AdamW
150
- lr: 0.0002
151
- weight_decay: 0.001
152
-
153
- scheduler:
154
- type: constant
155
-
156
40
  execution:
157
- enabled: true
158
- training_backend: xvars_videochatgpt_lora
159
- feature_backend: xvars_clip
160
- view_sampling_policy: random_train_deterministic_eval
161
- acc_grad_iter: 8
162
- log_interval: 1
163
- dry_run: false
164
-
165
- xvars:
166
- feature_mode: strict_xvars
167
- projection_path: null
168
-
169
41
  prompt:
170
- style: detailed
171
42
  system_prompt: You are a football video assistant. Answer the VQA question using the provided video context and referee priors.
172
- include_priors: true
173
- prediction_prior_adapter: xvars_referee
174
- prior_fields: [action, offence, contact, bodypart]
175
- video_token_len: 300
176
-
177
- generation:
178
- max_new_tokens: 128
179
- temperature: 0.0
180
-
181
- eval_profile:
182
- metric_set: [exact_match, contains_match, token_f1, referee_semantic]
183
- aggregation: mean
184
- exclusions: []
185
-
186
- sft:
187
- max_seq_length: 480
188
- include_video_tokens: true
189
- disable_tqdm: false
190
- reference_mode: all
191
- append_eos_token: true
192
- gradient_checkpointing: true
193
- save_strategy: epoch
194
43
 
195
44
  hf:
196
45
  tokenizer_id: ${MODEL.components.llm_decoder.params.repo_id}
@@ -198,31 +47,3 @@ TRAIN:
198
47
  local_files_only: false
199
48
  device_map: auto
200
49
  offload_folder: ./hf_offload_qwen
201
-
202
- lora:
203
- r: 16
204
- alpha: 32
205
- dropout: 0.05
206
- bias: none
207
- prepare_kbit: true
208
- target_modules: [mm_projector, upsample_features, up_proj, down_proj, gate_proj, k_proj, q_proj, v_proj, o_proj]
209
- exclude_modules: '^base_lm\.model\.mm_projector$'
210
-
211
- quantization:
212
- enabled: false
213
- load_in_4bit: true
214
- bnb_4bit_quant_type: nf4
215
- compute_dtype: float16
216
- bnb_4bit_use_double_quant: true
217
-
218
- checkpoint:
219
- save_adapter: true
220
- merge_and_save: false
221
-
222
- selection:
223
- monitor: loss
224
- mode: min
225
-
226
- checkpoint:
227
- save_every: 1
228
- save_best: true
@@ -0,0 +1,58 @@
1
+ SYSTEM:
2
+ paths:
3
+ save_dir: ./checkpoints_vqa_lora
4
+ gpu:
5
+ count: 4
6
+
7
+ MODEL:
8
+ components:
9
+ video_encoder:
10
+ kind: encoder
11
+ source:
12
+ provider: opensportslib
13
+ # XVARS-trained classifier weights to load into that architecture
14
+ load:
15
+ weights_path: /home/vorajv/X-VARS/weights/14_model.pth.tar
16
+ params:
17
+ # CLIP architecture and image processor to instantiate.
18
+ feature_source: indexed_or_raw_clip
19
+ vision_tower: openai/clip-vit-large-patch14
20
+ feature_dim: 1024
21
+ overrides: {}
22
+ mm_projector:
23
+ kind: projector
24
+ source:
25
+ provider: opensportslib
26
+ params:
27
+ input_dim: 1024
28
+ overrides: {}
29
+ llm_decoder:
30
+ kind: decoder
31
+ source:
32
+ provider: opensportslib
33
+ params:
34
+ repo_id: /home/vorajv/X-VARS/weights/base_model_videoChatGPT
35
+ overrides: {}
36
+ metadata:
37
+ backend: xvars_videochatgpt
38
+
39
+ TRAIN:
40
+ execution:
41
+ prompt:
42
+ system_prompt: You are Video-ChatGPT, a large vision-language assistant. You are able to understand the video content that the user provides, and assist the user with a variety of tasks using natural language.Follow the instructions carefully and explain your answers in detail based on the provided video.
43
+
44
+ # Optional XFoul-only generated-answer smoke test.
45
+ # During LoRA training, this runs generation on one known training sample and
46
+ # checks that the answer still contains referee-domain terms and avoids known
47
+ # code-like failure strings. Disable or replace these values for non-XFoul data.
48
+ generated_validation:
49
+ enabled: true
50
+ sample_id: action_0
51
+ every_steps: 25
52
+ max_new_tokens: 128
53
+ require_relevance: true
54
+ required_terms: [foul, card, challenge, spa, dogso, advantage]
55
+ forbidden_terms: [get_children, django, httpclient, "```python", "```php"]
56
+
57
+ hf:
58
+ tokenizer_id: ${MODEL.components.llm_decoder.params.repo_id}
@@ -18,7 +18,7 @@ from .migrate import migrate_config
18
18
  from .runtime_adapter import maybe_namespace, namespace_to_plain_dict
19
19
 
20
20
  _YAML_SUFFIXES = {".yaml", ".yml"}
21
- _TASK_DIRS = {"classification", "localization"}
21
+ _TASK_DIRS = {"classification", "localization", "vqa"}
22
22
  _INTERPOLATION_RE = re.compile(r"\$\{([^}]+)\}")
23
23
 
24
24
 
@@ -183,12 +183,12 @@ def verify():
183
183
  else:
184
184
  print("Running on CPU")
185
185
 
186
- def setup(dali=False, pyg=False, xvars=False, qwen=False):
186
+ def setup(dali=False, pyg=False, vqa_xvars=False, vqa_qwen=False):
187
187
  install_torch()
188
188
  install_extras(dali=dali, pyg=pyg)
189
- if xvars:
189
+ if vqa_xvars:
190
190
  install_xvars_dependencies(XVARS_DEPENDENCY_PINS)
191
- if qwen:
191
+ if vqa_qwen:
192
192
  install_xvars_dependencies(QWEN_DEPENDENCY_PINS)
193
193
  verify()
194
194
 
@@ -202,9 +202,9 @@ if __name__ == "__main__":
202
202
  parser = argparse.ArgumentParser()
203
203
  parser.add_argument("--dali", action="store_true")
204
204
  parser.add_argument("--pyg", action="store_true")
205
- parser.add_argument("--xvars", action="store_true")
206
- parser.add_argument("--qwen", action="store_true")
205
+ parser.add_argument("--vqa_xvars", action="store_true")
206
+ parser.add_argument("--vqa_qwen", action="store_true")
207
207
 
208
208
  args = parser.parse_args()
209
209
 
210
- setup(dali=args.dali, pyg=args.pyg, xvars=args.xvars, qwen=args.qwen)
210
+ setup(dali=args.dali, pyg=args.pyg, vqa_xvars=args.vqa_xvars, vqa_qwen=args.vqa_qwen)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: opensportslib
3
- Version: 0.2.0.dev2
3
+ Version: 0.2.0.dev4
4
4
  Summary: OpenSportsLib is the professional library, designed for advanced video understanding in sports. It provides state-of-the-art tools for action recognition, spotting, retrieval, and captioning, making it ideal for researchers, analysts, and developers working with sports video data.
5
5
  Author: Jeet Vora
6
6
  Requires-Python: >=3.12
@@ -93,13 +93,26 @@ opensportslib setup --pyg
93
93
 
94
94
  # Optional: install for DALI support
95
95
  opensportslib setup --dali
96
- ```
96
+
97
+ # Optional: install the X-VARS-compatible VQA dependency profile
98
+ opensportslib setup --vqa_xvars
99
+
100
+ # Optional: install the Qwen-compatible VQA dependency profile
101
+ opensportslib setup --vqa_qwen
102
+ ```
97
103
  ---
98
104
 
99
105
  **Note:**
100
106
  Run `opensportslib setup` to automatically configure dependencies.
101
107
  If issues occur, manually install compatible versions of `torch`, `torchvision`, and related libraries according to your CUDA version or system compatibility.
102
108
 
109
+ For VQA, use exactly one backend-specific dependency profile:
110
+
111
+ - `--vqa_xvars` installs the X-VARS-compatible Hugging Face stack from `XVARS_DEPENDENCY_PINS`
112
+ - `--vqa_qwen` installs the Qwen-compatible Hugging Face stack from `QWEN_DEPENDENCY_PINS`
113
+
114
+ The `vqa_qwen` config supports `Qwen/Qwen2.5-7B-Instruct` and `Qwen/Qwen3.5-9B-Base`.
115
+
103
116
  ---
104
117
 
105
118
  ## Data and pretrained models
@@ -287,7 +300,7 @@ metrics_from_file = my_model.evaluate(
287
300
  from opensportslib.apis import VQAModel
288
301
 
289
302
  my_model = VQAModel(
290
- config="/path/to/vqa.yaml",
303
+ config="opensportslib/configs/vqa/qwen.yaml",
291
304
  weights=None, # optional: path or Hugging Face model ID
292
305
  )
293
306
 
@@ -300,13 +313,20 @@ single_prediction = my_model.infer(
300
313
  video_path="/path/to/video.mp4",
301
314
  question="What card would you give? Why?",
302
315
  )
303
-
304
- metrics = my_model.evaluate(
305
- test_set="/path/to/test_annotations.json",
306
- predictions=predictions,
307
- )
308
316
  ```
309
317
 
318
+ Use `opensportslib/configs/vqa/xvars.yaml` with `opensportslib setup --vqa_xvars`
319
+ for the X-VARS-compatible backend, or `opensportslib/configs/vqa/qwen.yaml` with
320
+ `opensportslib setup --vqa_qwen` for the Qwen-compatible backend. The Qwen
321
+ backend currently supports `Qwen/Qwen2.5-7B-Instruct` and
322
+ `Qwen/Qwen3.5-9B-Base`.
323
+
324
+ For X-VARS, `feature_source: indexed_or_raw_clip` prefers indexed CLIP features
325
+ when available and falls back to extracting CLIP features from raw video during
326
+ `infer()`. Pre-extracted features remain the preferred path for parity, speed,
327
+ and reproducibility. See [docs/tools/vqa.md](docs/tools/vqa.md) for the full
328
+ VQA setup workflow.
329
+
310
330
 
311
331
  ---
312
332
 
@@ -414,6 +434,12 @@ opensportslib setup --pyg
414
434
 
415
435
  # Optional: install for DALI support
416
436
  opensportslib setup --dali
437
+
438
+ # Optional: install the X-VARS-compatible VQA dependency profile
439
+ opensportslib setup --vqa_xvars
440
+
441
+ # Optional: install the Qwen-compatible VQA dependency profile
442
+ opensportslib setup --vqa_qwen
417
443
  ```
418
444
 
419
445
  ### Git workflow