opensportslib 0.2.0.dev2__py3-none-any.whl → 0.2.0.dev4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- examples/quickstart/basic_vqa.py +37 -0
- opensportslib/cli.py +4 -4
- opensportslib/configs/vqa/{xvars_lora.yaml → default.yaml} +5 -76
- opensportslib/configs/vqa/qwen.yaml +2 -181
- opensportslib/configs/vqa/xvars.yaml +58 -0
- opensportslib/core/config/loader.py +1 -1
- opensportslib/setup/setup.py +6 -6
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/METADATA +34 -8
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/RECORD +18 -13
- tests/test_config_architecture.py +40 -11
- tests/test_setup_cli.py +101 -0
- tests/test_vqa_api.py +240 -0
- tests/test_vqa_qwen_xvars.py +251 -0
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/WHEEL +0 -0
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/entry_points.txt +0 -0
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/licenses/LICENSE +0 -0
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/licenses/LICENSE-COMMERCIAL +0 -0
- {opensportslib-0.2.0.dev2.dist-info → opensportslib-0.2.0.dev4.dist-info}/top_level.txt +0 -0
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from opensportslib.apis import VQAModel
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def main():
|
|
5
|
+
"""
|
|
6
|
+
Minimal VQA example.
|
|
7
|
+
Update config, question, and dataset paths before running.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
my_model = VQAModel(
|
|
11
|
+
config="examples/configs/vqa_qwen.yaml",
|
|
12
|
+
weights=None, # optional: path or Hugging Face model ID
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
predictions = my_model.infer(
|
|
16
|
+
test_set="/path/to/test_annotations.json",
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
print(predictions)
|
|
20
|
+
|
|
21
|
+
single_prediction = my_model.infer(
|
|
22
|
+
video_path="/path/to/video.mp4",
|
|
23
|
+
question="What card would you give? Why?",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
print(single_prediction)
|
|
27
|
+
|
|
28
|
+
metrics = my_model.evaluate(
|
|
29
|
+
test_set="/path/to/test_annotations.json",
|
|
30
|
+
predictions=predictions,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
print(metrics)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
if __name__ == "__main__":
|
|
37
|
+
main()
|
opensportslib/cli.py
CHANGED
|
@@ -11,8 +11,8 @@ def main(argv: Optional[list[str]] = None) -> int:
|
|
|
11
11
|
parser.add_argument("command", choices=["setup"])
|
|
12
12
|
parser.add_argument("--pyg", action="store_true")
|
|
13
13
|
parser.add_argument("--dali", action="store_true")
|
|
14
|
-
parser.add_argument("--
|
|
15
|
-
parser.add_argument("--
|
|
14
|
+
parser.add_argument("--vqa_xvars", action="store_true")
|
|
15
|
+
parser.add_argument("--vqa_qwen", action="store_true")
|
|
16
16
|
|
|
17
17
|
args = parser.parse_args(argv)
|
|
18
18
|
|
|
@@ -20,8 +20,8 @@ def main(argv: Optional[list[str]] = None) -> int:
|
|
|
20
20
|
setup(
|
|
21
21
|
pyg=args.pyg,
|
|
22
22
|
dali=args.dali,
|
|
23
|
-
|
|
24
|
-
|
|
23
|
+
vqa_xvars=args.vqa_xvars,
|
|
24
|
+
vqa_qwen=args.vqa_qwen
|
|
25
25
|
)
|
|
26
26
|
return 0
|
|
27
27
|
|
|
@@ -1,14 +1,13 @@
|
|
|
1
|
-
TASK:
|
|
1
|
+
TASK: vqa
|
|
2
2
|
VERSION: 2
|
|
3
3
|
|
|
4
4
|
SYSTEM:
|
|
5
5
|
paths:
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
work_dir: ./checkpoints_vqa_lora
|
|
6
|
+
save_dir: ./checkpoints_vqa
|
|
7
|
+
work_dir: ${SYSTEM.paths.save_dir}
|
|
9
8
|
device: cuda
|
|
10
9
|
gpu:
|
|
11
|
-
count:
|
|
10
|
+
count: 1
|
|
12
11
|
id: 0
|
|
13
12
|
reproducibility:
|
|
14
13
|
use_seed: true
|
|
@@ -90,41 +89,11 @@ MODEL:
|
|
|
90
89
|
strict: true
|
|
91
90
|
map_location: null
|
|
92
91
|
format: auto
|
|
93
|
-
components:
|
|
94
|
-
video_encoder:
|
|
95
|
-
kind: encoder
|
|
96
|
-
source:
|
|
97
|
-
provider: opensportslib
|
|
98
|
-
# XVARS-trained classifier weights to load into that architecture
|
|
99
|
-
load:
|
|
100
|
-
weights_path: /home/vorajv/X-VARS/weights/14_model.pth.tar
|
|
101
|
-
params:
|
|
102
|
-
# CLIP architecture and image processor to instantiate.
|
|
103
|
-
feature_source: indexed_or_raw_clip
|
|
104
|
-
vision_tower: openai/clip-vit-large-patch14
|
|
105
|
-
feature_dim: 1024
|
|
106
|
-
overrides: {}
|
|
107
|
-
mm_projector:
|
|
108
|
-
kind: projector
|
|
109
|
-
source:
|
|
110
|
-
provider: opensportslib
|
|
111
|
-
params:
|
|
112
|
-
input_dim: 1024
|
|
113
|
-
overrides: {}
|
|
114
|
-
llm_decoder:
|
|
115
|
-
kind: decoder
|
|
116
|
-
source:
|
|
117
|
-
provider: opensportslib
|
|
118
|
-
params:
|
|
119
|
-
repo_id: /home/vorajv/X-VARS/weights/base_model_videoChatGPT
|
|
120
|
-
overrides: {}
|
|
121
92
|
topology:
|
|
122
93
|
- from: video_encoder
|
|
123
94
|
to: mm_projector
|
|
124
95
|
- from: mm_projector
|
|
125
96
|
to: llm_decoder
|
|
126
|
-
metadata:
|
|
127
|
-
backend: xvars_videochatgpt
|
|
128
97
|
|
|
129
98
|
IO:
|
|
130
99
|
inputs:
|
|
@@ -137,20 +106,15 @@ IO:
|
|
|
137
106
|
TRAIN:
|
|
138
107
|
trainer:
|
|
139
108
|
type: vqa
|
|
140
|
-
|
|
141
109
|
epochs: 3
|
|
142
|
-
|
|
143
110
|
criterion:
|
|
144
111
|
type: CrossEntropyLoss
|
|
145
|
-
|
|
146
112
|
optimizer:
|
|
147
113
|
type: AdamW
|
|
148
114
|
lr: 0.0002
|
|
149
115
|
weight_decay: 0.001
|
|
150
|
-
|
|
151
116
|
scheduler:
|
|
152
117
|
type: constant
|
|
153
|
-
|
|
154
118
|
execution:
|
|
155
119
|
enabled: true
|
|
156
120
|
training_backend: xvars_videochatgpt_lora
|
|
@@ -159,45 +123,22 @@ TRAIN:
|
|
|
159
123
|
acc_grad_iter: 8
|
|
160
124
|
log_interval: 1
|
|
161
125
|
dry_run: false
|
|
162
|
-
|
|
163
126
|
xvars:
|
|
164
127
|
feature_mode: strict_xvars
|
|
165
|
-
# Optional: path to a separate mm_projector checkpoint.
|
|
166
|
-
# null means use the projector already embedded in base_model_videoChatGPT.
|
|
167
128
|
projection_path: null
|
|
168
|
-
|
|
169
129
|
prompt:
|
|
170
130
|
style: detailed
|
|
171
|
-
system_prompt: You are Video-ChatGPT, a large vision-language assistant. You are able to understand the video content that the user provides, and assist the user with a variety of tasks using natural language.Follow the instructions carefully and explain your answers in detail based on the provided video.
|
|
172
131
|
include_priors: true
|
|
173
132
|
prediction_prior_adapter: xvars_referee
|
|
174
133
|
prior_fields: [action, offence, contact, bodypart]
|
|
175
134
|
video_token_len: 300
|
|
176
|
-
|
|
177
135
|
generation:
|
|
178
136
|
max_new_tokens: 128
|
|
179
137
|
temperature: 0.0
|
|
180
|
-
|
|
181
|
-
# Optional XFoul-only generated-answer smoke test.
|
|
182
|
-
# During LoRA training, this runs generation on one known training sample and
|
|
183
|
-
# checks that the answer still contains referee-domain terms and avoids known
|
|
184
|
-
# code-like failure strings. Disable or replace these values for non-XFoul data.
|
|
185
|
-
generated_validation:
|
|
186
|
-
enabled: true
|
|
187
|
-
sample_id: action_0
|
|
188
|
-
every_steps: 25
|
|
189
|
-
max_new_tokens: 128
|
|
190
|
-
require_relevance: true
|
|
191
|
-
required_terms: [foul, card, challenge, spa, dogso, advantage]
|
|
192
|
-
forbidden_terms: [get_children, django, httpclient, "```python", "```php"]
|
|
193
|
-
|
|
194
|
-
# OpenSportsLib-native VQA evaluation config. This does not correspond to
|
|
195
|
-
# an upstream X-VARS benchmark scorer.
|
|
196
138
|
eval_profile:
|
|
197
139
|
metric_set: [exact_match, contains_match, token_f1, referee_semantic]
|
|
198
140
|
aggregation: mean
|
|
199
141
|
exclusions: []
|
|
200
|
-
|
|
201
142
|
sft:
|
|
202
143
|
max_seq_length: 480
|
|
203
144
|
include_video_tokens: true
|
|
@@ -205,11 +146,7 @@ TRAIN:
|
|
|
205
146
|
reference_mode: all
|
|
206
147
|
append_eos_token: true
|
|
207
148
|
gradient_checkpointing: true
|
|
208
|
-
save_strategy:
|
|
209
|
-
|
|
210
|
-
hf:
|
|
211
|
-
tokenizer_id: ${MODEL.components.llm_decoder.params.repo_id}
|
|
212
|
-
|
|
149
|
+
save_strategy: epoch
|
|
213
150
|
lora:
|
|
214
151
|
r: 16
|
|
215
152
|
alpha: 32
|
|
@@ -218,26 +155,18 @@ TRAIN:
|
|
|
218
155
|
prepare_kbit: true
|
|
219
156
|
target_modules: [mm_projector, upsample_features, up_proj, down_proj, gate_proj, k_proj, q_proj, v_proj, o_proj]
|
|
220
157
|
exclude_modules: '^base_lm\.model\.mm_projector$'
|
|
221
|
-
|
|
222
158
|
quantization:
|
|
223
159
|
enabled: false
|
|
224
160
|
load_in_4bit: true
|
|
225
161
|
bnb_4bit_quant_type: nf4
|
|
226
162
|
compute_dtype: float16
|
|
227
163
|
bnb_4bit_use_double_quant: true
|
|
228
|
-
|
|
229
|
-
# LoRA adapter save policy:
|
|
230
|
-
# save_adapter: true saves the LoRA adapter artifacts.
|
|
231
|
-
# merge_and_save: false keeps the base model and adapter separate.
|
|
232
|
-
# If you set merge_and_save: true, it would export a merged model for standalone inference.
|
|
233
164
|
checkpoint:
|
|
234
165
|
save_adapter: true
|
|
235
166
|
merge_and_save: false
|
|
236
|
-
|
|
237
167
|
selection:
|
|
238
168
|
monitor: loss
|
|
239
169
|
mode: min
|
|
240
|
-
|
|
241
170
|
checkpoint:
|
|
242
171
|
save_every: 1
|
|
243
172
|
save_best: true
|
|
@@ -1,95 +1,8 @@
|
|
|
1
|
-
TASK: VQA
|
|
2
|
-
VERSION: 2
|
|
3
|
-
|
|
4
1
|
SYSTEM:
|
|
5
2
|
paths:
|
|
6
|
-
log_dir: ./logs
|
|
7
3
|
save_dir: ./checkpoints_vqa_qwen
|
|
8
|
-
work_dir: ./checkpoints_vqa_qwen
|
|
9
|
-
device: cuda
|
|
10
|
-
gpu:
|
|
11
|
-
count: 1
|
|
12
|
-
id: 0
|
|
13
|
-
reproducibility:
|
|
14
|
-
use_seed: true
|
|
15
|
-
seed: 42
|
|
16
|
-
|
|
17
|
-
DATA:
|
|
18
|
-
common:
|
|
19
|
-
dataset_name: OSL-XFoul
|
|
20
|
-
data_root: /home/vorajv/dataset/OSL-XFoul
|
|
21
|
-
feature_index: /home/vorajv/dataset/OSL-XFoul/feature_index.json
|
|
22
|
-
prediction_index: /home/vorajv/dataset/OSL-XFoul/prediction_index.json
|
|
23
|
-
runtime:
|
|
24
|
-
loader_backend: opencv
|
|
25
|
-
splits:
|
|
26
|
-
train:
|
|
27
|
-
annotation_path: /home/vorajv/dataset/OSL-XFoul/train.json
|
|
28
|
-
source_path: /home/vorajv/dataset/OSL-XFoul
|
|
29
|
-
dataloader:
|
|
30
|
-
batch_size: 1
|
|
31
|
-
shuffle: true
|
|
32
|
-
num_workers: 0
|
|
33
|
-
pin_memory: false
|
|
34
|
-
mp_context: spawn
|
|
35
|
-
persistent_workers: false
|
|
36
|
-
valid:
|
|
37
|
-
annotation_path: /home/vorajv/dataset/OSL-XFoul/valid.json
|
|
38
|
-
source_path: /home/vorajv/dataset/OSL-XFoul
|
|
39
|
-
dataloader:
|
|
40
|
-
batch_size: 1
|
|
41
|
-
shuffle: false
|
|
42
|
-
num_workers: 0
|
|
43
|
-
pin_memory: false
|
|
44
|
-
mp_context: spawn
|
|
45
|
-
persistent_workers: false
|
|
46
|
-
test:
|
|
47
|
-
annotation_path: /home/vorajv/dataset/OSL-XFoul/test.json
|
|
48
|
-
source_path: /home/vorajv/dataset/OSL-XFoul
|
|
49
|
-
dataloader:
|
|
50
|
-
batch_size: 1
|
|
51
|
-
shuffle: false
|
|
52
|
-
num_workers: 0
|
|
53
|
-
pin_memory: false
|
|
54
|
-
mp_context: spawn
|
|
55
|
-
persistent_workers: false
|
|
56
|
-
inputs:
|
|
57
|
-
video:
|
|
58
|
-
modality: video
|
|
59
|
-
representation: raw
|
|
60
|
-
source:
|
|
61
|
-
format: mp4
|
|
62
|
-
sampling:
|
|
63
|
-
num_frames: 100
|
|
64
|
-
input_fps: 25
|
|
65
|
-
target_fps: 17
|
|
66
|
-
start_frame: 63
|
|
67
|
-
end_frame: 87
|
|
68
|
-
transform: {}
|
|
69
|
-
augmentations: {}
|
|
70
|
-
params: {}
|
|
71
|
-
question:
|
|
72
|
-
modality: text
|
|
73
|
-
representation: raw
|
|
74
|
-
source:
|
|
75
|
-
format: json
|
|
76
|
-
sampling: {}
|
|
77
|
-
transform: {}
|
|
78
|
-
augmentations: {}
|
|
79
|
-
params: {}
|
|
80
4
|
|
|
81
5
|
MODEL:
|
|
82
|
-
runtime:
|
|
83
|
-
dtype: fp16
|
|
84
|
-
device: auto
|
|
85
|
-
compile: false
|
|
86
|
-
freeze: false
|
|
87
|
-
load:
|
|
88
|
-
checkpoint_path: null
|
|
89
|
-
pretrained: false
|
|
90
|
-
strict: true
|
|
91
|
-
map_location: null
|
|
92
|
-
format: auto
|
|
93
6
|
components:
|
|
94
7
|
video_encoder:
|
|
95
8
|
kind: encoder
|
|
@@ -114,83 +27,19 @@ MODEL:
|
|
|
114
27
|
kind: decoder
|
|
115
28
|
source:
|
|
116
29
|
provider: huggingface
|
|
117
|
-
# Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
|
|
30
|
+
# Supported models: Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
|
|
118
31
|
name: Qwen/Qwen3.5-9B-Base
|
|
119
32
|
params:
|
|
120
|
-
# Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
|
|
33
|
+
# Supported models: Qwen/Qwen2.5-7B-Instruct or Qwen/Qwen3.5-9B-Base
|
|
121
34
|
repo_id: Qwen/Qwen3.5-9B-Base
|
|
122
35
|
overrides: {}
|
|
123
|
-
topology:
|
|
124
|
-
- from: video_encoder
|
|
125
|
-
to: mm_projector
|
|
126
|
-
- from: mm_projector
|
|
127
|
-
to: llm_decoder
|
|
128
36
|
metadata:
|
|
129
37
|
backend: qwen_xvars_infer
|
|
130
38
|
|
|
131
|
-
IO:
|
|
132
|
-
inputs:
|
|
133
|
-
video: video_encoder
|
|
134
|
-
question: llm_decoder
|
|
135
|
-
outputs:
|
|
136
|
-
answer_text: llm_decoder
|
|
137
|
-
explanation_text: llm_decoder
|
|
138
|
-
|
|
139
39
|
TRAIN:
|
|
140
|
-
trainer:
|
|
141
|
-
type: vqa
|
|
142
|
-
|
|
143
|
-
epochs: 3
|
|
144
|
-
|
|
145
|
-
criterion:
|
|
146
|
-
type: CrossEntropyLoss
|
|
147
|
-
|
|
148
|
-
optimizer:
|
|
149
|
-
type: AdamW
|
|
150
|
-
lr: 0.0002
|
|
151
|
-
weight_decay: 0.001
|
|
152
|
-
|
|
153
|
-
scheduler:
|
|
154
|
-
type: constant
|
|
155
|
-
|
|
156
40
|
execution:
|
|
157
|
-
enabled: true
|
|
158
|
-
training_backend: xvars_videochatgpt_lora
|
|
159
|
-
feature_backend: xvars_clip
|
|
160
|
-
view_sampling_policy: random_train_deterministic_eval
|
|
161
|
-
acc_grad_iter: 8
|
|
162
|
-
log_interval: 1
|
|
163
|
-
dry_run: false
|
|
164
|
-
|
|
165
|
-
xvars:
|
|
166
|
-
feature_mode: strict_xvars
|
|
167
|
-
projection_path: null
|
|
168
|
-
|
|
169
41
|
prompt:
|
|
170
|
-
style: detailed
|
|
171
42
|
system_prompt: You are a football video assistant. Answer the VQA question using the provided video context and referee priors.
|
|
172
|
-
include_priors: true
|
|
173
|
-
prediction_prior_adapter: xvars_referee
|
|
174
|
-
prior_fields: [action, offence, contact, bodypart]
|
|
175
|
-
video_token_len: 300
|
|
176
|
-
|
|
177
|
-
generation:
|
|
178
|
-
max_new_tokens: 128
|
|
179
|
-
temperature: 0.0
|
|
180
|
-
|
|
181
|
-
eval_profile:
|
|
182
|
-
metric_set: [exact_match, contains_match, token_f1, referee_semantic]
|
|
183
|
-
aggregation: mean
|
|
184
|
-
exclusions: []
|
|
185
|
-
|
|
186
|
-
sft:
|
|
187
|
-
max_seq_length: 480
|
|
188
|
-
include_video_tokens: true
|
|
189
|
-
disable_tqdm: false
|
|
190
|
-
reference_mode: all
|
|
191
|
-
append_eos_token: true
|
|
192
|
-
gradient_checkpointing: true
|
|
193
|
-
save_strategy: epoch
|
|
194
43
|
|
|
195
44
|
hf:
|
|
196
45
|
tokenizer_id: ${MODEL.components.llm_decoder.params.repo_id}
|
|
@@ -198,31 +47,3 @@ TRAIN:
|
|
|
198
47
|
local_files_only: false
|
|
199
48
|
device_map: auto
|
|
200
49
|
offload_folder: ./hf_offload_qwen
|
|
201
|
-
|
|
202
|
-
lora:
|
|
203
|
-
r: 16
|
|
204
|
-
alpha: 32
|
|
205
|
-
dropout: 0.05
|
|
206
|
-
bias: none
|
|
207
|
-
prepare_kbit: true
|
|
208
|
-
target_modules: [mm_projector, upsample_features, up_proj, down_proj, gate_proj, k_proj, q_proj, v_proj, o_proj]
|
|
209
|
-
exclude_modules: '^base_lm\.model\.mm_projector$'
|
|
210
|
-
|
|
211
|
-
quantization:
|
|
212
|
-
enabled: false
|
|
213
|
-
load_in_4bit: true
|
|
214
|
-
bnb_4bit_quant_type: nf4
|
|
215
|
-
compute_dtype: float16
|
|
216
|
-
bnb_4bit_use_double_quant: true
|
|
217
|
-
|
|
218
|
-
checkpoint:
|
|
219
|
-
save_adapter: true
|
|
220
|
-
merge_and_save: false
|
|
221
|
-
|
|
222
|
-
selection:
|
|
223
|
-
monitor: loss
|
|
224
|
-
mode: min
|
|
225
|
-
|
|
226
|
-
checkpoint:
|
|
227
|
-
save_every: 1
|
|
228
|
-
save_best: true
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
SYSTEM:
|
|
2
|
+
paths:
|
|
3
|
+
save_dir: ./checkpoints_vqa_lora
|
|
4
|
+
gpu:
|
|
5
|
+
count: 4
|
|
6
|
+
|
|
7
|
+
MODEL:
|
|
8
|
+
components:
|
|
9
|
+
video_encoder:
|
|
10
|
+
kind: encoder
|
|
11
|
+
source:
|
|
12
|
+
provider: opensportslib
|
|
13
|
+
# XVARS-trained classifier weights to load into that architecture
|
|
14
|
+
load:
|
|
15
|
+
weights_path: /home/vorajv/X-VARS/weights/14_model.pth.tar
|
|
16
|
+
params:
|
|
17
|
+
# CLIP architecture and image processor to instantiate.
|
|
18
|
+
feature_source: indexed_or_raw_clip
|
|
19
|
+
vision_tower: openai/clip-vit-large-patch14
|
|
20
|
+
feature_dim: 1024
|
|
21
|
+
overrides: {}
|
|
22
|
+
mm_projector:
|
|
23
|
+
kind: projector
|
|
24
|
+
source:
|
|
25
|
+
provider: opensportslib
|
|
26
|
+
params:
|
|
27
|
+
input_dim: 1024
|
|
28
|
+
overrides: {}
|
|
29
|
+
llm_decoder:
|
|
30
|
+
kind: decoder
|
|
31
|
+
source:
|
|
32
|
+
provider: opensportslib
|
|
33
|
+
params:
|
|
34
|
+
repo_id: /home/vorajv/X-VARS/weights/base_model_videoChatGPT
|
|
35
|
+
overrides: {}
|
|
36
|
+
metadata:
|
|
37
|
+
backend: xvars_videochatgpt
|
|
38
|
+
|
|
39
|
+
TRAIN:
|
|
40
|
+
execution:
|
|
41
|
+
prompt:
|
|
42
|
+
system_prompt: You are Video-ChatGPT, a large vision-language assistant. You are able to understand the video content that the user provides, and assist the user with a variety of tasks using natural language.Follow the instructions carefully and explain your answers in detail based on the provided video.
|
|
43
|
+
|
|
44
|
+
# Optional XFoul-only generated-answer smoke test.
|
|
45
|
+
# During LoRA training, this runs generation on one known training sample and
|
|
46
|
+
# checks that the answer still contains referee-domain terms and avoids known
|
|
47
|
+
# code-like failure strings. Disable or replace these values for non-XFoul data.
|
|
48
|
+
generated_validation:
|
|
49
|
+
enabled: true
|
|
50
|
+
sample_id: action_0
|
|
51
|
+
every_steps: 25
|
|
52
|
+
max_new_tokens: 128
|
|
53
|
+
require_relevance: true
|
|
54
|
+
required_terms: [foul, card, challenge, spa, dogso, advantage]
|
|
55
|
+
forbidden_terms: [get_children, django, httpclient, "```python", "```php"]
|
|
56
|
+
|
|
57
|
+
hf:
|
|
58
|
+
tokenizer_id: ${MODEL.components.llm_decoder.params.repo_id}
|
|
@@ -18,7 +18,7 @@ from .migrate import migrate_config
|
|
|
18
18
|
from .runtime_adapter import maybe_namespace, namespace_to_plain_dict
|
|
19
19
|
|
|
20
20
|
_YAML_SUFFIXES = {".yaml", ".yml"}
|
|
21
|
-
_TASK_DIRS = {"classification", "localization"}
|
|
21
|
+
_TASK_DIRS = {"classification", "localization", "vqa"}
|
|
22
22
|
_INTERPOLATION_RE = re.compile(r"\$\{([^}]+)\}")
|
|
23
23
|
|
|
24
24
|
|
opensportslib/setup/setup.py
CHANGED
|
@@ -183,12 +183,12 @@ def verify():
|
|
|
183
183
|
else:
|
|
184
184
|
print("Running on CPU")
|
|
185
185
|
|
|
186
|
-
def setup(dali=False, pyg=False,
|
|
186
|
+
def setup(dali=False, pyg=False, vqa_xvars=False, vqa_qwen=False):
|
|
187
187
|
install_torch()
|
|
188
188
|
install_extras(dali=dali, pyg=pyg)
|
|
189
|
-
if
|
|
189
|
+
if vqa_xvars:
|
|
190
190
|
install_xvars_dependencies(XVARS_DEPENDENCY_PINS)
|
|
191
|
-
if
|
|
191
|
+
if vqa_qwen:
|
|
192
192
|
install_xvars_dependencies(QWEN_DEPENDENCY_PINS)
|
|
193
193
|
verify()
|
|
194
194
|
|
|
@@ -202,9 +202,9 @@ if __name__ == "__main__":
|
|
|
202
202
|
parser = argparse.ArgumentParser()
|
|
203
203
|
parser.add_argument("--dali", action="store_true")
|
|
204
204
|
parser.add_argument("--pyg", action="store_true")
|
|
205
|
-
parser.add_argument("--
|
|
206
|
-
parser.add_argument("--
|
|
205
|
+
parser.add_argument("--vqa_xvars", action="store_true")
|
|
206
|
+
parser.add_argument("--vqa_qwen", action="store_true")
|
|
207
207
|
|
|
208
208
|
args = parser.parse_args()
|
|
209
209
|
|
|
210
|
-
setup(dali=args.dali, pyg=args.pyg,
|
|
210
|
+
setup(dali=args.dali, pyg=args.pyg, vqa_xvars=args.vqa_xvars, vqa_qwen=args.vqa_qwen)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensportslib
|
|
3
|
-
Version: 0.2.0.
|
|
3
|
+
Version: 0.2.0.dev4
|
|
4
4
|
Summary: OpenSportsLib is the professional library, designed for advanced video understanding in sports. It provides state-of-the-art tools for action recognition, spotting, retrieval, and captioning, making it ideal for researchers, analysts, and developers working with sports video data.
|
|
5
5
|
Author: Jeet Vora
|
|
6
6
|
Requires-Python: >=3.12
|
|
@@ -93,13 +93,26 @@ opensportslib setup --pyg
|
|
|
93
93
|
|
|
94
94
|
# Optional: install for DALI support
|
|
95
95
|
opensportslib setup --dali
|
|
96
|
-
|
|
96
|
+
|
|
97
|
+
# Optional: install the X-VARS-compatible VQA dependency profile
|
|
98
|
+
opensportslib setup --vqa_xvars
|
|
99
|
+
|
|
100
|
+
# Optional: install the Qwen-compatible VQA dependency profile
|
|
101
|
+
opensportslib setup --vqa_qwen
|
|
102
|
+
```
|
|
97
103
|
---
|
|
98
104
|
|
|
99
105
|
**Note:**
|
|
100
106
|
Run `opensportslib setup` to automatically configure dependencies.
|
|
101
107
|
If issues occur, manually install compatible versions of `torch`, `torchvision`, and related libraries according to your CUDA version or system compatibility.
|
|
102
108
|
|
|
109
|
+
For VQA, use exactly one backend-specific dependency profile:
|
|
110
|
+
|
|
111
|
+
- `--vqa_xvars` installs the X-VARS-compatible Hugging Face stack from `XVARS_DEPENDENCY_PINS`
|
|
112
|
+
- `--vqa_qwen` installs the Qwen-compatible Hugging Face stack from `QWEN_DEPENDENCY_PINS`
|
|
113
|
+
|
|
114
|
+
The `vqa_qwen` config supports `Qwen/Qwen2.5-7B-Instruct` and `Qwen/Qwen3.5-9B-Base`.
|
|
115
|
+
|
|
103
116
|
---
|
|
104
117
|
|
|
105
118
|
## Data and pretrained models
|
|
@@ -287,7 +300,7 @@ metrics_from_file = my_model.evaluate(
|
|
|
287
300
|
from opensportslib.apis import VQAModel
|
|
288
301
|
|
|
289
302
|
my_model = VQAModel(
|
|
290
|
-
config="/
|
|
303
|
+
config="opensportslib/configs/vqa/qwen.yaml",
|
|
291
304
|
weights=None, # optional: path or Hugging Face model ID
|
|
292
305
|
)
|
|
293
306
|
|
|
@@ -300,13 +313,20 @@ single_prediction = my_model.infer(
|
|
|
300
313
|
video_path="/path/to/video.mp4",
|
|
301
314
|
question="What card would you give? Why?",
|
|
302
315
|
)
|
|
303
|
-
|
|
304
|
-
metrics = my_model.evaluate(
|
|
305
|
-
test_set="/path/to/test_annotations.json",
|
|
306
|
-
predictions=predictions,
|
|
307
|
-
)
|
|
308
316
|
```
|
|
309
317
|
|
|
318
|
+
Use `opensportslib/configs/vqa/xvars.yaml` with `opensportslib setup --vqa_xvars`
|
|
319
|
+
for the X-VARS-compatible backend, or `opensportslib/configs/vqa/qwen.yaml` with
|
|
320
|
+
`opensportslib setup --vqa_qwen` for the Qwen-compatible backend. The Qwen
|
|
321
|
+
backend currently supports `Qwen/Qwen2.5-7B-Instruct` and
|
|
322
|
+
`Qwen/Qwen3.5-9B-Base`.
|
|
323
|
+
|
|
324
|
+
For X-VARS, `feature_source: indexed_or_raw_clip` prefers indexed CLIP features
|
|
325
|
+
when available and falls back to extracting CLIP features from raw video during
|
|
326
|
+
`infer()`. Pre-extracted features remain the preferred path for parity, speed,
|
|
327
|
+
and reproducibility. See [docs/tools/vqa.md](docs/tools/vqa.md) for the full
|
|
328
|
+
VQA setup workflow.
|
|
329
|
+
|
|
310
330
|
|
|
311
331
|
---
|
|
312
332
|
|
|
@@ -414,6 +434,12 @@ opensportslib setup --pyg
|
|
|
414
434
|
|
|
415
435
|
# Optional: install for DALI support
|
|
416
436
|
opensportslib setup --dali
|
|
437
|
+
|
|
438
|
+
# Optional: install the X-VARS-compatible VQA dependency profile
|
|
439
|
+
opensportslib setup --vqa_xvars
|
|
440
|
+
|
|
441
|
+
# Optional: install the Qwen-compatible VQA dependency profile
|
|
442
|
+
opensportslib setup --vqa_qwen
|
|
417
443
|
```
|
|
418
444
|
|
|
419
445
|
### Git workflow
|