wxyi088 commited on
Commit
98f6e07
·
0 Parent(s):

Weights release

Browse files
.gitattributes ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: cc-by-nc-sa-4.0
3
+ base_model: Qwen/Qwen3.5-9B
4
+ library_name: peft
5
+ pipeline_tag: video-text-to-text
6
+ language:
7
+ - en
8
+ tags:
9
+ - surgical-video
10
+ - video-question-answering
11
+ - long-video-understanding
12
+ - medical
13
+ - lora
14
+ - qwen3.5
15
+ - miccai-2026
16
+ datasets:
17
+ - orena-dkfz/heico-focus-vqa
18
+ - orena-dkfz/lapchole-focus-vqa
19
+ ---
20
+
21
+ # SurgScope
22
+
23
+ Our solution to the PROCEDURE track of the [ORena SAVE FOCUS Challenge](https://orena-focus-challenge.org/), MICCAI 2026.
24
+
25
+ LoRA adapter for [Qwen3.5-9B](https://huggingface.co/Qwen/Qwen3.5-9B) that answers questions
26
+ about foreign objects over entire surgical procedures of up to five hours.
27
+ **Code:** [github.com/wxyi057/orena-SurgScope](https://github.com/wxyi057/orena-SurgScope) ·
28
+ **Data:** [HeiCo-FOCUS-VQA](https://huggingface.co/datasets/orena-dkfz/heico-focus-vqa) ·
29
+ [LapChole-FOCUS-VQA](https://huggingface.co/datasets/orena-dkfz/lapchole-focus-vqa)
30
+
31
+ ## Method
32
+
33
+ 19 rules on the question text pick the time window. Every question gets ~46k visual tokens:
34
+ timestamp and counting questions spend them on 1536 frames, all others on 1152 sharper frames.
35
+ A timestamp answer is re-asked on ±10 min and ±2 min windows around it. The adapter averages
36
+ three fine-tunes trained with different frame-sampling regimes.
37
+
38
+ ## Results
39
+
40
+ | Pre-evaluation | In-distribution | Out-of-distribution |
41
+ |:-:|:-:|:-:|
42
+ | **0.6592** | 0.7195 | 0.5990 |
43
+
44
+ ## Usage
45
+
46
+ ```bash
47
+ git clone https://github.com/wxyi057/orena-SurgScope && cd orena-SurgScope
48
+ pip install -e .
49
+ hf auth login # after access to HeiCo-FOCUS-VQA is approved
50
+ bash scripts/make_examples.sh
51
+ surgscope infer --input examples/test --output answer.json
52
+ ```
53
+
54
+ The `surgscope` package reproduces the training-time inputs (windows, frame counts, prompt) and
55
+ merges the adapter into the base model at load time.
56
+
57
+ ## Training
58
+
59
+ PROCEDURE training split of HeiCo-FOCUS-VQA and LapChole-FOCUS-VQA (6,873 questions). LoRA
60
+ r 64 / α 128 on all linear layers, 15 epochs, ms-swift 4.3.2, three frame-sampling regimes
61
+ (768; 1536 / 1152; 2880 / 1512 frames). Full recipe: `surgscope_recipe.json`.
62
+
63
+ ## License
64
+
65
+ CC BY-NC-SA 4.0 (base model: Apache 2.0).
66
+
67
+ ## Citation
68
+
69
+ ```bibtex
70
+ @misc{surgscope2026,
71
+ title = {SurgScope: Question-Conditioned Windowing for Hour-Long Surgical Video Question Answering},
72
+ author = {Yi, Weixi and Zhang, Hanyuan and He, Runlong},
73
+ year = {2026},
74
+ note = {Solution to the PROCEDURE track, ORena SAVE FOCUS Challenge, MICCAI 2026},
75
+ url = {https://github.com/wxyi057/orena-SurgScope}
76
+ }
77
+ ```
adapter_config.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-9B",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 128,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "lora_ga_config": null,
23
+ "megatron_config": null,
24
+ "megatron_core": "megatron.core",
25
+ "modules_to_save": [],
26
+ "peft_type": "LORA",
27
+ "peft_version": "0.19.1",
28
+ "qalora_group_size": 16,
29
+ "r": 64,
30
+ "rank_pattern": {},
31
+ "revision": null,
32
+ "target_modules": "^(model\\.language_model(?=\\.).*\\.(in_proj_z|o_proj|k_proj|in_proj_qkv|out_proj|gate_proj|up_proj|q_proj|in_proj_b|v_proj|in_proj_a|down_proj)|(?!(model.visual.merger))model\\.visual(?=\\.).*\\.(linear_fc1|linear_fc2|qkv|attn.proj)|model\\.visual\\.merger(?=\\.).*\\.(linear_fc1|linear_fc2))$",
33
+ "target_parameters": null,
34
+ "task_type": "CAUSAL_LM",
35
+ "trainable_token_indices": null,
36
+ "use_bdlora": null,
37
+ "use_dora": false,
38
+ "use_qalora": false,
39
+ "use_rslora": false
40
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3ee1bd381a6af9a1d5863ea51ce3c2fc2479e860dbf75c50719870fecca4acfb
3
+ size 820346568
surgscope_recipe.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "base_model": "Qwen/Qwen3.5-9B",
3
+ "adapter": {"type": "LoRA", "rank": 64, "alpha": 128, "dropout": 0.05,
4
+ "target_modules": "all linear layers of vision encoder, patch merger and language model"},
5
+ "weights": "equal-weight fp32 mean of LoRA A and B of three adapters trained under three frame-sampling regimes (same base, seed and LoRA shape); merged into the base model in bf16 at load time",
6
+ "regimes": {
7
+ "r768": {"frames": "768 for every question", "tokens_per_frame_pair": "<= 128",
8
+ "lr": 1.4e-4, "global_batch": 64, "max_length": 57344, "checkpoint_epoch": 4},
9
+ "dense-mix": {"frames": "1536 time / counting / aggregation, 1152 otherwise", "tokens_per_frame_pair": "60 / 84",
10
+ "lr": 1e-4, "global_batch": 32, "max_length": 57344, "checkpoint_epoch": 14},
11
+ "dense-mix-2": {"frames": "2880 time / counting / aggregation, 1512 otherwise", "tokens_per_frame_pair": "32 / 64",
12
+ "lr": 1e-4, "global_batch": 32, "max_length": 65536, "checkpoint_epoch": 6}
13
+ },
14
+ "optimisation": {"optimizer": "AdamW", "weight_decay": 0.1, "adam_beta2": 0.95, "schedule": "cosine",
15
+ "warmup_ratio": 0.03, "epochs": 15, "precision": "bf16", "seed": 42, "framework": "ms-swift 4.3.2"},
16
+ "training_data": "official PROCEDURE train split of HeiCo-FOCUS-VQA and LapChole-FOCUS-VQA (6,873 question-answer pairs), no external data",
17
+ "inference": {"windows": "19 rules on the question text", "regime": "dense-mix",
18
+ "refinement": "timestamp answers re-asked on +-10 min and +-2 min windows",
19
+ "env": {"VIDEO_MAX_TOKEN_NUM": "128", "VIDEO_MIN_TOKEN_NUM": "32", "FORCE_QWENVL_VIDEO_READER": "torchvision"},
20
+ "decoding": "greedy, max 64 new tokens, thinking disabled"}
21
+ }