orena-SurgScope / surgscope_recipe.json
wxyi088's picture
Weights release
98f6e07
Raw History Blame Contribute Delete
1.81 kB
{
"base_model": "Qwen/Qwen3.5-9B",
"adapter": {"type": "LoRA", "rank": 64, "alpha": 128, "dropout": 0.05,
"target_modules": "all linear layers of vision encoder, patch merger and language model"},
"weights": "equal-weight fp32 mean of LoRA A and B of three adapters trained under three frame-sampling regimes (same base, seed and LoRA shape); merged into the base model in bf16 at load time",
"regimes": {
"r768": {"frames": "768 for every question", "tokens_per_frame_pair": "<= 128",
"lr": 1.4e-4, "global_batch": 64, "max_length": 57344, "checkpoint_epoch": 4},
"dense-mix": {"frames": "1536 time / counting / aggregation, 1152 otherwise", "tokens_per_frame_pair": "60 / 84",
"lr": 1e-4, "global_batch": 32, "max_length": 57344, "checkpoint_epoch": 14},
"dense-mix-2": {"frames": "2880 time / counting / aggregation, 1512 otherwise", "tokens_per_frame_pair": "32 / 64",
"lr": 1e-4, "global_batch": 32, "max_length": 65536, "checkpoint_epoch": 6}
},
"optimisation": {"optimizer": "AdamW", "weight_decay": 0.1, "adam_beta2": 0.95, "schedule": "cosine",
"warmup_ratio": 0.03, "epochs": 15, "precision": "bf16", "seed": 42, "framework": "ms-swift 4.3.2"},
"training_data": "official PROCEDURE train split of HeiCo-FOCUS-VQA and LapChole-FOCUS-VQA (6,873 question-answer pairs), no external data",
"inference": {"windows": "19 rules on the question text", "regime": "dense-mix",
"refinement": "timestamp answers re-asked on +-10 min and +-2 min windows",
"env": {"VIDEO_MAX_TOKEN_NUM": "128", "VIDEO_MIN_TOKEN_NUM": "32", "FORCE_QWENVL_VIDEO_READER": "torchvision"},
"decoding": "greedy, max 64 new tokens, thinking disabled"}
}