{ "base_model": "Qwen/Qwen3.5-9B", "adapter": {"type": "LoRA", "rank": 64, "alpha": 128, "dropout": 0.05, "target_modules": "all linear layers of vision encoder, patch merger and language model"}, "weights": "equal-weight fp32 mean of LoRA A and B of three adapters trained under three frame-sampling regimes (same base, seed and LoRA shape); merged into the base model in bf16 at load time", "regimes": { "r768": {"frames": "768 for every question", "tokens_per_frame_pair": "<= 128", "lr": 1.4e-4, "global_batch": 64, "max_length": 57344, "checkpoint_epoch": 4}, "dense-mix": {"frames": "1536 time / counting / aggregation, 1152 otherwise", "tokens_per_frame_pair": "60 / 84", "lr": 1e-4, "global_batch": 32, "max_length": 57344, "checkpoint_epoch": 14}, "dense-mix-2": {"frames": "2880 time / counting / aggregation, 1512 otherwise", "tokens_per_frame_pair": "32 / 64", "lr": 1e-4, "global_batch": 32, "max_length": 65536, "checkpoint_epoch": 6} }, "optimisation": {"optimizer": "AdamW", "weight_decay": 0.1, "adam_beta2": 0.95, "schedule": "cosine", "warmup_ratio": 0.03, "epochs": 15, "precision": "bf16", "seed": 42, "framework": "ms-swift 4.3.2"}, "training_data": "official PROCEDURE train split of HeiCo-FOCUS-VQA and LapChole-FOCUS-VQA (6,873 question-answer pairs), no external data", "inference": {"windows": "19 rules on the question text", "regime": "dense-mix", "refinement": "timestamp answers re-asked on +-10 min and +-2 min windows", "env": {"VIDEO_MAX_TOKEN_NUM": "128", "VIDEO_MIN_TOKEN_NUM": "32", "FORCE_QWENVL_VIDEO_READER": "torchvision"}, "decoding": "greedy, max 64 new tokens, thinking disabled"} }