{ "created_at": "2026-07-19T06:55:19", "global_optimizer_step": 7344, "git_commit": "e8ffda2aed348adb6b48591f2219b69582c4fc44", "package_versions": { "torch": "2.11.0+cu128", "transformers": "4.46.2", "accelerate": "0.33.0", "bitsandbytes": "0.49.2", "flash-attn": "2.6.3" }, "config_snapshot": { "experiment": { "name": "wad_internvl_pretrain_traj_cls", "version": "1.0", "description": "InternVL2-2B trajectory pretrain with cls_add fusion", "tags": [ "internvl2", "pretrain", "trajectory", "cls_add" ] }, "model": { "architecture": "internvl", "name": "OpenGVLab/InternVL2_5-2B", "trust_remote_code": true, "multimodal_max_length": 8192, "qformer": { "enabled": true, "source_model": "Salesforce/instructblip-flan-t5-xl", "cache_dir": "./qformer_cache", "num_query_tokens": 32, "freeze_qformer": true, "freeze_mlp1": true, "prompt_aware": true, "max_text_length": 128, "train_lora_llm": false, "bridge_mode": "prompt_aware_preproj_mlp1" }, "lora": { "enabled": false, "r": 16, "alpha": 32, "dropout": 0.05, "target_modules": [ "wqkv", "wo", "w1", "w2", "w3" ], "bias": "none", "task_type": "CAUSAL_LM" }, "quantization": { "enabled": true, "bits": 4, "type": "nf4", "double_quant": true, "compute_dtype": "bfloat16" }, "vision": { "freeze_encoder": true, "min_pixels": 256, "max_pixels": 1280, "image_size": [ 448, 448 ] } }, "trajectory": { "enabled": true, "fusion_mode": "cls_add", "source_file": "json", "num_objects": 6, "d_cat": 32, "d_numeric_hidden": 64, "d_traj": 384, "num_heads": 4, "num_layers": 4, "ffn_dim": 768, "dropout": 0.1 }, "pretrain": { "question_train_file": "./json/question_train_split_train.jsonl", "question_val_file": "./json/question_train_split_val.jsonl", "question_test_file": "./json/question_train_split_test.jsonl", "frame_index_file": "./wad_dataset/frame_index.pkl", "train_split_ratio": 0.8, "val_split_ratio": 0.1, "test_split_ratio": 0.1, "val_split_seed": 42, "movement_enabled": true }, "data": { "name": "minhdang0901/WAD_Images_All_Size", "num_frames": 1, "seed": 42, "response_format": "direct_text" }, "training": { "output_dir": "./outputs/pretrain_traj_cls", "num_epochs": 50, "batch_size": 24, "gradient_accumulation_steps": 1, "use_accelerate": true, "learning_rate": "1e-5", "trajectory_learning_rate": "3e-4", "bridge_learning_rate": "5e-5", "warmup_ratio": 0.1, "warmup_min_steps": 20, "warmup_max_steps": 100, "weight_decay": 0.02, "loss_mode": "label_smoothing", "label_smoothing": 0.1, "early_stopping_patience": 8, "early_stopping_min_delta": 0.001, "early_stopping_burn_in_epochs": 1, "restore_best_checkpoint": true, "bf16": true, "fp16": false, "gradient_checkpointing": false, "optimizer": "adamw", "lr_scheduler": "cosine", "max_grad_norm": 1.0, "logging_steps": 10, "log_token_stats": false, "token_log_batches": 0, "log_trainable_parameter_names": false, "profile_steps": 0, "gradient_debug_steps": 3, "gradient_health_steps": 0, "batch_debug_steps": 0, "memory_debug_steps": 0 }, "hardware": { "device": "cuda", "num_workers": 4, "pin_memory": true, "persistent_workers": true, "prefetch_factor": 2, "cuda_alloc_conf": "expandable_segments:True" } } }