{ "artifact": "qwen3-droped", "config": { "adam_beta1": 0.9, "adam_beta2": 0.95, "adam_eps": 1e-08, "base_model": "Qwen/Qwen3-0.6B", "base_revision": "c1899de289a04d12100db370d81485cdf75e47ca", "dataset_name": "sample-10BT", "dataset_path": "HuggingFaceFW/fineweb-edu", "dataset_split": "train", "dtype": "bf16", "eval_slice_rule": "stream HuggingFaceFW/fineweb-edu sample-10BT train in provider order; concatenate non-empty document text with one EOS token after each document; use the first eval_tokens tokens as the held-out eval slice; training starts immediately after that prefix", "eval_tokens": 5000000, "global_batch_tokens": 524288, "grad_clip": 1.0, "learning_rate": 0.001, "micro_batch_size": 8, "min_lr_fraction": 0.1, "seed": 0, "train_context": 2048, "train_tokens": 1000000000, "warmup_fraction": 0.02, "weight_decay": 0.1 }, "created_at": "2026-07-25T17:26:13Z", "grad_accumulation_steps": 32, "hardware": { "cuda_name": "NVIDIA H100 80GB HBM3", "device": "cuda" }, "output_dir": "/workspace/qwen3-droped", "rotary_identity_probe": { "identity_cos_max_abs_error": 0.0, "identity_sin_max_abs_error": 0.0, "pass": true, "true_rope_max_abs_delta_from_identity": 2.0 }, "token_cache": { "eval_path": "/workspace/rs1b-token-cache/fineweb_edu_qwen3_eval_5000000.uint32", "train_path": "/workspace/rs1b-token-cache/fineweb_edu_qwen3_train_after_eval5000000_1000000000.uint32" }, "total_steps": 1907 }