winkin119 commited on
Commit
1f6e2fb
·
verified ·
1 Parent(s): 694219a

upload via upload_folder 2025-08-12T19:20:16.313075+00:00

Browse files
README.md ADDED
@@ -0,0 +1,50 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ env_name: LunarLander-v3
3
+ tags:
4
+ - LunarLander-v3
5
+ - rainbow-dqn (with uniform sampling)
6
+ - reinforcement-learning
7
+ - custom-implementation
8
+ - deep-q-learning
9
+ - pytorch
10
+ - rainbow
11
+ - dqn
12
+ model-index:
13
+ - name: Rainbow-1d-LunarLander-v3-NoPer
14
+ results:
15
+ - task:
16
+ type: reinforcement-learning
17
+ name: reinforcement-learning
18
+ dataset:
19
+ name: LunarLander-v3
20
+ type: LunarLander-v3
21
+ metrics:
22
+ - type: mean_reward
23
+ value: 282.11 +/- 19.77
24
+ name: mean_reward
25
+ verified: false
26
+ ---
27
+
28
+ # **Rainbow-DQN (with uniform sampling)** Agent playing **LunarLander-v3**
29
+ This is a trained model of a **Rainbow-DQN (with uniform sampling)** agent playing **LunarLander-v3**.
30
+
31
+ ## Usage
32
+ ### create the conda env in https://github.com/GeneHit/drl_practice
33
+ ```bash
34
+ conda create -n drl python=3.12
35
+ conda activate drl
36
+ python -m pip install -r requirements.txt
37
+ ```
38
+
39
+ ### play with full model
40
+ ```python
41
+ # load the full model
42
+ model = load_from_hub(repo_id="winkin119/Rainbow-1d-LunarLander-v3-NoPer", filename="full_model.pt")
43
+
44
+ # Create the environment.
45
+ env = gym.make("LunarLander-v3")
46
+ state, _ = env.reset()
47
+ action = model.action(state)
48
+ ...
49
+ ```
50
+ There is also a state dict version of the model, you can check the corresponding definition in the repo.
eval_result.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "mean_reward": 282.1136285308617,
3
+ "std_reward": 19.771440248266718,
4
+ "datetime": "2025-08-10T20:31:50.282987+00:00",
5
+ "train_duration_min": "6.54"
6
+ }
full_model.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:46f828bd08bf9f5fe0b8143c8f1f65ca2138a0c592b6e97097648818a8422a78
3
+ size 807925
params.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "env_config": {
3
+ "env_id": "LunarLander-v3",
4
+ "env_kwargs": {},
5
+ "max_steps": null,
6
+ "normalize_obs": false,
7
+ "use_image": false,
8
+ "vector_env_num": 6,
9
+ "use_multi_processing": true,
10
+ "image_shape": null,
11
+ "frame_stack": 1,
12
+ "frame_skip": 1,
13
+ "training_render_mode": null
14
+ },
15
+ "device": "cpu",
16
+ "learning_rate": 0.0003,
17
+ "gamma": 0.99,
18
+ "checkpoint_pathname": "",
19
+ "max_grad_norm": 10.0,
20
+ "log_interval": 100,
21
+ "eval_episodes": 100,
22
+ "eval_random_seed": 42,
23
+ "eval_video_num": 10,
24
+ "timesteps": 125000,
25
+ "epsilon_schedule": {
26
+ "_type": "ConstantSchedule",
27
+ "_module": "practice.utils_for_coding.scheduler_utils",
28
+ "value": 0.0
29
+ },
30
+ "replay_buffer_capacity": 0,
31
+ "batch_size": 64,
32
+ "train_interval": 1,
33
+ "target_update_interval": 250,
34
+ "update_start_step": 2000,
35
+ "dqn_algorithm": "rainbow",
36
+ "noisy_std": 0.5,
37
+ "per_buffer_config": {
38
+ "capacity": 150000,
39
+ "n_step": 3,
40
+ "gamma": 0.99,
41
+ "use_uniform_sampling": true,
42
+ "alpha": 0.6,
43
+ "beta": 0.4,
44
+ "beta_increment": 3e-06
45
+ },
46
+ "v_min": -300.0,
47
+ "v_max": 300.0,
48
+ "num_atoms": 51
49
+ }
replay.mp4 ADDED
Binary file (28.7 kB). View file
 
state_dict.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:42b574f556a32b3316dde32946689497381e63c8fdfb85685dac0bc92ac65acc
3
+ size 806005
tensorboard/events.out.tfevents.1754857508.winkindeMacBook-Air.local.34042.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:92b91b010a452b2e77f1a42407f0a846697e5787d3811145778cb08159a01e77
3
+ size 568536