File size: 3,373 Bytes
3799788
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
{
  "project": "gomoku-maskable-ppo-stage3-h6",
  "board_size": 9,
  "win_length": 5,
  "best_checkpoint_timestep": 13350000,
  "last_checkpoint_timestep": 13850000,
  "best_checkpoint_mean_reward": 1.524874998186715,
  "last_checkpoint_mean_reward": 1.488199998936616,
  "training_eval_curve": [
    {
      "timestep": 9350000,
      "mean_reward": 0.12257499982602894,
      "std_reward": 1.2194251432896845,
      "min_reward": -0.9049999993294477,
      "max_reward": 2.355000013951212
    },
    {
      "timestep": 9850000,
      "mean_reward": 0.49527500024996696,
      "std_reward": 1.1900186925584642,
      "min_reward": -0.9325000001117587,
      "max_reward": 2.2100000074133277
    },
    {
      "timestep": 10350000,
      "mean_reward": 0.8255999991856515,
      "std_reward": 1.1751735037863182,
      "min_reward": -0.9125000010244548,
      "max_reward": 2.1725000077858567
    },
    {
      "timestep": 10850000,
      "mean_reward": 1.0603750000242145,
      "std_reward": 1.0538205125065332,
      "min_reward": -0.9050000002607703,
      "max_reward": 2.1825000075623393
    },
    {
      "timestep": 11350000,
      "mean_reward": 0.9657749989815057,
      "std_reward": 1.1312206173587287,
      "min_reward": -0.902500000782311,
      "max_reward": 2.182500012218952
    },
    {
      "timestep": 11850000,
      "mean_reward": 1.3115499984752388,
      "std_reward": 1.0438106135908165,
      "min_reward": -0.8949999990873039,
      "max_reward": 2.207500002812594
    },
    {
      "timestep": 12350000,
      "mean_reward": 1.2630499980971217,
      "std_reward": 1.0021463573939906,
      "min_reward": -0.8899999987334013,
      "max_reward": 2.2000000034458935
    },
    {
      "timestep": 12850000,
      "mean_reward": 1.3795499993255362,
      "std_reward": 0.9100868345368864,
      "min_reward": -0.8999999994412065,
      "max_reward": 2.334999999497086
    },
    {
      "timestep": 13350000,
      "mean_reward": 1.524874998186715,
      "std_reward": 0.6208355627758182,
      "min_reward": -0.8175000008195639,
      "max_reward": 2.210000006482005
    },
    {
      "timestep": 13850000,
      "mean_reward": 1.488199998936616,
      "std_reward": 0.8314822657740978,
      "min_reward": -0.8950000004842877,
      "max_reward": 2.324999989476055
    }
  ],
  "quick_benchmarks": {
    "best_model_vs_heuristic_depth1_radius2_max4_early6_games50": {
      "model_file": "best_model/best_model.zip",
      "games": 50,
      "wins": 47,
      "losses": 3,
      "draws": 0,
      "win_rate": 0.94,
      "loss_rate": 0.06,
      "draw_rate": 0.0
    },
    "final_model_vs_heuristic_depth1_radius2_max4_early6_games50": {
      "model_file": "gomoku_maskable_ppo_final.zip",
      "games": 50,
      "wins": 38,
      "losses": 10,
      "draws": 2,
      "win_rate": 0.76,
      "loss_rate": 0.2,
      "draw_rate": 0.04
    }
  },
  "training_command": "python train.py --resume-from models_stage3_h5/best_model/best_model.zip --opponent heuristic --heuristic-search-depth 1 --heuristic-max-candidates 4 --heuristic-early-max-candidates 6 --vec-env subproc --n-envs 8 --total-timesteps 5000000 --models-dir models_stage3_h6 --log-dir logs_stage3_h6 --eval-opponent heuristic --eval-freq 500000 --eval-games 100"
}