Instructions to use capser54/gomoku-maskable-ppo-stage3-h6 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- stable-baselines3
How to use capser54/gomoku-maskable-ppo-stage3-h6 with stable-baselines3:
from huggingface_sb3 import load_from_hub checkpoint = load_from_hub( repo_id="capser54/gomoku-maskable-ppo-stage3-h6", filename="{MODEL FILENAME}.zip", ) - Notebooks
- Google Colab
- Kaggle
File size: 3,373 Bytes
3799788 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 | {
"project": "gomoku-maskable-ppo-stage3-h6",
"board_size": 9,
"win_length": 5,
"best_checkpoint_timestep": 13350000,
"last_checkpoint_timestep": 13850000,
"best_checkpoint_mean_reward": 1.524874998186715,
"last_checkpoint_mean_reward": 1.488199998936616,
"training_eval_curve": [
{
"timestep": 9350000,
"mean_reward": 0.12257499982602894,
"std_reward": 1.2194251432896845,
"min_reward": -0.9049999993294477,
"max_reward": 2.355000013951212
},
{
"timestep": 9850000,
"mean_reward": 0.49527500024996696,
"std_reward": 1.1900186925584642,
"min_reward": -0.9325000001117587,
"max_reward": 2.2100000074133277
},
{
"timestep": 10350000,
"mean_reward": 0.8255999991856515,
"std_reward": 1.1751735037863182,
"min_reward": -0.9125000010244548,
"max_reward": 2.1725000077858567
},
{
"timestep": 10850000,
"mean_reward": 1.0603750000242145,
"std_reward": 1.0538205125065332,
"min_reward": -0.9050000002607703,
"max_reward": 2.1825000075623393
},
{
"timestep": 11350000,
"mean_reward": 0.9657749989815057,
"std_reward": 1.1312206173587287,
"min_reward": -0.902500000782311,
"max_reward": 2.182500012218952
},
{
"timestep": 11850000,
"mean_reward": 1.3115499984752388,
"std_reward": 1.0438106135908165,
"min_reward": -0.8949999990873039,
"max_reward": 2.207500002812594
},
{
"timestep": 12350000,
"mean_reward": 1.2630499980971217,
"std_reward": 1.0021463573939906,
"min_reward": -0.8899999987334013,
"max_reward": 2.2000000034458935
},
{
"timestep": 12850000,
"mean_reward": 1.3795499993255362,
"std_reward": 0.9100868345368864,
"min_reward": -0.8999999994412065,
"max_reward": 2.334999999497086
},
{
"timestep": 13350000,
"mean_reward": 1.524874998186715,
"std_reward": 0.6208355627758182,
"min_reward": -0.8175000008195639,
"max_reward": 2.210000006482005
},
{
"timestep": 13850000,
"mean_reward": 1.488199998936616,
"std_reward": 0.8314822657740978,
"min_reward": -0.8950000004842877,
"max_reward": 2.324999989476055
}
],
"quick_benchmarks": {
"best_model_vs_heuristic_depth1_radius2_max4_early6_games50": {
"model_file": "best_model/best_model.zip",
"games": 50,
"wins": 47,
"losses": 3,
"draws": 0,
"win_rate": 0.94,
"loss_rate": 0.06,
"draw_rate": 0.0
},
"final_model_vs_heuristic_depth1_radius2_max4_early6_games50": {
"model_file": "gomoku_maskable_ppo_final.zip",
"games": 50,
"wins": 38,
"losses": 10,
"draws": 2,
"win_rate": 0.76,
"loss_rate": 0.2,
"draw_rate": 0.04
}
},
"training_command": "python train.py --resume-from models_stage3_h5/best_model/best_model.zip --opponent heuristic --heuristic-search-depth 1 --heuristic-max-candidates 4 --heuristic-early-max-candidates 6 --vec-env subproc --n-envs 8 --total-timesteps 5000000 --models-dir models_stage3_h6 --log-dir logs_stage3_h6 --eval-opponent heuristic --eval-freq 500000 --eval-games 100"
}
|