Instructions to use capser54/gomoku-maskable-ppo-stage3-h6 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- stable-baselines3
How to use capser54/gomoku-maskable-ppo-stage3-h6 with stable-baselines3:
from huggingface_sb3 import load_from_hub checkpoint = load_from_hub( repo_id="capser54/gomoku-maskable-ppo-stage3-h6", filename="{MODEL FILENAME}.zip", ) - Notebooks
- Google Colab
- Kaggle
Download metrics_summary.json from capser54/gomoku-maskable-ppo-stage3-h6: direct link, hf CLI and curl.
- Browser
- Download file 3.37 kB
-
https://huggingface.co/capser54/gomoku-maskable-ppo-stage3-h6/resolve/main/metrics_summary.json
- Command line
-
hf download hf://capser54/gomoku-maskable-ppo-stage3-h6/metrics_summary.json
-
curl -L -o metrics_summary.json https://huggingface.co/capser54/gomoku-maskable-ppo-stage3-h6/resolve/main/metrics_summary.json
3.37 kB
| { | |
| "project": "gomoku-maskable-ppo-stage3-h6", | |
| "board_size": 9, | |
| "win_length": 5, | |
| "best_checkpoint_timestep": 13350000, | |
| "last_checkpoint_timestep": 13850000, | |
| "best_checkpoint_mean_reward": 1.524874998186715, | |
| "last_checkpoint_mean_reward": 1.488199998936616, | |
| "training_eval_curve": [ | |
| { | |
| "timestep": 9350000, | |
| "mean_reward": 0.12257499982602894, | |
| "std_reward": 1.2194251432896845, | |
| "min_reward": -0.9049999993294477, | |
| "max_reward": 2.355000013951212 | |
| }, | |
| { | |
| "timestep": 9850000, | |
| "mean_reward": 0.49527500024996696, | |
| "std_reward": 1.1900186925584642, | |
| "min_reward": -0.9325000001117587, | |
| "max_reward": 2.2100000074133277 | |
| }, | |
| { | |
| "timestep": 10350000, | |
| "mean_reward": 0.8255999991856515, | |
| "std_reward": 1.1751735037863182, | |
| "min_reward": -0.9125000010244548, | |
| "max_reward": 2.1725000077858567 | |
| }, | |
| { | |
| "timestep": 10850000, | |
| "mean_reward": 1.0603750000242145, | |
| "std_reward": 1.0538205125065332, | |
| "min_reward": -0.9050000002607703, | |
| "max_reward": 2.1825000075623393 | |
| }, | |
| { | |
| "timestep": 11350000, | |
| "mean_reward": 0.9657749989815057, | |
| "std_reward": 1.1312206173587287, | |
| "min_reward": -0.902500000782311, | |
| "max_reward": 2.182500012218952 | |
| }, | |
| { | |
| "timestep": 11850000, | |
| "mean_reward": 1.3115499984752388, | |
| "std_reward": 1.0438106135908165, | |
| "min_reward": -0.8949999990873039, | |
| "max_reward": 2.207500002812594 | |
| }, | |
| { | |
| "timestep": 12350000, | |
| "mean_reward": 1.2630499980971217, | |
| "std_reward": 1.0021463573939906, | |
| "min_reward": -0.8899999987334013, | |
| "max_reward": 2.2000000034458935 | |
| }, | |
| { | |
| "timestep": 12850000, | |
| "mean_reward": 1.3795499993255362, | |
| "std_reward": 0.9100868345368864, | |
| "min_reward": -0.8999999994412065, | |
| "max_reward": 2.334999999497086 | |
| }, | |
| { | |
| "timestep": 13350000, | |
| "mean_reward": 1.524874998186715, | |
| "std_reward": 0.6208355627758182, | |
| "min_reward": -0.8175000008195639, | |
| "max_reward": 2.210000006482005 | |
| }, | |
| { | |
| "timestep": 13850000, | |
| "mean_reward": 1.488199998936616, | |
| "std_reward": 0.8314822657740978, | |
| "min_reward": -0.8950000004842877, | |
| "max_reward": 2.324999989476055 | |
| } | |
| ], | |
| "quick_benchmarks": { | |
| "best_model_vs_heuristic_depth1_radius2_max4_early6_games50": { | |
| "model_file": "best_model/best_model.zip", | |
| "games": 50, | |
| "wins": 47, | |
| "losses": 3, | |
| "draws": 0, | |
| "win_rate": 0.94, | |
| "loss_rate": 0.06, | |
| "draw_rate": 0.0 | |
| }, | |
| "final_model_vs_heuristic_depth1_radius2_max4_early6_games50": { | |
| "model_file": "gomoku_maskable_ppo_final.zip", | |
| "games": 50, | |
| "wins": 38, | |
| "losses": 10, | |
| "draws": 2, | |
| "win_rate": 0.76, | |
| "loss_rate": 0.2, | |
| "draw_rate": 0.04 | |
| } | |
| }, | |
| "training_command": "python train.py --resume-from models_stage3_h5/best_model/best_model.zip --opponent heuristic --heuristic-search-depth 1 --heuristic-max-candidates 4 --heuristic-early-max-candidates 6 --vec-env subproc --n-envs 8 --total-timesteps 5000000 --models-dir models_stage3_h6 --log-dir logs_stage3_h6 --eval-opponent heuristic --eval-freq 500000 --eval-games 100" | |
| } | |