Focus project on JAX PPO

This commit is contained in:
2026-07-14 20:09:03 +09:00
parent 79273f7eb3
commit ef4b9d82b0
44 changed files with 302 additions and 503 deletions
+57
View File
@@ -0,0 +1,57 @@
run:
experiment_name: ismcts-default
max_iterations: 500
seed: 1
device: cuda
rules:
n_colors: 5
n_ranks: 9
min_rank: 2
n_handshakes: 3
hand_size: 8
expedition_penalty: -20
bonus_threshold: 8
bonus_amount: 20
encoding:
derived_playability: true
slot_aware_playability: true
network:
kind: mlp
hidden_size: 768
num_layers: 4
activation: relu
mcts:
n_simulations: 50
c_puct: 5.0
max_depth: 200
parallel_simulations: 64
virtual_loss_value: 5.0
eval_n_simulations: 16
rollout_policy: heuristic_balanced
use_rollout_value: false
root_dirichlet_alpha: 0.3
root_dirichlet_epsilon: 0.4
temperature:
training: 1.0
eval: 0.0
training:
games_per_iter: 10
gradient_steps_per_iter: 10
batch_size: 128
replay_capacity: 100000
interleave_games: 8
interleave_max_batch: 64
num_workers: 8
worker_device: cuda
optimization:
learning_rate: 0.0003
grad_clip: 5.0
checkpoint:
save_every: 20
save_latest: true
evaluation:
eval_every: 5
games: 5
opponents: [random, discard-only, heuristic-cautious]
max_steps: 500
num_workers: 8