Focus project on JAX PPO
This commit is contained in:
@@ -0,0 +1,57 @@
|
||||
run:
|
||||
experiment_name: ismcts-default
|
||||
max_iterations: 500
|
||||
seed: 1
|
||||
device: cuda
|
||||
rules:
|
||||
n_colors: 5
|
||||
n_ranks: 9
|
||||
min_rank: 2
|
||||
n_handshakes: 3
|
||||
hand_size: 8
|
||||
expedition_penalty: -20
|
||||
bonus_threshold: 8
|
||||
bonus_amount: 20
|
||||
encoding:
|
||||
derived_playability: true
|
||||
slot_aware_playability: true
|
||||
network:
|
||||
kind: mlp
|
||||
hidden_size: 768
|
||||
num_layers: 4
|
||||
activation: relu
|
||||
mcts:
|
||||
n_simulations: 50
|
||||
c_puct: 5.0
|
||||
max_depth: 200
|
||||
parallel_simulations: 64
|
||||
virtual_loss_value: 5.0
|
||||
eval_n_simulations: 16
|
||||
rollout_policy: heuristic_balanced
|
||||
use_rollout_value: false
|
||||
root_dirichlet_alpha: 0.3
|
||||
root_dirichlet_epsilon: 0.4
|
||||
temperature:
|
||||
training: 1.0
|
||||
eval: 0.0
|
||||
training:
|
||||
games_per_iter: 10
|
||||
gradient_steps_per_iter: 10
|
||||
batch_size: 128
|
||||
replay_capacity: 100000
|
||||
interleave_games: 8
|
||||
interleave_max_batch: 64
|
||||
num_workers: 8
|
||||
worker_device: cuda
|
||||
optimization:
|
||||
learning_rate: 0.0003
|
||||
grad_clip: 5.0
|
||||
checkpoint:
|
||||
save_every: 20
|
||||
save_latest: true
|
||||
evaluation:
|
||||
eval_every: 5
|
||||
games: 5
|
||||
opponents: [random, discard-only, heuristic-cautious]
|
||||
max_steps: 500
|
||||
num_workers: 8
|
||||
Reference in New Issue
Block a user