run: experiment_name: jax-ppo-smoke seed: 20260704 learner_seat: 0 total_updates: 1 log_every: 1 checkpoint_every: 1 artifact_root: runs/tmp/jax-ppo-artifacts opponent: name: discard_only network: hidden_size: 64 num_layers: 2 ppo: batch_games: 64 rollout_steps: 64 gamma: 1.0 gae_lambda: 0.95 clip_epsilon: 0.2 entropy_coef: 0.01 value_coef: 0.5 max_grad_norm: 0.5 learning_rate: 0.0003 epochs: 1 minibatches: 4 reward: terminal_scale: 50.0 potential_shaping_initial: 1.0 potential_shaping_final: 0.0 potential_shaping_anneal_steps: 100000 evaluation: games: 128 duplicate: true shuffle_bank_seed: 20260704 batch_games: 64