run: experiment_name: lost-cities-deep-cfr-selfplay-full-depth-slot-playability seed: 79 max_iterations: null max_minutes: 240 device: cuda use_amp: false rules: n_colors: 5 n_ranks: 9 min_rank: 2 n_handshakes: 3 hand_size: 8 expedition_penalty: -20 bonus_threshold: 8 bonus_amount: 20 encoding: derived_playability: true slot_aware_playability: true network: hidden_size: 256 num_layers: 3 activation: relu traversal: traversals_per_player: 70 strategy_sample_interval: 1 store_strategy_on_opponent_nodes: false store_strategy_on_traverser_nodes: true max_depth: null max_nodes_per_traversal: 1000 opponent_policy: self_play_league cutoff_value_mode: score_diff cutoff_rollouts: 0 cutoff_rollout_policy: random cutoff_rollout_max_steps: 300 progress_every_traversals: 10 num_workers: 8 worker_chunk_size: 8 regret_matching_epsilon: 0.0001 outcome_sampling_epsilon: 0.2 outcome_sampling_value_clip: 500 outcome_unsampled_regret: zero endpoint_depth_bucket_width: 100 endpoint_depth_bucket_max: 1000 self_play: current_weight: 0.5 recent_weight: 0.3 older_weight: 0.2 anchor_weight: 0.0 recent_window: 5 max_snapshots: 20 snapshot_every: 1 optimization: advantage_batch_size: 1024 strategy_batch_size: 1024 advantage_updates_per_iteration: 256 strategy_updates_per_iteration: 256 learning_rate: 0.00003 weight_decay: 0.0001 grad_clip: 1.0 memory: advantage_capacity: 2000000 strategy_capacity: 2000000 evaluation: eval_every: 5 games: 100 batch_size: 64 device: trainer num_workers: 4 max_steps: 1000 on_max_steps: score_diff opponents: - random - passive_discard - safe_heuristic - safe_heuristic_loose - safe_heuristic_strict - noisy_safe checkpoint: save_every: 10 progress_interval_seconds: 20.0