{ "_scheme": { "naming": "Astronomical names, alphabetically ordered. The first letter is the generation; a new letter means the observation space broke, not that the model got better.", "rule": "A codename never encodes quality. The record this replaces stored 'FINAL PPO', which stops meaning anything the moment there is a second final model.", "identity": "The hash is the truth -- it is what actually played. The codename is for humans, and it is assigned here, not derived. Records should carry the hash; look the codename up.", "next": "cygnus, deneb, ..." }, "models": { "altair": { "hash": "e8241e305c01", "hash_kind": "sha256 of web/public/models/jax-ppo.onnx", "generation": "a", "game": "single round", "observation_size": 454, "hidden_size": 512, "num_layers": 3, "trained": "self-play league, 122.6M learner actions", "source": "/mnt/2tbhdd/coolrl-lost-cities-artifacts/league/2026-07-05_052325_jax-ppo-league-v1/latest", "deployed": "web/public/models/jax-ppo.onnx", "displayed_as": "WASM ยท FINAL PPO", "note": "Every game in data/human-play/game-records.jsonl (format v1, 111 games, 2026-07-14) was played against this model. The v1 schema has no model field -- it stores the on-screen label -- so this line is the record of what they played." }, "borealis": { "hash": "4ae613b010ca", "hash_kind": "sha256 over the orbax checkpoint files (no ONNX export yet)", "generation": "b", "game": "three-round match (classic rules)", "observation_size": 501, "critic_observation_size": 681, "hidden_size": 512, "num_layers": 3, "trained": "self-play, 131.1M learner actions, linear total-score reward, both seats, privileged critic", "source": "runs/jax-ppo-match/2026-07-15_031529_match-scaled/latest", "deployed": null, "results": { "vs_altair_3round": "0.6094 win rate [0.599, 0.620], +22.0 points, 8192 duplicate matches", "exploitability": "a from-scratch exploiter funded to 131M reaches 0.4657 against it, and 0.6295 against altair -- lower bound, neither exploiter had plateaued" } } } }