{"id":"W4403816328","doi":"10.48550/arxiv.2409.20553","title":"Maia-2: A Unified Model for Human-AI Alignment in Chess","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Microsoft Research; John D. and Catherine T. MacArthur Foundation","keywords":"Computer science; Artificial intelligence; Cognitive science; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003961082,0.0003579893,0.0003761385,0.0004483212,0.0001094271,0.0002394059,0.002318647,0.000340049,0.0000126185],"category_scores_gemma":[0.00002178859,0.0004249997,0.0002582303,0.0005284378,0.0001204801,0.0002514014,0.003216667,0.0006651192,0.0001041956],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005216472,"about_ca_system_score_gemma":0.000304605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002556476,"about_ca_topic_score_gemma":0.0005758946,"domain_scores_codex":[0.9975495,0.0000671079,0.0003423048,0.001427019,0.0001179808,0.0004960677],"domain_scores_gemma":[0.9982663,0.000102267,0.0001457311,0.00122683,0.0001272507,0.0001316602],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001102121,0.00006215898,0.00008626233,0.00008943817,0.00003008426,0.0001016254,0.000783718,0.4758363,0.00008500629,0.5222111,0.0003584345,0.0003447579],"study_design_scores_gemma":[0.00006600306,0.00001903929,0.00001109424,0.00009909839,0.00002141074,5.367679e-7,0.00006207378,0.5496423,0.0008079758,0.4489217,0.000100644,0.0002482245],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1403293,0.00005738381,0.8554159,0.0004728448,0.0005898469,0.0006656969,0.00002348988,0.0002681885,0.00217729],"genre_scores_gemma":[0.9899586,0.00002915699,0.002819935,0.0001869833,0.00006671226,0.00001131959,0.000009989896,0.00002975383,0.006887484],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.852596,"threshold_uncertainty_score":0.9998202,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1681204725693926,"score_gpt":0.2585748532727388,"score_spread":0.09045438070334619,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}