{"id":"W4289356281","doi":"10.1007/978-3-031-11488-5_13","title":"On the Road to Perfection? Evaluating Leela Chess Zero Against Endgame Tablebases","year":2022,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Chess endgame; Heuristics; Computer science; Zero (linguistics); Perfection; Artificial intelligence; Epistemology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":[],"category_scores_codex":[0.002923218,0.0006519624,0.0005120751,0.0008471773,0.001190838,0.000976464,0.005864101,0.0001929105,0.0002778957],"category_scores_gemma":[0.0009263417,0.0005117917,0.0001974728,0.001340815,0.0006055762,0.000574403,0.002914842,0.00151108,0.0001984782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007844931,"about_ca_system_score_gemma":0.0007185711,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009839445,"about_ca_topic_score_gemma":0.00009809477,"domain_scores_codex":[0.993978,0.000184726,0.0006871995,0.002058396,0.002184516,0.0009072184],"domain_scores_gemma":[0.9944749,0.002419801,0.0003481325,0.002234588,0.0003159183,0.0002066862],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006596737,0.00002823297,0.00001338231,0.000008154469,0.000009145248,0.00003077524,0.001449767,0.3489779,0.000275121,0.04200754,0.0001195173,0.6070738],"study_design_scores_gemma":[0.00007195058,0.0006936329,0.00004938601,0.0003799298,0.00001060297,0.00003602009,0.000002124013,0.8542352,0.006259467,0.1322337,0.005102018,0.000926004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002250707,0.0002064288,0.9799568,0.003453994,0.003578821,0.0009447296,0.000007351107,0.0002491916,0.009351958],"genre_scores_gemma":[0.8154882,0.00005081227,0.1573184,0.02322468,0.001164819,0.0002610146,0.0000080108,0.0001401834,0.00234383],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8226384,"threshold_uncertainty_score":0.9997334,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05198616385661934,"score_gpt":0.3093362363009957,"score_spread":0.2573500724443763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}