{"id":"W4409053427","doi":"10.1007/978-3-031-86585-5_4","title":"Analysing KataGo: A Comparative Evaluation Against Perfect Play in the Game of Go","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Theoretical computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.003656388,0.0004351019,0.000673691,0.001288779,0.0001626036,0.0004012683,0.004344675,0.0002201243,0.00001408954],"category_scores_gemma":[0.0003248067,0.0003319913,0.000172375,0.001825621,0.0009468584,0.0005535569,0.0008037582,0.0008482387,0.00001935277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004040222,"about_ca_system_score_gemma":0.001005447,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007138575,"about_ca_topic_score_gemma":0.0006476759,"domain_scores_codex":[0.9954823,0.0002756604,0.0008464223,0.00123521,0.001670658,0.0004897745],"domain_scores_gemma":[0.9955971,0.001882126,0.0004605694,0.001496271,0.0005092782,0.00005467605],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008054037,0.00005161327,0.0002129527,0.00002842665,0.00001997048,0.00002339227,0.01499781,0.3474349,0.0002721467,0.01547729,0.00003465274,0.6214388],"study_design_scores_gemma":[0.00009122814,0.00009676372,0.0003210031,0.0005513048,0.00001829966,0.000006988763,0.000005452571,0.9251724,0.00360517,0.0695831,0.0002269584,0.0003212529],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002699593,0.0006564516,0.9851147,0.000522154,0.0007705439,0.0007749279,0.000004364987,0.00003908973,0.009418182],"genre_scores_gemma":[0.9491544,0.00002891009,0.049707,0.0008588624,0.0001344771,0.00001982577,0.000006451436,0.000009425334,0.00008067335],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9464548,"threshold_uncertainty_score":0.9999132,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06007977150295404,"score_gpt":0.3403553913993017,"score_spread":0.2802756198963476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}