{"id":"W2913781869","doi":"10.1016/j.artint.2019.103216","title":"The Hanabi challenge: A new frontier for AI research","year":2019,"lang":"en","type":"article","venue":"Artificial Intelligence","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":237,"is_retracted":false,"has_abstract":true,"ca_institutions":"Google (Canada)","funders":"","keywords":"Computer science; Domain (mathematical analysis); Imperfect; Artificial intelligence; Frontier; Perfect information; State (computer science); Cognitive science; Data science; Human–computer interaction; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003772688,0.0009807901,0.001178184,0.0007808294,0.00287164,0.00466689,0.003438527,0.001961228,0.01425428],"category_scores_gemma":[0.01375959,0.0005203729,0.0006330229,0.0009469746,0.003893171,0.008987051,0.004964871,0.005909472,0.002783474],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001957485,"about_ca_system_score_gemma":0.004760937,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008076472,"about_ca_topic_score_gemma":0.01018004,"domain_scores_codex":[0.9975441,0.001340994,0.00006950364,0.0003777783,0.0005119785,0.0001556889],"domain_scores_gemma":[0.9937477,0.00413742,0.0001985124,0.0006278144,0.0005415455,0.0007469786],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004423499,0.0002923063,0.002027617,0.0006542672,0.00007055992,0.0001113363,0.001481953,0.01693523,0.001465959,0.6769581,0.05038111,0.2491791],"study_design_scores_gemma":[0.00007940926,0.000133582,0.0007534909,0.0002309847,0.00001748429,0.0001068676,0.001154026,0.09532993,0.001659566,0.7796156,0.1208446,0.00007461296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05000562,0.01078606,0.7318398,0.06432147,0.002534381,0.0005885342,0.0007396562,0.001693113,0.1374914],"genre_scores_gemma":[0.426234,0.006526984,0.5133267,0.007562194,0.001438658,0.0009657363,0.001435594,0.0006643805,0.04184581],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01425428,"threshold_uncertainty_score":0.04768527,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1628522236355365,"score_gpt":0.4126734209769778,"score_spread":0.2498211973414413,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}