{"id":"W4309663019","doi":"10.1126/science.ade9097","title":"Human-level play in the game of<i>Diplomacy</i>by combining language models with strategic reasoning","year":2022,"lang":"en","type":"article","venue":"Science","topic":"Topic Modeling","field":"Computer Science","cited_by":201,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Cicero; Negotiation; Diplomacy; Computer science; Competition (biology); Reinforcement learning; Artificial intelligence; League; Natural language; Political science; Politics; Law; History; Ecology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002023539,0.0009239783,0.0003293536,0.0004117798,0.0005520736,0.002604939,0.0009987416,0.0006266536,0.00254311],"category_scores_gemma":[0.005216511,0.0002906271,0.0004451997,0.0001434492,0.001419059,0.001795759,0.001394529,0.001115064,0.000628232],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001257983,"about_ca_system_score_gemma":0.001765132,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01122816,"about_ca_topic_score_gemma":0.0146802,"domain_scores_codex":[0.9991267,0.000484011,0.00003491692,0.0001456753,0.00012477,0.00008384605],"domain_scores_gemma":[0.9978105,0.00135318,0.0002125171,0.0002050704,0.0001735138,0.0002450832],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001417558,0.001851911,0.06395772,0.0009490402,0.0004977329,0.0006135692,0.01291188,0.3445778,0.04746849,0.2019621,0.02061604,0.3031761],"study_design_scores_gemma":[0.0001033905,0.0007127125,0.007686193,0.0001022367,0.00007450008,0.0002130962,0.001956231,0.8798941,0.01065436,0.06081417,0.0376907,0.00009817799],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5091438,0.0003638365,0.4284849,0.002164998,0.0001082833,0.0004520252,0.0003166751,0.002415969,0.05654943],"genre_scores_gemma":[0.895644,0.0001225095,0.0994444,0.000163013,0.00001693646,0.0001134796,0.000256611,0.00006688955,0.00417208],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01122816,"threshold_uncertainty_score":0.02232563,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05109947060723923,"score_gpt":0.2899594067233681,"score_spread":0.2388599361161289,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}