{"id":"W4410087623","doi":"10.1109/wi-iat62293.2024.00074","title":"Evaluating and Enhancing LLMs Agent Based on Theory of Mind in Guandan: A Multi-Player Cooperative Game Under Imperfect Information","year":2024,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Imperfect; Perfect information; Computer science; Game theory; Mathematical economics; Economics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001351753,0.0001132786,0.000129627,0.0002292055,0.00003720125,0.000181793,0.0001706959,0.00004837712,0.00009412356],"category_scores_gemma":[0.0002683121,0.00008661595,0.00003117402,0.0003375403,0.0000542663,0.0007546815,0.00008150222,0.0001267541,0.00005923518],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007251975,"about_ca_system_score_gemma":0.0001404951,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005185237,"about_ca_topic_score_gemma":0.0001087953,"domain_scores_codex":[0.9988581,0.000162916,0.0003651963,0.0002145948,0.0002399262,0.0001592563],"domain_scores_gemma":[0.99889,0.0007449895,0.00005137977,0.0001924434,0.00008569752,0.00003552143],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006790926,0.0001069797,0.0003058981,0.00011376,0.00003072147,0.00001003172,0.04742376,0.1989244,0.04191089,0.06688054,0.00003207722,0.6441931],"study_design_scores_gemma":[0.00008650954,0.0001767341,0.0003236465,0.0001539215,0.000003070623,0.000001468768,0.0008808976,0.8264179,0.1710828,0.0007634985,0.00002089603,0.00008864004],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3440922,0.0000588624,0.65475,0.0001291899,0.0001293291,0.0001880344,0.000001099129,0.00002775113,0.0006235721],"genre_scores_gemma":[0.9809992,0.000003740127,0.01859564,0.0002957219,0.00001073013,0.00001631749,0.000001177494,0.000004356635,0.00007315793],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6441044,"threshold_uncertainty_score":0.3532096,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0727216316583185,"score_gpt":0.3662076669918495,"score_spread":0.293486035333531,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}