{"id":"W4413115901","doi":"10.1098/rstb.2023.0499","title":"Re-evaluating Theory of Mind evaluation in large language models","year":2025,"lang":"en","type":"article","venue":"Philosophical Transactions of the Royal Society B Biological Sciences","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Theory of mind; Computer science; Cognitive science; Linguistics; Natural language processing; Psychology; Philosophy; Neuroscience; Cognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004698081,0.0001015586,0.0002042,0.00004403607,0.0002376983,0.00002102057,0.001185546,0.000135195,0.00005506994],"category_scores_gemma":[0.0002321325,0.00006009723,0.0002617832,0.0009073731,0.0004586036,0.000144201,0.0001114331,0.0002381844,6.116333e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005036229,"about_ca_system_score_gemma":0.000129445,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003719898,"about_ca_topic_score_gemma":0.000007846133,"domain_scores_codex":[0.9980046,0.0004848351,0.0003944962,0.0003779924,0.0005069795,0.000231081],"domain_scores_gemma":[0.9988894,0.0005509401,0.0001159059,0.0003268486,0.00008784496,0.00002908918],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003127944,0.0006764745,0.004291635,0.00005593887,0.00006794171,3.184017e-7,0.004814783,0.4576706,0.002887244,0.4253032,0.000006975613,0.1041937],"study_design_scores_gemma":[0.0001832466,0.00005211747,0.001168058,0.00003825839,0.000009882788,9.979787e-8,0.0003773888,0.6552936,0.000853946,0.3419738,9.007721e-7,0.00004876157],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3251865,0.0003444387,0.6667424,0.004929221,0.0001406297,0.0002528503,0.000005959403,0.00001744407,0.002380466],"genre_scores_gemma":[0.9793215,0.00001097622,0.02041337,0.0001952073,0.00001682284,0.00001844601,3.597714e-7,0.000001272686,0.0000219756],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.654135,"threshold_uncertainty_score":0.2450694,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.122782212426812,"score_gpt":0.3560545157109268,"score_spread":0.2332723032841147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}