{"id":"W4413115901","doi":"10.1098/rstb.2023.0499","title":"Re-evaluating Theory of Mind evaluation in large language models","year":2025,"lang":"en","type":"article","venue":"Philosophical Transactions of the Royal Society B Biological Sciences","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Theory of mind; Computer science; Cognitive science; Linguistics; Natural language processing; Psychology; Philosophy; Neuroscience; Cognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07438702,0.001435344,0.002403827,0.003259209,0.00117198,0.01023751,0.003410929,0.002941636,0.003896373],"category_scores_gemma":[0.3595674,0.0009193723,0.001475032,0.001819501,0.005422107,0.01479071,0.00888461,0.004392946,0.0005830106],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006353087,"about_ca_system_score_gemma":0.00413437,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00592664,"about_ca_topic_score_gemma":0.007646807,"domain_scores_codex":[0.9260579,0.05875145,0.002453621,0.002502767,0.009290552,0.0009435776],"domain_scores_gemma":[0.6688031,0.2787719,0.008419112,0.02078051,0.01990024,0.003325087],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001566848,0.0004227164,0.02636769,0.002158239,0.001302767,0.0004970301,0.01118727,0.2246342,0.003963117,0.4937969,0.009965129,0.2241381],"study_design_scores_gemma":[0.0001143112,0.0002383637,0.00285387,0.0004109244,0.0001535649,0.0001020589,0.001408127,0.4579825,0.002248517,0.5281665,0.006188737,0.0001324093],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2977207,0.00499815,0.6337642,0.02770859,0.0005905599,0.0004234454,0.00070263,0.001965049,0.03212672],"genre_scores_gemma":[0.8749234,0.0004721063,0.1217018,0.0007994621,0.000145738,0.0001749774,0.0004462334,0.0005241852,0.0008121788],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.925613,"threshold_uncertainty_score":0.3934008,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.122782212426812,"score_gpt":0.3560545157109268,"score_spread":0.2332723032841147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}