{"id":"W4416035547","doi":"10.18653/v1/2025.emnlp-main.1471","title":"Agent-to-Agent Theory of Mind: Testing Interlocutor Awareness among Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Language and cultural evolution","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Bundesministerium für Bildung und Forschung","keywords":"Natural language; On Language; Language model; Comprehension; Field (mathematics)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02404187,0.0008629896,0.0007884845,0.001221394,0.001065614,0.003354979,0.002221322,0.001851327,0.004116233],"category_scores_gemma":[0.1363319,0.0009078204,0.001525101,0.0006631634,0.00314108,0.007910536,0.005844985,0.003375592,0.0005017718],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001541785,"about_ca_system_score_gemma":0.00188993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005164545,"about_ca_topic_score_gemma":0.002673642,"domain_scores_codex":[0.9885782,0.007738986,0.000398625,0.001841668,0.00111419,0.0003283437],"domain_scores_gemma":[0.8276895,0.1454892,0.007113034,0.01397941,0.003186034,0.002542886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006311497,0.00481547,0.2443688,0.001167534,0.00270809,0.0008714564,0.03242187,0.3582756,0.01730056,0.1081142,0.004886251,0.2187587],"study_design_scores_gemma":[0.000282813,0.001145387,0.01392787,0.00007849755,0.0001969232,0.0001448117,0.002036053,0.9054276,0.002680387,0.07279769,0.001181053,0.0001007637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8546547,0.0001497026,0.1336903,0.0008841736,0.00009480647,0.0004385189,0.0002431364,0.0006893204,0.009155286],"genre_scores_gemma":[0.957853,0.00004706834,0.0408528,0.0002190941,0.00002626041,0.0003255737,0.0002105825,0.00009459615,0.000370943],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02404187,"threshold_uncertainty_score":0.1271471,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0301710922244386,"score_gpt":0.3143922501280921,"score_spread":0.2842211579036535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}