{"id":"W4416035547","doi":"10.18653/v1/2025.emnlp-main.1471","title":"Agent-to-Agent Theory of Mind: Testing Interlocutor Awareness among Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Language and cultural evolution","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"Natural Sciences and Engineering Research Council of Canada; Bundesministerium für Bildung und Forschung","keywords":"Natural language; On Language; Language model; Comprehension; Field (mathematics)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001334577,0.000272415,0.0004016596,0.0001543646,0.0005428535,0.0001194784,0.0006037987,0.0002256658,0.001874535],"category_scores_gemma":[0.0006132004,0.000216611,0.0002126565,0.001241523,0.0002245958,0.0004991462,0.0003545858,0.0001914732,0.00008047496],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002309388,"about_ca_system_score_gemma":0.0003004635,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007434472,"about_ca_topic_score_gemma":0.005995777,"domain_scores_codex":[0.9972697,0.000472098,0.0005968607,0.0005202207,0.0004718557,0.0006692259],"domain_scores_gemma":[0.9986494,0.0002656303,0.0002104461,0.000362912,0.00031021,0.000201443],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001800005,0.0009629438,0.01762356,0.0004601844,0.0004717325,0.00007423457,0.6442952,0.0009889684,0.01615291,0.137242,0.003603252,0.177945],"study_design_scores_gemma":[0.001389483,0.0002744419,0.0122272,0.001879141,0.0005132085,0.000001608296,0.9286292,0.02078167,0.02104904,0.009363085,0.002719644,0.001172293],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8843624,0.001453947,0.01795156,0.0002283716,0.0007795117,0.000625818,0.0000391559,0.00006985018,0.09448937],"genre_scores_gemma":[0.949157,0.00002731391,0.0005256623,0.0004184365,0.0001892566,0.00002686641,0.00001049213,0.00001224,0.04963271],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.284334,"threshold_uncertainty_score":0.9991751,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0301710922244386,"score_gpt":0.3143922501280921,"score_spread":0.2842211579036535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}