{"id":"W4394647506","doi":"10.1162/tacl_a_00670","title":"Scope Ambiguities in Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"Transactions of the Association for Computational Linguistics","topic":"Topic Modeling","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; National Research Council Canada; Mila - Quebec Artificial Intelligence Institute","funders":"Fonds de Recherche du Québec-Société et Culture","keywords":"Scope (computer science); Computer science; Linguistics; Cognitive science; Psychology; Programming language; Philosophy","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01214811,0.0009581585,0.00114541,0.002415678,0.0008754414,0.003200091,0.001176603,0.00117239,0.001467629],"category_scores_gemma":[0.05126545,0.0007042747,0.001264532,0.002281595,0.001747151,0.004489352,0.001616941,0.002625595,0.0003685952],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001528014,"about_ca_system_score_gemma":0.0009840496,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007378241,"about_ca_topic_score_gemma":0.00832256,"domain_scores_codex":[0.9911237,0.006212833,0.0003981973,0.001172444,0.0008971487,0.0001958029],"domain_scores_gemma":[0.931223,0.06177311,0.002961238,0.002255571,0.001380658,0.000406449],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006892596,0.0001953489,0.02791137,0.0005172478,0.0004694706,0.001080486,0.004681053,0.7596234,0.004019022,0.1069712,0.007330769,0.08651127],"study_design_scores_gemma":[0.00002036531,0.00002158745,0.001482938,0.00003600601,0.00002620965,0.00007645391,0.0002319149,0.9092196,0.0005795248,0.08723713,0.001040671,0.00002761899],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.324225,0.001955675,0.6645086,0.001797179,0.00008947296,0.0001217236,0.001037984,0.001861909,0.0044025],"genre_scores_gemma":[0.9528204,0.0002983013,0.04485708,0.0001780687,0.00008096759,0.00006772448,0.001006534,0.0001398449,0.0005511562],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01214811,"threshold_uncertainty_score":0.06424612,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02432477583018319,"score_gpt":0.2884349640007543,"score_spread":0.2641101881705711,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}