{"id":"W4394673405","doi":"10.48550/arxiv.2404.05545","title":"Evaluating Interventional Reasoning Capabilities of Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01833837,0.001870558,0.0009670463,0.002123778,0.0006148018,0.002372626,0.002458119,0.002273202,0.003124578],"category_scores_gemma":[0.1324981,0.0007476367,0.001424491,0.001352797,0.00165739,0.004251687,0.001900402,0.004203304,0.0005795366],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002582454,"about_ca_system_score_gemma":0.002609508,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01041955,"about_ca_topic_score_gemma":0.01225246,"domain_scores_codex":[0.9902287,0.006907017,0.0005178169,0.001246715,0.0008645369,0.0002351886],"domain_scores_gemma":[0.7027739,0.2849973,0.003952688,0.005391756,0.001954499,0.0009298562],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001282512,0.001028349,0.01639995,0.0008336225,0.000520155,0.0002447137,0.0007105913,0.8719115,0.001748809,0.01113044,0.003101075,0.09108832],"study_design_scores_gemma":[0.0001016645,0.0001708242,0.0007366671,0.0000349402,0.00005139453,0.00002825881,0.00007057333,0.9840119,0.0009168814,0.0133912,0.0004693029,0.00001626651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6127363,0.003166043,0.3577323,0.005871224,0.0002243597,0.0008240373,0.003518031,0.008712031,0.007215611],"genre_scores_gemma":[0.8494945,0.0004626631,0.146155,0.0005790991,0.00005929963,0.0004304346,0.002051785,0.0002065434,0.0005607393],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01833837,"threshold_uncertainty_score":0.09698373,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07922288203065865,"score_gpt":0.2760822933143821,"score_spread":0.1968594112837234,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}