{"id":"W4389520485","doi":"10.18653/v1/2023.findings-emnlp.652","title":"NEWTON: Are Large Language Models Capable of Physical Reasoning?","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; National Science Foundation","keywords":"Benchmark (surveying); Computer science; Qualitative reasoning; Consistency (knowledge bases); Automated reasoning; Artificial intelligence; Mainstream; Natural language processing; Data science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00925414,0.001242631,0.0007854279,0.00175903,0.0007441027,0.005137324,0.002785238,0.002105539,0.01133849],"category_scores_gemma":[0.07197121,0.0008942854,0.001363673,0.00129205,0.001876777,0.01375645,0.003820915,0.002645015,0.004024735],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001783165,"about_ca_system_score_gemma":0.002655707,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008215504,"about_ca_topic_score_gemma":0.00941435,"domain_scores_codex":[0.9899232,0.005856794,0.0005784308,0.001440561,0.001911698,0.0002892246],"domain_scores_gemma":[0.9573694,0.03173773,0.001416247,0.00634485,0.002521965,0.0006097734],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001710488,0.0006641847,0.02354885,0.004650916,0.0005781141,0.0007523407,0.008691358,0.09741598,0.01243882,0.2212835,0.1268354,0.50143],"study_design_scores_gemma":[0.0003575528,0.0003626283,0.006869333,0.00101509,0.0002503609,0.0008562351,0.002327848,0.4781929,0.01118794,0.3506242,0.1477422,0.0002136527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1281385,0.003566686,0.7582056,0.007859998,0.0004166938,0.0007715374,0.01879703,0.04693982,0.0353041],"genre_scores_gemma":[0.5854079,0.001478324,0.3660783,0.001695252,0.0001526764,0.001089279,0.03487384,0.004120563,0.005103805],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01133849,"threshold_uncertainty_score":0.0489412,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02060342473632154,"score_gpt":0.2654812464257332,"score_spread":0.2448778216894117,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}