{"id":"W7106794174","doi":"10.48448/q2b3-x919","title":"Can LLMs Reason Abstractly Over Math Word Problems Without CoT? Disentangling Abstract Formulation From Arithmetic Computation","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Conflation; Computation; Causal reasoning; Word (group theory); Automated reasoning; Mental arithmetic; Multiplication (music)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003797642,0.001064398,0.0007312622,0.0009071274,0.0005994673,0.004727526,0.002013936,0.001455313,0.01802354],"category_scores_gemma":[0.03620738,0.0005667555,0.001400296,0.0006839617,0.002623493,0.01271944,0.004633369,0.002994011,0.005204631],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001745377,"about_ca_system_score_gemma":0.003048164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005689121,"about_ca_topic_score_gemma":0.01050759,"domain_scores_codex":[0.996102,0.001454288,0.0002532699,0.0007015289,0.00120237,0.0002865518],"domain_scores_gemma":[0.9842062,0.007999764,0.0007696425,0.004628829,0.001886627,0.0005089384],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00167962,0.000635716,0.02874622,0.001667979,0.0002451953,0.0005055351,0.003876994,0.09696802,0.02881998,0.4001418,0.04322693,0.393486],"study_design_scores_gemma":[0.0001050421,0.0003250683,0.003842979,0.0002597452,0.0001006548,0.0002339316,0.001201465,0.4489397,0.02921262,0.468282,0.04737646,0.0001203803],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2936799,0.0008130952,0.6069009,0.004821565,0.0005022825,0.0002293696,0.003838525,0.02762907,0.06158537],"genre_scores_gemma":[0.7495826,0.000228938,0.2322648,0.0008770367,0.00005273739,0.0001641987,0.004060405,0.003287074,0.009482308],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01802354,"threshold_uncertainty_score":0.06029481,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03509017438742097,"score_gpt":0.3214927460309684,"score_spread":0.2864025716435475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}