{"id":"W4415158403","doi":"10.48550/arxiv.2504.09858","title":"Reasoning Models Can Be Effective Without Thinking","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Logic, Reasoning, and Knowledge","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto","keywords":"Simple (philosophy); Thinking processes; Set (abstract data type); Latency (audio); Process (computing); Scaling; Automated reasoning; Analytic reasoning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0007791477,0.0005562424,0.0006878772,0.0002482459,0.0004227044,0.0003109285,0.002752861,0.0004777425,0.000007670972],"category_scores_gemma":[0.0002577633,0.0005231722,0.0003043901,0.00043738,0.00009476076,0.0003326079,0.004161705,0.001301841,0.0000399573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003462969,"about_ca_system_score_gemma":0.000608494,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008087862,"about_ca_topic_score_gemma":0.0002962568,"domain_scores_codex":[0.9967514,0.0003470716,0.0003510885,0.001484182,0.0004051937,0.0006611135],"domain_scores_gemma":[0.9972597,0.0003095021,0.000322973,0.00164392,0.0002760277,0.0001878821],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005955207,0.0003334197,0.3748273,0.0008300859,0.0008433434,0.0003059249,0.04299541,0.01512379,0.0001513736,0.5279483,0.002583848,0.03399761],"study_design_scores_gemma":[0.002627627,0.0003402375,0.09316228,0.006057996,0.0004681799,0.00007622432,0.0004254101,0.6899808,0.005209351,0.1828313,0.0143639,0.004456683],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.272403,0.001564499,0.5901329,0.001010683,0.003699199,0.001286078,0.00002562008,0.001429635,0.1284484],"genre_scores_gemma":[0.9784971,0.0001096206,0.01542819,0.0006099186,0.0003529664,0.00009799097,0.00001979358,0.00003102362,0.004853327],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7060942,"threshold_uncertainty_score":0.999722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03665691938393383,"score_gpt":0.2781942094752528,"score_spread":0.241537290091319,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}