{"id":"W7084446458","doi":"","title":"Recursive Self-Aggregation Unlocks Deep Thinking in Large Language Models","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Climate Change and Health Impacts","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Lawrence Livermore National Laboratory; Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada; Office of Science; Alliance de recherche numérique du Canada; National Energy Research Scientific Computing Center; Canadian Institute for Advanced Research; Nvidia; Laboratory Directed Research and Development; U.S. Department of Energy","keywords":"Inference; Scaling; Language model; Population; Deep learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002312897,0.001175573,0.0008948509,0.000617193,0.0006118736,0.001439238,0.002479138,0.0009103396,0.003382695],"category_scores_gemma":[0.0100116,0.0007484248,0.0009854467,0.0006010964,0.001435776,0.00418081,0.002823083,0.00301914,0.0008357992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001147252,"about_ca_system_score_gemma":0.001972932,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007406911,"about_ca_topic_score_gemma":0.01728118,"domain_scores_codex":[0.9987853,0.0004229594,0.00007069275,0.0003086664,0.0002876302,0.0001247049],"domain_scores_gemma":[0.9940506,0.003486747,0.0002893319,0.001508556,0.000448762,0.0002159344],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003529949,0.0003793941,0.004796975,0.0002011403,0.0001382372,0.0001878257,0.0003910997,0.6405857,0.01124358,0.01688568,0.007910055,0.3169274],"study_design_scores_gemma":[0.0000251008,0.00003272566,0.0001051745,0.000004109819,0.000009804478,0.00001037092,0.00001504451,0.9893009,0.001661201,0.008271658,0.0005587844,0.00000503804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.183805,0.0006637497,0.7898131,0.001176478,0.0001447243,0.000165822,0.0002787541,0.01686823,0.007084055],"genre_scores_gemma":[0.687238,0.000139508,0.308284,0.0004551744,0.00006876174,0.000192562,0.0005079822,0.0008183265,0.002295704],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007406911,"threshold_uncertainty_score":0.01472759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04325180611905798,"score_gpt":0.2170000575218499,"score_spread":0.1737482514027919,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}