{"id":"W7106830348","doi":"10.48448/bd0k-3k96","title":"Unleashing the Reasoning Potential of LLMs by Critique Fine-Tuning on One Problem","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Verifiable secret sharing; Robustness (evolution); Set (abstract data type); Automated reasoning; Analytic reasoning; Ranging; Matching (statistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003613468,0.001388586,0.0009752816,0.0005836071,0.0006734305,0.002027454,0.002276537,0.001908998,0.007191548],"category_scores_gemma":[0.02898297,0.0005040329,0.0009725504,0.0003967739,0.001845907,0.003001659,0.002458262,0.003991331,0.002216573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001241969,"about_ca_system_score_gemma":0.002159804,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003743398,"about_ca_topic_score_gemma":0.008286349,"domain_scores_codex":[0.9974147,0.001174352,0.00009903116,0.0006210503,0.0005345536,0.0001561888],"domain_scores_gemma":[0.9896128,0.006833046,0.000456452,0.002008739,0.0007929924,0.0002959705],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009145108,0.0004842759,0.009574874,0.001076429,0.0002219444,0.0005488651,0.001348021,0.4532054,0.02845045,0.0461215,0.03440984,0.4236439],"study_design_scores_gemma":[0.0001012357,0.0001384998,0.0005253138,0.00006040592,0.0000299786,0.00008720331,0.0001596131,0.9435521,0.006118834,0.04046379,0.008734339,0.00002874207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1880593,0.001923473,0.7560727,0.003591716,0.0004686579,0.0003518249,0.001082261,0.02465198,0.02379811],"genre_scores_gemma":[0.7087919,0.0002273819,0.2803345,0.001275477,0.00008530684,0.00026165,0.001216912,0.001857988,0.005948917],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007191548,"threshold_uncertainty_score":0.0240581,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0219723540371234,"score_gpt":0.3054132628634634,"score_spread":0.28344090882634,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}