{"id":"W4417004307","doi":"10.48550/arxiv.2512.02304","title":"When Does Verification Pay Off? A Closer Look at LLMs as Solution Verifiers","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Natural Sciences and Engineering Research Council of Canada; Ministry of Science and ICT, South Korea","keywords":"Metric (unit); Domain (mathematical analysis); Base (topology); Solver; Scale (ratio); Problem solver; Logical framework","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","research_integrity","insufficient_payload"],"consensus_categories":["metaepi_narrow","insufficient_payload"],"category_scores_codex":[0.005011958,0.001467964,0.001592985,0.0004944821,0.001798082,0.0009853508,0.003870624,0.001327212,0.01661995],"category_scores_gemma":[0.002365788,0.001140085,0.0005023314,0.0006908515,0.001371762,0.0009941825,0.004874414,0.001496899,0.0129843],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001580821,"about_ca_system_score_gemma":0.001454189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003797577,"about_ca_topic_score_gemma":0.0004453026,"domain_scores_codex":[0.9877708,0.002048709,0.002292163,0.004193225,0.001770786,0.001924348],"domain_scores_gemma":[0.9921668,0.0004466191,0.00216653,0.004019824,0.0006514745,0.0005487989],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006520556,0.0002916927,0.1727766,0.001094625,0.00009302511,0.00002624826,0.004740381,0.004305053,0.8107722,0.0005578628,0.002672944,0.002017233],"study_design_scores_gemma":[0.001951659,0.0005087909,0.4219785,0.002245219,0.0006575212,0.00005475807,0.0005674417,0.02026331,0.4332421,0.004918513,0.1096848,0.003927435],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9618044,0.000346487,0.003778193,0.004692285,0.02164082,0.002063922,0.0002539376,0.0005626201,0.004857344],"genre_scores_gemma":[0.9388571,0.001240825,0.006332167,0.001099651,0.001124766,0.000512322,0.0002857786,0.0001144804,0.05043289],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3775302,"threshold_uncertainty_score":0.9999693,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02335765223738391,"score_gpt":0.2864959467195923,"score_spread":0.2631382944822084,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}