{"id":"W4409362112","doi":"10.1609/aaai.v39i26.34987","title":"Evaluating Mathematical Reasoning Beyond Accuracy","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Mathematics Education and Teaching Techniques","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Management science; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002548391,0.0001382923,0.0002126204,0.0001319297,0.0006327286,0.0002380847,0.001021165,0.00009942098,0.0005346425],"category_scores_gemma":[0.01070781,0.0001074367,0.000100743,0.0005686184,0.0004423886,0.0001740002,0.0001537722,0.0003052149,0.00005369887],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008363154,"about_ca_system_score_gemma":0.0003171507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001030041,"about_ca_topic_score_gemma":0.00002140214,"domain_scores_codex":[0.998368,0.00005733547,0.0005055302,0.0002569751,0.0005400351,0.0002721656],"domain_scores_gemma":[0.9982886,0.000488857,0.0003458841,0.0001991533,0.0006136842,0.00006384352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00001140392,0.0001126937,0.00007596875,0.00005008157,0.000008434893,2.830695e-8,0.00821337,0.000001177857,0.003078172,0.9457522,0.000483217,0.04221324],"study_design_scores_gemma":[0.00001019161,0.00002964097,0.00001821623,0.0005527501,0.00001983561,3.039297e-7,0.005558763,0.003813632,0.07366435,0.9157993,0.0004303107,0.0001027063],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1400261,0.00002187168,0.01916307,0.01297217,0.0004536448,0.0009171183,0.00000219218,0.0002606701,0.8261831],"genre_scores_gemma":[0.9823474,0.00002426214,0.01495835,0.0002621822,0.00006261111,0.00004848476,2.378646e-7,0.000008218778,0.002288296],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8423212,"threshold_uncertainty_score":0.9976254,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1780688572621074,"score_gpt":0.4662585735755815,"score_spread":0.2881897163134741,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}