{"id":"W4406281472","doi":"10.1016/j.ipm.2025.104059","title":"A diversity-enhanced knowledge distillation model for practical math word problem solving","year":2025,"lang":"en","type":"article","venue":"Information Processing & Management","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Diversity (politics); Distillation; Mathematics; Mathematics education; Computer science; Chemistry; Sociology; Chromatography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006543701,0.0004477395,0.0008794801,0.0004794179,0.0004668492,0.001089601,0.001796318,0.00114982,0.005713671],"category_scores_gemma":[0.002795485,0.0003280347,0.0005456972,0.0006120933,0.0006040274,0.001783568,0.001473669,0.001596019,0.0006083168],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008422578,"about_ca_system_score_gemma":0.001285121,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01052403,"about_ca_topic_score_gemma":0.01065195,"domain_scores_codex":[0.9997334,0.00006828011,0.00001528619,0.00007632657,0.00004775234,0.00005898381],"domain_scores_gemma":[0.9990255,0.0005720341,0.00005531139,0.00007407802,0.0002008038,0.00007237899],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002430223,0.0001922961,0.0009399664,0.00005860339,0.00004766426,0.00007275867,0.0000992174,0.9005085,0.001465109,0.02019078,0.00122911,0.07495296],"study_design_scores_gemma":[0.000006941038,0.00001280807,0.0000472963,0.000001987335,0.000003929332,0.000003601249,0.000003000588,0.9966614,0.0001364995,0.003019979,0.0001002201,0.000002444258],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1798401,0.0006417702,0.8060085,0.0009367326,0.0001266202,0.0001095518,0.0003773345,0.0008440159,0.01111542],"genre_scores_gemma":[0.9477334,0.0001482998,0.04692083,0.0001187668,0.00003209553,0.0001041545,0.0001937143,0.00003767806,0.004711092],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01052403,"threshold_uncertainty_score":0.02092558,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02701293583463594,"score_gpt":0.2946960030078121,"score_spread":0.2676830671731762,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}