{"id":"W4412496569","doi":"10.1007/978-3-031-98462-4_52","title":"From Recall to Reasoning: Automated Question Generation for Deeper Math Learning Through Large Language Models","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Recall; Automated reasoning; Artificial intelligence; Natural language processing; Cognitive science; Programming language; Mathematics education; Cognitive psychology; Mathematics; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0008967022,0.0004165524,0.0004330562,0.0003934725,0.0003237055,0.0005948443,0.001684991,0.0003595884,0.000004784723],"category_scores_gemma":[0.0002208365,0.0004079283,0.0001048822,0.000375509,0.00006192445,0.0007801527,0.0008879037,0.0005424126,0.00001221674],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003623328,"about_ca_system_score_gemma":0.0003707133,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002151805,"about_ca_topic_score_gemma":0.0001486216,"domain_scores_codex":[0.9965933,0.00006138886,0.0004767354,0.001686379,0.0006050433,0.0005771724],"domain_scores_gemma":[0.9980668,0.0003095496,0.0001987682,0.001044477,0.0002727835,0.000107609],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000056437,0.00001599529,0.000004786601,0.00002480782,0.00001132384,0.0000158698,0.006494326,0.6809497,0.0005596501,0.1577084,0.0001147608,0.1540946],"study_design_scores_gemma":[0.0002115361,0.00006664624,0.00000293563,0.0004416128,0.000008917698,0.000004794428,0.00000108561,0.9276273,0.0005875087,0.06973451,0.0009234091,0.0003896933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0005917337,0.0002855994,0.9943671,0.0007461531,0.001540615,0.0006245477,0.00001471304,0.0007506955,0.001078829],"genre_scores_gemma":[0.05378985,0.00001170138,0.9427542,0.001708031,0.0007424921,0.00003468148,0.00005281121,0.00002896755,0.0008772683],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2466776,"threshold_uncertainty_score":0.9998373,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02278628825185522,"score_gpt":0.2863770605732907,"score_spread":0.2635907723214355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}