{"id":"W4412496569","doi":"10.1007/978-3-031-98462-4_52","title":"From Recall to Reasoning: Automated Question Generation for Deeper Math Learning Through Large Language Models","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Recall; Automated reasoning; Artificial intelligence; Natural language processing; Cognitive science; Programming language; Mathematics education; Cognitive psychology; Mathematics; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001940077,0.001492605,0.001091915,0.001583699,0.0006796796,0.002813532,0.002966801,0.001674748,0.02101435],"category_scores_gemma":[0.01060295,0.001062021,0.002334566,0.0009089772,0.0009642573,0.00700479,0.003820779,0.003758725,0.008161385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001160487,"about_ca_system_score_gemma":0.001481169,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002407034,"about_ca_topic_score_gemma":0.003902917,"domain_scores_codex":[0.9981647,0.000656132,0.0001213645,0.0005641173,0.0003678206,0.0001258397],"domain_scores_gemma":[0.9929529,0.005156416,0.0001832108,0.0009193656,0.0006178601,0.0001701827],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004942051,0.0004285543,0.001840506,0.0004330286,0.000118026,0.0002492852,0.000741308,0.02649656,0.01568383,0.03386461,0.03984351,0.8798066],"study_design_scores_gemma":[0.00008948696,0.00009725193,0.0004266819,0.00007081176,0.00008837561,0.0001282492,0.0002771374,0.8410614,0.01964882,0.1248905,0.01318368,0.00003769552],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02013516,0.0003907149,0.9453595,0.001041226,0.0001541029,0.000247392,0.001719645,0.02603751,0.004914817],"genre_scores_gemma":[0.2700806,0.0002890763,0.710479,0.0004957631,0.0001513344,0.0002747301,0.007877134,0.001834794,0.008517652],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02101435,"threshold_uncertainty_score":0.07030004,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02278628825185522,"score_gpt":0.2863770605732907,"score_spread":0.2635907723214355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}