{"id":"W4382567310","doi":"10.1007/978-3-031-36336-8_83","title":"How Useful Are Educational Questions Generated by Large Language Models?","year":2023,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Topic Modeling","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University; Canadian Institute for International Peace and Security; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Computer science; Quality (philosophy); Domain (mathematical analysis); Mathematics education; Natural language processing; Psychology; Epistemology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005180091,0.0008207376,0.0005983028,0.001459534,0.0005916091,0.005197005,0.001467986,0.001760042,0.0159877],"category_scores_gemma":[0.06333567,0.0005357384,0.0007121948,0.001169103,0.001307674,0.01563389,0.001524326,0.002994431,0.008263087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009655217,"about_ca_system_score_gemma":0.001040224,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001852751,"about_ca_topic_score_gemma":0.001753236,"domain_scores_codex":[0.9936321,0.004626018,0.0001512252,0.0004751422,0.0009438508,0.0001716704],"domain_scores_gemma":[0.9448494,0.04883485,0.0009542558,0.002437186,0.002345103,0.0005793119],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005404684,0.0007012066,0.01411366,0.0009716665,0.0001447652,0.0003474106,0.003846821,0.03073166,0.00553899,0.1974429,0.1048126,0.6408079],"study_design_scores_gemma":[0.0001268075,0.0001874415,0.004084247,0.0004800508,0.0001699976,0.0004379898,0.004372238,0.2547721,0.01098942,0.573613,0.1506635,0.0001031555],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06052357,0.003236686,0.7784804,0.03514402,0.0007298935,0.0003143553,0.003116467,0.006501023,0.1119536],"genre_scores_gemma":[0.7176228,0.002726549,0.2469161,0.003023678,0.0008104962,0.0004383774,0.004558095,0.001291598,0.02261239],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0159877,"threshold_uncertainty_score":0.05348414,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05845123054466652,"score_gpt":0.2962198031426312,"score_spread":0.2377685725979647,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}