{"id":"W4399434480","doi":"10.1007/s10639-024-12771-3","title":"Exploring quality criteria and evaluation methods in automated question generation: A comprehensive survey","year":2024,"lang":"en","type":"article","venue":"Education and Information Technologies","topic":"Topic Modeling","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Quality (philosophy); Computer science; Educational technology; Evaluation methods; Management science; Data science; Mathematics education; Psychology; Engineering; Reliability engineering; Epistemology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1584981,0.001717329,0.002878889,0.01911767,0.001284569,0.008432834,0.004152553,0.002756027,0.003004049],"category_scores_gemma":[0.3652562,0.001133377,0.002020766,0.01373755,0.002801107,0.01278045,0.003513431,0.002254186,0.0008952125],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003881694,"about_ca_system_score_gemma":0.006207896,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004433669,"about_ca_topic_score_gemma":0.004098973,"domain_scores_codex":[0.8661284,0.07727612,0.01449127,0.006341276,0.03414722,0.001615591],"domain_scores_gemma":[0.2530116,0.6787577,0.01597935,0.01084885,0.03972883,0.001673647],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0003965672,0.0005416819,0.03648796,0.01090015,0.000411314,0.00003778559,0.003880513,0.003076609,0.001972536,0.007818716,0.004448436,0.9300279],"study_design_scores_gemma":[0.0007424949,0.004931375,0.2394919,0.07300104,0.004261856,0.002417433,0.03397663,0.1642808,0.04241057,0.09791141,0.3354325,0.001141913],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.1563324,0.3882958,0.4190487,0.0108259,0.0003812217,0.002374778,0.002206457,0.00193284,0.01860194],"genre_scores_gemma":[0.4796125,0.1335085,0.3752099,0.002044893,0.0005584013,0.002016478,0.004002997,0.001040579,0.002005798],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.1584981,"threshold_uncertainty_score":0.8382282,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3294129269693888,"score_gpt":0.4810801104680509,"score_spread":0.1516671834986622,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}