{"id":"W4285155190","doi":"10.18653/v1/2022.findings-acl.168","title":"Question Generation for Reading Comprehension Assessment by Modeling How and What to Ask","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: ACL 2022","topic":"Topic Modeling","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading comprehension; Computer science; Comprehension; Reading (process); Focus (optics); Artificial intelligence; Natural language processing; Mathematics education; Linguistics; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003277139,0.001959847,0.0006173209,0.002861357,0.0006599366,0.001675362,0.002768022,0.003081611,0.005406877],"category_scores_gemma":[0.02201586,0.0004409969,0.001813869,0.001502872,0.0006542341,0.004015435,0.001844676,0.003114311,0.005135349],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00144906,"about_ca_system_score_gemma":0.001316939,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006337006,"about_ca_topic_score_gemma":0.01428678,"domain_scores_codex":[0.9968746,0.001666036,0.0002225404,0.000853925,0.0002914026,0.00009143002],"domain_scores_gemma":[0.9868009,0.009019402,0.0005819158,0.001919951,0.001373178,0.0003045892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001366534,0.002315366,0.06082597,0.003394434,0.0006029955,0.0006955085,0.003123555,0.07085049,0.02249245,0.009141075,0.1324114,0.6927803],"study_design_scores_gemma":[0.0003295262,0.0007555957,0.02340589,0.00032725,0.0002600733,0.0006765441,0.0009786909,0.8290964,0.02397893,0.0310219,0.08902275,0.0001464381],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2221991,0.006494463,0.6018617,0.004943758,0.0007422758,0.003065375,0.08773612,0.05824472,0.01471258],"genre_scores_gemma":[0.3923664,0.0007003239,0.4157056,0.00116739,0.000195863,0.002426306,0.1812088,0.0008799477,0.005349403],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006337006,"threshold_uncertainty_score":0.01808786,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02239636680911099,"score_gpt":0.2829373594837503,"score_spread":0.2605409926746393,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}