{"id":"W4416673024","doi":"10.1007/s10462-025-11421-5","title":"Exploring unanswerability in machine reading comprehension: approaches, benchmarks, and open challenges","year":2025,"lang":"en","type":"article","venue":"Artificial Intelligence Review","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University; Ted Rogers Centre for Heart Research; University of Guelph","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Key (lock); Comprehension; Reading (process); Work (physics); Reading comprehension","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02399274,0.002304655,0.002374422,0.01101005,0.001184269,0.008471979,0.003612074,0.002512349,0.004197866],"category_scores_gemma":[0.1184601,0.0009390438,0.001788096,0.009437413,0.002678751,0.018632,0.004488919,0.004552603,0.002615258],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002458365,"about_ca_system_score_gemma":0.00375355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007243303,"about_ca_topic_score_gemma":0.008748617,"domain_scores_codex":[0.9787433,0.01260182,0.001553676,0.003124537,0.003559481,0.0004171673],"domain_scores_gemma":[0.8143226,0.1580751,0.005269864,0.008138278,0.01286198,0.001332164],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002695665,0.0003367141,0.03046492,0.01123346,0.0007712789,0.000167359,0.009716607,0.01049815,0.003356868,0.02014232,0.02462258,0.8884202],"study_design_scores_gemma":[0.0001601429,0.0009032781,0.1067776,0.01245027,0.001342252,0.001412508,0.02774475,0.2056783,0.01455938,0.378007,0.2502317,0.0007328111],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.1240524,0.3612666,0.4343042,0.02830788,0.001051628,0.0009868178,0.007569443,0.007834865,0.03462609],"genre_scores_gemma":[0.5911626,0.08470377,0.295208,0.002764307,0.001786289,0.001592332,0.01764116,0.001239011,0.003902572],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.02399274,"threshold_uncertainty_score":0.1268873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5294704198234306,"score_gpt":0.3733258316109949,"score_spread":0.1561445882124357,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}