{"id":"W4385732413","doi":"10.1109/tkde.2023.3303916","title":"XMQAs: Constructing Complex-Modified Question-Answering Dataset for Robust Question Understanding","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Knowledge and Data Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"Science and Technology Commission of Shanghai Municipality; National Natural Science Foundation of China","keywords":"Computer science; Question answering; Robustness (evolution); Construct (python library); Semantics (computer science); Simple (philosophy); Artificial intelligence; Machine learning; Information retrieval; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004721664,0.002148067,0.001097163,0.005414131,0.001327712,0.001945244,0.004168367,0.002875477,0.004294377],"category_scores_gemma":[0.01887224,0.0004930668,0.00236399,0.00342423,0.0008997871,0.004453781,0.004437661,0.003073293,0.003198414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001656036,"about_ca_system_score_gemma":0.002418405,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01293327,"about_ca_topic_score_gemma":0.01617249,"domain_scores_codex":[0.9949043,0.001672831,0.0007685109,0.001507215,0.0009143542,0.0002328267],"domain_scores_gemma":[0.9922562,0.002872504,0.0005330967,0.002080187,0.001807467,0.000450582],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001389541,0.002624623,0.03651801,0.006241007,0.0007466947,0.00105363,0.00402556,0.02866528,0.04338966,0.02144578,0.3684694,0.4854308],"study_design_scores_gemma":[0.0007635216,0.001145755,0.04400479,0.0005688471,0.0004089464,0.001312518,0.00373418,0.4841541,0.04673247,0.04040712,0.3763759,0.000391793],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1683941,0.004817163,0.4124681,0.003460807,0.0009519354,0.006481222,0.3227157,0.07076926,0.009941677],"genre_scores_gemma":[0.1141763,0.0005218745,0.365207,0.001109604,0.0001413396,0.003661773,0.5119044,0.000412882,0.002864808],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01293327,"threshold_uncertainty_score":0.02571595,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1505179283078189,"score_gpt":0.3267831472600723,"score_spread":0.1762652189522534,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}