{"id":"W4389518861","doi":"10.18653/v1/2023.findings-emnlp.616","title":"Knowledge Corpus Error in Question Answering","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Pipeline (software); Computer science; Context (archaeology); Natural language processing; String (physics); Artificial intelligence; Space (punctuation); Question answering; Domain (mathematical analysis); Information retrieval; History; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01914785,0.0008220784,0.001367545,0.001975678,0.001706501,0.002876962,0.001833341,0.002689633,0.002332343],"category_scores_gemma":[0.1560918,0.0006889832,0.0006955704,0.002553293,0.002637687,0.00606366,0.004434144,0.002497535,0.001136487],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001789309,"about_ca_system_score_gemma":0.001840775,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005228142,"about_ca_topic_score_gemma":0.003293485,"domain_scores_codex":[0.9668444,0.01973366,0.001955958,0.005270053,0.005541323,0.0006546209],"domain_scores_gemma":[0.8003322,0.1685341,0.005271078,0.01528493,0.009794607,0.0007831314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002047808,0.0006624845,0.03890014,0.004211879,0.0005380386,0.002116336,0.01497716,0.1035584,0.02859482,0.06612428,0.04155021,0.6967185],"study_design_scores_gemma":[0.0003044251,0.001041955,0.01907667,0.001492444,0.0006089928,0.004743411,0.00468837,0.5724072,0.1155616,0.1629079,0.1168577,0.0003090776],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3365144,0.01241481,0.6157781,0.006686293,0.001042375,0.0006393422,0.001902508,0.01035817,0.01466408],"genre_scores_gemma":[0.85131,0.001235865,0.1372571,0.001585313,0.0003272223,0.0003491775,0.003223784,0.001243514,0.00346812],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01914785,"threshold_uncertainty_score":0.1012648,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05140478904870115,"score_gpt":0.3137755038093568,"score_spread":0.2623707147606556,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}