{"id":"W4389518635","doi":"10.18653/v1/2023.findings-emnlp.776","title":"Detrimental Contexts in Open-Domain Question Answering","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Computer science; Question answering; Leverage (statistics); Pipeline (software); Open domain; Context (archaeology); Artificial intelligence; Information retrieval; Natural language processing; Matching (statistics); Machine learning; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004543874,0.001138574,0.0009958365,0.001164956,0.001224758,0.001719443,0.001369346,0.001982445,0.001290919],"category_scores_gemma":[0.0225749,0.0007962885,0.0008812207,0.0007603293,0.001297033,0.004247567,0.00278617,0.002527478,0.0008039537],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008135197,"about_ca_system_score_gemma":0.0009326276,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00441245,"about_ca_topic_score_gemma":0.008500522,"domain_scores_codex":[0.9966992,0.001892395,0.0001438083,0.0007916483,0.0002969142,0.0001761109],"domain_scores_gemma":[0.9872801,0.01016333,0.0005480044,0.001090348,0.0006405113,0.0002777403],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002136987,0.0006319404,0.0359892,0.001395654,0.0004438486,0.001525239,0.005734622,0.5105265,0.02901641,0.0365295,0.01083215,0.3652379],"study_design_scores_gemma":[0.00005397746,0.0002648331,0.003686402,0.0001176514,0.0001441688,0.0003658833,0.000476027,0.9254527,0.009163505,0.05365146,0.006564648,0.00005865873],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3926543,0.01128348,0.5820554,0.001806119,0.000221051,0.0002537096,0.0006807885,0.005552718,0.005492446],"genre_scores_gemma":[0.9178931,0.0009694737,0.07770699,0.0004475776,0.000177911,0.0001441844,0.0008390996,0.000246985,0.001574757],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004543874,"threshold_uncertainty_score":0.02403057,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03086295426828448,"score_gpt":0.3123762498647927,"score_spread":0.2815132955965082,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}