{"id":"W2912817604","doi":"10.18653/v1/n19-4013","title":"End-to-End Open-Domain Question Answering with","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":368,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Question answering; Computer science; Open domain; Benchmark (surveying); Information retrieval; Reading (process); End-to-end principle; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Reading comprehension; Language model; Open source; Programming language; Linguistics; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004099092,0.002476443,0.001933936,0.002387334,0.001346115,0.003964356,0.003276713,0.003357423,0.0267848],"category_scores_gemma":[0.01264427,0.0009680649,0.001991447,0.001846122,0.0007920894,0.007636747,0.009914985,0.003138681,0.03433625],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007565882,"about_ca_system_score_gemma":0.00105096,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002672359,"about_ca_topic_score_gemma":0.006292839,"domain_scores_codex":[0.9950918,0.001703465,0.0003339094,0.001730843,0.0008010856,0.0003389],"domain_scores_gemma":[0.9934837,0.003339895,0.0001484884,0.001863637,0.0008465429,0.0003177177],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002543512,0.001018007,0.002410186,0.001192325,0.0005067444,0.0009380208,0.001902396,0.007788813,0.02027892,0.01885165,0.3589087,0.5836607],"study_design_scores_gemma":[0.0005034279,0.0003750497,0.001926737,0.0002125182,0.0002614712,0.0006683469,0.001901749,0.5731444,0.04403492,0.1470115,0.2297667,0.0001931923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01717217,0.001689264,0.7188481,0.001406013,0.000807245,0.0007644944,0.01472607,0.2309439,0.01364269],"genre_scores_gemma":[0.1843385,0.0005287297,0.727059,0.001332212,0.0004135293,0.001071957,0.05978991,0.005131938,0.02033425],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0267848,"threshold_uncertainty_score":0.08960408,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02837405652156753,"score_gpt":0.2801999427192826,"score_spread":0.2518258861977151,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}