{"id":"W2889787757","doi":"10.18653/v1/d18-1259","title":"HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering","year":2018,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1582,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Office of Naval Research; Defense Advanced Research Projects Agency; Université de Montréal; Nvidia; National Science Foundation","keywords":"Zhàng; Question answering; Computer science; Artificial intelligence; Natural language processing; Information retrieval; History; China; Archaeology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001910017,0.003035023,0.00185651,0.004365229,0.001779899,0.001912627,0.004397795,0.005018116,0.01366162],"category_scores_gemma":[0.01088954,0.0006876175,0.002027706,0.003273329,0.0007104411,0.003933081,0.003996583,0.002343436,0.01112113],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001890202,"about_ca_system_score_gemma":0.002338488,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03047901,"about_ca_topic_score_gemma":0.06105387,"domain_scores_codex":[0.9976252,0.0006557292,0.0003007678,0.000658599,0.0005466761,0.0002130992],"domain_scores_gemma":[0.9956402,0.001926495,0.0002421895,0.0009267486,0.000837849,0.0004264489],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006742972,0.0004935014,0.005785217,0.002421837,0.0002927175,0.0005369638,0.0004589832,0.003229815,0.003928535,0.002575256,0.9368536,0.04274923],"study_design_scores_gemma":[0.001821499,0.0006766857,0.03553101,0.0008116472,0.0004916476,0.001493544,0.002508841,0.09806535,0.01014418,0.02200793,0.8260829,0.0003647891],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.03094041,0.004568587,0.01204815,0.002165876,0.0004489257,0.00085351,0.9264646,0.01579567,0.006714235],"genre_scores_gemma":[0.02339396,0.000307163,0.0136263,0.0004085373,0.00007070362,0.0005091894,0.9589603,0.0002326344,0.002491262],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03047901,"threshold_uncertainty_score":0.0606032,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06108907148323083,"score_gpt":0.3121959901327288,"score_spread":0.2511069186494979,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}