{"id":"W4385569780","doi":"10.18653/v1/2023.acl-long.307","title":"Evaluating Open-Domain Question Answering in the Era of Large Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":99,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Matching (statistics); Benchmark (surveying); Question answering; Open domain; Natural language processing; Domain (mathematical analysis); Artificial intelligence; Language model; Information retrieval; Machine learning; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01595927,0.001906341,0.001433956,0.001866915,0.0009443567,0.003375608,0.003042781,0.003093237,0.003356559],"category_scores_gemma":[0.06972561,0.0006116046,0.001346534,0.001112759,0.001457657,0.006397056,0.004011751,0.003825752,0.002240415],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002278064,"about_ca_system_score_gemma":0.001903163,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00746053,"about_ca_topic_score_gemma":0.01061652,"domain_scores_codex":[0.9834009,0.0103824,0.000706792,0.00243803,0.002691573,0.000380287],"domain_scores_gemma":[0.9555085,0.03268451,0.001137496,0.00547475,0.004070427,0.001124318],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00164161,0.001457948,0.02413613,0.003061755,0.0009552458,0.0005852648,0.00299081,0.3154905,0.0256848,0.02413087,0.04191532,0.5579497],"study_design_scores_gemma":[0.00009332826,0.00039737,0.002963431,0.0001199156,0.00009079894,0.0001825627,0.0004747044,0.9539455,0.009180645,0.02221101,0.0102848,0.00005605655],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3941059,0.00784396,0.5459175,0.004311198,0.0006794541,0.0007620757,0.004411997,0.0290997,0.0128683],"genre_scores_gemma":[0.8275643,0.0005027548,0.1586239,0.001138926,0.0001789546,0.0002918067,0.008642368,0.0007809546,0.002275992],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01595927,"threshold_uncertainty_score":0.08440173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07837739210868339,"score_gpt":0.3860292843682005,"score_spread":0.3076518922595171,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}