{"id":"W4284682639","doi":"10.1145/3477495.3531749","title":"Document Expansion Baselines and Learned Sparse Lexical Representations for MS MARCO V1 and V2","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","topic":"Topic Modeling","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund; Compute Canada","keywords":"Computer science; Weighting; Ranking (information retrieval); Information retrieval; Natural language processing; Question answering; Artificial intelligence; Sequence (biology); Term (time); Language model; Artificial neural network","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00772287,0.002449933,0.001534681,0.00425943,0.001688837,0.002584245,0.004069519,0.001987484,0.00724272],"category_scores_gemma":[0.02344943,0.0007176053,0.001585168,0.003422052,0.001053906,0.004640449,0.003160348,0.0032781,0.005814982],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002732092,"about_ca_system_score_gemma":0.002216835,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02789957,"about_ca_topic_score_gemma":0.04361722,"domain_scores_codex":[0.9949443,0.001712237,0.0003936671,0.001173496,0.001365131,0.0004111512],"domain_scores_gemma":[0.9931986,0.002066365,0.0003017843,0.002379315,0.001702224,0.0003517068],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003159715,0.00249938,0.009409162,0.001910427,0.0007441691,0.000442362,0.0004646883,0.1081345,0.02580998,0.007744473,0.2554309,0.5842503],"study_design_scores_gemma":[0.001047815,0.002086426,0.01199925,0.0002341421,0.0003681083,0.0005968324,0.0006397059,0.8538268,0.04859704,0.01052175,0.06980337,0.0002787798],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.4710049,0.01512766,0.2673469,0.003781539,0.003184319,0.004031927,0.07434092,0.09053209,0.07064975],"genre_scores_gemma":[0.4689226,0.001648336,0.3058983,0.001295903,0.0005076048,0.003468305,0.1935888,0.004401187,0.02026894],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02789957,"threshold_uncertainty_score":0.05547434,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1205216399447767,"score_gpt":0.3619357305030352,"score_spread":0.2414140905582585,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}