{"id":"W3160270149","doi":"10.48550/arxiv.2105.04021","title":"MS MARCO: Benchmarking Ranking Models in the Large-Data Regime","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Microsoft (Canada)","funders":"","keywords":"Benchmarking; Clef; Computer science; Ranking (information retrieval); Field (mathematics); Data science; Best practice; Track (disk drive); Artificial intelligence; Information retrieval; Political science; Task (project management); Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":["open_science"],"category_scores_codex":[0.01171833,0.0003153431,0.000517085,0.000501022,0.000279588,0.0009747268,0.009074244,0.0002477906,0.0005682747],"category_scores_gemma":[0.0006700647,0.0002654622,0.0002215878,0.00159354,0.0001324014,0.001387658,0.01369201,0.0008982055,0.00009865966],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001236434,"about_ca_system_score_gemma":0.0001971987,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009805718,"about_ca_topic_score_gemma":0.003359728,"domain_scores_codex":[0.9945143,0.00150484,0.0005877637,0.002148507,0.0007269398,0.0005176184],"domain_scores_gemma":[0.9921086,0.001237716,0.0004691771,0.00591639,0.0001762599,0.00009181022],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001267065,0.0005544931,0.003730084,0.0001424571,0.0002387295,0.003140658,0.004791928,0.4446398,0.000003179176,0.4793936,0.05401755,0.009220836],"study_design_scores_gemma":[0.0006372753,0.00001481885,0.001769683,0.0002080318,0.0001108171,0.00000412749,0.008904108,0.656535,0.000001605568,0.2904474,0.04088265,0.0004845564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1851752,0.0004502095,0.7552918,0.001736931,0.001551464,0.0009682503,0.0005169199,0.0001026982,0.05420657],"genre_scores_gemma":[0.9952391,0.0004534873,0.0005675934,0.0008274651,0.0001271202,0.000001283427,0.0005288464,0.00001463427,0.00224052],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8100639,"threshold_uncertainty_score":0.9999797,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4743132750901781,"score_gpt":0.3087619711037293,"score_spread":0.1655513039864488,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}