{"id":"W3156861296","doi":"10.1145/3404835.3463034","title":"Significant Improvements over the State of the Art? A Case Study of the MS MARCO Document Ranking Leaderboard","year":2021,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada); University of Waterloo","funders":"","keywords":"Ranking (information retrieval); Metric (unit); Context (archaeology); Computer science; State (computer science); sort; Artificial intelligence; Information retrieval; Order (exchange); Machine learning; Algorithm; Engineering; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0143508,0.0008303591,0.001045241,0.002348239,0.00215821,0.003765316,0.001148717,0.001325884,0.002681391],"category_scores_gemma":[0.04247739,0.0002135978,0.0005753082,0.003473142,0.001636523,0.004110443,0.001551899,0.002072487,0.001572033],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00144804,"about_ca_system_score_gemma":0.0007938112,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0060429,"about_ca_topic_score_gemma":0.01216147,"domain_scores_codex":[0.9897702,0.006212505,0.0004048445,0.001042099,0.002129935,0.0004404926],"domain_scores_gemma":[0.9579327,0.03018628,0.001795277,0.004633567,0.003822064,0.001630085],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005446102,0.002148064,0.08959089,0.002260603,0.0007641343,0.002354987,0.01111343,0.05422395,0.01908841,0.03354703,0.1442326,0.6352299],"study_design_scores_gemma":[0.001086534,0.007595553,0.2487293,0.0007231378,0.0005787442,0.002849225,0.02035654,0.3653162,0.0471049,0.06572622,0.2392849,0.0006486862],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9294089,0.006197678,0.02264371,0.006645226,0.0003822792,0.0002500753,0.003199575,0.002858229,0.02841428],"genre_scores_gemma":[0.9555382,0.0005924931,0.03405662,0.0004092772,0.0003296387,0.00009657381,0.003168545,0.0006083824,0.005200249],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0143508,"threshold_uncertainty_score":0.07589519,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02407949558039306,"score_gpt":0.2593646241959989,"score_spread":0.2352851286156058,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}