{"id":"W3195010973","doi":"10.1145/3459637.3482011","title":"MS MARCO Chameleons: Challenging the MS MARCO Leaderboard with Extremely Obstinate Queries","year":2021,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University; Microsoft (Canada); University of Waterloo","funders":"","keywords":"Computer science; Set (abstract data type); Task (project management); Information retrieval; Perspective (graphical); Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002532748,0.001867154,0.001588721,0.002457549,0.001736212,0.002357007,0.001915203,0.002598504,0.007369923],"category_scores_gemma":[0.01310502,0.0002998901,0.0008838944,0.002421192,0.001169484,0.003568735,0.00212921,0.002276623,0.0041803],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001700138,"about_ca_system_score_gemma":0.001968276,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02815417,"about_ca_topic_score_gemma":0.06114214,"domain_scores_codex":[0.9969521,0.0007720973,0.000228308,0.000689915,0.001040267,0.0003172921],"domain_scores_gemma":[0.9940594,0.002790802,0.0004119436,0.001174278,0.0008610043,0.000702543],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004493756,0.001597877,0.01971878,0.003035524,0.0003829596,0.002577178,0.00132742,0.02766301,0.01643337,0.007591695,0.6358856,0.2792929],"study_design_scores_gemma":[0.001800862,0.003518289,0.0408888,0.0005435601,0.0002934483,0.005225515,0.007931583,0.3479013,0.05083969,0.01535406,0.5251489,0.0005540647],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7731647,0.01650948,0.0223208,0.009595353,0.002833804,0.001578167,0.0985291,0.01891243,0.05655602],"genre_scores_gemma":[0.643014,0.002283831,0.08234582,0.002761462,0.001283907,0.0005848436,0.2350793,0.001356849,0.03128986],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02815417,"threshold_uncertainty_score":0.05598056,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03365566842748198,"score_gpt":0.2226192605793842,"score_spread":0.1889635921519022,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}