{"id":"W6910418724","doi":"10.48448/csby-n907","title":"Mr. TyDi: A Multi-lingual Benchmark for Dense Retrieval","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Benchmark (surveying); Relevance (law); Ranking (information retrieval); Representation (politics); Resource (disambiguation); Learning to rank","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005880358,0.003728736,0.002299367,0.008104828,0.00280726,0.0042153,0.00501309,0.003292266,0.02233187],"category_scores_gemma":[0.01939213,0.0007699347,0.002307639,0.008903781,0.001628435,0.006100279,0.00597463,0.002541719,0.02816583],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002867966,"about_ca_system_score_gemma":0.003757693,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.05151963,"about_ca_topic_score_gemma":0.08449682,"domain_scores_codex":[0.9923075,0.002018121,0.0009739226,0.001436901,0.00248244,0.0007810038],"domain_scores_gemma":[0.9916379,0.001890303,0.000410932,0.002935785,0.002441229,0.0006838946],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008570059,0.000878382,0.00299162,0.002952663,0.0004045664,0.0003491458,0.000280875,0.007309209,0.006286209,0.003382578,0.8341921,0.1401158],"study_design_scores_gemma":[0.001403221,0.001741128,0.01869397,0.0007947348,0.0004590065,0.003162998,0.00205041,0.1263749,0.02997411,0.01305319,0.8016406,0.0006518969],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.1108734,0.01774528,0.04731552,0.003762897,0.003882034,0.003090971,0.6707342,0.05731559,0.08528018],"genre_scores_gemma":[0.05308247,0.001160349,0.0459165,0.000784207,0.0002695533,0.0008017802,0.8825693,0.001790237,0.01362566],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.05151963,"threshold_uncertainty_score":0.1024395,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05996990092066781,"score_gpt":0.3599443139427626,"score_spread":0.2999744130220948,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}