{"id":"W6910418724","doi":"10.48448/csby-n907","title":"Mr. TyDi: A Multi-lingual Benchmark for Dense Retrieval","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Benchmark (surveying); Relevance (law); Ranking (information retrieval); Representation (politics); Resource (disambiguation); Learning to rank","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002304625,0.0007347667,0.0007976608,0.001503853,0.0004943822,0.0004184758,0.001779192,0.0005561859,0.004062335],"category_scores_gemma":[0.003468432,0.0007132924,0.0002512747,0.00306673,0.002291561,0.0001814336,0.0005383826,0.000590931,0.001182622],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006375675,"about_ca_system_score_gemma":0.003885953,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000254616,"about_ca_topic_score_gemma":0.001950126,"domain_scores_codex":[0.9939637,0.0001011164,0.0006442463,0.002011306,0.001794778,0.001484858],"domain_scores_gemma":[0.9963632,0.0002632097,0.0006317404,0.001434755,0.0007876487,0.0005194348],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002856568,0.001417182,0.0003359041,0.0005657472,0.0003082003,0.0003275774,0.0009291552,0.0001452035,0.1053881,0.003640341,0.875654,0.01100293],"study_design_scores_gemma":[0.00322449,0.0003136998,0.00008565423,0.0008628473,0.0002126256,0.000119963,0.0005294327,0.02947498,0.007333067,0.0003909923,0.9555371,0.001915156],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.01171831,0.02142661,0.04412244,0.001340131,0.02410804,0.01456752,0.009996026,0.006225347,0.8664956],"genre_scores_gemma":[0.007014043,0.00007942302,0.2196734,0.0005281945,0.002559691,0.00006432537,0.001160619,0.002184357,0.766736],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1755509,"threshold_uncertainty_score":0.999595,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05996990092066781,"score_gpt":0.3599443139427626,"score_spread":0.2999744130220948,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}