{"id":"W6910477211","doi":"10.48448/mccv-hq07","title":"Leveraging LLMs for Synthesizing Training Data Across Many Languages in Multilingual Dense Retrieval","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Training set; Training (meteorology); Construct (python library); Language model; Data retrieval; Multilingualism; Labeled data; Document retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.007820487,0.0006930172,0.0007836767,0.001608659,0.0003002746,0.000894284,0.003792522,0.0003749706,0.0001761551],"category_scores_gemma":[0.004212447,0.0006724125,0.0001011418,0.002184764,0.001565888,0.0005534895,0.001730228,0.0008818399,0.001104746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005611902,"about_ca_system_score_gemma":0.001309998,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006533638,"about_ca_topic_score_gemma":0.002965517,"domain_scores_codex":[0.9933037,0.00009142978,0.0007329024,0.002615191,0.001437803,0.001818958],"domain_scores_gemma":[0.9966806,0.000513298,0.0004018684,0.002030624,0.0001206221,0.0002529843],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008145076,0.001057353,0.000869853,0.004858109,0.0008437558,0.004535343,0.1285315,0.001338496,0.1667327,0.003517261,0.2869953,0.3999058],"study_design_scores_gemma":[0.003558471,0.0001976387,0.000139425,0.01008904,0.0003805194,0.0003805699,0.08903802,0.4601132,0.007070747,0.001186875,0.4234283,0.004417186],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2543531,0.06621618,0.03582032,0.003737353,0.0268142,0.02399995,0.0650469,0.02391733,0.5000947],"genre_scores_gemma":[0.6521655,0.00007921851,0.1071852,0.0004362106,0.003585503,0.00006413521,0.001132883,0.005142804,0.2302085],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4587747,"threshold_uncertainty_score":0.999673,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1674384594459624,"score_gpt":0.4350408651803633,"score_spread":0.2676024057344009,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}