{"id":"W4385574225","doi":"10.18653/v1/2022.findings-emnlp.31","title":"Lexicon-Enhanced Self-Supervised Training for Multilingual Dense Retrieval","year":2022,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Artificial intelligence; Relevance (law); Lexicon; Generator (circuit theory); Training set; Labeled data; Natural language processing; Machine learning; Information retrieval","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001799941,0.001240898,0.001387334,0.00156487,0.0006689076,0.0006740225,0.002361206,0.001148886,0.003108555],"category_scores_gemma":[0.005777094,0.0005972087,0.0009057861,0.001498758,0.0008484785,0.002659564,0.001620381,0.00177452,0.002891102],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007558067,"about_ca_system_score_gemma":0.001529588,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007168817,"about_ca_topic_score_gemma":0.01529708,"domain_scores_codex":[0.9988067,0.0004341151,0.00008804171,0.0003615068,0.0001960324,0.0001136521],"domain_scores_gemma":[0.997243,0.001293449,0.0001497436,0.0005993778,0.0006195084,0.00009489078],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005609181,0.0009486948,0.004464115,0.0004191724,0.0001852883,0.0002670153,0.0003537678,0.1094213,0.02769933,0.003446026,0.02555792,0.8266765],"study_design_scores_gemma":[0.00006554585,0.0001568864,0.0007183444,0.00001629209,0.00003344853,0.0001602343,0.00008439586,0.9820971,0.009451978,0.004075851,0.003110969,0.0000289484],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09239738,0.001344562,0.8826784,0.0003257326,0.0001149158,0.0003483134,0.001064171,0.01721428,0.004512372],"genre_scores_gemma":[0.6265485,0.0004243368,0.3522423,0.0008131978,0.0001849371,0.0006037871,0.01038027,0.0008827541,0.007919778],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007168817,"threshold_uncertainty_score":0.01425421,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03598622992703411,"score_gpt":0.3098080522508624,"score_spread":0.2738218223238283,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}