{"id":"W4200635123","doi":"10.18653/v1/2022.naacl-main.168","title":"GPL: Generative Pseudo Labeling for Unsupervised Domain Adaptation of Dense Retrieval","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","topic":"Topic Modeling","field":"Computer Science","cited_by":69,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Deutsche Forschungsgemeinschaft","keywords":"Generative grammar; Computer science; Domain (mathematical analysis); Natural language processing; Artificial intelligence; Adaptation (eye); Domain adaptation; Mathematics; Psychology; Neuroscience","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002229995,0.00161355,0.001861393,0.002169705,0.001208644,0.002236912,0.004828881,0.002815567,0.01605406],"category_scores_gemma":[0.006543548,0.001589802,0.002024218,0.002628008,0.001330216,0.004006315,0.005076965,0.003219398,0.01513595],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001376708,"about_ca_system_score_gemma":0.001873253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01199869,"about_ca_topic_score_gemma":0.02420058,"domain_scores_codex":[0.9984508,0.0005295406,0.00006834036,0.0004533828,0.0003448781,0.0001531047],"domain_scores_gemma":[0.9976262,0.0008956933,0.0000877597,0.0009163746,0.0003576635,0.000116268],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000510035,0.0002600571,0.0008468848,0.0003692035,0.0001963368,0.0002277142,0.0003116797,0.111398,0.01121998,0.02425112,0.1098229,0.7405862],"study_design_scores_gemma":[0.00007727899,0.00004517292,0.000158255,0.00002153402,0.00002568621,0.00009861207,0.00004695216,0.9491575,0.00461115,0.03448422,0.01123946,0.00003410686],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001569802,0.0002628575,0.974918,0.0001177334,0.00007475739,0.00007363468,0.0006333081,0.02133956,0.001010363],"genre_scores_gemma":[0.07191544,0.0003785165,0.9042158,0.0006378339,0.0001712604,0.0004557137,0.008243828,0.006243865,0.007737737],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01605406,"threshold_uncertainty_score":0.05370617,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02453907150408344,"score_gpt":0.2515881201371103,"score_spread":0.2270490486330269,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}