{"id":"W4404782892","doi":"10.18653/v1/2024.emnlp-main.250","title":"PromptReps: Prompting Large Language Models to Generate Dense and Sparse Representations for Zero-Shot Document Retrieval","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Zero (linguistics); Computer science; Shot (pellet); Natural language processing; Artificial intelligence; Language model; Information retrieval; Linguistics; Materials science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001416388,0.001464024,0.001221063,0.001097132,0.0004534886,0.001142046,0.002306383,0.001303279,0.004308442],"category_scores_gemma":[0.005600112,0.0005506129,0.001087123,0.0009707942,0.0007413907,0.003831504,0.002312667,0.001967804,0.003952468],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007937212,"about_ca_system_score_gemma":0.00126902,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004099715,"about_ca_topic_score_gemma":0.008311663,"domain_scores_codex":[0.999047,0.0003239305,0.00006068118,0.0002455655,0.0002312456,0.00009151067],"domain_scores_gemma":[0.9982698,0.0007022652,0.0001241436,0.000522085,0.0002854263,0.00009633419],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006339011,0.0004870893,0.001218778,0.0005707678,0.0001868203,0.0002860805,0.0003850613,0.08603788,0.04548468,0.01414429,0.03530804,0.8152566],"study_design_scores_gemma":[0.00008684145,0.0002212614,0.0003006808,0.00001725678,0.00003392093,0.000158816,0.00008739586,0.9634942,0.01804151,0.01148031,0.00602794,0.00004977866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01878803,0.0004940772,0.9639849,0.0002136802,0.0001159365,0.0001679404,0.0004815003,0.01448847,0.001265515],"genre_scores_gemma":[0.296905,0.0005335297,0.6839032,0.0006630901,0.0002587124,0.0006405446,0.00479703,0.001241036,0.01105773],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004308442,"threshold_uncertainty_score":0.01441312,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06476820452331428,"score_gpt":0.3327538031602711,"score_spread":0.2679855986369569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}