{"id":"W4396820721","doi":"10.48550/arxiv.2404.18424","title":"PromptReps: Prompting Large Language Models to Generate Dense and Sparse Representations for Zero-Shot Document Retrieval","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo","keywords":"Zero (linguistics); Computer science; Shot (pellet); Natural language processing; Artificial intelligence; Information retrieval; Linguistics; Philosophy; Chemistry","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005124928,0.0003064316,0.0003092453,0.000304889,0.0001987411,0.0003823438,0.0008526015,0.0001780736,0.000006941906],"category_scores_gemma":[0.00005025057,0.0003472133,0.0001456834,0.0004091004,0.00002868032,0.0003278877,0.003514959,0.0003993355,0.00002454901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000193857,"about_ca_system_score_gemma":0.0002000363,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001488912,"about_ca_topic_score_gemma":0.00005778552,"domain_scores_codex":[0.9973518,0.00008933456,0.0002843562,0.001678384,0.0001309604,0.0004651692],"domain_scores_gemma":[0.9981967,0.00008654411,0.000143788,0.001180234,0.0001610926,0.0002316978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005596986,0.00006083753,0.0001089286,0.0003423954,0.0001431321,0.0003996103,0.006385092,0.6894807,0.000654999,0.3012898,0.0004234881,0.000655105],"study_design_scores_gemma":[0.0003495369,0.0000366512,0.00001614616,0.0001447987,0.00007796663,0.00000729465,0.0002244781,0.8704498,0.000698941,0.127557,0.00009195632,0.0003454192],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4360956,0.0001184493,0.561905,0.0002172245,0.0003108236,0.000807756,0.00003259251,0.000196504,0.0003160552],"genre_scores_gemma":[0.9561689,0.00003515659,0.03988521,0.0001278253,0.0001238243,0.00001087728,0.00002058118,0.00002937499,0.003598196],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5220198,"threshold_uncertainty_score":0.999898,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1266048190329702,"score_gpt":0.2514864844701213,"score_spread":0.1248816654371512,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}