{"id":"W4298095693","doi":"10.48550/arxiv.2109.10739","title":"Predicting Efficiency/Effectiveness Trade-offs for Dense vs. Sparse\\n Retrieval Strategy Selection","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Leverage (statistics); Embedding; Classifier (UML); Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002859715,0.001012841,0.001196317,0.002094495,0.0003526053,0.001613252,0.0009994479,0.001339627,0.001943601],"category_scores_gemma":[0.01542845,0.0003656017,0.0006359233,0.00166785,0.0006057246,0.003131769,0.0006173468,0.0008028701,0.00193682],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008640054,"about_ca_system_score_gemma":0.001042609,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00551549,"about_ca_topic_score_gemma":0.008560661,"domain_scores_codex":[0.9980298,0.0004303114,0.0002633297,0.0004685799,0.0005337568,0.0002743685],"domain_scores_gemma":[0.9912548,0.006733705,0.0004376307,0.0007528916,0.0006131622,0.0002077471],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002961874,0.001408002,0.04823323,0.001771276,0.0004794248,0.000422767,0.0002667518,0.1808037,0.04174025,0.00393032,0.02183915,0.6961432],"study_design_scores_gemma":[0.0001592692,0.0009427351,0.007227033,0.00004878703,0.0002051106,0.0004824148,0.000193642,0.9657531,0.01815921,0.003814882,0.002966579,0.0000472445],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8487743,0.01555498,0.1166263,0.001360832,0.0001363561,0.0005380664,0.00231541,0.004643183,0.01005058],"genre_scores_gemma":[0.9278104,0.002386238,0.06292322,0.0002935738,0.0001295631,0.0001678418,0.003169143,0.0002757489,0.00284429],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.00551549,"threshold_uncertainty_score":0.01512378,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08046011134342035,"score_gpt":0.2106495405814781,"score_spread":0.1301894292380577,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}