{"id":"W4384828687","doi":"10.1145/3539618.3591902","title":"SPRINT: A Unified Toolkit for Evaluating and Demystifying Zero-shot Neural Sparse Retrieval","year":2023,"lang":"en","type":"article","venue":"","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; Deutsche Forschungsgemeinschaft; Compute Canada","keywords":"Computer science; Information retrieval; Benchmark (surveying); Question answering; Artificial intelligence; Weighting; Python (programming language); Language model; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00702439,0.003774155,0.002059179,0.004511799,0.001026884,0.003302562,0.006999409,0.002165701,0.01793527],"category_scores_gemma":[0.02372904,0.001305738,0.002283452,0.002989745,0.001271065,0.005811732,0.005625881,0.003671392,0.01924241],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002289169,"about_ca_system_score_gemma":0.004196605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01948258,"about_ca_topic_score_gemma":0.03130217,"domain_scores_codex":[0.9948851,0.001331017,0.0006025544,0.0008502458,0.00194658,0.0003844059],"domain_scores_gemma":[0.9940255,0.002091537,0.0003625797,0.00174718,0.001414262,0.0003589168],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001943519,0.0008738277,0.004206425,0.003911362,0.0009898192,0.0004985427,0.0003469388,0.083544,0.01501195,0.007362467,0.4613844,0.4199267],"study_design_scores_gemma":[0.0006334888,0.001046433,0.002835345,0.0002523454,0.0001911356,0.0006057309,0.0001773191,0.885012,0.03149815,0.01681225,0.06064668,0.0002890865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.02357071,0.003175616,0.2922195,0.000637412,0.0007384454,0.001184117,0.0303135,0.6357872,0.01237346],"genre_scores_gemma":[0.1839446,0.002723899,0.5628893,0.00161131,0.0002646674,0.003214579,0.1934384,0.03416813,0.01774502],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01948258,"threshold_uncertainty_score":0.05999947,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1771123195196433,"score_gpt":0.3614235738250197,"score_spread":0.1843112543053763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}