{"id":"W4416712105","doi":"10.48550/arxiv.2509.02558","title":"Lighting the Way for BRIGHT: Reproducible Baselines with Anserini, Pyserini, and RankLLM","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Natural Sciences and Engineering Research Council of Canada; Ministry of Science and ICT, South Korea","keywords":"Benchmark (surveying); Construct (python library); Range (aeronautics); Query expansion; Baseline (sea); Search engine; Question answering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001180115,0.0002967755,0.0003282471,0.0001365481,0.0004342708,0.0003583281,0.001155506,0.0001496151,0.00001227597],"category_scores_gemma":[0.0001981736,0.0001773432,0.0001123113,0.0003564732,0.0001032851,0.0003982998,0.001235754,0.0004622019,0.00001636075],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003191261,"about_ca_system_score_gemma":0.0002805844,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009370063,"about_ca_topic_score_gemma":0.00001970298,"domain_scores_codex":[0.9979591,0.00007072507,0.0004530152,0.0008178165,0.0003113697,0.0003879969],"domain_scores_gemma":[0.9974911,0.0002516454,0.0002248246,0.001508709,0.0004303211,0.00009345153],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001063686,0.0006134271,0.6550457,0.005358707,0.0008501187,0.0001248909,0.02048966,0.003059573,0.00167603,0.04282549,0.0158224,0.2530703],"study_design_scores_gemma":[0.00401633,0.0008385267,0.5144541,0.001707425,0.0003172714,0.0001252815,0.0005161226,0.08569686,0.07093727,0.003106982,0.3156033,0.002680534],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8833434,0.0003950115,0.1030714,0.008629331,0.0009909174,0.001613412,0.00004726732,0.0003537504,0.001555484],"genre_scores_gemma":[0.9577322,0.0003247201,0.02936635,0.001828361,0.0005650622,0.0005961863,0.00006171467,0.00003419728,0.00949119],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.299781,"threshold_uncertainty_score":0.7231844,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05225261252582453,"score_gpt":0.2928951518894453,"score_spread":0.2406425393636207,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}