{"id":"W4380687254","doi":"10.48550/arxiv.2306.07471","title":"Resources for Brewing BEIR: Reproducible Reference Models and an Official Leaderboard","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund","keywords":"Computer science; Sophistication; Popularity; Artificial intelligence; Benchmark (surveying); CLARITY; Optimal distinctiveness theory; Publication; Information retrieval; Data science; Political science; Geography; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006282984,0.0002881378,0.0003562472,0.0002845874,0.0002646843,0.0002399335,0.001695635,0.0003016145,0.000001892005],"category_scores_gemma":[0.00006686888,0.0003495839,0.0001079498,0.0003122688,0.0000844153,0.0006839865,0.002186542,0.0004601359,0.000009929396],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009472821,"about_ca_system_score_gemma":0.000153781,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004643788,"about_ca_topic_score_gemma":0.0001570413,"domain_scores_codex":[0.9967583,0.0001009197,0.0002436217,0.002361263,0.0001116631,0.0004242768],"domain_scores_gemma":[0.9973358,0.0001254079,0.0001929142,0.002018156,0.0001611309,0.000166611],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003521714,0.00003540925,0.0002757693,0.0001423253,0.00004384412,0.00005077713,0.00168962,0.8303139,0.00004699161,0.164896,0.0001317459,0.002338413],"study_design_scores_gemma":[0.0002264981,0.00005050952,0.000133985,0.00008610269,0.00003283254,0.000001788012,0.0001767202,0.8408533,0.00004792418,0.1577154,0.0003373711,0.0003375753],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2943044,0.00005695774,0.70382,0.0002076711,0.0003579657,0.0003045634,0.00001377657,0.0004703813,0.0004642367],"genre_scores_gemma":[0.9792103,0.00008598092,0.01846774,0.00007243227,0.0002635422,0.000002962436,0.00001149481,0.00003070876,0.001854852],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6853523,"threshold_uncertainty_score":0.9998956,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3203948243550147,"score_gpt":0.2444573021915565,"score_spread":0.07593752216345812,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}