{"id":"W2940263831","doi":"10.48550/arxiv.1904.09171","title":"Critically Examining the \"Neural Hype\": Weak Baselines and the Additivity of Effectiveness Gains from Neural Ranking Models","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Skepticism; Ranking (information retrieval); Computer science; Artificial neural network; Post hoc; Artificial intelligence; Additive function; Deep neural networks; Machine learning; Epistemology; Mathematics; Philosophy; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07587998,0.002061319,0.002581986,0.004064695,0.001119398,0.005175572,0.002977035,0.001985121,0.004017774],"category_scores_gemma":[0.1718257,0.0007065274,0.001504588,0.002994447,0.003318268,0.01327034,0.003209798,0.006622699,0.00175167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00196316,"about_ca_system_score_gemma":0.001601253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002270573,"about_ca_topic_score_gemma":0.004341376,"domain_scores_codex":[0.9477515,0.0387299,0.002045624,0.003061867,0.007877219,0.0005339237],"domain_scores_gemma":[0.8247496,0.1472914,0.004148928,0.01471873,0.008228357,0.0008630356],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00416438,0.0006105772,0.01391562,0.006906126,0.009943246,0.0001605675,0.0007456404,0.06241634,0.00404087,0.1224095,0.04175715,0.73293],"study_design_scores_gemma":[0.001504367,0.005763822,0.0174706,0.003660448,0.008852546,0.0005124704,0.0009371135,0.3451596,0.01730185,0.5383293,0.05995678,0.0005510719],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1703522,0.2280092,0.401588,0.08864591,0.004512569,0.0007925386,0.003707597,0.003220182,0.09917193],"genre_scores_gemma":[0.8810022,0.01095435,0.09115081,0.007265763,0.002757388,0.0003851337,0.00133269,0.0005552337,0.004596408],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.92412,"threshold_uncertainty_score":0.4012964,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1098757386317768,"score_gpt":0.2078641927835763,"score_spread":0.09798845415179945,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}