{"id":"W2940263831","doi":"10.48550/arxiv.1904.09171","title":"Critically Examining the \"Neural Hype\": Weak Baselines and the Additivity of Effectiveness Gains from Neural Ranking Models","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Skepticism; Ranking (information retrieval); Computer science; Artificial neural network; Post hoc; Artificial intelligence; Additive function; Deep neural networks; Machine learning; Epistemology; Mathematics; Philosophy; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001105575,0.0003213204,0.0005289746,0.000100737,0.0002112265,0.0001515501,0.001880024,0.0001834285,0.000005040974],"category_scores_gemma":[0.0002023126,0.000232117,0.0001896849,0.0002313,0.0003893852,0.0004663754,0.002893455,0.0007153115,0.000002747095],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000557532,"about_ca_system_score_gemma":0.0001084806,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006159135,"about_ca_topic_score_gemma":0.00003242207,"domain_scores_codex":[0.997059,0.001162237,0.0002623083,0.001020718,0.0001738234,0.0003218872],"domain_scores_gemma":[0.9942629,0.00372173,0.0002324364,0.001462759,0.0002464039,0.00007379667],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001220753,0.00003238098,0.002596203,0.0001320494,0.0001002947,0.00003946797,0.0005580258,0.8630971,0.0001146142,0.1310179,0.000004964519,0.002184913],"study_design_scores_gemma":[0.0009810063,0.0000206085,0.004257977,0.0001897555,0.00009683385,0.000003903569,0.0001090675,0.9445677,0.00007236652,0.04945888,0.000004997979,0.0002369209],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4942086,0.0001274663,0.5044992,0.0001244965,0.0003620251,0.0002986944,0.00002280209,0.00005257163,0.0003041394],"genre_scores_gemma":[0.9985968,0.00006559742,0.0009947604,0.0001585036,0.0001100229,0.000002634613,0.000008178683,0.0000167481,0.00004674506],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5043882,"threshold_uncertainty_score":0.9465455,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1098757386317768,"score_gpt":0.2078641927835763,"score_spread":0.09798845415179945,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}