{"id":"W3003709258","doi":"10.1029/2020ms002203","title":"WeatherBench: A Benchmark Data Set for Data‐Driven Weather Forecasting","year":2020,"lang":"en","type":"article","venue":"Journal of Advances in Modeling Earth Systems","topic":"Meteorological Phenomena and Simulations","field":"Earth and Planetary Sciences","cited_by":445,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Seventh Framework Programme; Deutsche Forschungsgemeinschaft","keywords":"Benchmark (surveying); Set (abstract data type); Data set; Weather forecasting; Baseline (sea); Numerical weather prediction; Deep learning; Code (set theory); Simple (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002454435,0.001796497,0.0007706577,0.0026656,0.0005386028,0.001347133,0.002508753,0.001350344,0.006012019],"category_scores_gemma":[0.00988953,0.0003763194,0.0009772484,0.003853747,0.0004155439,0.001981852,0.001051162,0.00147074,0.004147522],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008631854,"about_ca_system_score_gemma":0.001167988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0275567,"about_ca_topic_score_gemma":0.02020673,"domain_scores_codex":[0.9984999,0.0003284356,0.0002469982,0.000332418,0.000460545,0.0001317218],"domain_scores_gemma":[0.9957836,0.001396878,0.0002903227,0.001016629,0.001180973,0.0003316348],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001129854,0.0009084802,0.02551785,0.00142025,0.0005460313,0.0003744919,0.000120391,0.2635166,0.003941446,0.003641937,0.6232617,0.07562096],"study_design_scores_gemma":[0.001046618,0.0004385418,0.03701307,0.0002502364,0.0001351596,0.0002807932,0.0003195693,0.7497872,0.01454274,0.009134785,0.1868311,0.0002201406],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.1431094,0.001497788,0.01960514,0.001882169,0.001007411,0.0005087386,0.792927,0.02897334,0.01048904],"genre_scores_gemma":[0.1109309,0.0003334587,0.01923913,0.0001898228,0.00009319074,0.000343661,0.8664839,0.0009929255,0.001393015],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.0275567,"threshold_uncertainty_score":0.05479264,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2384365425349394,"score_gpt":0.3205473835112972,"score_spread":0.08211084097635782,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}