{"id":"W2146676240","doi":"10.1002/qj.268","title":"Impact of observational error on the validation of ensemble prediction systems","year":2008,"lang":"en","type":"article","venue":"Quarterly Journal of the Royal Meteorological Society","topic":"Meteorological Phenomena and Simulations","field":"Earth and Planetary Sciences","cited_by":60,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Brier score; Reliability (semiconductor); Observational study; Statistics; Receiver operating characteristic; Computer science; Histogram; Mathematics; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01909349,0.0007799995,0.0008153415,0.001026905,0.00047765,0.001704123,0.001151473,0.001211221,0.0004542648],"category_scores_gemma":[0.09268411,0.000267045,0.0005928845,0.0007203714,0.00157628,0.002058661,0.002461408,0.001326469,0.0001199395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006431178,"about_ca_system_score_gemma":0.0009308422,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002999508,"about_ca_topic_score_gemma":0.0009441368,"domain_scores_codex":[0.9827105,0.01034583,0.0008323004,0.001431821,0.004281591,0.0003978856],"domain_scores_gemma":[0.9018075,0.07135703,0.00799005,0.01046485,0.007599253,0.0007813395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005810065,0.00008950434,0.03483823,0.000195936,0.0003105885,0.0002036881,0.0002184429,0.896383,0.004531491,0.009984184,0.0003443571,0.05231946],"study_design_scores_gemma":[0.00001628507,0.0001251575,0.006044801,0.00004035074,0.00002637073,0.00005058101,0.00003366422,0.9840789,0.004474465,0.004785721,0.0002989602,0.00002480864],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4006917,0.001044718,0.5949228,0.0003531985,0.0001149816,0.00005898631,0.0002313977,0.0005034175,0.002078739],"genre_scores_gemma":[0.980647,0.00008937887,0.0189092,0.00002907485,0.00003058641,0.00002157312,0.0001385316,0.00002305276,0.0001116819],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01909349,"threshold_uncertainty_score":0.1009772,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07906041572139631,"score_gpt":0.262450098756735,"score_spread":0.1833896830353386,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}