{"id":"W4416388507","doi":"10.1002/met.70129","title":"On the Reliability of Surface Observations and the Pitfalls of Verification Against Own Analyses","year":2025,"lang":"en","type":"article","venue":"Meteorological Applications","topic":"Meteorological Phenomena and Simulations","field":"Earth and Planetary Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Environment and Climate Change Canada","funders":"","keywords":"Radiosonde; Data assimilation; Representativeness heuristic; Humidity; Numerical weather prediction; Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03557836,0.0007330557,0.0006291662,0.001342597,0.0007872502,0.002264839,0.001165123,0.001186174,0.0007201678],"category_scores_gemma":[0.1368685,0.0004155647,0.0008087453,0.001171175,0.001504248,0.002445256,0.002095769,0.00134739,0.0005665743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008032772,"about_ca_system_score_gemma":0.001015099,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01347117,"about_ca_topic_score_gemma":0.00836056,"domain_scores_codex":[0.9685645,0.0166599,0.002154703,0.004218125,0.007720331,0.000682404],"domain_scores_gemma":[0.797684,0.113326,0.02107087,0.03713436,0.02968294,0.001101784],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00300419,0.0002580652,0.583747,0.0006926777,0.001852596,0.0006634918,0.002662698,0.1346596,0.02367137,0.006052834,0.007425446,0.2353099],"study_design_scores_gemma":[0.0002291155,0.001277641,0.3966119,0.001190343,0.0008511813,0.0006448951,0.001931433,0.5095364,0.06277274,0.009115627,0.01551183,0.0003268391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8957623,0.002301489,0.08782758,0.001185248,0.0005625163,0.0001249985,0.001849287,0.001333595,0.009052941],"genre_scores_gemma":[0.9874558,0.0001104075,0.01123458,0.00009035781,0.0000600164,0.00001676434,0.00065017,0.0001430082,0.0002388632],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03557836,"threshold_uncertainty_score":0.1881586,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06125127307162563,"score_gpt":0.2842994448043285,"score_spread":0.2230481717327029,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}