{"id":"W2156892295","doi":"10.5194/gmd-5-819-2012","title":"Towards a public, standardized, diagnostic benchmarking system for land surface models","year":2012,"lang":"en","type":"article","venue":"Geoscientific model development","topic":"Soil Geostatistics and Mapping","field":"Environmental Science","cited_by":124,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Lawrence Berkeley National Laboratory; Biological and Environmental Research; Natural Sciences and Engineering Research Council of Canada; Oak Ridge National Laboratory; University of New South Wales; Microsoft Research; Canadian Foundation for Climate and Atmospheric Sciences; Natural Resources Canada; Université Laval; U.S. Department of Energy; National Science Foundation","keywords":"Benchmarking; A priori and a posteriori; Computer science; Protocol (science); Work (physics); Business; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1879943,0.001809436,0.002515909,0.01314987,0.003560001,0.01458062,0.006803548,0.004046711,0.007981543],"category_scores_gemma":[0.222072,0.00140587,0.001476319,0.009990349,0.003418187,0.01994509,0.01370621,0.005704609,0.005258063],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006598612,"about_ca_system_score_gemma":0.0157176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004982563,"about_ca_topic_score_gemma":0.003251199,"domain_scores_codex":[0.89779,0.05236646,0.01600768,0.006557221,0.02496191,0.002316638],"domain_scores_gemma":[0.761237,0.0414898,0.01499864,0.1104534,0.06570861,0.006112485],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008510376,0.003322285,0.02154027,0.001078734,0.0003774861,0.0006210846,0.003679764,0.05068335,0.01159434,0.2486739,0.09230205,0.5652757],"study_design_scores_gemma":[0.0006366655,0.002085616,0.02242129,0.003146094,0.0003623909,0.0008786344,0.004283348,0.391197,0.04661002,0.2501477,0.2772933,0.0009380398],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01445997,0.0001440134,0.9332265,0.002227267,0.0002242957,0.002620744,0.003241099,0.02998053,0.01387555],"genre_scores_gemma":[0.1545553,0.0002284028,0.8072244,0.0007362821,0.0001739593,0.005390656,0.02136719,0.006052165,0.004271664],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1879943,"threshold_uncertainty_score":0.9942209,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03648892098034696,"score_gpt":0.2361041233209256,"score_spread":0.1996152023405786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}