{"id":"W4408364771","doi":"10.5194/essd-2025-83","title":"CY-Bench: A comprehensive benchmark dataset for sub-national crop yield forecasting","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Remote Sensing and Land Use","field":"Earth and Planetary Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba","funders":"HORIZON EUROPE Digital, Industry and Space","keywords":"Benchmark (surveying); Yield (engineering); Crop; Agricultural engineering; Environmental science; Engineering; Geography; Forestry; Cartography; Materials science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001517254,0.001362409,0.0006462021,0.001914906,0.0004275917,0.001150807,0.002378345,0.001498432,0.003995772],"category_scores_gemma":[0.005710871,0.0002609872,0.0009451908,0.003775254,0.0003369883,0.00134695,0.001186868,0.001166442,0.004093221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001476735,"about_ca_system_score_gemma":0.00134192,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03945117,"about_ca_topic_score_gemma":0.04043127,"domain_scores_codex":[0.9991836,0.0001551427,0.0000965141,0.0002694601,0.0001994156,0.00009595096],"domain_scores_gemma":[0.9977788,0.0005303741,0.000223066,0.0004980945,0.0008039823,0.000165737],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007241499,0.0005034668,0.04537154,0.0009944388,0.0004552896,0.0002991729,0.0001132832,0.1214862,0.003874103,0.003236283,0.7525116,0.07043057],"study_design_scores_gemma":[0.0005835464,0.0004667045,0.1314446,0.0003318775,0.0001415254,0.0003018322,0.0004968861,0.4593809,0.007917566,0.007816105,0.3908583,0.0002603344],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0560305,0.0006850537,0.008601271,0.0007828459,0.0002810446,0.0001512002,0.9230248,0.006131013,0.00431225],"genre_scores_gemma":[0.03863955,0.000163507,0.008109178,0.0001223946,0.00002730039,0.0001490167,0.9518165,0.0001905028,0.0007821132],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.03945117,"threshold_uncertainty_score":0.07844311,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1108729015104403,"score_gpt":0.2862418928305192,"score_spread":0.1753689913200789,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}