{"id":"W4408931974","doi":"10.2480/agrmet.d-24-00033","title":"Toward improving global rice yield reference dataset compilation through machine learning: Insights from training data selection and random forest analysis","year":2025,"lang":"en","type":"article","venue":"Journal of Agricultural Meteorology","topic":"Smart Agriculture and AI","field":"Agricultural and Biological Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; University of British Columbia; U.S. Geological Survey; University of Minnesota; National Aeronautics and Space Administration","keywords":"Random forest; Selection (genetic algorithm); Computer science; Machine learning; Artificial intelligence; Yield (engineering); Training (meteorology); Training set; Geography; Meteorology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000287239,0.0002292821,0.0006112451,0.00004405215,0.000293251,0.0001310336,0.000506129,0.0002133652,0.00005068411],"category_scores_gemma":[0.0003410962,0.00007490838,0.0001213923,0.001424789,0.00004898091,0.0007518937,0.0002267751,0.0004474301,0.000002345618],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004509988,"about_ca_system_score_gemma":0.00001732491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001589229,"about_ca_topic_score_gemma":0.009896402,"domain_scores_codex":[0.9982491,0.0002769151,0.0005875887,0.0004042045,0.0002360049,0.0002461883],"domain_scores_gemma":[0.9983553,0.0006847501,0.0005782056,0.00007631617,0.0002135353,0.00009190701],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.001428957,0.0003190142,0.3462259,0.00003820304,0.003834635,0.00004420976,0.000897311,0.001713727,0.5983359,0.001269112,0.01079048,0.03510249],"study_design_scores_gemma":[0.0006476956,0.0004341964,0.9828143,0.00002481381,0.001226478,0.00007829527,0.0009001326,0.0008742329,0.0003753631,0.0005749476,0.01184514,0.0002044302],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9944714,0.00162383,0.0004247358,0.002203849,0.0002342449,0.000123676,0.0006782489,0.00002759787,0.0002123941],"genre_scores_gemma":[0.9933252,0.000222748,0.0007437119,0.0003080667,0.0004145751,0.000002234031,0.00495485,5.859779e-7,0.00002806632],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6365883,"threshold_uncertainty_score":0.5522425,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06292614207723515,"score_gpt":0.2690514841701182,"score_spread":0.2061253420928831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}