{"id":"W4408931974","doi":"10.2480/agrmet.d-24-00033","title":"Toward improving global rice yield reference dataset compilation through machine learning: Insights from training data selection and random forest analysis","year":2025,"lang":"en","type":"article","venue":"Journal of Agricultural Meteorology","topic":"Smart Agriculture and AI","field":"Agricultural and Biological Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; University of British Columbia; U.S. Geological Survey; University of Minnesota; National Aeronautics and Space Administration","keywords":"Random forest; Selection (genetic algorithm); Computer science; Machine learning; Artificial intelligence; Yield (engineering); Training (meteorology); Training set; Geography; Meteorology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008793777,0.001132057,0.0008736985,0.002380448,0.0003290617,0.001099823,0.001014271,0.0006224634,0.0004375634],"category_scores_gemma":[0.01477946,0.0002294013,0.0008551723,0.002925268,0.0004180223,0.001557776,0.001269866,0.0007947319,0.0003293903],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004189246,"about_ca_system_score_gemma":0.001081972,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005407464,"about_ca_topic_score_gemma":0.007249317,"domain_scores_codex":[0.9983236,0.000846425,0.0001307333,0.0003675079,0.0002535011,0.00007821163],"domain_scores_gemma":[0.9956504,0.002050124,0.0003726785,0.0008851388,0.0009757135,0.00006600758],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001996093,0.0002794491,0.08795837,0.0005938198,0.0004525436,0.0003341439,0.0004209634,0.3511977,0.0132815,0.009161575,0.01018635,0.525934],"study_design_scores_gemma":[0.00004702365,0.0001334716,0.04258438,0.0001458891,0.0001297369,0.00009820714,0.000298826,0.9231017,0.01053036,0.01128694,0.01157352,0.00007005769],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1654145,0.001699469,0.8243328,0.0008864828,0.00008385774,0.0001422581,0.002865279,0.002553364,0.00202197],"genre_scores_gemma":[0.4411332,0.0009896012,0.5452081,0.0002106902,0.00009559979,0.0002393702,0.01125574,0.0003687567,0.0004989232],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008793777,"threshold_uncertainty_score":0.04650646,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06292614207723515,"score_gpt":0.2690514841701182,"score_spread":0.2061253420928831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}