{"id":"W4391661497","doi":"10.1109/jstars.2024.3364020","title":"Fuzzy Similarity Analysis of Effective Training Samples to Improve Machine Learning Estimations of Water Quality Parameters Using Sentinel-2 Remote Sensing Data","year":2024,"lang":"en","type":"article","venue":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing","topic":"Hydrological Forecasting Using AI","field":"Environmental Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Regina; University of Waterloo","funders":"","keywords":"Computer science; Mean absolute percentage error; Support vector machine; Mean squared error; Water quality; Artificial intelligence; Regression; Similarity (geometry); Data mining; Machine learning; Remote sensing; Statistics; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001625976,0.0004650838,0.0006049241,0.001309566,0.0003154974,0.0005119187,0.0006329688,0.0005247477,0.0004420059],"category_scores_gemma":[0.00514813,0.0002124848,0.000717285,0.0007707194,0.0002809236,0.0009864931,0.0005118864,0.0004430613,0.0001518587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000369897,"about_ca_system_score_gemma":0.0005019282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003798991,"about_ca_topic_score_gemma":0.002971797,"domain_scores_codex":[0.9992328,0.0001817539,0.00006438985,0.0001522538,0.0003188087,0.00005006377],"domain_scores_gemma":[0.9984118,0.0006975606,0.0001956306,0.0001157127,0.0005458266,0.00003338588],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004706506,0.0003351225,0.01551271,0.0001640965,0.0001621054,0.0001734447,0.0002631551,0.3612896,0.04855862,0.002531208,0.001191716,0.5693476],"study_design_scores_gemma":[0.000006696446,0.00003842482,0.00209274,0.000003327137,0.00001430459,0.00001820606,0.0000200782,0.9910512,0.006056348,0.000465106,0.0002255846,0.000008024746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2912955,0.0002612266,0.7065706,0.0001108676,0.00004728712,0.00005617306,0.00008285135,0.000528961,0.001046545],"genre_scores_gemma":[0.8499116,0.000100564,0.1491762,0.00005191689,0.00003553539,0.00004797272,0.0001967603,0.00002829502,0.0004511223],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003798991,"threshold_uncertainty_score":0.008599043,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09748984460820852,"score_gpt":0.3131189630593307,"score_spread":0.2156291184511221,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}