{"id":"W4394575778","doi":"10.22541/essoar.171255743.38640796/v1","title":"Benchmark Framework for Global River Model (Version 1.0)","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Hydrology and Watershed Management Studies","field":"Environmental Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Japan Society for the Promotion of Science","keywords":"Benchmark (surveying); Metric (unit); Computer science; Set (abstract data type); Quantile; Data mining; Benchmarking; Flood myth; Hydrological modelling; Data pre-processing; Scale (ratio); Preprocessor; Environmental science; Surface runoff; Hydrology (agriculture); Statistics; Cartography; Artificial intelligence; Mathematics; Geography; Climatology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007693488,0.00209315,0.001335408,0.002632915,0.0007860332,0.003227888,0.005862453,0.002023206,0.0148333],"category_scores_gemma":[0.01368103,0.001145818,0.001915398,0.003177083,0.000576999,0.002866385,0.002868335,0.002922013,0.008378492],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001466964,"about_ca_system_score_gemma":0.003504006,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01955805,"about_ca_topic_score_gemma":0.01055141,"domain_scores_codex":[0.9971999,0.001006907,0.0003407352,0.0003193688,0.0008794522,0.0002536971],"domain_scores_gemma":[0.9960014,0.001128039,0.000278402,0.001052101,0.001351914,0.0001880868],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005480505,0.0003338266,0.006172498,0.001449703,0.0005895888,0.0002988236,0.0003107036,0.5094379,0.004090197,0.154126,0.2014605,0.1211822],"study_design_scores_gemma":[0.0002997499,0.000214504,0.00213567,0.0003474455,0.0001198567,0.0001378086,0.00009851834,0.6445127,0.00533314,0.07947638,0.2671641,0.0001601635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008599923,0.0009297088,0.7954411,0.0006106721,0.0004345291,0.001037101,0.07979129,0.09041861,0.02273704],"genre_scores_gemma":[0.09288607,0.001251926,0.6534832,0.0005705747,0.0001789989,0.005294811,0.2085621,0.03127409,0.006498372],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01955805,"threshold_uncertainty_score":0.04962236,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01473909092047777,"score_gpt":0.2638916563328623,"score_spread":0.2491525654123845,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}