{"id":"W4408428954","doi":"10.5194/egusphere-egu25-14649","title":"Evaluating the Spatial Generalizability of ML- and DL-Based Surrogate Models for Flood Depth Prediction","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Hydrology and Watershed Management Studies","field":"Environmental Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; McGill University","funders":"","keywords":"Generalizability theory; Flood myth; Surrogate model; Surrogate endpoint; Computer science; Environmental science; Statistics; Geography; Mathematics; Machine learning; Archaeology; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003799529,0.0008721036,0.0006606319,0.0007472343,0.000329147,0.0009365884,0.001006063,0.001103009,0.0006327939],"category_scores_gemma":[0.01226576,0.0003545962,0.0008581124,0.0005497348,0.0005875601,0.00131401,0.0011926,0.001150255,0.0001866428],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007627244,"about_ca_system_score_gemma":0.001320747,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02158271,"about_ca_topic_score_gemma":0.01392436,"domain_scores_codex":[0.999303,0.000289579,0.00007006073,0.0001410564,0.0001228717,0.00007355562],"domain_scores_gemma":[0.9953577,0.002768477,0.0003997184,0.0005846243,0.0007437787,0.0001457584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001096339,0.00005279455,0.005775304,0.00003113643,0.00004473135,0.00002276643,0.00001888064,0.9833043,0.0006041576,0.0006289231,0.0002424604,0.009164947],"study_design_scores_gemma":[0.000006460547,0.00003150504,0.0008029657,0.000004713682,0.000004459646,0.000005517015,0.000009482579,0.9984878,0.0003508548,0.0002234985,0.0000683246,0.000004349904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8306774,0.0006570142,0.1630227,0.0008174502,0.0001015779,0.00009524188,0.001122352,0.001071533,0.002434714],"genre_scores_gemma":[0.9772466,0.0001453963,0.02102169,0.00007581059,0.00001652069,0.00004781159,0.001019825,0.0000407455,0.0003854833],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02158271,"threshold_uncertainty_score":0.04291421,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06571808562358776,"score_gpt":0.3192573972834447,"score_spread":0.2535393116598569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}