{"id":"W6921140991","doi":"10.6084/m9.figshare.27800394.v2","title":"A Comprehensive Surface Water Quality Monitoring Dataset (1940-2023): 2.82Million Record Resource for Empirical and ML-Based Research","year":2025,"lang":"en","type":"dataset","venue":"Figshare","topic":"","field":"","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Outlier; Water quality; Index (typography); Missing data; Quality (philosophy); Resource (disambiguation); Sample (material); Data quality; Data collection","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001583917,0.001002907,0.001335021,0.0008443659,0.00098833,0.0006120558,0.001800613,0.001250097,0.01421955],"category_scores_gemma":[0.006012666,0.0008916746,0.0002524735,0.0009852746,0.0001207521,0.0002459603,0.00254759,0.002633766,0.004836793],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006743264,"about_ca_system_score_gemma":0.000544856,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009073669,"about_ca_topic_score_gemma":0.0002067461,"domain_scores_codex":[0.9907092,0.002229386,0.00111334,0.002385354,0.001757574,0.001805104],"domain_scores_gemma":[0.9895874,0.004967296,0.0003746786,0.003158378,0.001393805,0.0005184864],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008926773,0.0001595397,0.00002462969,0.007130923,0.000135877,0.0000813576,0.00003460509,0.00005161009,0.0001727246,2.705422e-8,0.9912269,0.00008910997],"study_design_scores_gemma":[0.001518779,0.0002241124,0.0001774146,0.007768858,0.00009056352,0.000004584522,0.00009709477,0.00007761451,0.001481905,0.000009759078,0.9875578,0.0009914972],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0002842433,0.001083511,2.196833e-7,0.0002872638,0.0002540539,0.002907433,0.9949999,0.0001682011,0.00001520617],"genre_scores_gemma":[0.00001983733,0.00001665681,0.0001641738,0.0002161274,0.0008597958,0.00120435,0.9970398,0.0001442082,0.0003350766],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.009382756,"threshold_uncertainty_score":0.9996672,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5001938191741394,"score_gpt":0.5108734306667351,"score_spread":0.01067961149259566,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}