{"id":"W3196374756","doi":"10.1029/2021ea001896","title":"Labeling Poststorm Coastal Imagery for Machine Learning: Measurement of Interrater Agreement","year":2021,"lang":"en","type":"article","venue":"Earth and Space Science","topic":"Tropical and Extratropical Cyclones Research","field":"Earth and Planetary Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Gulf Research Program; U.S. Geological Survey; Directorate for Geosciences; Office of the Director; Leverhulme Trust","keywords":"Computer science; Inter-rater reliability; Set (abstract data type); Artificial intelligence; Process (computing); Data set; Focus (optics); Training set; Machine learning; Supervised learning; Statistics; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09007285,0.0006430777,0.0007311685,0.004838349,0.002109847,0.003192662,0.00144871,0.001159903,0.001664681],"category_scores_gemma":[0.2926269,0.0004790207,0.0009323992,0.003082734,0.00253619,0.002777127,0.005082633,0.001415261,0.000851012],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001187819,"about_ca_system_score_gemma":0.000983305,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00189352,"about_ca_topic_score_gemma":0.002921991,"domain_scores_codex":[0.9120185,0.0573458,0.009391258,0.007705146,0.01231994,0.001219349],"domain_scores_gemma":[0.4709731,0.3810262,0.03841027,0.02812543,0.07829457,0.003170362],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002065435,0.0005336093,0.7710919,0.00155883,0.00120204,0.0002887685,0.03643047,0.009003935,0.01365941,0.00388474,0.0105541,0.1497268],"study_design_scores_gemma":[0.0002120417,0.001183613,0.7196908,0.0009429454,0.0006174511,0.0005847387,0.02747896,0.156764,0.0556219,0.01981709,0.0165564,0.0005300685],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9222269,0.0003856956,0.06620463,0.0003011481,0.0001924604,0.001063737,0.001325153,0.0006929588,0.007607243],"genre_scores_gemma":[0.9606668,0.0001188156,0.03614837,0.0001271211,0.00006984035,0.001121098,0.0009663291,0.0001484839,0.0006331046],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9099271,"threshold_uncertainty_score":0.4763564,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02925973294962399,"score_gpt":0.2447320153598032,"score_spread":0.2154722824101792,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}