{"id":"W4289657037","doi":"10.2196/35150","title":"Assessing the Generalizability of Deep Learning Models Trained on Standardized and Nonstandardized Images and Their Performance Against Teledermatologists: Retrospective Comparative Study","year":2022,"lang":"en","type":"article","venue":"JMIR Dermatology","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Health and Medical Research Council; Medical Research Council; Australian Government; Nvidia","keywords":"Generalizability theory; Artificial intelligence; Computer science; Machine learning; Psychology; Medical physics; Medicine; Developmental psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01474681,0.0006223826,0.000575617,0.001461916,0.0003953991,0.001090149,0.001255811,0.001181845,0.002069729],"category_scores_gemma":[0.06295963,0.0004499974,0.00109742,0.0007317035,0.001232747,0.001453281,0.001103057,0.001012728,0.0009564474],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007583014,"about_ca_system_score_gemma":0.0004100795,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002703658,"about_ca_topic_score_gemma":0.002176555,"domain_scores_codex":[0.9931929,0.001840763,0.0006694215,0.002692221,0.001295354,0.000309413],"domain_scores_gemma":[0.9478246,0.02371132,0.008878497,0.01111281,0.007543315,0.0009294192],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002519244,0.000263289,0.9579155,0.0001500399,0.001258162,0.0003170406,0.0006680641,0.003713575,0.002315493,0.0002094449,0.001543067,0.02912718],"study_design_scores_gemma":[0.0001286169,0.002314625,0.9663459,0.0001259574,0.000733886,0.001797825,0.0009089907,0.01896344,0.004699129,0.0007563247,0.003157646,0.00006757404],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9911591,0.001140598,0.004616995,0.0001856974,0.00009588354,0.0001129627,0.00132514,0.00008325368,0.001280248],"genre_scores_gemma":[0.996559,0.0001838446,0.0009717885,0.0001126824,0.00004488143,0.00006015624,0.001688175,0.00004572597,0.0003338514],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01474681,"threshold_uncertainty_score":0.07798952,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03285068772704493,"score_gpt":0.314296387602292,"score_spread":0.2814456998752471,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}