{"id":"W4200578556","doi":"10.2196/35391","title":"Assessing Generalizability of Deep Learning Models Trained on Standardized and Nonstandardized Images and Their Performance Against Teledermatologists","year":2021,"lang":"en","type":"article","venue":"Iproceedings","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Convolutional neural network; Generalizability theory; Artificial intelligence; Standardization; Computer science; Deep learning; Pattern recognition (psychology); Receiver operating characteristic; Machine learning; Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003639804,0.0001744037,0.0004513414,0.00009442969,0.0001886918,0.00009661796,0.00003198203,0.0000734089,0.0000184854],"category_scores_gemma":[0.0001961203,0.0001402005,0.00006381521,0.0001563276,0.0001538506,0.00019735,0.00006422526,0.0001945532,3.650419e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005800073,"about_ca_system_score_gemma":0.0000448965,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000003371885,"about_ca_topic_score_gemma":0.000001156061,"domain_scores_codex":[0.9989561,0.00002999647,0.0002766742,0.0003336181,0.0002069789,0.0001966274],"domain_scores_gemma":[0.999366,0.00006025112,0.0001136959,0.000103366,0.0002558269,0.0001008617],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001882795,0.0002779401,0.02162103,0.002510439,0.0003677001,0.0001809986,0.005706876,0.0004776547,0.4655582,0.0007540849,0.0001886189,0.5004736],"study_design_scores_gemma":[0.01675324,0.001349225,0.02470338,0.001162237,0.0003556982,0.002669503,0.02173769,0.1789084,0.7416576,0.0009218829,0.00874344,0.001037672],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9844785,0.000427266,0.003707684,0.0004156597,0.00005120286,0.0002151754,0.000004286321,0.00009632424,0.01060387],"genre_scores_gemma":[0.997057,0.0006428465,0.00183516,0.0002611125,0.00002942001,0.000009630087,0.000006306311,0.000018758,0.0001397201],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.499436,"threshold_uncertainty_score":0.5717212,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02582374461614296,"score_gpt":0.2677876922085836,"score_spread":0.2419639475924406,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}