{"id":"W4214552201","doi":"10.2196/36902","title":"Testing Artificial Intelligence Algorithms in the Real World: Lessons From the SMARTI Trial","year":2022,"lang":"en","type":"article","venue":"Iproceedings","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Health and Medical Research Council; Medical Research Council; Monash University","keywords":"Medicine; Medical diagnosis; Interim; Laptop; Algorithm; Skin cancer; Skin lesion; Medical physics; Artificial intelligence; Computer science; Radiology; Pathology; Cancer","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000781582,0.000111263,0.000152767,0.0001093237,0.0004359401,0.00007986672,0.0002647394,0.00002163343,0.0003002016],"category_scores_gemma":[0.0002833572,0.00007176603,0.00006417352,0.0009494785,0.000053664,0.00003669689,0.0001539677,0.0004747253,0.00002806328],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001188009,"about_ca_system_score_gemma":0.00004462016,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002157253,"about_ca_topic_score_gemma":0.0003928789,"domain_scores_codex":[0.9987122,0.00004176586,0.0003027495,0.0002663466,0.0004465881,0.0002303548],"domain_scores_gemma":[0.999307,0.0003470781,0.00009652376,0.0001698707,0.00003985549,0.0000396834],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.01243416,0.001009495,0.007112121,0.00003186189,0.0001184948,0.0005011327,0.02175387,0.00009064718,0.006578289,0.02320436,0.01563338,0.9115322],"study_design_scores_gemma":[0.01780547,0.00546856,0.0523742,0.0002559905,0.0008401186,0.000645118,0.1975292,0.03948979,0.003708028,0.05111365,0.6293827,0.001387171],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9522198,0.00003806012,0.0001696484,0.0230485,0.001024629,0.00118704,0.000009543239,0.0001286184,0.02217412],"genre_scores_gemma":[0.995836,0.000005680021,0.0003690142,0.001892303,0.0009719323,0.0002056036,0.000006253972,0.00001518299,0.000698061],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.910145,"threshold_uncertainty_score":0.3352943,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1083153469008021,"score_gpt":0.3350657810509829,"score_spread":0.2267504341501809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}