{"id":"W4214552201","doi":"10.2196/36902","title":"Testing Artificial Intelligence Algorithms in the Real World: Lessons From the SMARTI Trial","year":2022,"lang":"en","type":"article","venue":"Iproceedings","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Health and Medical Research Council; Medical Research Council; Monash University","keywords":"Medicine; Medical diagnosis; Interim; Laptop; Algorithm; Skin cancer; Skin lesion; Medical physics; Artificial intelligence; Computer science; Radiology; Pathology; Cancer","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09173524,0.001052198,0.002638327,0.0004910109,0.0006109987,0.002789889,0.002208113,0.002895871,0.004193161],"category_scores_gemma":[0.2472431,0.0004644492,0.001967331,0.0009652803,0.002475649,0.003899625,0.001421804,0.005362714,0.001081613],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001338087,"about_ca_system_score_gemma":0.002692284,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001439181,"about_ca_topic_score_gemma":0.001607651,"domain_scores_codex":[0.944276,0.04895197,0.001949809,0.001310591,0.002871466,0.0006402492],"domain_scores_gemma":[0.7656147,0.2033824,0.005466234,0.01646176,0.006265855,0.002809009],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.1134609,0.007901029,0.04230492,0.005526053,0.008577238,0.000752361,0.002751832,0.0335718,0.001218495,0.02341636,0.08789501,0.672624],"study_design_scores_gemma":[0.1687976,0.1258832,0.06255602,0.0135256,0.01205581,0.002216464,0.002646162,0.1943631,0.00826401,0.2221851,0.1864604,0.001046638],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5895137,0.06096774,0.1230888,0.152639,0.007452435,0.009960721,0.004937733,0.001873497,0.04956638],"genre_scores_gemma":[0.8613578,0.01128156,0.0939488,0.01675809,0.002596864,0.007622231,0.002438917,0.0006510608,0.003344718],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.09173524,"threshold_uncertainty_score":0.4851481,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1083153469008021,"score_gpt":0.3350657810509829,"score_spread":0.2267504341501809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}