{"id":"W4411687990","doi":"10.1093/bjd/ljaf250","title":"Assessing the performance of artificial intelligence models in evaluating inflammatory skin disease severity: a systematic review and meta-analysis","year":2025,"lang":"en","type":"review","venue":"British Journal of Dermatology","topic":"Cutaneous Melanoma Detection and Management","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Innovaderm (Canada); McGill University Health Centre; Université de Montréal","funders":"National Institute of Arthritis and Musculoskeletal and Skin Diseases; National Institute of Mental Health; National Institutes of Health","keywords":"Medicine; Meta-analysis; Contingency table; Bivariate analysis; Atopic dermatitis; Internal medicine; Systematic review; MEDLINE; Disease; Dermatology; Machine learning; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02937458,0.00289431,0.01648302,0.009918426,0.0006935499,0.00422537,0.002530502,0.002306237,0.002319099],"category_scores_gemma":[0.07346673,0.001411646,0.04144912,0.009099661,0.001018746,0.002450653,0.001609716,0.001899444,0.0002932361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002810058,"about_ca_system_score_gemma":0.003816854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005220014,"about_ca_topic_score_gemma":0.01110913,"domain_scores_codex":[0.9797061,0.009927511,0.006073834,0.001682316,0.002244615,0.0003656677],"domain_scores_gemma":[0.935093,0.05242674,0.007188036,0.001675858,0.003265521,0.00035077],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"meta_analysis","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.001432362,0.00003321404,0.009527884,0.3681997,0.599531,0.00009995511,0.00009622303,0.0009189283,0.0001683327,0.0001576396,0.0006674391,0.01916729],"study_design_scores_gemma":[0.000371603,0.0002045924,0.00466439,0.03934598,0.9532905,0.00008650948,0.00004525536,0.0004116708,0.0001135818,0.0002803111,0.00115523,0.00003040086],"study_design_candidate":"meta_analysis","study_design_consensus":"meta_analysis","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.004737017,0.9927474,0.000997012,0.0002687423,0.0001237323,0.0003523919,0.0005158162,0.00002919442,0.0002286625],"genre_scores_gemma":[0.2177457,0.7718843,0.005200161,0.00110165,0.0004272012,0.002097542,0.001221733,0.00004506863,0.0002767302],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.02937458,"threshold_uncertainty_score":0.1553494,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0970255586441627,"score_gpt":0.3885096566984381,"score_spread":0.2914840980542754,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}