{"id":"W4402527587","doi":"10.1016/j.jaci.2024.09.006","title":"Clustering of clinical symptoms using large language models reveals low diagnostic specificity of proposed alternatives to consensus mast cell activation syndrome criteria","year":2024,"lang":"en","type":"article","venue":"Journal of Allergy and Clinical Immunology","topic":"Mast cells and histamine","field":"Immunology and Microbiology","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"Institute of Infection and Immunity","funders":"National Institute of Allergy and Infectious Diseases; Eunice Kennedy Shriver National Institute of Child Health and Human Development; Bill and Melinda Gates Foundation; National Institute of Child Health and Human Development; U.S. Department of Defense","keywords":"Cluster analysis; Mast cell; Computer science; Artificial intelligence; Natural language processing; Medicine; Immunology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001474584,0.0001928683,0.00107129,0.0002150308,0.00006017425,0.0000145588,0.0002453791,0.000427173,0.0003896224],"category_scores_gemma":[0.0006909993,0.0001563353,0.0003060469,0.00012498,0.0004774205,0.0001067344,0.0001927157,0.0007081374,0.0000103496],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000234299,"about_ca_system_score_gemma":0.000123452,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004781039,"about_ca_topic_score_gemma":0.00002063876,"domain_scores_codex":[0.9964093,0.0006963575,0.002252497,0.0002921631,0.00007047183,0.0002792083],"domain_scores_gemma":[0.9962651,0.002317584,0.0009011814,0.0002216637,0.0002367651,0.000057776],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.005327424,0.003048065,0.003694766,0.0005737788,0.006055191,0.0006409335,0.002479749,0.000582954,0.9610868,0.001971515,0.0006482647,0.01389054],"study_design_scores_gemma":[0.05472285,0.05784843,0.5321804,0.01916132,0.003040763,0.009459781,0.01452307,0.0250503,0.2475696,0.006274809,0.02611438,0.004054355],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9873602,0.005012966,0.005047988,0.0001938378,0.001934225,0.000167569,0.00004858043,0.00001181101,0.0002228532],"genre_scores_gemma":[0.9967318,0.001623678,0.001002766,0.000063466,0.0001106463,8.937286e-7,0.000009624213,0.00002025919,0.000436921],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7135172,"threshold_uncertainty_score":0.6375167,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03658835648434897,"score_gpt":0.3422283631494035,"score_spread":0.3056400066650545,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}