{"id":"W4402527587","doi":"10.1016/j.jaci.2024.09.006","title":"Clustering of clinical symptoms using large language models reveals low diagnostic specificity of proposed alternatives to consensus mast cell activation syndrome criteria","year":2024,"lang":"en","type":"article","venue":"Journal of Allergy and Clinical Immunology","topic":"Mast cells and histamine","field":"Immunology and Microbiology","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"Institute of Infection and Immunity","funders":"National Institute of Allergy and Infectious Diseases; Eunice Kennedy Shriver National Institute of Child Health and Human Development; Bill and Melinda Gates Foundation; National Institute of Child Health and Human Development; U.S. Department of Defense","keywords":"Cluster analysis; Mast cell; Computer science; Artificial intelligence; Natural language processing; Medicine; Immunology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005580994,0.001207066,0.0009094285,0.002746804,0.000681547,0.002003348,0.001609137,0.001441905,0.002448094],"category_scores_gemma":[0.02645951,0.0002741505,0.001939031,0.0008725637,0.0007286682,0.001218286,0.001647772,0.001377878,0.001555488],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00058126,"about_ca_system_score_gemma":0.001397845,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004068518,"about_ca_topic_score_gemma":0.004705066,"domain_scores_codex":[0.9944292,0.003030594,0.0005083411,0.001160446,0.0005048442,0.0003665836],"domain_scores_gemma":[0.9779608,0.01835226,0.0008125939,0.001143346,0.001114726,0.0006163242],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01090569,0.001021976,0.5136999,0.001723208,0.003593317,0.003160058,0.002167509,0.07801815,0.03243512,0.005136129,0.02165156,0.3264875],"study_design_scores_gemma":[0.0004945348,0.0008340942,0.1252215,0.0002946945,0.0009878544,0.002952581,0.001239314,0.8316912,0.00659164,0.02555937,0.003948482,0.0001847778],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8754931,0.002103037,0.1093715,0.001767572,0.0001973996,0.0003634816,0.005277738,0.002040755,0.003385407],"genre_scores_gemma":[0.9847995,0.000109588,0.009909215,0.0001751173,0.00004901103,0.00007457065,0.004573233,0.0000902836,0.0002195902],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005580994,"threshold_uncertainty_score":0.02951545,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03658835648434897,"score_gpt":0.3422283631494035,"score_spread":0.3056400066650545,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}