{"id":"W4391643387","doi":"10.1007/s00405-024-08464-9","title":"Is generative pre-trained transformer artificial intelligence (Chat-GPT) a reliable tool for guidelines synthesis? A preliminary evaluation for biologic CRSwNP therapy","year":2024,"lang":"en","type":"article","venue":"European Archives of Oto-Rhino-Laryngology","topic":"Sinusitis and nasal conditions","field":"Medicine","cited_by":20,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University","funders":"","keywords":"Mepolizumab; Medicine; Omalizumab; Dupilumab; Nasal polyps; Systematic review; Intensive care medicine; Population; Asthma; MEDLINE; Immunology; Immunoglobulin E; Antibody; Eosinophil","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002088069,0.0007838503,0.0006464186,0.0006009323,0.0002653694,0.001101303,0.001783513,0.001090343,0.01476837],"category_scores_gemma":[0.01050196,0.0003393762,0.0006704818,0.0003965331,0.0003989964,0.001021023,0.001074593,0.0013775,0.002429449],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006829622,"about_ca_system_score_gemma":0.001438528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004449657,"about_ca_topic_score_gemma":0.005446404,"domain_scores_codex":[0.9991653,0.0003553951,0.00005477657,0.0001845752,0.0001776061,0.00006232362],"domain_scores_gemma":[0.9936559,0.004736802,0.0001539397,0.0005566159,0.0007100928,0.0001865878],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001347458,0.0004773049,0.004220383,0.0005802062,0.0001966675,0.0003322858,0.0004183712,0.1156825,0.009210415,0.003806285,0.006667417,0.8570606],"study_design_scores_gemma":[0.0001492789,0.0004605138,0.001398939,0.00008006271,0.00009049559,0.0002424715,0.0001473045,0.975968,0.01052358,0.005248674,0.005661911,0.00002874141],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1073772,0.0008649388,0.8541316,0.0004942737,0.0002768632,0.0006027861,0.0009858323,0.02657952,0.008687105],"genre_scores_gemma":[0.6657802,0.0002471742,0.3275106,0.0002797444,0.00004412191,0.0003143752,0.00123511,0.0005277932,0.004060799],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9979119,"threshold_uncertainty_score":0.0494051,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1328556145027562,"score_gpt":0.3877791342628926,"score_spread":0.2549235197601364,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}