{"id":"W4406385218","doi":"10.1089/fpsam.2024.0206","title":"Comparative Performance of the Leading Large Language Models in Answering Complex Rhinoplasty Consultation Questions","year":2025,"lang":"en","type":"article","venue":"Facial Plastic Surgery & Aesthetic Medicine","topic":"Nasal Surgery and Airway Studies","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; University of Ottawa","funders":"","keywords":"Rhinoplasty; Natural language processing; Computer science; Linguistics; Psychology; Medicine; Nose; Philosophy; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009121939,0.001671727,0.0009369266,0.002345083,0.0005477723,0.002584142,0.001267568,0.001511752,0.004571615],"category_scores_gemma":[0.03838101,0.0003444188,0.001459291,0.001057628,0.0004590608,0.002706214,0.001955973,0.001203729,0.002164474],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001420776,"about_ca_system_score_gemma":0.001824837,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006713523,"about_ca_topic_score_gemma":0.008513947,"domain_scores_codex":[0.9941068,0.003798848,0.0004961956,0.0007087096,0.0006941389,0.0001953328],"domain_scores_gemma":[0.9601529,0.03474758,0.0008638898,0.001070159,0.002263658,0.0009018178],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01684806,0.003041583,0.1234859,0.007441391,0.001790138,0.0008431336,0.01010064,0.0537056,0.01713238,0.001964075,0.02897622,0.7346709],"study_design_scores_gemma":[0.001771964,0.01076447,0.129986,0.00164374,0.003342804,0.001575675,0.0130378,0.7730446,0.02222876,0.006872644,0.03493465,0.0007969092],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9362296,0.003981761,0.03021699,0.001586039,0.0002956075,0.001085491,0.004263729,0.008584822,0.01375589],"genre_scores_gemma":[0.9415259,0.0009350906,0.04663963,0.0004008432,0.0001105581,0.0008971994,0.006152047,0.0003265815,0.003012148],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009121939,"threshold_uncertainty_score":0.04824203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04048520541200501,"score_gpt":0.3090169258263897,"score_spread":0.2685317204143847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}