{"id":"W4401538229","doi":"10.1016/j.aohep.2024.101537","title":"Evaluation of four chatbots in autoimmune liver disease: A comparative analysis","year":2024,"lang":"en","type":"article","venue":"Annals of Hepatology","topic":"Liver Diseases and Immunity","field":"Medicine","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Chinesisch-Deutsche Zentrum für Wissenschaftsförderung; Bundesministerium für Bildung und Forschung","keywords":"Medicine; Autoimmune disease; Health professionals; Autoimmune hepatitis; Disease; Liver disease; Health care; Intensive care medicine; Pathology; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000735583,0.00008730273,0.0004993508,0.0004450338,0.00001342424,0.000003263509,0.00006785619,0.00004930192,0.0005705595],"category_scores_gemma":[0.0001235365,0.00007485868,0.0002419258,0.0006979593,0.0001085988,0.00007414298,0.00003853295,0.00009424366,0.00001511653],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002311698,"about_ca_system_score_gemma":0.0003462533,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008005779,"about_ca_topic_score_gemma":0.00009988359,"domain_scores_codex":[0.998699,0.0003301959,0.0003246728,0.0001613452,0.0003421145,0.0001426764],"domain_scores_gemma":[0.9989954,0.000108313,0.00008163863,0.0002619885,0.0004893436,0.00006328858],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009442657,0.001185668,0.9653718,0.000740662,0.00608154,0.0002086987,0.002741111,0.0003911614,0.0009813081,0.00312994,0.001961465,0.01626236],"study_design_scores_gemma":[0.0003797429,0.0001472355,0.8679986,0.0001150466,0.002844884,0.00000348214,0.00008958233,0.126757,0.0005846114,0.0007112524,0.0003123283,0.00005623034],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9680631,0.02978982,0.00002950277,0.001048467,0.00005660837,0.0002355521,0.00003243594,0.00001354161,0.0007309515],"genre_scores_gemma":[0.9989301,0.0007825899,0.00002437431,0.0001147255,0.00001148331,0.00001856396,0.00006161052,0.000004298039,0.00005223801],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1263658,"threshold_uncertainty_score":0.6247227,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2490873102605842,"score_gpt":0.4347216936920845,"score_spread":0.1856343834315002,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}