{"id":"W4413932853","doi":"10.1111/ctr.70303","title":"Accuracy, Clarity, and Comprehensiveness of ChatGPT Outputs for Commonly Asked Questions About Living Kidney Donation","year":2025,"lang":"en","type":"article","venue":"Clinical Transplantation","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Hospital; University of Ottawa","funders":"","keywords":"Medicine; Kappa; Kidney donation; CLARITY; Cohen's kappa; Kidney; Family medicine; Likert scale; Kidney transplantation; Inter-rater reliability; Donation; Internal medicine; Consistency (knowledge bases); Rating scale; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006471802,0.0000870912,0.0002779452,0.0001122876,0.0001324117,0.0000151094,0.00005042485,0.0001924007,0.000009722503],"category_scores_gemma":[0.001554265,0.00008610455,0.00008859839,0.0001528449,0.0001328021,0.0001089434,0.000007282819,0.000176029,0.000002526984],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002526841,"about_ca_system_score_gemma":0.0003352295,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000190383,"about_ca_topic_score_gemma":0.0001049769,"domain_scores_codex":[0.9986761,0.0001259675,0.000756534,0.0002145188,0.0001045383,0.0001223067],"domain_scores_gemma":[0.9957773,0.003347218,0.0001589331,0.0001449955,0.0004604354,0.0001110913],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009786016,0.0006803816,0.6862454,0.003993249,0.0001120485,0.000002145558,0.00274228,0.00002198618,0.001774835,0.02165318,0.0005446717,0.2812513],"study_design_scores_gemma":[0.0003806402,0.0003204787,0.9812936,0.002297312,0.0003468722,0.000008255043,0.0003947273,0.002849294,0.003801392,0.005809869,0.002387165,0.0001103789],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.906203,0.0003943774,0.08258226,0.008167919,0.001106906,0.001048336,0.00006771011,0.00005284243,0.000376663],"genre_scores_gemma":[0.9935889,0.002260844,0.00219017,0.001414573,0.0001472531,0.00004997568,0.0002523565,0.000007777759,0.00008812815],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2950482,"threshold_uncertainty_score":0.3511241,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.190274519640396,"score_gpt":0.5057505621121946,"score_spread":0.3154760424717986,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}