{"id":"W4415645090","doi":"10.1080/00016489.2025.2577834","title":"Evaluating the accuracy and reproducibility of ChatGPT responses in the context of cochlear implantation","year":2025,"lang":"en","type":"article","venue":"Acta Oto-Laryngologica","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Western Hospital; Western University","funders":"","keywords":"Cochlear implantation; Context (archaeology); Reproducibility; Cochlear implant; Hearing loss","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07775272,0.0005548606,0.0008578451,0.002325241,0.0008016606,0.001294816,0.001074994,0.001212374,0.002100724],"category_scores_gemma":[0.2896017,0.0003335149,0.0009473437,0.001293758,0.00143483,0.00108924,0.003157225,0.0007013899,0.0009399246],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009580704,"about_ca_system_score_gemma":0.001766297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000958949,"about_ca_topic_score_gemma":0.001719911,"domain_scores_codex":[0.895677,0.06878964,0.0137292,0.005392252,0.01486589,0.001545984],"domain_scores_gemma":[0.5320762,0.314186,0.04316906,0.02083781,0.08744456,0.002286406],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003155139,0.0004733652,0.7209782,0.004648711,0.0004399341,0.001029449,0.05276915,0.001050543,0.0140405,0.0005888956,0.00476298,0.1960632],"study_design_scores_gemma":[0.0001787342,0.002567241,0.9282212,0.002068135,0.0005869487,0.003277654,0.01952815,0.006832019,0.01857856,0.001244238,0.01667024,0.0002467375],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9650006,0.001592343,0.02472058,0.0008259562,0.0004804788,0.001707407,0.0007209342,0.0002410752,0.004710667],"genre_scores_gemma":[0.9781099,0.0004957209,0.0172696,0.0003782577,0.0001762185,0.002020685,0.0005282817,0.0000999843,0.0009214083],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07775272,"threshold_uncertainty_score":0.4112006,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2512198779434433,"score_gpt":0.501128908778339,"score_spread":0.2499090308348957,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}