{"id":"W4400981147","doi":"10.7759/cureus.65343","title":"A Blinded Comparison of Three Generative Artificial Intelligence Chatbots for Orthopaedic Surgery Therapeutic Questions","year":2024,"lang":"en","type":"article","venue":"Cureus","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; McMaster University","funders":"","keywords":"Medicine; Logistic regression; Family medicine; Physical therapy; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004725922,0.0001363487,0.0003346247,0.000163009,0.0001157593,0.00003248021,0.00006967969,0.0001379418,0.0001662807],"category_scores_gemma":[0.0003998135,0.0001171882,0.0002117338,0.0003898315,0.0001264174,0.00009798726,0.00001214897,0.0001902024,0.00005180691],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009993833,"about_ca_system_score_gemma":0.0006448814,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000258372,"about_ca_topic_score_gemma":0.0008269637,"domain_scores_codex":[0.9985445,0.00004236728,0.0006843584,0.0002785767,0.0002014155,0.0002487769],"domain_scores_gemma":[0.9983364,0.0008732091,0.00009479885,0.0002442518,0.0003392507,0.0001120968],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002589102,0.0005290252,0.01019701,0.0005666604,0.0001702516,0.000005603988,0.005252374,0.00009861756,0.003895268,0.05269997,0.001704661,0.9246216],"study_design_scores_gemma":[0.00005986913,0.002241077,0.005821739,0.002187298,0.0009483087,0.00006604406,0.01082232,0.2293171,0.3853113,0.348839,0.01356739,0.0008185797],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4784565,0.0253143,0.4777685,0.01106562,0.005144014,0.001700721,0.00003934514,0.0002340647,0.0002769679],"genre_scores_gemma":[0.9969097,0.0002409147,0.001357567,0.0001784204,0.001003292,0.0001640372,0.00005020053,0.00002451881,0.00007130595],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9238031,"threshold_uncertainty_score":0.4778796,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4698047920402946,"score_gpt":0.5041710116963164,"score_spread":0.03436621965602182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}