{"id":"W4400981147","doi":"10.7759/cureus.65343","title":"A Blinded Comparison of Three Generative Artificial Intelligence Chatbots for Orthopaedic Surgery Therapeutic Questions","year":2024,"lang":"en","type":"article","venue":"Cureus","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; McMaster University","funders":"","keywords":"Medicine; Logistic regression; Family medicine; Physical therapy; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06679808,0.001009503,0.001728379,0.001589642,0.001219219,0.00207597,0.00119027,0.002110864,0.005307408],"category_scores_gemma":[0.2466118,0.0009594566,0.001462033,0.0008227975,0.001569076,0.001505528,0.002073316,0.0007819653,0.0009998265],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002131113,"about_ca_system_score_gemma":0.003644331,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001028831,"about_ca_topic_score_gemma":0.001576611,"domain_scores_codex":[0.9243727,0.04938551,0.0130085,0.00459944,0.007714632,0.0009192232],"domain_scores_gemma":[0.710799,0.1884657,0.04440945,0.01896384,0.03239667,0.004965404],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"randomized_trial","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.4818845,0.0128493,0.1192811,0.01426335,0.003969033,0.0004480086,0.01505361,0.002764195,0.01750856,0.001441767,0.004814271,0.3257223],"study_design_scores_gemma":[0.2154666,0.3586549,0.3222702,0.005317896,0.008338808,0.001263244,0.01066304,0.01466283,0.02861478,0.005081966,0.02845596,0.001209859],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9551312,0.002179441,0.009198581,0.0003516726,0.0008132686,0.02691974,0.0008464094,0.000277035,0.004282603],"genre_scores_gemma":[0.9442491,0.0005921304,0.01785449,0.0005027357,0.0002458064,0.03464748,0.000415424,0.0000551926,0.00143766],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06679808,"threshold_uncertainty_score":0.3532662,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4698047920402946,"score_gpt":0.5041710116963164,"score_spread":0.03436621965602182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}