{"id":"W4403866520","doi":"10.1002/jso.27966","title":"Assessment of Artificial Intelligence Chatbot Responses to Common Patient Questions on Bone Sarcoma","year":2024,"lang":"en","type":"article","venue":"Journal of Surgical Oncology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"National Institute of Arthritis and Musculoskeletal and Skin Diseases","keywords":"Readability; Medicine; Likert scale; Empathy; Relevance (law); Artificial intelligence; Psychology; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001132909,0.0001075612,0.0004996666,0.0004717119,0.00005600918,0.00001625806,0.00008169557,0.0001481903,0.0002812829],"category_scores_gemma":[0.0002692478,0.00008053169,0.000175179,0.0003993123,0.0001062473,0.00006284931,0.00002722328,0.0005574995,0.00005162328],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004771157,"about_ca_system_score_gemma":0.00106164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007914113,"about_ca_topic_score_gemma":0.00003061306,"domain_scores_codex":[0.9976957,0.0002727048,0.001282465,0.0001534283,0.0003899593,0.0002057739],"domain_scores_gemma":[0.9972163,0.001695223,0.0002469394,0.0001512521,0.0004031603,0.0002871684],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002061012,0.001498426,0.001443375,0.00008621797,0.0000575461,0.001636662,0.001457783,0.0002645443,0.002437225,0.02160644,0.001228626,0.9662222],"study_design_scores_gemma":[0.00018024,0.08040549,0.0177921,0.00361387,0.0003928571,0.007305573,0.007514716,0.00425594,0.05035013,0.02350271,0.8042189,0.0004674645],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9540957,0.000625535,0.001624293,0.03934813,0.001953689,0.0002526536,0.000006238999,0.00001683612,0.002076909],"genre_scores_gemma":[0.9959033,0.0005879535,0.002219782,0.0005372427,0.0006536805,0.000009835977,0.000002404741,0.00001147964,0.00007428993],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9657547,"threshold_uncertainty_score":0.3283987,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1801518825300124,"score_gpt":0.51038109222645,"score_spread":0.3302292096964376,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}