{"id":"W6926487900","doi":"10.25384/sage.23961354","title":"Supplemental Material - Comparative Performance of ChatGPT and Bard in a Text-Based Radiology Knowledge Assessment","year":2023,"lang":"en","type":"article","venue":"Sage Journals Data","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Juravinski Hospital","funders":"","keywords":"Association (psychology); Medical knowledge; Competency assessment; Risk assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003685935,0.0009198398,0.0007122427,0.006181268,0.0008322091,0.00217798,0.001365438,0.0009036095,0.430287],"category_scores_gemma":[0.07475609,0.0003345037,0.0008318361,0.003823443,0.0003228628,0.001714154,0.001772263,0.0009488721,0.0970349],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001359963,"about_ca_system_score_gemma":0.00205431,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006874555,"about_ca_topic_score_gemma":0.009967553,"domain_scores_codex":[0.9962436,0.0008839592,0.0006615483,0.000335363,0.001694251,0.0001813682],"domain_scores_gemma":[0.8842618,0.07634804,0.003000443,0.004376807,0.02971003,0.002302886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009323893,0.001186446,0.01347973,0.002038239,0.00008198858,0.0002339972,0.0006827933,0.0007890873,0.001572569,0.0008419649,0.7789117,0.1992491],"study_design_scores_gemma":[0.001351694,0.001339557,0.2456677,0.00413526,0.0004223692,0.002684936,0.006233582,0.02245821,0.01831825,0.01275241,0.68409,0.0005459971],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.05292372,0.0004627512,0.03639742,0.001899782,0.001350002,0.00428955,0.7945998,0.0418769,0.06619998],"genre_scores_gemma":[0.1527269,0.0006068308,0.1794592,0.001735379,0.0005866847,0.01105359,0.5876124,0.00682759,0.0593913],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.430287,"threshold_uncertainty_score":0.8126265,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04947658560565904,"score_gpt":0.351542836562135,"score_spread":0.3020662509564759,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}