{"id":"W4404902547","doi":"10.1016/j.clinimag.2024.110382","title":"Comparative diagnostic accuracy of GPT-4o and LLaMA 3-70b: Proprietary vs. open-source large language models in radiology","year":2024,"lang":"en","type":"article","venue":"Clinical Imaging","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":19,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University; McGill University Health Centre; London Health Sciences Centre","funders":"","keywords":"Medicine; Radiology; Nuclear medicine; Medical physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00819016,0.001425438,0.001023066,0.003184575,0.0005381713,0.003254036,0.002353195,0.003061933,0.004017627],"category_scores_gemma":[0.05580711,0.0005680879,0.001660726,0.001131407,0.0009168583,0.003036065,0.002391013,0.001541268,0.003101418],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001147667,"about_ca_system_score_gemma":0.001556641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01154237,"about_ca_topic_score_gemma":0.01231504,"domain_scores_codex":[0.9948192,0.00293958,0.0004352285,0.00084022,0.0007313605,0.0002344088],"domain_scores_gemma":[0.9722269,0.02196838,0.0007453457,0.002293693,0.001999801,0.0007658562],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.02052852,0.001386857,0.1248299,0.002154952,0.003518952,0.00146795,0.001101691,0.0954116,0.02030706,0.0035674,0.06504825,0.6606769],"study_design_scores_gemma":[0.001791774,0.002189865,0.0666744,0.0006128786,0.002104364,0.005046412,0.001504409,0.8457552,0.03706858,0.01084428,0.02585987,0.0005478523],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8553638,0.01109199,0.07053298,0.005260276,0.0007801712,0.0003280627,0.01935021,0.02607074,0.01122182],"genre_scores_gemma":[0.9004642,0.002032103,0.06050727,0.001067356,0.000349277,0.0002045983,0.02939933,0.00316899,0.002806891],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01154237,"threshold_uncertainty_score":0.04331422,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2348965275187137,"score_gpt":0.5358115507626536,"score_spread":0.3009150232439399,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}