{"id":"W4404065884","doi":"10.1148/radiol.241421","title":"Laterality: A Potential Pitfall in Applying Multimodal Large Language Models to Radiology","year":2024,"lang":"en","type":"letter","venue":"Radiology","topic":"Interpreting and Communication in Healthcare","field":"Health Professions","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"London Health Sciences Centre; Western University","funders":"","keywords":"Medicine; Laterality; Radiology; Medical physics; Linguistics; Audiology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity","insufficient_payload"],"consensus_categories":["research_integrity"],"category_scores_codex":[0.001536181,0.0004649881,0.001116963,0.0006636492,0.0003810927,0.00001537335,0.0009894435,0.00315042,0.0004513946],"category_scores_gemma":[0.0001691071,0.0004383631,0.0002026702,0.0002620817,0.00008508506,0.0000531855,0.0006928884,0.009167542,0.001552168],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006631884,"about_ca_system_score_gemma":0.000416745,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003307447,"about_ca_topic_score_gemma":0.0007335233,"domain_scores_codex":[0.9912236,0.004728633,0.001289045,0.0009248811,0.0001917719,0.001642012],"domain_scores_gemma":[0.9970311,0.001131474,0.0002726308,0.001302428,0.0001265087,0.0001358712],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001108576,0.00002369266,0.0009955467,0.001215998,0.00009384996,0.00095181,0.02795464,0.0003136517,0.00005673222,0.002062993,0.9656775,0.0005427288],"study_design_scores_gemma":[0.001120172,0.0001933847,0.0003056322,0.00158435,0.00006704429,0.0002805103,0.003082857,0.02341983,0.000001338758,0.006605541,0.9625054,0.0008339664],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.05041543,0.003483131,0.002578372,0.9200079,0.009344058,0.003695164,0.0007587141,0.0006677272,0.009049502],"genre_scores_gemma":[0.2582239,0.0001326289,0.000988348,0.7229848,0.005221473,0.002798037,0.00126623,0.0001506077,0.008234024],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.2078084,"threshold_uncertainty_score":0.9998068,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05654591465774114,"score_gpt":0.4125511802588636,"score_spread":0.3560052656011225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}