{"id":"W4410523294","doi":"10.1101/2025.05.16.652427","title":"Assessing Large Language Model Alignment Towards Radiological Myths and Misconceptions","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Nuclear Laboratories","funders":"Atomic Energy of Canada Limited","keywords":"Mythology; Computer science; Linguistics; Epistemology; Cognitive science; Psychology; History; Philosophy; Classics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01462327,0.0009517885,0.0004838013,0.00227111,0.001053831,0.003705846,0.0008619645,0.001145921,0.003728566],"category_scores_gemma":[0.08194915,0.0003864335,0.001074536,0.001687655,0.001030058,0.002768941,0.002530253,0.002130359,0.0012326],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002142776,"about_ca_system_score_gemma":0.001999337,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009101235,"about_ca_topic_score_gemma":0.009018186,"domain_scores_codex":[0.990512,0.006337063,0.0007006777,0.0009815756,0.001247348,0.0002213986],"domain_scores_gemma":[0.8349245,0.1465976,0.005860301,0.004376506,0.007424845,0.0008162584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004274474,0.002404782,0.4330395,0.00354994,0.001343797,0.001182969,0.05497243,0.1358743,0.02407566,0.02084814,0.02084317,0.2975909],"study_design_scores_gemma":[0.0003201066,0.001326155,0.1012191,0.0006760829,0.000661982,0.0005016867,0.01947522,0.82208,0.01180065,0.02222231,0.01939133,0.00032545],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9128481,0.0002641237,0.0701874,0.001070389,0.0001150195,0.0009616117,0.004899244,0.001415212,0.008238981],"genre_scores_gemma":[0.9420997,0.0000910245,0.0468858,0.0002155385,0.00002605141,0.001022631,0.008520228,0.0001570385,0.0009820685],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01462327,"threshold_uncertainty_score":0.07733619,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02794521837437856,"score_gpt":0.2748842983420117,"score_spread":0.2469390799676331,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}