{"id":"W4404169684","doi":"10.1101/2024.11.08.24316949","title":"Reliability of large language model knowledge across brand and generic cancer drug names","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Reliability (semiconductor); Brand names; Drug; Computer science; Natural language processing; Business; Medicine; Advertising; Pharmacology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02468227,0.001819523,0.001323226,0.002456635,0.0008018097,0.003439888,0.002082708,0.001668587,0.001809931],"category_scores_gemma":[0.0982244,0.0006689793,0.002193057,0.00109779,0.001181604,0.003846403,0.003138906,0.002388624,0.001169157],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00165003,"about_ca_system_score_gemma":0.001888396,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01152558,"about_ca_topic_score_gemma":0.009130032,"domain_scores_codex":[0.9820425,0.01009937,0.001565391,0.004423386,0.001403783,0.000465537],"domain_scores_gemma":[0.8865282,0.09640326,0.004552091,0.007259747,0.004005139,0.001251635],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007657603,0.001240718,0.320244,0.001046402,0.004324483,0.0008275167,0.001704539,0.2268359,0.01306656,0.001537844,0.007870069,0.4136444],"study_design_scores_gemma":[0.0003129821,0.001109691,0.04320863,0.000180598,0.0008443436,0.0005310653,0.001045757,0.9334618,0.01136316,0.00550615,0.002255634,0.0001801293],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9001064,0.001706998,0.08344209,0.001381179,0.0002359683,0.0003706715,0.003858512,0.004502596,0.004395687],"genre_scores_gemma":[0.9815113,0.0001285524,0.01378875,0.0002557,0.000041704,0.00009101687,0.003638825,0.0001525542,0.0003916381],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02468227,"threshold_uncertainty_score":0.1305339,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02513666307188248,"score_gpt":0.3695623369574255,"score_spread":0.344425673885543,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}