{"id":"W4417132303","doi":"10.1109/bhi67747.2025.11269553","title":"PatientSafeBench: Evaluating the Safety of Medical LLMs for Patient Use","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"HORIZON EUROPE Health","keywords":"Patient safety; Risk assessment; Benchmark (surveying); Health care; Strengths and weaknesses; Scale (ratio)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02131125,0.001500846,0.0006995715,0.001709541,0.0006709066,0.002452011,0.00165809,0.002213432,0.002427028],"category_scores_gemma":[0.1279361,0.0003799372,0.0009923441,0.0008199153,0.001176964,0.002894319,0.002974276,0.00174872,0.001326393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001449178,"about_ca_system_score_gemma":0.001985373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003591305,"about_ca_topic_score_gemma":0.003595912,"domain_scores_codex":[0.9712802,0.01851277,0.002637276,0.002075923,0.004915246,0.0005786104],"domain_scores_gemma":[0.8715191,0.1033285,0.005252522,0.01008298,0.007702124,0.002114747],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01437869,0.005471335,0.247354,0.006541933,0.001624907,0.001562188,0.01338111,0.128177,0.03877268,0.007713355,0.05806642,0.4769563],"study_design_scores_gemma":[0.001328951,0.01433731,0.1175221,0.001331714,0.0007399492,0.002421345,0.005880721,0.7059138,0.07114229,0.01639202,0.06223587,0.0007539899],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.904753,0.001426153,0.0648352,0.001853754,0.0002846756,0.001490949,0.005821014,0.01222402,0.007311318],"genre_scores_gemma":[0.9392686,0.0002688665,0.04978395,0.0006913684,0.00005035872,0.0006989402,0.007108562,0.0005737563,0.001555644],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02131125,"threshold_uncertainty_score":0.112706,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07083018934792984,"score_gpt":0.4076169843503667,"score_spread":0.3367867950024369,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}