{"id":"W4408672934","doi":"10.1101/2025.03.17.25324157","title":"Real-World Evaluation of Large Language Models in Healthcare (RWE-LLM): A New Realm of AI Safety &amp; Validation","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Realm; Health care; Computer science; Political science; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09902756,0.001301955,0.0008936964,0.001963738,0.001150963,0.004869104,0.003262173,0.001896859,0.00238169],"category_scores_gemma":[0.262105,0.0006492095,0.001467563,0.001221591,0.002472991,0.004391088,0.006873477,0.002854264,0.000779118],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003842044,"about_ca_system_score_gemma":0.006827692,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006225386,"about_ca_topic_score_gemma":0.007734806,"domain_scores_codex":[0.8663019,0.106682,0.005572293,0.005703098,0.01430286,0.00143788],"domain_scores_gemma":[0.7526317,0.1721481,0.01256119,0.03067927,0.02877052,0.003209206],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002977574,0.002989787,0.1066416,0.004737643,0.001426487,0.0008633493,0.01323198,0.282342,0.02074286,0.04344016,0.03421632,0.4863903],"study_design_scores_gemma":[0.0006699396,0.003988691,0.02488669,0.002197013,0.0002939323,0.0003816462,0.004929864,0.8486016,0.02711761,0.04495192,0.04162292,0.000358223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3607115,0.001906648,0.5985313,0.008926955,0.0005162723,0.002869002,0.002924419,0.01049459,0.01311931],"genre_scores_gemma":[0.674755,0.0002512426,0.3182536,0.001482221,0.00006047292,0.001141353,0.002531236,0.0007024331,0.0008223581],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9009724,"threshold_uncertainty_score":0.5237141,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2160216630493233,"score_gpt":0.4999122890355706,"score_spread":0.2838906259862473,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}