{"id":"W6979286987","doi":"","title":"BRIDGE: Benchmarking Large Language Models for Understanding Real-world Clinical Practice Text","year":2025,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Military Technology and Strategies","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; National Institute of Allergy and Infectious Diseases; National Institute of Arthritis and Musculoskeletal and Skin Diseases; Hamilton Health Sciences Foundation; Stanford Bio-X; Patient-Centered Outcomes Research Institute; Brigham Research Institute; National Heart, Lung, and Blood Institute; Korea Health Industry Development Institute; Gordon and Betty Moore Foundation; National Institutes of Health; National Science Foundation","keywords":"Benchmarking; Generalizability theory; Triage; Clinical Practice; Inference; Health care; Benchmark (surveying); Bridge (graph theory)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01222142,0.003851578,0.001227688,0.00402189,0.001066145,0.003218908,0.004606825,0.003258963,0.007266664],"category_scores_gemma":[0.04580885,0.0009386584,0.002760319,0.003156868,0.001333385,0.005328244,0.004111362,0.004342225,0.006490947],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003042208,"about_ca_system_score_gemma":0.005201264,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02284087,"about_ca_topic_score_gemma":0.0302947,"domain_scores_codex":[0.9899356,0.005142083,0.0009523076,0.002430901,0.001121423,0.0004176467],"domain_scores_gemma":[0.977051,0.01714405,0.0006670597,0.002426713,0.001883933,0.0008272849],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001908789,0.002058656,0.02108799,0.003741117,0.002091659,0.0007609862,0.001008573,0.3595491,0.003929604,0.005961613,0.2259931,0.3719088],"study_design_scores_gemma":[0.0005294749,0.0006850706,0.004285248,0.0002727304,0.0002188251,0.0002809909,0.0003965123,0.9531496,0.004671548,0.01260033,0.02276799,0.0001416344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3369753,0.01967209,0.3377392,0.00973692,0.003594654,0.00375493,0.1213259,0.146294,0.02090705],"genre_scores_gemma":[0.4751383,0.002913478,0.2048822,0.003260597,0.0005754085,0.002399704,0.300149,0.004548872,0.006132466],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02284087,"threshold_uncertainty_score":0.06463385,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1001626789662388,"score_gpt":0.261044732082406,"score_spread":0.1608820531161673,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}