{"id":"W4417013798","doi":"10.1182/blood-2025-4349","title":"Evaluating large language models in real-world hematologic clinical decision-making: Performance, limitations, and clinical implications","year":2025,"lang":"en","type":"article","venue":"Blood","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Health Sciences Centre; Sunnybrook Health Science Centre","funders":"","keywords":"Subspecialty; Hematology; Hematologic Neoplasms; Test (biology); MEDLINE; Diagnostic test; Set (abstract data type); Scale (ratio); Specialty","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01666172,0.001697282,0.0009785192,0.001667851,0.0006181637,0.003136723,0.002230866,0.002138762,0.002870823],"category_scores_gemma":[0.05953853,0.0006052485,0.001321912,0.001196177,0.0009207065,0.002359587,0.001725697,0.002264878,0.0007944411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002769941,"about_ca_system_score_gemma":0.002944882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01904388,"about_ca_topic_score_gemma":0.01864583,"domain_scores_codex":[0.9900582,0.006912523,0.0007866647,0.001317549,0.0006487977,0.0002763341],"domain_scores_gemma":[0.9086608,0.08418688,0.001481126,0.002151161,0.002315004,0.001204897],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003084342,0.001811414,0.0569802,0.00166073,0.00104116,0.0006955055,0.0009127307,0.7413868,0.003554865,0.003412796,0.0117208,0.1737386],"study_design_scores_gemma":[0.0001450434,0.0002696275,0.001972283,0.00008092815,0.00008658444,0.00008824375,0.000230489,0.9908607,0.001525083,0.003580487,0.001126209,0.0000343526],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8661669,0.004173891,0.1049419,0.006348423,0.0004205237,0.001028382,0.006140382,0.005871094,0.004908372],"genre_scores_gemma":[0.9162816,0.0004148741,0.07762156,0.0007204504,0.00007835106,0.0002744626,0.004038406,0.000100866,0.0004694299],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01904388,"threshold_uncertainty_score":0.08811665,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3906290804132259,"score_gpt":0.5832482353174215,"score_spread":0.1926191549041956,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}