{"id":"W4391973028","doi":"10.1016/j.compbiomed.2024.108189","title":"A comprehensive evaluation of large Language models on benchmark biomedical text processing tasks","year":2024,"lang":"en","type":"article","venue":"Computers in Biology and Medicine","topic":"Topic Modeling","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; York University","keywords":"Benchmark (surveying); Computer science; Task (project management); Set (abstract data type); Domain (mathematical analysis); Work (physics); Artificial intelligence; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01164097,0.003574848,0.001809423,0.003410971,0.001307438,0.00194832,0.002814957,0.002691403,0.002892871],"category_scores_gemma":[0.02432127,0.0007013856,0.002201469,0.002488438,0.001022,0.00410569,0.002118642,0.002830857,0.003099655],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002051093,"about_ca_system_score_gemma":0.003300952,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01315187,"about_ca_topic_score_gemma":0.0198573,"domain_scores_codex":[0.9935166,0.003088065,0.0006754448,0.001550409,0.0008909978,0.0002784999],"domain_scores_gemma":[0.9873375,0.008553077,0.0004428038,0.001438354,0.001653216,0.0005750195],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00311009,0.003088768,0.0119886,0.00607045,0.002715651,0.0007319757,0.000655553,0.1851284,0.01893824,0.002264123,0.08785744,0.6774507],"study_design_scores_gemma":[0.0006985343,0.003866164,0.01105463,0.0006018034,0.00112919,0.0008058197,0.0008298782,0.9140688,0.03149708,0.005883839,0.02927183,0.000292331],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5585918,0.08004539,0.2356261,0.00830177,0.004752516,0.002549568,0.0307562,0.05780489,0.02157171],"genre_scores_gemma":[0.650445,0.01035773,0.2232631,0.003180753,0.001149,0.001541841,0.0984834,0.001808945,0.009770343],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01315187,"threshold_uncertainty_score":0.06156403,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04517738477580747,"score_gpt":0.377081677449967,"score_spread":0.3319042926741596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}