{"id":"W6996832240","doi":"","title":"Studying The Effectiveness Of Large Language Models In Benchmark Biomedical Tasks","year":2024,"lang":"en","type":"other","venue":"York University Digital Library (York University)","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; York University; Compute Canada","keywords":"Benchmark (surveying); Set (abstract data type); Language model; Work (physics); Domain (mathematical analysis)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007755735,0.003193851,0.001269556,0.002754486,0.0009927242,0.002292303,0.002219615,0.002026119,0.003318792],"category_scores_gemma":[0.01920194,0.0006259108,0.001654628,0.00178755,0.0007727377,0.004182009,0.001888981,0.002606466,0.003909026],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001947201,"about_ca_system_score_gemma":0.002519114,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01427604,"about_ca_topic_score_gemma":0.02418252,"domain_scores_codex":[0.9958867,0.001941087,0.0003084561,0.001152428,0.0005057519,0.0002056054],"domain_scores_gemma":[0.9893595,0.007822976,0.000388704,0.001151831,0.0009322182,0.0003447119],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002325879,0.00163956,0.01402643,0.002954674,0.001787986,0.0005443407,0.0005181282,0.2036418,0.01418489,0.003844157,0.07241651,0.6821156],"study_design_scores_gemma":[0.0003067348,0.001588421,0.00558745,0.0003540055,0.0006065698,0.0004669891,0.0005786889,0.9337595,0.02482704,0.008805467,0.02293733,0.0001818507],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5605058,0.05300387,0.2666316,0.007997358,0.004201302,0.001291466,0.02517994,0.0467182,0.03447057],"genre_scores_gemma":[0.7314489,0.0064985,0.1836008,0.002538234,0.000845424,0.0007495601,0.05984404,0.001690206,0.01278436],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01427604,"threshold_uncertainty_score":0.04101676,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009574604453611068,"score_gpt":0.1918108259559743,"score_spread":0.1822362215023632,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}