{"id":"W4400261110","doi":"10.2196/57674","title":"Data Set and Benchmark (MedGPTEval) to Evaluate Responses From Large Language Models in Medicine: Evaluation Development and Validation","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Benchmark (surveying); Set (abstract data type); Data set; Data mining; Natural language processing; Artificial intelligence; Data science; Programming language; Cartography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01966895,0.002134612,0.0009968156,0.003697473,0.0009332591,0.001435093,0.002240453,0.001941345,0.002638493],"category_scores_gemma":[0.04984771,0.0003676549,0.001529493,0.002410663,0.00100026,0.001503049,0.003032014,0.001214838,0.001610265],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002117116,"about_ca_system_score_gemma":0.002823362,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006389601,"about_ca_topic_score_gemma":0.006671231,"domain_scores_codex":[0.9787572,0.01231787,0.003280599,0.002026343,0.00321144,0.0004065119],"domain_scores_gemma":[0.9648373,0.01793228,0.002696582,0.003461874,0.009564011,0.001507856],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0128317,0.01060967,0.1525429,0.0232018,0.003801378,0.002116696,0.005230629,0.08323152,0.02623783,0.007247536,0.2003134,0.472635],"study_design_scores_gemma":[0.006147021,0.01514104,0.3281161,0.002926534,0.002242893,0.002427971,0.004939892,0.4079227,0.067154,0.006638321,0.1554891,0.0008545595],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7308983,0.003808332,0.09392529,0.001955218,0.0005851059,0.02603749,0.1133745,0.01178186,0.01763383],"genre_scores_gemma":[0.6280479,0.0007124188,0.1416439,0.001205325,0.0001279197,0.03104293,0.1926525,0.0005829,0.003984265],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01966895,"threshold_uncertainty_score":0.1040206,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2744553495825037,"score_gpt":0.5143539930829046,"score_spread":0.239898643500401,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}