{"id":"W4415641333","doi":"10.1101/2025.10.27.25338910","title":"Human Evaluators vs. LLM-as-a-Judge: Toward Scalable, Real-Time Evaluation of GenAI in Global Health","year":2025,"lang":"","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Programs for Assessment of Technology in Health Research Institute","funders":"Bill and Melinda Gates Foundation","keywords":"Context (archaeology); Empathy; Bottleneck; Reliability (semiconductor); Health care; Gold standard (test); Operationalization; Global health","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1106977,0.001109594,0.0009855598,0.002087614,0.001187996,0.004739011,0.001644006,0.001589907,0.003327137],"category_scores_gemma":[0.2022198,0.0005341257,0.0007117033,0.001259028,0.002102398,0.002417199,0.005666129,0.00190726,0.001054601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002252504,"about_ca_system_score_gemma":0.002206138,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002749766,"about_ca_topic_score_gemma":0.004791773,"domain_scores_codex":[0.8631085,0.1224677,0.003136964,0.004788101,0.005343616,0.001155132],"domain_scores_gemma":[0.8200312,0.1357307,0.01399514,0.009111452,0.01705525,0.004076183],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01052235,0.003043588,0.2093463,0.003156741,0.0009961214,0.0006277057,0.05316664,0.06822775,0.02745023,0.01036238,0.02130822,0.5917919],"study_design_scores_gemma":[0.001412133,0.008184543,0.1580471,0.001703174,0.0004646525,0.0004430591,0.02733061,0.7108621,0.03731698,0.02622515,0.02721645,0.0007939709],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8376737,0.0008447006,0.1393285,0.001981403,0.0002807856,0.00276899,0.0006107112,0.002172204,0.01433903],"genre_scores_gemma":[0.9103001,0.0001184523,0.08687907,0.0004016221,0.00005899148,0.001070997,0.0002866754,0.0001178726,0.0007661252],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8893023,"threshold_uncertainty_score":0.5854323,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.173656723964162,"score_gpt":0.5036308675150175,"score_spread":0.3299741435508555,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}