{"id":"W4415591465","doi":"10.1101/2025.10.25.25338798","title":"Benchmarking Large Language Models and Clinicians Using Locally Generated Primary Healthcare Vignettes in Kenya","year":2025,"lang":"","type":"preprint","venue":"medRxiv","topic":"Migration, Health and Trauma","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Programs for Assessment of Technology in Health Research Institute","funders":"Bill and Melinda Gates Foundation","keywords":"Rubric; Benchmarking; Likert scale; Pairwise comparison; Harm; Health care; Scale (ratio); Relevance (law); Logistic regression","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03375245,0.00077605,0.0006170645,0.001259848,0.001066074,0.002107915,0.001585973,0.001528444,0.002818445],"category_scores_gemma":[0.1231073,0.000530553,0.0007759631,0.001402138,0.001240836,0.002007561,0.00223426,0.001866992,0.0007518094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006042328,"about_ca_system_score_gemma":0.003330533,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02580542,"about_ca_topic_score_gemma":0.0362772,"domain_scores_codex":[0.9727302,0.02312695,0.001013313,0.001598657,0.001116853,0.0004140253],"domain_scores_gemma":[0.8726097,0.1132243,0.003930268,0.003877981,0.00475874,0.001598996],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006507296,0.004823051,0.4598102,0.002818224,0.001090549,0.0009738364,0.01869262,0.2755045,0.002136424,0.007292025,0.02151053,0.1988407],"study_design_scores_gemma":[0.001668876,0.00381867,0.2073172,0.002011413,0.0004596149,0.0007069675,0.01317332,0.726091,0.005224184,0.01583004,0.0232323,0.000466393],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9855829,0.0006805367,0.006844413,0.001477674,0.00004383493,0.0006429277,0.002010631,0.0001237389,0.002593196],"genre_scores_gemma":[0.9803103,0.000188375,0.01547978,0.0002882336,0.00001642582,0.0006864899,0.002683954,0.00002843,0.0003179441],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03375245,"threshold_uncertainty_score":0.1785021,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05269265279545338,"score_gpt":0.3673083311388636,"score_spread":0.3146156783434102,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}