{"id":"W4413800245","doi":"10.1101/2025.08.26.25333975","title":"How Good Are Large Language Models at Supporting Frontline Healthcare Workers in Low-Resource Settings – A Benchmarking Study &amp; Dataset","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Programs for Assessment of Technology in Health Research Institute","funders":"Bill and Melinda Gates Foundation","keywords":"Rubric; Benchmarking; Likert scale; Metric (unit); Health care; Resource (disambiguation); Medicine; Psychology; Computer science; Business; Marketing; Economics; Economic growth; Developmental psychology; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01187243,0.001049147,0.0005377803,0.001207105,0.0006871229,0.001511346,0.001635641,0.001407196,0.002403536],"category_scores_gemma":[0.02956498,0.0002654595,0.0008905419,0.001214474,0.0007102993,0.002037911,0.001850443,0.001494425,0.001487208],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001767611,"about_ca_system_score_gemma":0.001844913,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01427909,"about_ca_topic_score_gemma":0.01884174,"domain_scores_codex":[0.9906623,0.006692906,0.0006063514,0.001069919,0.0006291011,0.0003394591],"domain_scores_gemma":[0.978396,0.01487384,0.0009849289,0.002326038,0.002671131,0.0007479729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003908814,0.00509585,0.3922757,0.00324274,0.001106922,0.001029647,0.004232964,0.1384679,0.007085019,0.003592782,0.165582,0.2743796],"study_design_scores_gemma":[0.00157445,0.004270519,0.2322438,0.0009715336,0.0006287366,0.0008983825,0.008254446,0.6478724,0.01943724,0.007113266,0.07637408,0.0003610234],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9336516,0.00126504,0.01395174,0.002686827,0.0002399823,0.0007640664,0.04016537,0.002484665,0.004790646],"genre_scores_gemma":[0.8821861,0.0002548737,0.02365655,0.0005717467,0.00007360648,0.0008343954,0.09063247,0.0001340135,0.001656276],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01427909,"threshold_uncertainty_score":0.06278813,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09466266633977251,"score_gpt":0.4145056247089073,"score_spread":0.3198429583691348,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}