{"id":"W7082249071","doi":"10.48448/f320-g937","title":"AfriMed-QA: A Pan-African, Multi-Specialty, Medical Question-Answering Benchmark Dataset","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Economic shortage; Correctness; Benchmark (surveying); Health care; Sophistication; Preference; Healthcare system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003035769,0.001214579,0.0005785784,0.003120105,0.0009717937,0.001006692,0.001590943,0.00240584,0.01103454],"category_scores_gemma":[0.01877537,0.0002742203,0.001054096,0.002484631,0.0005071674,0.001382643,0.002013318,0.00144462,0.006987369],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001733396,"about_ca_system_score_gemma":0.002396289,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02301287,"about_ca_topic_score_gemma":0.03641169,"domain_scores_codex":[0.9975681,0.001044611,0.0003139479,0.0004778782,0.0004251273,0.0001703302],"domain_scores_gemma":[0.9936268,0.00331204,0.0004620817,0.0007490395,0.001342664,0.0005073886],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001293583,0.0009926946,0.04392819,0.00403161,0.0002893301,0.0005292668,0.001139427,0.007535684,0.004169227,0.003449061,0.8414967,0.09114517],"study_design_scores_gemma":[0.001138449,0.0006513629,0.1408503,0.001313113,0.0002045055,0.001554367,0.002317014,0.0393853,0.008565504,0.008573849,0.7952431,0.0002031818],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.06009388,0.00210208,0.005127671,0.002889712,0.0002530755,0.0005744888,0.9183148,0.003135656,0.007508713],"genre_scores_gemma":[0.04927515,0.0002915597,0.01071598,0.0006565501,0.00006000744,0.0006122349,0.9362029,0.0001266267,0.002058904],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02301287,"threshold_uncertainty_score":0.04575783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01806332233446785,"score_gpt":0.2990691703336037,"score_spread":0.2810058479991359,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}