{"id":"W4416037816","doi":"10.18653/v1/2025.arabicnlp-sharedtasks.20","title":"MedLingua at MedArabiQ2025: Zero- and Few-Shot Prompting of Large Language Models for Arabic Medical QA","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Natural language; Language model; Language identification","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003967254,0.001620321,0.0008748314,0.0008470014,0.0006818713,0.001978436,0.001901431,0.002199179,0.0275318],"category_scores_gemma":[0.01331002,0.000576189,0.000894366,0.0003416559,0.000590449,0.003844578,0.003474481,0.002742783,0.01381455],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001208529,"about_ca_system_score_gemma":0.001634071,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003706682,"about_ca_topic_score_gemma":0.004204811,"domain_scores_codex":[0.998106,0.0009095414,0.0001045089,0.0005548246,0.000235301,0.00008983591],"domain_scores_gemma":[0.9958954,0.002486535,0.0001281304,0.0006001779,0.0005821669,0.0003076532],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004070614,0.0007377277,0.003127254,0.002573315,0.0003059831,0.001281395,0.003166643,0.01903412,0.05768139,0.009576716,0.2937363,0.6047086],"study_design_scores_gemma":[0.00105925,0.002085642,0.003813126,0.0003385014,0.0001992492,0.002244993,0.002190562,0.6049017,0.1146071,0.02919415,0.2389687,0.0003971726],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0969383,0.002129095,0.4448197,0.002919387,0.001644651,0.001315517,0.01542498,0.4207845,0.01402386],"genre_scores_gemma":[0.4897586,0.000465395,0.4517946,0.00246314,0.0003738053,0.0008650653,0.03208698,0.006378181,0.01581422],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0275318,"threshold_uncertainty_score":0.09210306,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02693583828171458,"score_gpt":0.3626651479881062,"score_spread":0.3357293097063916,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}