{"id":"W4416037816","doi":"10.18653/v1/2025.arabicnlp-sharedtasks.20","title":"MedLingua at MedArabiQ2025: Zero- and Few-Shot Prompting of Large Language Models for Arabic Medical QA","year":2025,"lang":"","type":"article","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Arabic; Natural language; Language model; Language identification","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.004108855,0.000507392,0.0009933576,0.0003859871,0.0006207713,0.0001398824,0.001568699,0.0005499815,0.0003839743],"category_scores_gemma":[0.003369102,0.0004625283,0.0002323395,0.0008211641,0.0002604055,0.0003655691,0.002291422,0.0008814562,0.000008714973],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001796777,"about_ca_system_score_gemma":0.001152802,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000621066,"about_ca_topic_score_gemma":0.0006656231,"domain_scores_codex":[0.9941452,0.0004861452,0.001470105,0.001459104,0.001227316,0.001212112],"domain_scores_gemma":[0.9953329,0.002046613,0.0004679905,0.001168497,0.0004616085,0.0005224144],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004046343,0.0008451966,0.02909717,0.01715831,0.0005011748,0.0001414278,0.04538522,0.003901781,0.001020163,0.6008051,0.004879248,0.2958606],"study_design_scores_gemma":[0.001803393,0.000264304,0.0008614503,0.001304714,0.00005367952,0.00003078126,0.0004041061,0.9857146,0.001459312,0.00640986,0.001295003,0.0003988045],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2089382,0.008498117,0.7498838,0.0178278,0.001639513,0.00209162,0.0000550861,0.0002751195,0.01079083],"genre_scores_gemma":[0.9622238,0.0001729702,0.0289649,0.002219232,0.0001591027,0.0000957846,0.00001758323,0.00003766523,0.006108982],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9818128,"threshold_uncertainty_score":0.9997826,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02693583828171458,"score_gpt":0.3626651479881062,"score_spread":0.3357293097063916,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}