{"id":"W4416037765","doi":"10.18653/v1/2025.arabicnlp-sharedtasks.107","title":"PalmX 2025: The First Shared Task on Benchmarking LLMs on Arabic and Islamic Culture","year":2025,"lang":"","type":"article","venue":"","topic":"Mathematics, Computing, and Information Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Task (project management); Benchmarking; Islam; Arabic; Islamic culture","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0008592976,0.0004632894,0.0003933533,0.000259804,0.001748731,0.002323336,0.001355279,0.0001931507,0.0002558078],"category_scores_gemma":[0.0002025149,0.0003000038,0.000136887,0.0009089144,0.0001641859,0.0008649832,0.0006687224,0.0005944318,0.0001487918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009457331,"about_ca_system_score_gemma":0.0001641307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001602424,"about_ca_topic_score_gemma":0.00001780949,"domain_scores_codex":[0.9972451,0.00007523432,0.0008983085,0.0006715964,0.0005566547,0.0005531445],"domain_scores_gemma":[0.9979224,0.0004060343,0.0004688177,0.0009156081,0.0002163745,0.00007082145],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002679108,0.0002442248,0.0002022303,0.001393528,0.0001235829,0.000005182816,0.04869727,0.00262016,0.00001908489,0.5924267,0.1178853,0.2363559],"study_design_scores_gemma":[0.0005503353,0.0002018264,0.0008934197,0.002120529,0.00002888118,0.00000955638,0.0005764797,0.9375442,0.0002350631,0.02335894,0.03406849,0.0004122555],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.00843521,0.001020398,0.4749243,0.01079853,0.002817559,0.0009736672,0.000008021309,0.0002593299,0.500763],"genre_scores_gemma":[0.9634309,0.0003016586,0.00556522,0.01685649,0.0002689017,0.00002123287,0.000007181331,0.00001505943,0.01353334],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9549957,"threshold_uncertainty_score":0.9999452,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01208333510318101,"score_gpt":0.2448959339539454,"score_spread":0.2328125988507644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}