{"id":"W7106801623","doi":"10.48448/4d9q-yh22","title":"Pearl: A Multimodal Culturally-Aware Arabic Instruction Dataset","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; University of British Columbia","funders":"","keywords":"Pearl; Arabic; Benchmark (surveying); Workflow; Benchmarking; Mainstream; Semantics (computer science)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009027494,0.002161665,0.0005252496,0.001907772,0.001165721,0.001433836,0.002421032,0.002035589,0.02059393],"category_scores_gemma":[0.006332749,0.0003348103,0.0009212357,0.001658405,0.0006196972,0.00239044,0.00294055,0.002082638,0.01509824],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001602799,"about_ca_system_score_gemma":0.001664988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02308828,"about_ca_topic_score_gemma":0.04069257,"domain_scores_codex":[0.9991436,0.0002728011,0.00007368416,0.0002562597,0.0001831837,0.00007043574],"domain_scores_gemma":[0.9987071,0.0004152074,0.00006547406,0.0003509668,0.0003538563,0.0001075079],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000585841,0.0005195476,0.007586636,0.002047777,0.0001320588,0.0007620692,0.001084099,0.01258202,0.006311931,0.00699144,0.7382494,0.2231472],"study_design_scores_gemma":[0.0002825016,0.0003471335,0.01690897,0.0009177694,0.0001248732,0.0009217173,0.003377103,0.110032,0.01761544,0.01772342,0.8314569,0.0002921521],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.09775669,0.003807528,0.06180714,0.003382475,0.001097201,0.00148561,0.7132603,0.05895343,0.05844958],"genre_scores_gemma":[0.09149011,0.0007007829,0.06753009,0.0008900768,0.00007508894,0.001700078,0.8228848,0.001338933,0.01338997],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02308828,"threshold_uncertainty_score":0.06889361,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03434384877265304,"score_gpt":0.3359097414307182,"score_spread":0.3015658926580651,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}