{"id":"W6966818977","doi":"10.48448/ryqh-5c02","title":"AraMUS: Pushing the Limits of Data and Model Scale for Arabic Natural Language Processing","year":2022,"lang":"en","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Mennonite University","funders":"","keywords":"Arabic; Set (abstract data type); Generative grammar; Natural language; Generative model; Scale (ratio); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002495558,0.002974249,0.0008301202,0.001285749,0.001002602,0.002764046,0.003273123,0.001544871,0.03020544],"category_scores_gemma":[0.01243126,0.001015121,0.001286698,0.0009558024,0.0007518321,0.006541496,0.004915744,0.002957124,0.03060115],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008342051,"about_ca_system_score_gemma":0.001390178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01025272,"about_ca_topic_score_gemma":0.0129201,"domain_scores_codex":[0.998605,0.0003501231,0.00009597024,0.0005328592,0.0003285556,0.0000875762],"domain_scores_gemma":[0.9969411,0.001100098,0.00006540793,0.001197792,0.0005304303,0.0001651235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001412027,0.0002775643,0.001761797,0.0007245535,0.0003588451,0.0004033528,0.0005534916,0.03052238,0.02162586,0.006813416,0.3502269,0.5853198],"study_design_scores_gemma":[0.0003045543,0.0003018349,0.001841224,0.000176708,0.0001185289,0.0003861107,0.0004270155,0.6676115,0.05777629,0.02784845,0.2430134,0.0001943521],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04547846,0.00313196,0.4613639,0.002956693,0.001978459,0.0004405777,0.02245169,0.4384792,0.02371891],"genre_scores_gemma":[0.2215576,0.001390491,0.6232135,0.001817708,0.0004614639,0.00117792,0.09688777,0.02475083,0.02874271],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03020544,"threshold_uncertainty_score":0.1010472,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09527861069448368,"score_gpt":0.3779555220247786,"score_spread":0.2826769113302949,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}