{"id":"W4385572948","doi":"10.18653/v1/2022.blackboxnlp-1.8","title":"Post-hoc analysis of Arabic transformer models","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Transformer; Natural language processing; Arabic; Artificial intelligence; Vocabulary; Semitic languages; Arabic languages; Linguistics; Speech recognition; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001908248,0.00005221684,0.0001448389,0.0002588032,0.00006647007,0.00001439757,0.0005919162,0.00001150979,0.0003337619],"category_scores_gemma":[0.000001495337,0.00004750682,0.0001387325,0.0008871015,0.000007948818,0.000236876,0.00009054745,0.00006538188,0.000001957084],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002392679,"about_ca_system_score_gemma":0.00003769772,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001553293,"about_ca_topic_score_gemma":0.0000321172,"domain_scores_codex":[0.9991672,0.00003108687,0.0001823568,0.00020896,0.0002852008,0.0001251682],"domain_scores_gemma":[0.9994712,0.00002302943,0.0000321402,0.0004003996,0.00004153891,0.00003167978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006638741,0.00004637103,0.00007978002,0.000003472612,0.0001884311,0.000002132115,0.002016228,0.7194895,0.001126849,0.2475729,0.00003301897,0.02943468],"study_design_scores_gemma":[0.00008182484,0.00005796532,0.00008927123,3.181507e-7,0.00004992778,0.000001129877,0.00008856011,0.9952666,0.000253173,0.003862837,0.0001866957,0.00006176875],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1210306,0.00005583574,0.8701816,0.0006363367,0.00005656889,0.00004599238,0.000004636406,0.00004957616,0.00793882],"genre_scores_gemma":[0.9693077,0.000003268375,0.02955643,0.0004112838,0.000003998543,0.000009446235,0.000002331538,0.000002394195,0.0007031436],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8482771,"threshold_uncertainty_score":0.3654459,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02316346400162472,"score_gpt":0.2343240879897945,"score_spread":0.2111606239881697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}