{"id":"W2340990333","doi":"10.1177/0165551515625030","title":"The impact of indexing approaches on Arabic text classification","year":2016,"lang":"en","type":"article","venue":"Journal of Information Science","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Classifier (UML); Search engine indexing; Arabic; Artificial intelligence; Naive Bayes classifier; Natural language processing; Pattern recognition (psychology); Word (group theory); Mathematics; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01002092,0.001663661,0.001440257,0.004119969,0.001964013,0.004069255,0.001203585,0.001331207,0.001262582],"category_scores_gemma":[0.04268224,0.000314755,0.001078802,0.004588072,0.001007941,0.005105977,0.001754612,0.001416223,0.002071179],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001282804,"about_ca_system_score_gemma":0.001486847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006092732,"about_ca_topic_score_gemma":0.00605629,"domain_scores_codex":[0.9861135,0.005568518,0.002086437,0.001580959,0.004045274,0.0006053484],"domain_scores_gemma":[0.9625334,0.02616159,0.001832206,0.002572795,0.006363461,0.0005366582],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002294645,0.0009363249,0.03883403,0.0009303804,0.0003842525,0.0002083059,0.0008870443,0.0110615,0.02534007,0.0009733319,0.004488742,0.9136614],"study_design_scores_gemma":[0.0004333887,0.00944893,0.1683647,0.001047567,0.002156492,0.003922848,0.009303888,0.5481142,0.2160703,0.01091584,0.02953406,0.0006877108],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9017443,0.01549091,0.06092926,0.001202547,0.0008277833,0.0006419016,0.001166588,0.0026184,0.01537821],"genre_scores_gemma":[0.895936,0.003308946,0.09432859,0.0002772021,0.0003515175,0.0001883304,0.002883717,0.0002385144,0.002487407],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01002092,"threshold_uncertainty_score":0.05299634,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06173573844272501,"score_gpt":0.3028807703285067,"score_spread":0.2411450318857817,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}