{"id":"W3199808029","doi":"10.36227/techrxiv.16569420.v1","title":"Soft Computing Approaches for tagging Arabic text","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Association of Canadian Archivists","funders":"","keywords":"Computer science; Artificial intelligence; Arabic; Support vector machine; Hindi; Natural language processing; Artificial neural network; Process (computing); Multilayer perceptron; Perceptron; Part of speech; Machine learning; Speech recognition; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0007475162,0.0003768514,0.0005179472,0.0002373842,0.00019171,0.001162299,0.001629801,0.0003503808,0.00003877188],"category_scores_gemma":[0.0001447457,0.000376599,0.0003587703,0.0002330674,0.00005073472,0.0003013712,0.003002969,0.0005778217,0.00002028596],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009382009,"about_ca_system_score_gemma":0.0002713878,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003045731,"about_ca_topic_score_gemma":0.000009271869,"domain_scores_codex":[0.9973764,0.0001028694,0.0005145177,0.001224363,0.000304903,0.0004768829],"domain_scores_gemma":[0.9979353,0.0003256391,0.0002789916,0.001070247,0.0002728607,0.0001169829],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001842381,0.0001187992,0.00004918643,0.0005456283,0.00008519043,0.00001264398,0.0008720876,0.0004326564,0.0001389871,0.02216716,0.00328866,0.9722872],"study_design_scores_gemma":[0.0002859923,0.00005055046,0.0001148582,0.0006412054,0.00003071025,0.00004776173,0.000232127,0.9319678,0.02555168,0.03645814,0.003568149,0.001050999],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001206368,0.0002706597,0.9848357,0.001287747,0.0005377834,0.0008630952,0.000006054121,0.00197201,0.009020533],"genre_scores_gemma":[0.2978784,0.00001471628,0.7003844,0.000798295,0.0002029536,0.0001799335,0.00006918635,0.00003114212,0.0004410294],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9712362,"threshold_uncertainty_score":0.9998746,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06861444804767225,"score_gpt":0.28092561340744,"score_spread":0.2123111653597677,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}