{"id":"W4417312283","doi":"10.18280/isi.301008","title":"Enhancing Sentiment Analysis Accuracy Through Intelligent Spelling Correction Using Damerau-Levenshtein Distance and N-Gram with Random Forest Classifier","year":2025,"lang":"","type":"article","venue":"Ingénierie des systèmes d information","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Random forest; Classifier (UML); Sentiment analysis; Pattern recognition (psychology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001254052,0.0006312953,0.001026498,0.001259177,0.001266388,0.002655522,0.0004738443,0.0002232379,0.00006209788],"category_scores_gemma":[0.0003688114,0.0005864597,0.0003939457,0.005006291,0.0003244343,0.006889738,0.0003254868,0.0003895306,0.00002664629],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008210096,"about_ca_system_score_gemma":0.000296884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005796208,"about_ca_topic_score_gemma":0.0004184085,"domain_scores_codex":[0.9954708,0.0002187345,0.002083557,0.000707642,0.0007752818,0.0007440147],"domain_scores_gemma":[0.9960698,0.0004431523,0.001610955,0.0007610497,0.0009615765,0.0001534563],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008771025,0.0002642987,0.08215135,0.001304951,0.007529838,0.00001440669,0.05016619,0.5655611,0.0003630123,0.01079591,0.000355712,0.2806161],"study_design_scores_gemma":[0.001214424,0.0001127826,0.002451185,0.001545882,0.001738925,0.000011255,0.004942712,0.9780207,0.006400561,0.0007933076,0.002142584,0.0006257394],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1568394,0.001154318,0.8382526,0.0000564018,0.001691159,0.0006608197,0.000006929828,0.00008848324,0.001249856],"genre_scores_gemma":[0.9722452,0.0005932978,0.02625415,0.0002187029,0.0001315531,0.00003013564,0.00009436647,0.0000176482,0.0004149022],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8154058,"threshold_uncertainty_score":0.9996587,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02123104573779639,"score_gpt":0.2740604898328334,"score_spread":0.252829444095037,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}