{"id":"W4386753352","doi":"10.21203/rs.3.rs-3343151/v1","title":"Homogenous Ensemble Boosting Approach to Improve the Consistency in the Accuracy of Text Data Classification","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; New Brunswick Innovation Foundation","keywords":"Sentiment analysis; Computer science; Consistency (knowledge bases); Artificial intelligence; Globe; The Internet; Machine learning; Boosting (machine learning); Unstructured data; Data science; Natural language processing; Data mining; Big data; World Wide Web","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006070892,0.0009299408,0.002390389,0.001586782,0.0008102637,0.001224763,0.001953901,0.001188195,0.001430233],"category_scores_gemma":[0.008718853,0.0003618195,0.00117419,0.001236452,0.000463502,0.001527801,0.001037602,0.001383638,0.0008106974],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006333484,"about_ca_system_score_gemma":0.0009821693,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003011832,"about_ca_topic_score_gemma":0.002858941,"domain_scores_codex":[0.9979054,0.000694623,0.0001684123,0.0004543301,0.0005443231,0.000232914],"domain_scores_gemma":[0.9950491,0.001677962,0.0002909111,0.0006201963,0.002199571,0.0001623187],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000865884,0.0006832053,0.01353514,0.0001656564,0.0004332472,0.0002206971,0.0002766362,0.3322865,0.02159481,0.005265501,0.008457698,0.6162151],"study_design_scores_gemma":[0.000007688519,0.00005235901,0.0007525454,0.000006749411,0.00003496358,0.00001968835,0.00001467098,0.9959829,0.001722669,0.0009093576,0.0004914984,0.000004881661],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1310668,0.001566779,0.8614131,0.000467109,0.0003711141,0.0001763632,0.0002141087,0.001321928,0.003402678],"genre_scores_gemma":[0.8206493,0.0003811762,0.1751178,0.0003123335,0.0003478765,0.0001304849,0.0006438189,0.0001249645,0.002292198],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006070892,"threshold_uncertainty_score":0.03210634,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3689404806849977,"score_gpt":0.4456225909170176,"score_spread":0.07668211023201982,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}