{"id":"W2982404528","doi":"10.1109/dasc/picom/cbdcom/cyberscitech.2019.00185","title":"Evaluating the Performance of Machine Learning Sentiment Analysis Algorithms in Software Engineering","year":2019,"lang":"en","type":"article","venue":"","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Sentiment analysis; Computer science; Lexicon; Machine learning; Artificial intelligence; Domain (mathematical analysis); Software; Algorithm; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00100095,0.00009514566,0.0002178898,0.0003068181,0.00004646311,0.00005454473,0.0004252027,0.00002144363,0.0001859848],"category_scores_gemma":[0.00003341334,0.00006551559,0.000132134,0.001522389,0.00000654632,0.0002020183,0.0002000125,0.0001289594,0.00002605377],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002383992,"about_ca_system_score_gemma":0.00001436246,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006440654,"about_ca_topic_score_gemma":0.000004494681,"domain_scores_codex":[0.9988427,0.00005395093,0.0003069405,0.0002417517,0.0003756981,0.0001789941],"domain_scores_gemma":[0.9993191,0.0001561284,0.000121141,0.0003358065,0.00004526106,0.00002253741],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001102148,0.0000124112,0.4735081,0.000007955407,0.0001271481,2.486599e-7,0.0004080303,0.5175853,0.0006656094,0.0001284247,0.000001167231,0.007554485],"study_design_scores_gemma":[0.0001530785,0.00006393807,0.04497671,0.00001602851,0.00003308895,4.206545e-7,0.00004800047,0.9526036,0.00199166,0.000001270933,0.00003188017,0.00008032307],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9101099,0.0001326675,0.08943706,0.00005344655,0.00007565079,0.00007772644,1.606054e-7,0.0000368294,0.00007649902],"genre_scores_gemma":[0.9446515,0.0000111626,0.05459686,0.0000229648,0.00001056209,0.000003936313,0.000003578251,0.000004150387,0.0006953043],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4350182,"threshold_uncertainty_score":0.2671648,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02107854380345215,"score_gpt":0.2850877187195719,"score_spread":0.2640091749161198,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}