{"id":"W2982404528","doi":"10.1109/dasc/picom/cbdcom/cyberscitech.2019.00185","title":"Evaluating the Performance of Machine Learning Sentiment Analysis Algorithms in Software Engineering","year":2019,"lang":"en","type":"article","venue":"","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Sentiment analysis; Computer science; Lexicon; Machine learning; Artificial intelligence; Domain (mathematical analysis); Software; Algorithm; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008404131,0.0009233232,0.0006784788,0.002803925,0.0005969168,0.001458398,0.0006109442,0.001072133,0.0007206629],"category_scores_gemma":[0.022903,0.0001440298,0.0006029925,0.002065541,0.0003334513,0.001555831,0.0005964086,0.0006172736,0.0005092992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008654477,"about_ca_system_score_gemma":0.0005662757,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001579372,"about_ca_topic_score_gemma":0.001714684,"domain_scores_codex":[0.9935014,0.003017265,0.0008272397,0.000567737,0.001793331,0.00029303],"domain_scores_gemma":[0.985789,0.008227636,0.00100458,0.0007561985,0.003977969,0.0002444658],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003642451,0.002542805,0.0953308,0.001609411,0.001119875,0.0002169833,0.0007089057,0.08560658,0.04313553,0.003717117,0.01723573,0.7451337],"study_design_scores_gemma":[0.0001600319,0.002261017,0.06048073,0.0001078157,0.0001796715,0.0001610284,0.0006003164,0.8916954,0.03645647,0.002809862,0.005023597,0.00006402202],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9336615,0.001795788,0.05508986,0.0004952702,0.0003501719,0.000461833,0.001518472,0.001107901,0.005519153],"genre_scores_gemma":[0.8914573,0.0005055219,0.1024862,0.0001236818,0.0001169931,0.0002508005,0.004126349,0.00007928807,0.0008537984],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008404131,"threshold_uncertainty_score":0.04444587,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02107854380345215,"score_gpt":0.2850877187195719,"score_spread":0.2640091749161198,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}