{"id":"W2741691725","doi":"10.18653/v1/p17-2074","title":"Best-Worst Scaling More Reliable than Rating Scales: A Case Study on Sentiment Intensity Annotation","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Annotation; Consistency (knowledge bases); Rating scale; Computer science; Set (abstract data type); Scaling; Scale (ratio); Quality (philosophy); Data mining; Data set; Information retrieval; Artificial intelligence; Natural language processing; Machine learning; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02233895,0.001315068,0.001449378,0.002253723,0.002393853,0.002697636,0.00162876,0.002300571,0.002515937],"category_scores_gemma":[0.08800716,0.0004670388,0.001010194,0.003544354,0.002012017,0.003243357,0.002325208,0.002409124,0.002174807],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001266027,"about_ca_system_score_gemma":0.0008218929,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003276371,"about_ca_topic_score_gemma":0.004440242,"domain_scores_codex":[0.9673405,0.02072316,0.001825159,0.003545298,0.005954995,0.0006107969],"domain_scores_gemma":[0.8967538,0.07572263,0.004005219,0.01236827,0.009920527,0.001229556],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003927668,0.001138864,0.04146016,0.004563628,0.0006166853,0.004310302,0.02039791,0.02644286,0.1018478,0.02167257,0.06688321,0.7067383],"study_design_scores_gemma":[0.0007610424,0.001991317,0.07430875,0.001021631,0.0005527759,0.005666588,0.01471845,0.5221007,0.127573,0.06837029,0.1823769,0.0005586281],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5030389,0.004503364,0.4420683,0.00549476,0.001347344,0.001001108,0.003049759,0.008143158,0.03135328],"genre_scores_gemma":[0.6842792,0.0005730095,0.304696,0.0006559998,0.0003741427,0.0005007833,0.002328824,0.001774907,0.0048171],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02233895,"threshold_uncertainty_score":0.1181411,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0678353615902395,"score_gpt":0.3486248772234626,"score_spread":0.2807895156332231,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}