{"id":"W7126433153","doi":"10.21428/594757db.4bc3ecf6","title":"Instruction Tuning of LLMs for Multi-label EmotionClassification in Social Media Content","year":2024,"lang":"en","type":"article","venue":"","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Controllability; Inference; Social media; Generative model; Test (biology); Order (exchange); Training set; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002412017,0.00004799452,0.00009358375,0.0001997265,0.00004452555,0.00006713082,0.0001288427,0.00003261283,0.00001106813],"category_scores_gemma":[0.00002828087,0.00004225054,0.00005128557,0.000371118,0.00001557504,0.0002873497,0.00002929042,0.00003746961,0.000004441822],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003314765,"about_ca_system_score_gemma":0.00002536494,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001203734,"about_ca_topic_score_gemma":0.00002554161,"domain_scores_codex":[0.9993755,0.00001895689,0.0002316123,0.0001753325,0.0001169948,0.00008159214],"domain_scores_gemma":[0.9997397,0.00006658378,0.00005027385,0.00007295837,0.00005577942,0.00001467242],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004923736,0.00008002717,0.001521835,0.00003976045,0.00003724293,5.374718e-7,0.004356273,0.00001930601,0.02603424,0.4355829,0.0004119719,0.531911],"study_design_scores_gemma":[0.0004422811,0.00001094859,0.008589439,0.00002759055,0.000005752947,4.368133e-7,0.0004804734,0.9870753,0.002636671,0.0005000255,0.000177852,0.00005319296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.101073,0.00006899469,0.8975078,0.0005122739,0.0005486147,0.00007785171,0.000001038683,0.00005038639,0.0001600699],"genre_scores_gemma":[0.8743929,0.000008197954,0.1252975,0.00001697645,0.00008281269,0.00001082862,0.00001255212,0.000003506815,0.0001746526],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.987056,"threshold_uncertainty_score":0.1722927,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2369569092403301,"score_gpt":0.3485233860888012,"score_spread":0.1115664768484711,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}