{"id":"W4406499798","doi":"10.1109/cascon62161.2024.10838185","title":"Sentiment Analysis with LLMs: Evaluating QLoRA Fine-Tuning, Instruction Strategies, and Prompt Sensitivity","year":2024,"lang":"en","type":"article","venue":"","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Sensitivity (control systems); Computer science; Sentiment analysis; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004421943,0.002015737,0.001176564,0.0008746706,0.0004876012,0.001532757,0.001906374,0.001482045,0.003585462],"category_scores_gemma":[0.02296299,0.0005705917,0.0008965416,0.0005399869,0.0005251517,0.003143851,0.001505961,0.002460624,0.002107857],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001361349,"about_ca_system_score_gemma":0.001411366,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008273629,"about_ca_topic_score_gemma":0.01058573,"domain_scores_codex":[0.9980251,0.000844321,0.0001743327,0.000527483,0.0002653563,0.0001635579],"domain_scores_gemma":[0.9935412,0.00417571,0.000263052,0.0008641338,0.0009047601,0.000251027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004596216,0.001626736,0.02524746,0.00131661,0.0005822915,0.0004484683,0.001206061,0.3269033,0.03358744,0.002459706,0.02314819,0.5788775],"study_design_scores_gemma":[0.0001916653,0.0004258573,0.001763903,0.00003943389,0.00007564546,0.00005475312,0.0002325795,0.9831166,0.009877712,0.001339173,0.002838861,0.00004379811],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7098505,0.005080078,0.1875589,0.001542185,0.0008243271,0.0007738817,0.00243196,0.08512034,0.006817759],"genre_scores_gemma":[0.8959363,0.0003993406,0.09593073,0.0006840522,0.0000783864,0.0004779586,0.002902626,0.001013675,0.002576834],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008273629,"threshold_uncertainty_score":0.02338576,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0271359459686862,"score_gpt":0.2991807555376677,"score_spread":0.2720448095689815,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}