{"id":"W4416539290","doi":"10.48550/arxiv.2504.06166","title":"Assessing how hyperparameters impact Large Language Models' sarcasm detection performance","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sarcasm; Hyperparameter; Natural language understanding; Language model; Computational linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009200376,0.003004139,0.001263261,0.001382514,0.001104472,0.003221471,0.002018451,0.002868659,0.002561277],"category_scores_gemma":[0.03151817,0.0009128341,0.001382537,0.0007545635,0.001017168,0.004734131,0.001465742,0.005337833,0.002834695],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001864376,"about_ca_system_score_gemma":0.001580635,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01253602,"about_ca_topic_score_gemma":0.02183955,"domain_scores_codex":[0.9966774,0.001544461,0.0002118406,0.0009688456,0.0003350876,0.0002623567],"domain_scores_gemma":[0.9893404,0.007630148,0.0003740599,0.00133836,0.0009215155,0.000395495],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003262066,0.001654212,0.07269409,0.001524265,0.002225677,0.0007861487,0.001611688,0.4009055,0.02414463,0.004423384,0.07272146,0.4140468],"study_design_scores_gemma":[0.0002488838,0.0006543518,0.007779907,0.0002853347,0.0003422727,0.0003386503,0.0005747118,0.9651429,0.01063653,0.007408922,0.006461535,0.0001259535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.79761,0.01330699,0.1311246,0.007031496,0.002105333,0.0006758129,0.006039693,0.02553329,0.01657281],"genre_scores_gemma":[0.9422528,0.0009148717,0.04312495,0.001332048,0.0002200668,0.0003745141,0.007460866,0.0009304929,0.003389421],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01253602,"threshold_uncertainty_score":0.04865682,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0628139093902101,"score_gpt":0.327784563086382,"score_spread":0.264970653696172,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}