{"id":"W4416539290","doi":"10.48550/arxiv.2504.06166","title":"Assessing how hyperparameters impact Large Language Models' sarcasm detection performance","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Sentiment Analysis and Opinion Mining","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sarcasm; Hyperparameter; Natural language understanding; Language model; Computational linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005588326,0.0003773372,0.0004600164,0.0004320087,0.0002809136,0.00093743,0.001068388,0.0002476372,0.0000197098],"category_scores_gemma":[0.00004280093,0.0003430198,0.0004216277,0.0005355364,0.00002735619,0.00133263,0.001313245,0.0006812053,0.0000336517],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001982914,"about_ca_system_score_gemma":0.0001649134,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009434081,"about_ca_topic_score_gemma":0.00001325283,"domain_scores_codex":[0.9977774,0.0001312301,0.0003238769,0.0008770949,0.0003847039,0.0005056878],"domain_scores_gemma":[0.9982187,0.00008730137,0.0003194236,0.001160403,0.0001064831,0.000107668],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003716401,0.0004488904,0.6458542,0.0006413381,0.001538431,0.00008923354,0.008797065,0.0835688,0.009893743,0.0004137124,0.0008820986,0.2478354],"study_design_scores_gemma":[0.0002594919,0.00003265007,0.04976938,0.0002346857,0.00006932368,0.000004699005,0.0002475666,0.9430835,0.00567472,0.00007629221,0.00012678,0.0004208764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7215838,0.0004021586,0.2760612,0.000176701,0.0008070542,0.000125036,0.000005127791,0.0001772018,0.0006617357],"genre_scores_gemma":[0.9923378,0.0001071719,0.006488769,0.0001732988,0.0001697244,0.00002299956,0.00003797034,0.00001665135,0.0006456603],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8595147,"threshold_uncertainty_score":0.9999022,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0628139093902101,"score_gpt":0.327784563086382,"score_spread":0.264970653696172,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}