{"id":"W4399364455","doi":"10.1145/3630106.3658993","title":"A Robot Walks into a Bar: Can Language Models Serve as Creativity SupportTools for Comedy? An Evaluation of LLMs’ Humour Alignment with Comedians","year":2024,"lang":"en","type":"article","venue":"","topic":"Humor Studies and Applications","field":"Psychology","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"Google (Canada)","funders":"","keywords":"Comedy; Creativity; Offensive; Sociology; Value (mathematics); Censorship; Media studies; Psychology; Visual arts; Computer science; Law; Engineering; Social psychology; Political science; Art","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02117656,0.0006095271,0.0003678556,0.001290322,0.002413976,0.006232543,0.001401181,0.001558592,0.003659287],"category_scores_gemma":[0.05740946,0.0002880116,0.0003593171,0.0006116632,0.004643086,0.003762853,0.005221678,0.001486331,0.0008262261],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002061771,"about_ca_system_score_gemma":0.001350095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001851825,"about_ca_topic_score_gemma":0.003028287,"domain_scores_codex":[0.9854999,0.01165035,0.0002950327,0.0005763352,0.001541805,0.0004365749],"domain_scores_gemma":[0.9488635,0.03815473,0.002564602,0.004390999,0.003370199,0.002655909],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.001644286,0.001322007,0.05600519,0.0009436551,0.0000768229,0.001123609,0.7460319,0.002226599,0.01578578,0.01011323,0.00278599,0.1619408],"study_design_scores_gemma":[0.0003460884,0.006845196,0.05058479,0.0008063244,0.0001780071,0.001177235,0.7719689,0.02697387,0.01690149,0.009842747,0.1140984,0.0002769531],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9835385,0.00008248721,0.005175541,0.0008654886,0.00003055548,0.0001150158,0.00002809338,0.0001847341,0.009979451],"genre_scores_gemma":[0.9923091,0.00004245654,0.005454957,0.0002437994,0.00001173898,0.0000754136,0.00003451633,0.00005564162,0.001772431],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02117656,"threshold_uncertainty_score":0.1119937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0760012151814271,"score_gpt":0.4138802537248433,"score_spread":0.3378790385434162,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}