{"id":"W7106815133","doi":"10.48448/whrs-et52","title":"Can Large Language Models Outperform Non-Experts in Poetry Evaluation? A Comparative Study Using the Consensual Assessment Technique","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Poetry; Creativity; Matching (statistics); Language model; Ground truth; Expression (computer science)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01748245,0.001689214,0.001014901,0.00371393,0.0006799887,0.003955699,0.001574683,0.001706959,0.003411732],"category_scores_gemma":[0.0571577,0.0002644634,0.0008246008,0.001844924,0.000882162,0.004839601,0.002251988,0.001784608,0.002267728],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008831586,"about_ca_system_score_gemma":0.0009874132,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003281323,"about_ca_topic_score_gemma":0.007002628,"domain_scores_codex":[0.9860753,0.009373768,0.0006952785,0.001762392,0.001788883,0.0003043585],"domain_scores_gemma":[0.9419035,0.04656217,0.001430662,0.004594426,0.004220261,0.001288851],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003789068,0.00146056,0.06661185,0.001527271,0.001379345,0.0005078241,0.003168167,0.06523227,0.01125271,0.004479663,0.0266052,0.8139861],"study_design_scores_gemma":[0.0004095063,0.002162749,0.04131849,0.0003544017,0.0004124346,0.0006925478,0.003357032,0.9060186,0.01307647,0.01492063,0.01705533,0.0002218009],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8440443,0.007808164,0.1103156,0.001687877,0.0007006503,0.0005266745,0.002671966,0.004820875,0.02742406],"genre_scores_gemma":[0.9533444,0.0004017877,0.04011843,0.0002461136,0.0001118039,0.000113836,0.003209277,0.000190762,0.002263667],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01748245,"threshold_uncertainty_score":0.09245712,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1357440755282679,"score_gpt":0.4845655092340808,"score_spread":0.3488214337058129,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}