{"id":"W7092183426","doi":"10.5281/zenodo.17347846","title":"AImoclips: A Benchmark for Evaluating Emotion Conveyance in Text-to-Music Generation","year":2025,"lang":"","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Fidelity; Rendering (computer graphics); Benchmark (surveying); Valence (chemistry); Arousal; Affective computing; Emotional valence; Likert scale","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002169278,0.0011159,0.0003975586,0.00115685,0.0002995449,0.001434268,0.0008029596,0.0008376502,0.004146448],"category_scores_gemma":[0.009969625,0.0001758918,0.0005378126,0.0004377467,0.0003359101,0.0009461273,0.001646238,0.0005801746,0.001790461],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002916566,"about_ca_system_score_gemma":0.0002089291,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006008656,"about_ca_topic_score_gemma":0.0007607022,"domain_scores_codex":[0.9986482,0.0004292259,0.0001392773,0.0002614855,0.0004312129,0.00009057534],"domain_scores_gemma":[0.9972402,0.001565368,0.0002117081,0.0003081083,0.0004685223,0.0002061009],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008794329,0.001790711,0.03431844,0.004964055,0.0006765677,0.0005762054,0.002637423,0.03224073,0.3124826,0.002877732,0.03620437,0.5624368],"study_design_scores_gemma":[0.001315367,0.01046546,0.2001706,0.0003808903,0.0006214042,0.002093452,0.002929465,0.4707045,0.2436292,0.008576413,0.05866271,0.000450494],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7585723,0.002413922,0.1856462,0.0003351402,0.0007146772,0.001280341,0.008975835,0.01797611,0.02408553],"genre_scores_gemma":[0.8840436,0.000537848,0.09200338,0.000214492,0.0001318259,0.001206586,0.01565276,0.001440198,0.004769355],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004146448,"threshold_uncertainty_score":0.01387125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1193823718522467,"score_gpt":0.3577327939924501,"score_spread":0.2383504221402034,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}