{"id":"W7092203361","doi":"10.5281/zenodo.17347847","title":"AImoclips: A Benchmark for Evaluating Emotion Conveyance in Text-to-Music Generation","year":2025,"lang":"","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Fidelity; Rendering (computer graphics); Benchmark (surveying); Valence (chemistry); Arousal; Affective computing; Emotional valence; Likert scale","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.003416635,0.0003527755,0.0003905641,0.001201495,0.002575536,0.0008914918,0.0008229222,0.0003192063,0.02558427],"category_scores_gemma":[0.003320095,0.0004432291,0.0001325836,0.002395141,0.0001610768,0.0003038771,0.0006197172,0.000552026,0.007242668],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005671001,"about_ca_system_score_gemma":0.00003192638,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003500261,"about_ca_topic_score_gemma":0.000009188639,"domain_scores_codex":[0.9948348,0.001702909,0.0009868267,0.001224769,0.000448456,0.0008022218],"domain_scores_gemma":[0.9968973,0.0001206776,0.0003132302,0.0006807917,0.001747213,0.0002407258],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004905678,0.0006925589,0.00001477487,0.0003486061,0.0001133002,0.000005237197,0.005364482,0.0004158425,0.02443987,0.01199726,0.2953636,0.6607539],"study_design_scores_gemma":[0.005626279,0.001637726,0.003706299,0.0007127285,0.0001232841,0.00004826583,0.002155642,0.02872581,0.001575685,0.001195292,0.9538381,0.0006549034],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.257381,0.0009747881,0.2160145,0.01061894,0.005731103,0.01146754,0.001183053,0.001258674,0.4953704],"genre_scores_gemma":[0.9820547,0.0001692544,0.001104473,0.001903787,0.0006333241,0.000002156643,0.005288441,0.001260325,0.007583563],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7246737,"threshold_uncertainty_score":0.9998019,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1193823718522467,"score_gpt":0.3577327939924501,"score_spread":0.2383504221402034,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}