{"id":"W4413372625","doi":"10.1250/ast.e25.42","title":"Quantitative analysis of singing expression and facial gestures using image recognition with artificial intelligence","year":2025,"lang":"en","type":"article","venue":"Nippon Onkyo Gakkaishi/Acoustical science and technology/Nihon Onkyo Gakkaishi","topic":"Educational Technology and Pedagogy","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Education and Early Childhood Development","funders":"Japan Society for the Promotion of Science","keywords":"Singing; Gesture; Facial expression; Artificial intelligence; Speech recognition; Expression (computer science); Pattern recognition (psychology); Computer science; Image (mathematics); Computer vision; Communication; Psychology; Acoustics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.001140059,0.0003544108,0.0006058072,0.003045516,0.0008691915,0.0003004386,0.001143891,0.0004812363,0.00002697798],"category_scores_gemma":[0.002314361,0.0003058421,0.00006691404,0.009308672,0.00440723,0.001144073,0.0006967667,0.0006812911,0.000004730031],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001227972,"about_ca_system_score_gemma":0.0006289022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001008532,"about_ca_topic_score_gemma":0.00008865479,"domain_scores_codex":[0.9966294,0.0000920121,0.0006200114,0.00126583,0.0006934053,0.0006993951],"domain_scores_gemma":[0.9971348,0.0005673284,0.0002981154,0.0006513403,0.001184981,0.0001634121],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001871441,0.0004934904,0.01265974,0.0001053018,0.0002455364,0.00003826066,0.0008798768,0.0002030358,0.5164393,0.2387571,0.00008944119,0.2299018],"study_design_scores_gemma":[0.0006288268,0.001497593,0.03003193,0.0008940211,0.001302698,0.0001298985,0.008508489,0.2972189,0.3023525,0.3554284,0.0003555778,0.001651189],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6361796,0.0002916603,0.3596975,0.002621423,0.0002251387,0.0002604596,0.00001923678,0.0002659033,0.0004390992],"genre_scores_gemma":[0.869156,0.00008851786,0.1304969,0.000173784,0.00002472678,0.00002399192,0.000006497783,0.000009765305,0.00001983806],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2970158,"threshold_uncertainty_score":0.9999394,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04359240028760244,"score_gpt":0.3506230857921549,"score_spread":0.3070306855045525,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}