{"id":"W4390905962","doi":"10.1109/aciiw59127.2023.10388150","title":"How (not) to Evaluate Computational Empathy: Testing the Assumptions of the Evaluation Methods in a Use-Case","year":2023,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Empathy; Computer science; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005402465,0.00009276866,0.0001056181,0.0001656177,0.0002126781,0.00026331,0.0007130554,0.0000328481,0.00001276705],"category_scores_gemma":[0.006019021,0.00005540178,0.00005654112,0.002411635,0.00007127709,0.0004047432,0.0004337478,0.0001235583,0.00004734217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007137904,"about_ca_system_score_gemma":0.0002015616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000215195,"about_ca_topic_score_gemma":0.0002513947,"domain_scores_codex":[0.9975391,0.0009975845,0.000311698,0.0002766152,0.0006742834,0.000200744],"domain_scores_gemma":[0.9950758,0.003691281,0.0001201971,0.0005372751,0.000538768,0.00003671525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002037081,0.00002558163,0.002637073,0.000003751148,0.00001008597,0.000007833869,0.007441456,0.6363499,0.002315828,0.01822809,0.0004414346,0.332537],"study_design_scores_gemma":[0.00003685212,0.00001847071,0.05236475,0.00001647535,0.000008309929,0.0000257251,0.0008334992,0.9128363,0.003404611,0.03033962,0.00004847477,0.00006687142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3631913,0.00000451246,0.6269182,0.008829982,0.0002528593,0.0005303955,0.000002222253,0.00008387704,0.0001866683],"genre_scores_gemma":[0.7376357,2.565781e-7,0.2617749,0.0003167792,0.00002075789,0.00005802969,7.455257e-7,0.000005198124,0.0001876382],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3744445,"threshold_uncertainty_score":0.7205765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4511124691341144,"score_gpt":0.4947184787365493,"score_spread":0.04360600960243494,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}