{"id":"W4390905962","doi":"10.1109/aciiw59127.2023.10388150","title":"How (not) to Evaluate Computational Empathy: Testing the Assumptions of the Evaluation Methods in a Use-Case","year":2023,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Empathy; Computer science; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07141942,0.001112799,0.000649484,0.001436745,0.0008575448,0.004055553,0.001913153,0.002677077,0.002111743],"category_scores_gemma":[0.2570061,0.0004549363,0.0007838644,0.0007128745,0.001509248,0.004163705,0.002431238,0.001771122,0.0006552069],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001900523,"about_ca_system_score_gemma":0.001435547,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001349446,"about_ca_topic_score_gemma":0.001423997,"domain_scores_codex":[0.9198491,0.05973827,0.005062949,0.004433763,0.009895992,0.001019901],"domain_scores_gemma":[0.7615726,0.1787061,0.01344689,0.01869329,0.0253607,0.002220413],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.009445536,0.005844746,0.1573915,0.005622772,0.002063047,0.0006571494,0.01496456,0.04089994,0.04293363,0.02975354,0.007533385,0.6828902],"study_design_scores_gemma":[0.002656929,0.0174259,0.1342229,0.003079257,0.001363973,0.001687356,0.009527002,0.6634924,0.1121941,0.0262542,0.02741396,0.0006820495],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6416067,0.001004529,0.3386612,0.002192185,0.0003104689,0.004037193,0.0003736939,0.001241802,0.01057221],"genre_scores_gemma":[0.8039454,0.0001529766,0.1916996,0.0002770189,0.0000347517,0.002619307,0.0002135027,0.0001626388,0.0008947496],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9285806,"threshold_uncertainty_score":0.3777065,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4511124691341144,"score_gpt":0.4947184787365493,"score_spread":0.04360600960243494,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}