{"id":"W4416034065","doi":"10.18653/v1/2025.findings-emnlp.1067","title":"Stop Playing the Guessing Game! Evaluating Conversational Recommender Systems via Target-free User Simulation","year":2025,"lang":"","type":"article","venue":"","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Recommender system; User modeling; Key (lock); Matching (statistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.005326831,0.0005818834,0.0006454593,0.0004027527,0.001644721,0.002328853,0.002405879,0.0003567827,0.0002331568],"category_scores_gemma":[0.0004001026,0.0004475565,0.0002661332,0.001102896,0.0001261516,0.001641579,0.001480621,0.0006664451,0.00004065345],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005888459,"about_ca_system_score_gemma":0.0005138175,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000950056,"about_ca_topic_score_gemma":0.00001788612,"domain_scores_codex":[0.9936476,0.001558033,0.001718285,0.001144495,0.001139929,0.0007916369],"domain_scores_gemma":[0.9938695,0.002186322,0.0008399278,0.00218622,0.000787342,0.0001307244],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001037112,0.0004776804,0.01332579,0.001269157,0.001361637,0.000013931,0.01430993,0.3603969,0.001099283,0.3171072,0.09081564,0.1997192],"study_design_scores_gemma":[0.0008898361,0.0000845477,0.001143225,0.0006565441,0.00006211788,0.00001009772,0.0008292809,0.9695659,0.0002816966,0.009869632,0.01612698,0.0004801452],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001132048,0.001534977,0.9629628,0.008059756,0.006717356,0.00171066,0.000007703744,0.0004541641,0.01742055],"genre_scores_gemma":[0.9594201,0.00002407064,0.03412478,0.001742666,0.0004401053,0.0001420009,0.00001221799,0.00003556007,0.004058488],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9582881,"threshold_uncertainty_score":0.9997976,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06331105928617485,"score_gpt":0.3454273776207131,"score_spread":0.2821163183345383,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}