{"id":"W2881301346","doi":"10.1145/3213586.3226216","title":"Bridging the Gap Between User-centric and Offline Evaluation of Personalized Recommendation Systems","year":2018,"lang":"en","type":"article","venue":"","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Bridging (networking); Novelty; Complement (music); Recommender system; Online and offline; Quality (philosophy); Machine learning; Data mining; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02454388,0.001547194,0.002196189,0.002339811,0.0009472538,0.005519069,0.001845353,0.001958414,0.002224698],"category_scores_gemma":[0.08094083,0.0005409708,0.0006146855,0.0020454,0.001115913,0.00633915,0.001696633,0.001933368,0.0008950061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001537291,"about_ca_system_score_gemma":0.001353577,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003785441,"about_ca_topic_score_gemma":0.004017321,"domain_scores_codex":[0.9430451,0.03787769,0.002122283,0.00305238,0.01314699,0.000755534],"domain_scores_gemma":[0.844835,0.1067815,0.0067215,0.01966734,0.01928706,0.002707647],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003654145,0.003546609,0.05717641,0.002116055,0.00154604,0.0002549446,0.001865057,0.06219829,0.02385322,0.01162899,0.008722121,0.8234382],"study_design_scores_gemma":[0.0005739905,0.01219007,0.1132984,0.001011887,0.0009021998,0.00143397,0.003100437,0.7610201,0.05253671,0.02045901,0.03284773,0.0006255655],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4205843,0.01218807,0.5340256,0.001999449,0.0003524293,0.0009083703,0.0007918965,0.002972359,0.02617763],"genre_scores_gemma":[0.863865,0.001251188,0.1306592,0.0005182309,0.0002721027,0.000319589,0.0006800002,0.000311222,0.002123557],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02454388,"threshold_uncertainty_score":0.129802,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09851703829333276,"score_gpt":0.3337958116595153,"score_spread":0.2352787733661826,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}