{"id":"W2881301346","doi":"10.1145/3213586.3226216","title":"Bridging the Gap Between User-centric and Offline Evaluation of Personalized Recommendation Systems","year":2018,"lang":"en","type":"article","venue":"","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Bridging (networking); Novelty; Complement (music); Recommender system; Online and offline; Quality (philosophy); Machine learning; Data mining; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002857851,0.00008198944,0.0001521725,0.00008933385,0.000120994,0.0001256953,0.0002706946,0.00003807955,0.00002714122],"category_scores_gemma":[0.00004297194,0.00005307554,0.00002795083,0.0002627446,0.00004293789,0.0002954933,0.0001152821,0.00004857238,0.000004449375],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004188038,"about_ca_system_score_gemma":0.00003727001,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000355684,"about_ca_topic_score_gemma":0.00001074853,"domain_scores_codex":[0.9987178,0.0003421333,0.0002935578,0.0002057541,0.000317392,0.0001233991],"domain_scores_gemma":[0.9990498,0.0001046127,0.0001907176,0.0002842961,0.0003379067,0.00003264759],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007043059,0.00006006535,0.01921459,0.00009604891,0.0001281892,2.68073e-7,0.0023735,0.000004739753,0.0008695663,0.09630468,0.02749255,0.8534487],"study_design_scores_gemma":[0.001226345,0.0002472403,0.01555583,0.0001304829,0.00008352404,0.00002731422,0.0002790289,0.9309682,0.00709592,0.002005466,0.04209681,0.0002837985],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04229118,0.0002191957,0.9494802,0.002213394,0.0003234149,0.0005694549,0.00000310074,0.0001195724,0.004780478],"genre_scores_gemma":[0.9958457,0.00001992912,0.00366559,0.00006474402,0.0002199095,0.00002239358,0.00000544178,0.000004813851,0.0001514308],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9535546,"threshold_uncertainty_score":0.2164358,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09851703829333276,"score_gpt":0.3337958116595153,"score_spread":0.2352787733661826,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}