{"id":"W4366598314","doi":"10.1145/3591109","title":"A Framework and Toolkit for Testing the Correctness of Recommendation Algorithms","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Recommender Systems","topic":"Recommender Systems and Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Test suite; Correctness; Unit testing; Python (programming language); Algorithm; Recommender system; Suite; White-box testing; Implementation; Surprise; Regression testing; Code coverage; Integration testing; Test case; Software engineering; Software; Programming language; Machine learning; Software system; Software construction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01217337,0.00274486,0.001580146,0.003944758,0.00101834,0.003707825,0.006693104,0.003251955,0.00879672],"category_scores_gemma":[0.05016431,0.002606544,0.004613273,0.001591801,0.003807958,0.007837695,0.006339679,0.006572065,0.005369583],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001853635,"about_ca_system_score_gemma":0.004390904,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006409619,"about_ca_topic_score_gemma":0.004287396,"domain_scores_codex":[0.9841468,0.005094115,0.002300253,0.001917462,0.00553268,0.001008649],"domain_scores_gemma":[0.9704329,0.01544132,0.002168676,0.007810023,0.003340453,0.0008066139],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001098128,0.001674696,0.02780375,0.003516985,0.0008399277,0.002639573,0.002219132,0.1645521,0.03149631,0.242332,0.08674587,0.4350815],"study_design_scores_gemma":[0.0003959337,0.000908388,0.006075243,0.001042908,0.000221332,0.002256132,0.0003779146,0.6330798,0.03816308,0.2080752,0.1089739,0.0004302575],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003989222,0.0002438279,0.9137149,0.0004407983,0.0001006978,0.0004448996,0.0007193979,0.0779419,0.002404286],"genre_scores_gemma":[0.09268761,0.0006234641,0.884574,0.0008130341,0.0001354075,0.002058681,0.003354991,0.01236504,0.003387754],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01217337,"threshold_uncertainty_score":0.06437969,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1268538343049634,"score_gpt":0.3261905453913853,"score_spread":0.1993367110864219,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}