{"id":"W4399486984","doi":"10.1109/tnnls.2024.3373749","title":"Off-Policy Prediction Learning: An Empirical Study of Online Algorithms","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Neural Networks and Learning Systems","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"DeepMind; Alberta Machine Intelligence Institute; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Machine learning; Artificial intelligence; Empirical research; Algorithm; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01975538,0.001442124,0.001316473,0.001574953,0.0007861583,0.001620354,0.002171364,0.002086599,0.001621873],"category_scores_gemma":[0.1207085,0.0005990166,0.001027668,0.001220489,0.002681976,0.005311863,0.001433142,0.003946389,0.000247452],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001968049,"about_ca_system_score_gemma":0.001299705,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003542805,"about_ca_topic_score_gemma":0.001676787,"domain_scores_codex":[0.991546,0.004900846,0.0005692848,0.001377717,0.001252152,0.0003540687],"domain_scores_gemma":[0.806066,0.1760144,0.004770668,0.007866136,0.004475765,0.0008069164],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000841257,0.001033187,0.03244305,0.0007938758,0.0003471013,0.0001448406,0.0003817535,0.755572,0.001139379,0.03777149,0.003551195,0.165981],"study_design_scores_gemma":[0.00004271676,0.0002985415,0.002561915,0.00009574348,0.00002750497,0.00009435255,0.00008163182,0.9820346,0.0008557993,0.01312361,0.000763048,0.00002057202],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4432941,0.009360685,0.5355562,0.002379221,0.0002308351,0.0004321119,0.0004739186,0.0008528509,0.007420085],"genre_scores_gemma":[0.9186867,0.001485402,0.0772456,0.0002978339,0.0001159279,0.0003653124,0.0005532732,0.0001798893,0.001070098],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01975538,"threshold_uncertainty_score":0.1044777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09340242677426551,"score_gpt":0.4280186097560809,"score_spread":0.3346161829818153,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}