{"id":"W75140964","doi":"","title":"Agnostic KWIK learning and efficient approximate reinforcement learning","year":2011,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Simple (philosophy); Artificial intelligence; Algorithm; Theoretical computer science; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001860294,0.0008768067,0.001381903,0.0004678743,0.0003741589,0.001017998,0.001933067,0.001330218,0.002672602],"category_scores_gemma":[0.009071812,0.000592013,0.0004464098,0.0005466783,0.001569277,0.002378393,0.002060032,0.002584088,0.0006149946],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001039835,"about_ca_system_score_gemma":0.001168304,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002373248,"about_ca_topic_score_gemma":0.002285784,"domain_scores_codex":[0.9985847,0.0005723143,0.00007579694,0.0002244502,0.000380422,0.0001622024],"domain_scores_gemma":[0.997453,0.001362525,0.0002675944,0.000496868,0.0003151932,0.0001047711],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001534725,0.000127022,0.0007214198,0.0001103225,0.00005612589,0.00008379658,0.00009197497,0.8176853,0.001736016,0.1252333,0.001486861,0.05251442],"study_design_scores_gemma":[0.00001087032,0.00001922954,0.00003401297,0.000003815441,0.000003934866,0.00001402536,0.000003655421,0.9684824,0.0002943736,0.03080655,0.0003233275,0.000003849695],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01018046,0.0001510358,0.9876179,0.000153561,0.00002335543,0.00002352155,0.00001520471,0.0002682369,0.001566832],"genre_scores_gemma":[0.8092794,0.0002704483,0.1827669,0.0002323637,0.00005829943,0.0001744431,0.000125358,0.0001303064,0.006962366],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002672602,"threshold_uncertainty_score":0.009838283,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03446888166875992,"score_gpt":0.247999907469349,"score_spread":0.2135310258005891,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}