{"id":"W2735649811","doi":"10.1007/s10994-017-5657-1","title":"Generalized exploration in policy search","year":2017,"lang":"en","type":"article","venue":"Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"Seventh Framework Programme; Technische Universität Darmstadt","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Entropy (arrow of time); Policy learning; Machine learning; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002850001,0.001129885,0.002562839,0.0009920279,0.00072785,0.001896706,0.001539206,0.002466837,0.003806002],"category_scores_gemma":[0.01474179,0.0009859467,0.0009160815,0.001463541,0.004248259,0.004194782,0.003093397,0.002885158,0.0003184006],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001642487,"about_ca_system_score_gemma":0.001498478,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004322237,"about_ca_topic_score_gemma":0.003214812,"domain_scores_codex":[0.9983747,0.00110925,0.00005469117,0.0001774003,0.0001808149,0.0001031537],"domain_scores_gemma":[0.9935149,0.005430716,0.00026674,0.0003400532,0.000234772,0.000212855],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001169629,0.00003192277,0.0004452133,0.0001495599,0.00007603399,0.00007792293,0.0001340773,0.5573142,0.0002431821,0.4182181,0.001693722,0.02149911],"study_design_scores_gemma":[0.00002482516,0.00001955026,0.00006201423,0.00001853264,0.000009647194,0.000012115,0.00001235938,0.6827138,0.00006106619,0.3165457,0.0005121586,0.000008205477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03779016,0.003500808,0.9485067,0.001644339,0.000185119,0.00003963711,0.00007793396,0.0002466478,0.008008659],"genre_scores_gemma":[0.8800489,0.00163164,0.1086218,0.0003838997,0.0002619137,0.0002223197,0.0001430373,0.0001859178,0.008500661],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004322237,"threshold_uncertainty_score":0.01507241,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05004398310810728,"score_gpt":0.3319969426844179,"score_spread":0.2819529595763107,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}