{"id":"W1538703232","doi":"10.1007/978-3-642-04428-1_39","title":"Anytime Self-play Learning to Satisfy Functional Optimality Criteria","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Artificial intelligence; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001514949,0.0007320727,0.0006594423,0.001030431,0.0005270005,0.001249267,0.003314747,0.0004042342,0.0001476235],"category_scores_gemma":[0.0002455883,0.0007364476,0.0001701442,0.0008789998,0.0002912324,0.000853812,0.00173208,0.001542335,0.0003634329],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006447828,"about_ca_system_score_gemma":0.00065478,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001541416,"about_ca_topic_score_gemma":0.000008457592,"domain_scores_codex":[0.99437,0.00009159259,0.0007675806,0.001996173,0.00175945,0.001015211],"domain_scores_gemma":[0.9968368,0.0005031934,0.0003737565,0.001506491,0.0003986368,0.0003811311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000005912207,0.0000145429,0.0001211702,0.00001742139,0.00001318049,0.00004052279,0.0007436466,0.8626161,0.0001337775,0.003989379,0.0001024007,0.132202],"study_design_scores_gemma":[0.0002760796,0.000569318,0.002661989,0.0002870183,0.00001284591,0.00009877104,1.882377e-7,0.974615,0.0003451159,0.01067295,0.009410883,0.001049796],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00008186098,0.00008124345,0.9876983,0.001112414,0.001964722,0.0004687057,0.000001623896,0.0005771336,0.008013955],"genre_scores_gemma":[0.07073344,0.00001828676,0.9226894,0.002993336,0.0007692907,0.000009393733,0.00001325725,0.00005086192,0.002722687],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1311522,"threshold_uncertainty_score":0.9997875,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01797993096656769,"score_gpt":0.2577936049483812,"score_spread":0.2398136739818135,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}