{"id":"W1538703232","doi":"10.1007/978-3-642-04428-1_39","title":"Anytime Self-play Learning to Satisfy Functional Optimality Criteria","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Artificial intelligence; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001441393,0.00114534,0.001010749,0.0003932073,0.0005311954,0.001163395,0.001483124,0.001221652,0.007843879],"category_scores_gemma":[0.004461483,0.0003849897,0.0006230886,0.0003540667,0.001201504,0.002010508,0.002089858,0.001935974,0.0007953274],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000978022,"about_ca_system_score_gemma":0.001233272,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001447681,"about_ca_topic_score_gemma":0.001557315,"domain_scores_codex":[0.9992641,0.0001911317,0.00004822061,0.0001276885,0.0002284926,0.0001404217],"domain_scores_gemma":[0.9984419,0.0008671642,0.00009000379,0.0001250613,0.0003053577,0.0001705298],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002291657,0.0001626298,0.0004567035,0.0002096423,0.00004508768,0.0001227093,0.0003012112,0.1217953,0.006802552,0.8059226,0.005164598,0.05878782],"study_design_scores_gemma":[0.00003852604,0.0001546231,0.0001600215,0.00002677939,0.00001114794,0.00006349423,0.00004644867,0.6301996,0.002046979,0.3649491,0.002291528,0.00001172973],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04364802,0.0002528976,0.9266351,0.0004061176,0.0001285056,0.0001096238,0.0001159734,0.0003093429,0.0283944],"genre_scores_gemma":[0.8041073,0.0003069969,0.1538417,0.0002513455,0.00009149772,0.000368772,0.0002872002,0.0003801651,0.040365],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007843879,"threshold_uncertainty_score":0.02624041,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01797993096656769,"score_gpt":0.2577936049483812,"score_spread":0.2398136739818135,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}