{"id":"W291243768","doi":"","title":"Efficient Reinforcement Learning with Multiple Reward Functions for Randomized Controlled Trial Analysis","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Generalization; Set (abstract data type); Extension (predicate logic); Reinforcement; State (computer science); Artificial intelligence; Mathematical optimization; Mathematics; Algorithm; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007460658,0.001628556,0.002531657,0.0009627765,0.0005114351,0.001644998,0.002834133,0.001686886,0.004959986],"category_scores_gemma":[0.02938749,0.001000622,0.001130694,0.001037513,0.002128343,0.002976449,0.002482821,0.004133789,0.0009796185],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002062509,"about_ca_system_score_gemma":0.002714852,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002091743,"about_ca_topic_score_gemma":0.002492737,"domain_scores_codex":[0.9940922,0.003303097,0.0003287348,0.0006757624,0.001281837,0.0003183832],"domain_scores_gemma":[0.9870085,0.01004263,0.0007515086,0.001035267,0.0008574819,0.0003046358],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002769802,0.0001692562,0.0005733698,0.0002442845,0.0001141894,0.0001102396,0.00009509108,0.6730787,0.002365258,0.1841857,0.002375663,0.1364113],"study_design_scores_gemma":[0.00005674532,0.00004172439,0.00004110741,0.00001678019,0.00001046321,0.00001672293,0.000002797387,0.9508876,0.0005665319,0.04771642,0.0006319138,0.00001112803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0005676456,0.00007654783,0.9988179,0.00004163159,0.00001098659,0.00003855356,0.000009166466,0.0001495459,0.0002880819],"genre_scores_gemma":[0.1045065,0.0002101119,0.8926597,0.0001146616,0.00006778909,0.0006769114,0.00008051581,0.000207003,0.001476795],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9925393,"threshold_uncertainty_score":0.03945619,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00903513300406085,"score_gpt":0.2400708871062105,"score_spread":0.2310357541021497,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}