{"id":"W4414360785","doi":"10.24963/ijcai.2024/1181","title":"The Evolving Landscape of LLM- and VLM-Integrated Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Elevator Systems and Control","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Key (lock); Action (physics); Reinforcement learning; Taxonomy (biology); Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004170506,0.001072186,0.001559971,0.001145818,0.000615649,0.003393715,0.002889737,0.002081587,0.003872484],"category_scores_gemma":[0.01455242,0.0008497949,0.001177909,0.001158837,0.002728,0.003767557,0.002507726,0.003810724,0.0008637477],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003112247,"about_ca_system_score_gemma":0.002660324,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005705567,"about_ca_topic_score_gemma":0.00476577,"domain_scores_codex":[0.9978362,0.000923679,0.0001465408,0.0005579551,0.0004105618,0.000125052],"domain_scores_gemma":[0.993947,0.004417203,0.0002425402,0.0005571872,0.0005894974,0.000246562],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008317726,0.0001042324,0.001504432,0.0007748512,0.0001580714,0.00007858859,0.0002597394,0.3877476,0.0008433091,0.3315857,0.004144469,0.2727158],"study_design_scores_gemma":[0.00002439256,0.00008561611,0.00028884,0.0001683698,0.00002876227,0.00005459172,0.00005604922,0.7627997,0.0005506784,0.2242232,0.01168505,0.00003477256],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.009200325,0.01966222,0.9500872,0.004886508,0.0001775396,0.00006171984,0.0001234097,0.0005104584,0.01529061],"genre_scores_gemma":[0.5176955,0.02081088,0.4512566,0.001670728,0.0005583144,0.0004398436,0.0003720038,0.0003043012,0.006891933],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.005705567,"threshold_uncertainty_score":0.02258098,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.003285814898221388,"score_gpt":0.1797980436110342,"score_spread":0.1765122287128129,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}