{"id":"W4414359780","doi":"10.24963/ijcai.2025/1181","title":"The Evolving Landscape of LLM- and VLM-Integrated Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Elevator Systems and Control","field":"Engineering","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Key (lock); Action (physics); Reinforcement learning; Taxonomy (biology); Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004155688,0.001072602,0.001562638,0.001141092,0.0006128261,0.003380622,0.002891129,0.00207447,0.003855662],"category_scores_gemma":[0.01450087,0.0008498832,0.001177887,0.001158391,0.002722795,0.003754679,0.002508305,0.003800387,0.000861533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003100997,"about_ca_system_score_gemma":0.002653886,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005699205,"about_ca_topic_score_gemma":0.004755441,"domain_scores_codex":[0.9978403,0.0009214446,0.0001467514,0.000555735,0.0004111969,0.0001245586],"domain_scores_gemma":[0.9939595,0.004406289,0.0002424126,0.0005569917,0.0005892079,0.000245473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008297994,0.0001041441,0.001495592,0.0007751369,0.0001577023,0.00007843775,0.0002591231,0.3901979,0.0008455119,0.3290978,0.004128565,0.2727771],"study_design_scores_gemma":[0.00002434411,0.00008536028,0.0002867467,0.0001675106,0.00002855178,0.00005433772,0.00005564201,0.7650713,0.0005509844,0.2220297,0.01161086,0.00003465808],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.009110473,0.01943523,0.9506605,0.004821412,0.0001765654,0.0000614381,0.0001223925,0.0005098027,0.01510222],"genre_scores_gemma":[0.5170467,0.02074344,0.4519997,0.001664971,0.0005569281,0.0004394359,0.0003712843,0.0003042241,0.006873276],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.005699205,"threshold_uncertainty_score":0.02249938,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.002396828550662751,"score_gpt":0.1784749064194855,"score_spread":0.1760780778688228,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}