{"id":"W4402667043","doi":"10.18653/v1/2024.acl-long.37","title":"OPEx: A Component-Wise Analysis of LLM-Centric Agents in Embodied Instruction Following","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Component (thermodynamics); Embodied cognition; Computer science; Artificial intelligence; Physics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002659036,0.0001031286,0.0002376037,0.001327061,0.0000281683,0.0001514154,0.0005468564,0.00005584575,0.00002198044],"category_scores_gemma":[0.00003123971,0.00008619518,0.0001666708,0.004378959,0.00001429283,0.0005757896,0.0002239391,0.0001264066,0.000005909672],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009397585,"about_ca_system_score_gemma":0.00003550931,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002714423,"about_ca_topic_score_gemma":0.00004776477,"domain_scores_codex":[0.9988865,0.00004626991,0.0002873969,0.0003335579,0.000274428,0.0001718659],"domain_scores_gemma":[0.9995429,0.00004494454,0.00004901621,0.0002953273,0.00003020288,0.00003761941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002561891,0.0004856527,0.02700184,0.0003633669,0.00179821,0.0007463741,0.004493496,0.0004568235,0.0159511,0.3299992,0.001073405,0.6176049],"study_design_scores_gemma":[0.0004546594,0.00005429914,0.007027775,0.000256246,0.0003221435,0.000009752932,0.00005625692,0.9523384,0.02002679,0.01876361,0.0003129618,0.0003771127],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4271988,0.003106287,0.565749,0.0002952685,0.0005082224,0.0002004838,0.000002156697,0.0009419737,0.001997845],"genre_scores_gemma":[0.8642673,0.00001219889,0.1355563,0.00005506213,0.000008576066,0.000005482533,0.000005384804,0.000004969228,0.00008474808],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9518816,"threshold_uncertainty_score":0.3514937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02009801514821536,"score_gpt":0.3025195860109078,"score_spread":0.2824215708626924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}