{"id":"W4389650673","doi":"10.48550/arxiv.2312.05822","title":"Toward Open-ended Embodied Tasks Solving","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Microsoft Research","keywords":"Embodied cognition; Task (project management); Computer science; Robot; Human–computer interaction; Artificial intelligence; Control (management); State (computer science); Plan (archaeology); Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science","insufficient_payload"],"consensus_categories":["open_science"],"category_scores_codex":[0.0005252946,0.0004452148,0.0005099284,0.0003929507,0.0002892461,0.0008924401,0.0078121,0.0003841296,0.00005082222],"category_scores_gemma":[0.0001443392,0.0005438408,0.0002398107,0.000818481,0.0001072942,0.0007768567,0.01728145,0.001078082,0.0009281358],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003616072,"about_ca_system_score_gemma":0.0004005652,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003052285,"about_ca_topic_score_gemma":0.00002467967,"domain_scores_codex":[0.9970316,0.000181702,0.0003424175,0.001585838,0.0002028446,0.0006556523],"domain_scores_gemma":[0.996691,0.0001964936,0.0004495699,0.002232543,0.0001935384,0.0002367925],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009469233,0.00002054082,0.000544984,0.00006793738,0.0001024044,0.0004474471,0.0003400182,0.8300888,0.00001058458,0.1662102,0.00204409,0.0001135453],"study_design_scores_gemma":[0.0005403694,0.00005570442,0.0006207081,0.0001600043,0.00005441519,0.000003605777,0.000100029,0.9703391,0.00004875233,0.02605478,0.001372845,0.0006497447],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002448974,0.00001436989,0.9616395,0.0002973884,0.001855501,0.000542903,0.000005945401,0.0009348637,0.03226052],"genre_scores_gemma":[0.9697731,0.0001303776,0.008403449,0.0001794354,0.00009359023,0.000002352867,0.00003502953,0.0000463845,0.02133624],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9673242,"threshold_uncertainty_score":0.9998497,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1926690095202335,"score_gpt":0.2343394282668352,"score_spread":0.04167041874660163,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}