{"id":"W4385430633","doi":"10.15607/rss.2023.xix.041","title":"FurnitureBench: Reproducible Real-World Benchmark for Long-Horizon Complex Manipulation","year":2023,"lang":"en","type":"article","venue":"","topic":"Robotic Path Planning Algorithms","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology; National Research Foundation of Korea; National Research Foundation","keywords":"Benchmark (surveying); Computer science; Horizon; Artificial intelligence; Mathematics; Geology; Geodesy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001426849,0.00261046,0.0008734263,0.001178965,0.0006248594,0.001029045,0.002863355,0.001570824,0.01465089],"category_scores_gemma":[0.005228286,0.0004620276,0.0008146794,0.00137339,0.0006022835,0.001129273,0.001521669,0.001442879,0.005596311],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009553732,"about_ca_system_score_gemma":0.0009683782,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01001926,"about_ca_topic_score_gemma":0.02146669,"domain_scores_codex":[0.9985124,0.0002809586,0.000109238,0.0003241832,0.0005672137,0.0002059642],"domain_scores_gemma":[0.9977405,0.0006652047,0.0001686677,0.000527905,0.0006117183,0.0002860863],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003298486,0.003064229,0.01153679,0.003571813,0.0005402883,0.001159594,0.0002305887,0.3997724,0.03394907,0.00663088,0.298859,0.2373868],"study_design_scores_gemma":[0.0006303966,0.002061907,0.01810012,0.0002223706,0.00008943731,0.000483423,0.0002759967,0.8481487,0.03165957,0.006086654,0.09209877,0.0001425892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4686185,0.005958448,0.2301866,0.002007697,0.001485295,0.00254937,0.126463,0.07967351,0.08305769],"genre_scores_gemma":[0.600695,0.0008933448,0.1571473,0.0005331252,0.0000747083,0.001554888,0.2187303,0.003878851,0.01649239],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01465089,"threshold_uncertainty_score":0.04901206,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1058822593118378,"score_gpt":0.3353532847107676,"score_spread":0.2294710253989298,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}