{"id":"W4313562939","doi":"10.1109/humanoids53995.2022.10000141","title":"A Standardized Benchmark for Humanoid Whole-Body Manipulation","year":2022,"lang":"en","type":"article","venue":"2022 IEEE-RAS 21st International Conference on Humanoid Robots (Humanoids)","topic":"Robotic Locomotion and Control","field":"Engineering","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Benchmark (surveying); Benchmarking; Humanoid robot; Computer science; Set (abstract data type); Key (lock); Human–computer interaction; Focus (optics); Artificial intelligence; Robot; Simulation; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00313605,0.001378357,0.000654101,0.001864998,0.0005618099,0.001070146,0.002060034,0.001049106,0.002980321],"category_scores_gemma":[0.01136006,0.0002492877,0.0005619903,0.001355877,0.0006983731,0.00109763,0.001854055,0.0007629494,0.001766325],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007280879,"about_ca_system_score_gemma":0.001173785,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002017514,"about_ca_topic_score_gemma":0.001945077,"domain_scores_codex":[0.9960276,0.00106583,0.0005263711,0.0004239235,0.001645461,0.0003109084],"domain_scores_gemma":[0.9941135,0.00154923,0.000699731,0.0009718032,0.00227271,0.0003928912],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003123629,0.003128587,0.02492265,0.006577451,0.0003966495,0.001129199,0.001349808,0.2131601,0.1964776,0.01766083,0.04726431,0.4848092],"study_design_scores_gemma":[0.0004500512,0.01577405,0.07684917,0.001452871,0.0003045394,0.002727905,0.002437585,0.478424,0.269176,0.01407776,0.1378958,0.0004302499],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4339887,0.00263155,0.5020426,0.0005446287,0.0007174763,0.004588607,0.009064243,0.01351899,0.03290314],"genre_scores_gemma":[0.7566072,0.001034754,0.2056584,0.0002180428,0.00004865713,0.004275466,0.02401043,0.001051168,0.007095925],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00313605,"threshold_uncertainty_score":0.01658523,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03545382935288508,"score_gpt":0.2791332350138127,"score_spread":0.2436794056609276,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}