{"id":"W4416878674","doi":"10.48550/arxiv.2507.00435","title":"RoboEval: Where Robotic Manipulation Meets Structured and Scalable Evaluation","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Robot Manipulation and Learning","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Army Research Laboratory; Office of Naval Research; Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Edward Via College of Osteopathic Medicine; National Science Foundation","keywords":"Task (project management); Suite; Benchmark (surveying); Scalability; Imitation; Strengths and weaknesses; Binary number","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02092905,0.003340193,0.001993438,0.003014813,0.0009847902,0.005486854,0.004538905,0.00261223,0.006586919],"category_scores_gemma":[0.07477974,0.001076542,0.001367251,0.001462507,0.002595956,0.007354183,0.009274399,0.004157355,0.004713206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001561253,"about_ca_system_score_gemma":0.003940326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003796363,"about_ca_topic_score_gemma":0.004911697,"domain_scores_codex":[0.9713254,0.01298532,0.002095062,0.003497293,0.008635817,0.001461166],"domain_scores_gemma":[0.9636368,0.01680656,0.002083694,0.0101815,0.005497726,0.001793575],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004358255,0.003169048,0.01843746,0.004340538,0.0010136,0.0005972527,0.001189252,0.2001462,0.04057718,0.09981228,0.1195687,0.5067903],"study_design_scores_gemma":[0.0006863992,0.002520616,0.007981049,0.0008098464,0.000126743,0.000455787,0.0004130941,0.7567192,0.02608462,0.1428518,0.06106226,0.0002885402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04493608,0.002140944,0.8638524,0.001757674,0.0005192018,0.00164263,0.003912026,0.06410535,0.0171337],"genre_scores_gemma":[0.3299868,0.0007203454,0.6384884,0.0009415817,0.0002605418,0.002708399,0.01170752,0.01235815,0.002828337],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02092905,"threshold_uncertainty_score":0.1106848,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05706522865858058,"score_gpt":0.2875441420298847,"score_spread":0.2304789133713041,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}