{"id":"W4416878674","doi":"10.48550/arxiv.2507.00435","title":"RoboEval: Where Robotic Manipulation Meets Structured and Scalable Evaluation","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Robot Manipulation and Learning","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Army Research Laboratory; Office of Naval Research; Natural Sciences and Engineering Research Council of Canada; Defense Advanced Research Projects Agency; Edward Via College of Osteopathic Medicine; National Science Foundation","keywords":"Task (project management); Suite; Benchmark (surveying); Scalability; Imitation; Strengths and weaknesses; Binary number","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003401281,0.0003397139,0.0003547706,0.0002200951,0.0001212032,0.0001147108,0.0001635705,0.0004044324,0.000355213],"category_scores_gemma":[0.00008838143,0.000379926,0.00007971177,0.000176042,0.00002226986,0.0001644917,0.0001786025,0.0006290245,0.00004901519],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002132115,"about_ca_system_score_gemma":0.00005895126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008582536,"about_ca_topic_score_gemma":0.0001517401,"domain_scores_codex":[0.9984144,0.0001262584,0.0004060863,0.000460833,0.0003339463,0.0002584713],"domain_scores_gemma":[0.9991714,0.00005053559,0.0001137222,0.0004402579,0.0001436848,0.00008044415],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003810767,0.000004746801,0.0855948,0.000380863,0.00007031498,0.000001117295,0.0001755302,0.909943,0.0004058524,0.0001105129,0.0003379086,0.002971521],"study_design_scores_gemma":[0.0002412383,0.00000530218,0.3812224,0.0002170451,0.0001200318,0.000001643446,0.00001853982,0.6170813,0.0001012327,0.0003241656,0.0004296583,0.0002374406],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9507117,0.00517051,0.03267789,0.0004457287,0.002806642,0.001219291,0.000002993756,0.0009343396,0.006030902],"genre_scores_gemma":[0.9971988,0.0002061512,0.001244971,0.00003765962,0.00019795,0.00005157317,0.0002039546,0.00005192035,0.0008070793],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2956276,"threshold_uncertainty_score":0.9998653,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05706522865858058,"score_gpt":0.2875441420298847,"score_spread":0.2304789133713041,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}