{"id":"W4293676943","doi":"10.3389/frobt.2022.951663","title":"Making Bipedal Robot Experiments Reproducible and Comparable: The Eurobench Software Approach","year":2022,"lang":"en","type":"article","venue":"Frontiers in Robotics and AI","topic":"Prosthetics and Rehabilitation Robotics","field":"Engineering","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Horizon 2020 Framework Programme; European Commission","keywords":"Computer science; Benchmarking; Executable; Benchmark (surveying); Software; Metric (unit); Documentation; Process (computing); Software engineering; Consistency (knowledge bases); Protocol (science); Standardization; Machine learning; Data mining; Artificial intelligence; Programming language; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05227548,0.00239353,0.001855316,0.006917753,0.00116341,0.007012685,0.0057544,0.001652629,0.004085099],"category_scores_gemma":[0.1354088,0.001835203,0.001958925,0.003446533,0.00293538,0.007111221,0.009585227,0.003099572,0.00226023],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001799639,"about_ca_system_score_gemma":0.005302949,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001832138,"about_ca_topic_score_gemma":0.00150851,"domain_scores_codex":[0.9532702,0.02111083,0.007368302,0.005005314,0.01210929,0.001136043],"domain_scores_gemma":[0.8913853,0.03245911,0.007128334,0.04807965,0.01935357,0.001593939],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001606807,0.002028743,0.01490722,0.002775351,0.001197189,0.0006586647,0.005465942,0.08635499,0.04020384,0.1364434,0.02676747,0.6815904],"study_design_scores_gemma":[0.0009917438,0.004326136,0.01762128,0.002585465,0.000613387,0.001085305,0.0022652,0.4329777,0.1029658,0.2024203,0.2313214,0.0008263294],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.01465436,0.00022336,0.9507015,0.0002778514,0.000166508,0.0008857797,0.0007642528,0.02831539,0.004011034],"genre_scores_gemma":[0.1091844,0.0002701379,0.8674941,0.000273083,0.00007335553,0.003380555,0.005801108,0.01140882,0.002114477],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.9477245,"threshold_uncertainty_score":0.2764624,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02169815831347332,"score_gpt":0.2459880041892696,"score_spread":0.2242898458757963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}