{"id":"W4411449683","doi":"10.1145/3729343","title":"VLATest: Testing and Evaluating Vision-Language-Action Models for Robotic Manipulation","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Artificial intelligence; Software deployment; Machine learning; Generative grammar; Human–computer interaction; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003202415,0.001674858,0.0006235217,0.000867647,0.0005213654,0.0009809466,0.003414438,0.002177025,0.003036358],"category_scores_gemma":[0.013307,0.0007530313,0.001365145,0.0003395785,0.001572145,0.00190759,0.001931716,0.002147024,0.0006799424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00190229,"about_ca_system_score_gemma":0.001669172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01230621,"about_ca_topic_score_gemma":0.01243409,"domain_scores_codex":[0.9983253,0.0005734492,0.0001157777,0.0004069807,0.0004446495,0.0001338779],"domain_scores_gemma":[0.9915685,0.006456992,0.0004107166,0.0008390776,0.000481562,0.0002431029],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007334704,0.0008501097,0.005736214,0.0006167298,0.0002056875,0.0001915491,0.0002624578,0.8998533,0.01245523,0.005071458,0.004045818,0.06997789],"study_design_scores_gemma":[0.00004961224,0.0002820326,0.0005378685,0.00001974922,0.000014182,0.00003475682,0.00002699867,0.9930728,0.004365382,0.001000293,0.0005819615,0.00001429495],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6558537,0.0009764071,0.3155743,0.0006797121,0.0003432333,0.001207726,0.001987693,0.01643031,0.006946956],"genre_scores_gemma":[0.8313949,0.0002398218,0.1629666,0.0002470521,0.00002649137,0.0006390127,0.002486316,0.0006298784,0.001369996],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01230621,"threshold_uncertainty_score":0.02446914,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04418603467234083,"score_gpt":0.3291355321729803,"score_spread":0.2849494975006395,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}