{"id":"W4391998095","doi":"10.1007/s10664-023-10433-5","title":"Evaluating the impact of flaky simulators on testing autonomous driving systems","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Replication (statistics); Key (lock); Scope (computer science); Code coverage; Simulation; Machine learning; Operating system; Statistics; Software; Mathematics; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004974374,0.0008984674,0.0002683462,0.001146142,0.0003428161,0.0005359555,0.001481061,0.001014408,0.00133956],"category_scores_gemma":[0.08499081,0.0004397492,0.0003372274,0.0007095672,0.0008432636,0.001684078,0.0007400524,0.0007697117,0.00013683],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001216504,"about_ca_system_score_gemma":0.001222274,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00739804,"about_ca_topic_score_gemma":0.01011196,"domain_scores_codex":[0.9949168,0.003314687,0.0002769813,0.0003475549,0.0008796799,0.0002643871],"domain_scores_gemma":[0.8266977,0.1556795,0.004792721,0.006087015,0.005376136,0.001367038],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004246696,0.006938372,0.09320131,0.0005485198,0.0003404789,0.0003117403,0.001118159,0.6489603,0.02102511,0.0026878,0.0009309366,0.2196905],"study_design_scores_gemma":[0.0004684874,0.01017544,0.03598637,0.00008884721,0.0002204018,0.0001855521,0.0007044156,0.9236166,0.02551823,0.00220806,0.0007682617,0.00005932057],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9921436,0.00009029847,0.006487836,0.00007151493,0.00001024833,0.00005309893,0.00006417737,0.0001893619,0.0008898234],"genre_scores_gemma":[0.9934149,0.00002793405,0.006305403,0.00001136169,0.00000203089,0.00001473408,0.00005856738,0.00001716885,0.0001479647],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00739804,"threshold_uncertainty_score":0.02630734,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08166117695653764,"score_gpt":0.382740428437695,"score_spread":0.3010792514811574,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}