{"id":"W4367692815","doi":"10.48550/arxiv.2305.00083","title":"Reflections on Surrogate-Assisted Search-Based Testing: A Taxonomy and Two Replication Studies based on Industrial ADAS and Simulink Models","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Computer science; Benchmark (surveying); Replication (statistics); Taxonomy (biology); Generalization; Machine learning; Heuristic; Artificial intelligence; Domain (mathematical analysis); Reliability engineering; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04917714,0.002060707,0.002094162,0.00875272,0.001878016,0.008310295,0.007024018,0.004290255,0.001576355],"category_scores_gemma":[0.1258494,0.001235058,0.002316134,0.01044209,0.007455493,0.01642573,0.00449155,0.00800766,0.0008357824],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006736976,"about_ca_system_score_gemma":0.007828368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008400298,"about_ca_topic_score_gemma":0.003932627,"domain_scores_codex":[0.9390085,0.02789262,0.006101395,0.005892171,0.01980544,0.001299866],"domain_scores_gemma":[0.8291401,0.09396771,0.007007294,0.02560027,0.04254265,0.001742057],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003251612,0.0006356777,0.008301437,0.004409291,0.000175931,0.0004859228,0.009886399,0.03608562,0.007372092,0.214484,0.01052514,0.7073134],"study_design_scores_gemma":[0.0002655781,0.005079482,0.009894758,0.01755805,0.0004493475,0.004356578,0.01734613,0.2676513,0.03408074,0.2514226,0.3909199,0.0009755675],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04053159,0.09428142,0.8256965,0.0158055,0.001249661,0.001450891,0.0002538574,0.001141611,0.01958889],"genre_scores_gemma":[0.2947365,0.05712125,0.6370188,0.002961586,0.0008107215,0.001742152,0.0008257908,0.0005882017,0.004195102],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9508228,"threshold_uncertainty_score":0.2600767,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6868043205396952,"score_gpt":0.3359967428355055,"score_spread":0.3508075777041897,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}