{"id":"W4411552045","doi":"10.1109/icse55347.2025.00220","title":"Mock Deep Testing: Toward Separate Development of Data and Models for Deep Learning","year":2025,"lang":"en","type":"article","venue":"","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"National Science Foundation","keywords":"Computer science; Deep learning; Artificial intelligence; Data modeling; Data science; Machine learning; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0297366,0.002129112,0.0008798874,0.002296481,0.001012548,0.004751471,0.006949397,0.00253563,0.002307333],"category_scores_gemma":[0.1625127,0.002672456,0.002189115,0.0009719733,0.005366696,0.01469796,0.01214614,0.00721259,0.001020367],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002878801,"about_ca_system_score_gemma":0.006602254,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004319895,"about_ca_topic_score_gemma":0.005541993,"domain_scores_codex":[0.9640129,0.02010295,0.002755521,0.003769713,0.007846645,0.001512189],"domain_scores_gemma":[0.8638681,0.07140093,0.00718053,0.04248355,0.01264919,0.002417696],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001722371,0.001631794,0.052611,0.001510152,0.0004458914,0.00158009,0.009163085,0.1205452,0.04476476,0.09559547,0.01391556,0.6565146],"study_design_scores_gemma":[0.0002135383,0.001298037,0.003787874,0.0005617458,0.0001392157,0.000715046,0.001213498,0.7981043,0.07705925,0.09207801,0.0246099,0.0002195619],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05737947,0.0002486006,0.921627,0.001021092,0.00006861292,0.0004519936,0.0001591135,0.01719497,0.001849058],"genre_scores_gemma":[0.2713622,0.0001576715,0.7224997,0.0009031887,0.00002578558,0.0005032681,0.0007041873,0.002927206,0.0009168083],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0297366,"threshold_uncertainty_score":0.157264,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09641629346456665,"score_gpt":0.3300717844408592,"score_spread":0.2336554909762925,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}