{"id":"W7053699420","doi":"","title":"Who Tests the Testers? Assessing the Effectiveness and Trustworthiness of Deep Learning Model Testing Techniques","year":2024,"lang":"fr","type":"other","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Laser Design and Applications","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Consortium de Recherche et d’innovation en Aérospatiale au Québec","keywords":"Trustworthiness; Deep learning; Context (archaeology); Validation test","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07144249,0.001163079,0.001021505,0.002633444,0.0007208005,0.003166065,0.002185269,0.002939956,0.001680219],"category_scores_gemma":[0.3687018,0.0007159669,0.0009770427,0.001447296,0.001923079,0.003958952,0.002527541,0.002092065,0.001163482],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00111319,"about_ca_system_score_gemma":0.0027244,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00384772,"about_ca_topic_score_gemma":0.003112936,"domain_scores_codex":[0.9129783,0.05023339,0.008416012,0.009563058,0.01717162,0.001637562],"domain_scores_gemma":[0.4271259,0.4461202,0.04011363,0.04266983,0.03872037,0.005250109],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004880087,0.001289865,0.4447902,0.00119027,0.001425272,0.001232787,0.007319601,0.04795146,0.01572793,0.007418652,0.005387121,0.4613867],"study_design_scores_gemma":[0.0007741893,0.005492815,0.1163326,0.001367388,0.00100668,0.002500206,0.005331397,0.7646093,0.06269399,0.02161948,0.01791449,0.0003573812],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8051257,0.003076037,0.1781013,0.002424185,0.0005132545,0.0006433775,0.0004477134,0.001736708,0.007931725],"genre_scores_gemma":[0.9638066,0.0002676744,0.03372167,0.0002907034,0.00007697212,0.0001998221,0.0002920118,0.0002025448,0.001142065],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9285575,"threshold_uncertainty_score":0.3778285,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01136557139903286,"score_gpt":0.2433983145462487,"score_spread":0.2320327431472159,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}