{"id":"W4404660366","doi":"10.2139/ssrn.5031810","title":"Evaluating and Enhancing Segmentation Model Robustness with Metamorphic Testing","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Robustness (evolution); Segmentation; Computer science; Artificial intelligence; Robustness testing; Computer vision; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009459553,0.00219006,0.001670909,0.002259217,0.0008499475,0.002946381,0.003868149,0.004721879,0.003086656],"category_scores_gemma":[0.03916801,0.0009754614,0.001479008,0.001152928,0.002129044,0.002865379,0.003850174,0.002387912,0.001124362],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00131824,"about_ca_system_score_gemma":0.001839668,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004381417,"about_ca_topic_score_gemma":0.004379159,"domain_scores_codex":[0.9952999,0.001565405,0.0004251286,0.001170393,0.0011448,0.0003942464],"domain_scores_gemma":[0.9734207,0.01666861,0.00168408,0.00498488,0.002546481,0.0006951464],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00262638,0.0004097464,0.01225489,0.0003935374,0.0005792204,0.0004068116,0.0002195251,0.61295,0.04351053,0.006886056,0.003463094,0.3163002],"study_design_scores_gemma":[0.00002731438,0.000143992,0.0006493278,0.00001713585,0.00004415397,0.0001052143,0.00002398509,0.9850135,0.01169243,0.002003732,0.0002688144,0.00001030449],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3430278,0.001660525,0.6390803,0.001186167,0.0003209178,0.0002241925,0.0005175563,0.009912882,0.004069692],"genre_scores_gemma":[0.8517459,0.0002026214,0.1437481,0.0003355042,0.00007950736,0.00005657316,0.001061003,0.00108498,0.00168583],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009459553,"threshold_uncertainty_score":0.05002749,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05007778346795639,"score_gpt":0.3265554576501277,"score_spread":0.2764776741821713,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}