{"id":"W2994711059","doi":"10.1002/stvr.1721","title":"Leveraging metamorphic testing to automatically detect inconsistencies in code generator families","year":2019,"lang":"en","type":"article","venue":"Software Testing Verification and Reliability","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Agence Nationale de la Recherche","keywords":"Computer science; Software quality; Oracle; Software; Leverage (statistics); Regression testing; Code (set theory); Code coverage; Unreachable code; Software development; Code generation; Programming language; Redundant code; Software construction; Set (abstract data type); Operating system; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002627858,0.0005398613,0.0006672531,0.003716693,0.0003046484,0.00086493,0.001199489,0.0006385205,0.0005350219],"category_scores_gemma":[0.01769063,0.0003904501,0.0005939456,0.001064441,0.0007113995,0.0008786203,0.001135405,0.0006720379,0.0002029628],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004512732,"about_ca_system_score_gemma":0.0006587433,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001142817,"about_ca_topic_score_gemma":0.001276282,"domain_scores_codex":[0.9964659,0.0009975854,0.0003163737,0.0008288821,0.001206952,0.000184247],"domain_scores_gemma":[0.976344,0.01254667,0.005570724,0.002482028,0.002674083,0.0003825188],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007035004,0.0004319251,0.2123737,0.0003667823,0.0002590825,0.001517168,0.0007784913,0.1701235,0.1156935,0.00846123,0.002252987,0.4870381],"study_design_scores_gemma":[0.00002966075,0.0002227055,0.01686741,0.00004501565,0.00004496876,0.0006775658,0.00005409483,0.9500443,0.02626222,0.004777249,0.0009448613,0.00002991711],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4932825,0.0002832314,0.4961875,0.0001657549,0.00002440856,0.000150696,0.0002188367,0.008405702,0.001281309],"genre_scores_gemma":[0.8781351,0.00005324172,0.1208248,0.00005792157,0.00001292644,0.00006927011,0.000387458,0.000206606,0.0002528178],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003716693,"threshold_uncertainty_score":0.01389766,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03641849049665075,"score_gpt":0.2574023251371866,"score_spread":0.2209838346405358,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}