{"id":"W1505105576","doi":"10.1109/iolts.2015.7229856","title":"Mining simulation metrics for failure triage in regression testing","year":2015,"lang":"en","type":"article","venue":"","topic":"VLSI and Analog Circuit Testing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Bottleneck; Debugging; Cluster analysis; Regression testing; Root cause; Triage; Data mining; Reliability engineering; Machine learning; Engineering; Embedded system; Software; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008244636,0.0000768611,0.0001127831,0.0002227253,0.00006172952,0.0001022532,0.0002669368,0.00003831882,6.610825e-7],"category_scores_gemma":[0.004887803,0.00006234736,0.00002365491,0.001156636,0.000005450965,0.0004431273,0.00007323368,0.00005823304,0.000003574094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000450265,"about_ca_system_score_gemma":0.00007807471,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002174944,"about_ca_topic_score_gemma":0.00001149397,"domain_scores_codex":[0.9991696,0.00003406395,0.0002071864,0.0002283319,0.0001649053,0.0001958958],"domain_scores_gemma":[0.9983263,0.00119331,0.00008648542,0.0001707012,0.0001503842,0.00007283599],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[8.507764e-7,0.00004467398,0.04407181,0.0000217632,0.000002883953,0.00001848704,0.001020005,0.0620981,0.0005527067,0.003287438,0.0008845791,0.8879967],"study_design_scores_gemma":[0.000557126,0.00005512692,0.0005169282,0.00005208983,0.000001499557,0.000003133402,0.0001310039,0.9959849,0.0002341281,0.001947303,0.0004205087,0.00009629803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04872963,0.00004700533,0.9477419,0.0001953612,0.0001170599,0.0001396627,2.199032e-7,0.0001657519,0.002863382],"genre_scores_gemma":[0.8485414,9.030029e-8,0.1512129,0.00007363492,0.0000456205,0.00000545174,0.000001306135,0.000004693102,0.0001149012],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9338868,"threshold_uncertainty_score":0.585151,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1916625425581596,"score_gpt":0.3384561760901678,"score_spread":0.1467936335320082,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}