{"id":"W1505105576","doi":"10.1109/iolts.2015.7229856","title":"Mining simulation metrics for failure triage in regression testing","year":2015,"lang":"en","type":"article","venue":"","topic":"VLSI and Analog Circuit Testing","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Bottleneck; Debugging; Cluster analysis; Regression testing; Root cause; Triage; Data mining; Reliability engineering; Machine learning; Engineering; Embedded system; Software; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002391316,0.001526339,0.001334625,0.005815955,0.000489452,0.001118392,0.001296263,0.0008156699,0.0007570199],"category_scores_gemma":[0.01709961,0.0004326261,0.0008846438,0.002410971,0.0006583742,0.001419056,0.001054201,0.0007103548,0.0003721767],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001110171,"about_ca_system_score_gemma":0.001340652,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002830725,"about_ca_topic_score_gemma":0.004476888,"domain_scores_codex":[0.9975232,0.000872476,0.0002873218,0.0003799529,0.0007894771,0.0001475183],"domain_scores_gemma":[0.9904174,0.005475791,0.001583104,0.0008317024,0.001449718,0.0002422027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002210043,0.0002728334,0.0555232,0.0002825651,0.0001448789,0.0002192441,0.0002401482,0.6894706,0.01315337,0.007419926,0.001747574,0.2313047],"study_design_scores_gemma":[0.000006772895,0.00004559672,0.00212763,0.00001040891,0.00000982906,0.00004482487,0.0000292974,0.9900396,0.00333936,0.003959163,0.0003783161,0.000009177502],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1063655,0.0003572182,0.8898501,0.0001683353,0.00001529811,0.0001417376,0.0005617468,0.001974965,0.000564985],"genre_scores_gemma":[0.7419148,0.0001643472,0.2550231,0.00005009973,0.00002000055,0.0002628139,0.002076013,0.0001669322,0.0003218989],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005815955,"threshold_uncertainty_score":0.01264662,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1916625425581596,"score_gpt":0.3384561760901678,"score_spread":0.1467936335320082,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}