{"id":"W2303463411","doi":"10.1007/s10836-016-5577-1","title":"Exemplar-based Failure Triage for Regression Design Debugging","year":2016,"lang":"en","type":"article","venue":"Journal of Electronic Testing","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Debugging; Computer science; Metric (unit); Triage; Flexibility (engineering); Process (computing); Reliability engineering; Data mining; Engineering; Programming language; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00300231,0.001419364,0.001296916,0.002276022,0.0005410397,0.001122133,0.002412073,0.001746444,0.004983092],"category_scores_gemma":[0.02678007,0.0006442487,0.0008130934,0.0008604464,0.0006734189,0.00187386,0.001918712,0.002531423,0.001402498],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004993667,"about_ca_system_score_gemma":0.0009658424,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001309107,"about_ca_topic_score_gemma":0.002699645,"domain_scores_codex":[0.9964281,0.001323252,0.0002924645,0.0005724332,0.001123278,0.0002604545],"domain_scores_gemma":[0.9815305,0.009717355,0.00177967,0.00380221,0.002666566,0.0005036882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001308197,0.000611514,0.009874947,0.0003350258,0.0001670712,0.0006786411,0.0003885437,0.2457263,0.02820259,0.01492107,0.008504826,0.6892813],"study_design_scores_gemma":[0.00002427438,0.0001159965,0.0004339154,0.00002891834,0.00001960172,0.0001524732,0.00002452873,0.9874153,0.006502667,0.00421525,0.001051321,0.00001563766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02171027,0.0001501928,0.9701737,0.0001078549,0.0000433193,0.00006669106,0.00009130814,0.00683873,0.0008179261],"genre_scores_gemma":[0.4337605,0.00008108254,0.5640063,0.0001217927,0.00003320206,0.00009052591,0.0003173705,0.0004912351,0.001097984],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004983092,"threshold_uncertainty_score":0.01667011,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05114569072665122,"score_gpt":0.2914108569765113,"score_spread":0.2402651662498601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}