{"id":"W4389337865","doi":"10.1007/s10664-023-10400-0","title":"Bug characterization in machine learning-based systems","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University; Polytechnique Montréal","funders":"","keywords":"Computer science; Software bug; Software; Component (thermodynamics); Software system; Task (project management); Process (computing); Software engineering; Software development; Focus (optics); Software maintenance; Corrective maintenance; Machine learning; Reliability engineering; Operating system; Systems engineering; Engineering; Preventive maintenance","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004665167,0.0005314131,0.0005260096,0.006278696,0.0005330492,0.001476008,0.000950951,0.001007989,0.001189203],"category_scores_gemma":[0.06844034,0.0003419853,0.0005509327,0.002488303,0.0009455484,0.00311223,0.0009171902,0.001031561,0.0002426255],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008380294,"about_ca_system_score_gemma":0.001224789,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003376165,"about_ca_topic_score_gemma":0.004399724,"domain_scores_codex":[0.9933207,0.001931078,0.0009575678,0.001100091,0.002269346,0.0004211822],"domain_scores_gemma":[0.9147225,0.05359252,0.0135661,0.008267517,0.008839646,0.001011755],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005831433,0.0005868163,0.5865269,0.0005684799,0.0002104451,0.0004159231,0.001126734,0.06986167,0.008586907,0.01041597,0.003020404,0.3180966],"study_design_scores_gemma":[0.00006571769,0.0004946084,0.1505489,0.0002164536,0.0001565505,0.001233612,0.0005820282,0.8016824,0.01181043,0.03109302,0.002048967,0.00006732902],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8627934,0.001051442,0.1309927,0.0004291732,0.00004607887,0.0001227786,0.0007064349,0.001808361,0.00204964],"genre_scores_gemma":[0.9758084,0.00008541236,0.02301563,0.00002995895,0.00001479789,0.000027836,0.0005729179,0.00008867311,0.0003564461],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006278696,"threshold_uncertainty_score":0.02467203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02440577382411849,"score_gpt":0.2681081263815844,"score_spread":0.2437023525574659,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}