{"id":"W3178956269","doi":"10.1007/s10664-021-10004-6","title":"Evaluating the impact of falsely detected performance bug-inducing changes in JIT models","year":2021,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Queen's University; Polytechnique Montréal; Concordia University","funders":"","keywords":"Leverage (statistics); Software bug; Computer science; Software; Software quality; Empirical research; Software quality assurance; Quality (philosophy); Capability Maturity Model; Source code; Software engineering; Software development; Operating system; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01566727,0.001007623,0.0006448692,0.001464305,0.00063575,0.001535996,0.001591823,0.002086743,0.001437132],"category_scores_gemma":[0.1943422,0.0005431952,0.001063975,0.0008711187,0.001305906,0.002350028,0.001165968,0.002172668,0.000258705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001267899,"about_ca_system_score_gemma":0.001833159,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006997013,"about_ca_topic_score_gemma":0.01181846,"domain_scores_codex":[0.9863189,0.006900731,0.0009060206,0.002266545,0.002918123,0.0006895311],"domain_scores_gemma":[0.5707826,0.383166,0.01653843,0.01820471,0.009001589,0.002306707],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006643844,0.002700547,0.4733267,0.0007251075,0.001526198,0.0006788447,0.0007320732,0.3768537,0.01499764,0.003885767,0.003050113,0.1148794],"study_design_scores_gemma":[0.0002604687,0.003124334,0.1007369,0.0001014658,0.0008157821,0.0004066728,0.0003482192,0.8775497,0.0120962,0.003691827,0.0007873857,0.00008108625],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9907867,0.0003316338,0.006762076,0.0002936788,0.00007941871,0.00003298,0.000305711,0.0004710425,0.0009367098],"genre_scores_gemma":[0.993928,0.00004762413,0.005263796,0.00005358387,0.00001908743,0.00001181373,0.000388008,0.00006602788,0.0002221672],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01566727,"threshold_uncertainty_score":0.08285743,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09454756759896381,"score_gpt":0.3660511802990926,"score_spread":0.2715036127001288,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}