{"id":"W2888962833","doi":"","title":"Poster: Designing Bug Detection Rules for Fewer False Alarms","year":2018,"lang":"en","type":"article","venue":"International Conference on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"False positive paradox; Computer science; Software bug; Constant false alarm rate; Open source; Static analysis; False alarm; True positive rate; False positives and false negatives; False positive rate; Data mining; Machine learning; Artificial intelligence; Software; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0146696,0.001899881,0.001311439,0.004973724,0.001207339,0.003112522,0.003267763,0.002507379,0.002487945],"category_scores_gemma":[0.08199285,0.001115795,0.00195366,0.00157722,0.001674072,0.00439275,0.002145511,0.003413991,0.001911905],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001168336,"about_ca_system_score_gemma":0.003017332,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003994431,"about_ca_topic_score_gemma":0.005983684,"domain_scores_codex":[0.9851521,0.003610146,0.002043324,0.003491233,0.004962365,0.0007407001],"domain_scores_gemma":[0.9188291,0.03745852,0.008203489,0.01389479,0.02007631,0.001537789],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009373903,0.0009071291,0.09283316,0.0008839223,0.0004302057,0.0006116946,0.0009210322,0.05392163,0.03023074,0.00751196,0.03461472,0.7761963],"study_design_scores_gemma":[0.0003018584,0.001003942,0.02405151,0.0004246886,0.0005226955,0.001734837,0.0003407519,0.8402403,0.0755586,0.02566508,0.02985664,0.0002990871],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09025367,0.0007372524,0.8714585,0.002831923,0.0003780464,0.000878726,0.001109779,0.02892187,0.003430212],"genre_scores_gemma":[0.3262157,0.0002238652,0.6655598,0.001043647,0.0001749949,0.0003918562,0.002664203,0.001652732,0.002073294],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0146696,"threshold_uncertainty_score":0.07758117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03837499704633768,"score_gpt":0.2935458189583928,"score_spread":0.2551708219120551,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}