{"id":"W2952903800","doi":"10.1109/icst.2019.00019","title":"BugsJS: a Benchmark of JavaScript Bugs","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":94,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"JavaScript; Unobtrusive JavaScript; Computer science; Benchmark (surveying); Programming language; Test case; Software bug; Unit testing; Rich Internet application; Operating system; Database; Software; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00566814,0.001664238,0.0006491041,0.005360328,0.0006792318,0.001141416,0.00290446,0.001272778,0.001066714],"category_scores_gemma":[0.03439905,0.0005437106,0.0009441368,0.004233945,0.001132881,0.001815648,0.00171177,0.001414875,0.0009295581],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007938062,"about_ca_system_score_gemma":0.001774942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004403885,"about_ca_topic_score_gemma":0.006278822,"domain_scores_codex":[0.9892736,0.002322168,0.001513904,0.001497792,0.004823578,0.0005689814],"domain_scores_gemma":[0.967399,0.0149638,0.00405836,0.00504528,0.0072895,0.001244148],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002865691,0.003297372,0.1990172,0.01433499,0.001277747,0.00260231,0.003616604,0.0783501,0.0882781,0.01070513,0.1664803,0.4291745],"study_design_scores_gemma":[0.001067933,0.004410102,0.2517635,0.001921781,0.0008372967,0.005122528,0.002112906,0.3928814,0.1320781,0.01524384,0.1920065,0.0005542094],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.7875907,0.009624546,0.08969591,0.001048601,0.0006564929,0.00136284,0.03692102,0.05961674,0.01348318],"genre_scores_gemma":[0.7034864,0.002238874,0.1726481,0.0006038807,0.0001463137,0.00108743,0.1071049,0.008819089,0.003865],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.00566814,"threshold_uncertainty_score":0.02997631,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004992930306730839,"score_gpt":0.2021300423044523,"score_spread":0.1971371119977214,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}