{"id":"W4284693828","doi":"10.1145/3510003.3510106","title":"Nessie","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 44th International Conference on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; National Science Foundation","keywords":"Callback; Computer science; Asynchronous communication; Unit testing; Test case; Generator (circuit theory); JavaScript; Test (biology); Operating system; Programming language; Software; Machine learning; Computer network","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002116689,0.001465737,0.000815244,0.001943411,0.0006462858,0.001706337,0.003176105,0.001249223,0.02858166],"category_scores_gemma":[0.01005825,0.000799001,0.001405785,0.0009766886,0.001022645,0.002469677,0.002347674,0.001700204,0.01473524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00111521,"about_ca_system_score_gemma":0.001917829,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00183555,"about_ca_topic_score_gemma":0.003866971,"domain_scores_codex":[0.9973827,0.0006271236,0.0002313973,0.0005701202,0.0009875647,0.0002011124],"domain_scores_gemma":[0.9950222,0.001924681,0.0003603475,0.001287353,0.001254398,0.0001510959],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008457471,0.0003555059,0.003748155,0.001389729,0.0001250198,0.0004925689,0.0004182515,0.05202567,0.02191524,0.1046228,0.0651587,0.7489026],"study_design_scores_gemma":[0.0002689071,0.000530723,0.00119232,0.0003284471,0.0001004358,0.001445174,0.0001589082,0.5655947,0.05674078,0.09388532,0.2796387,0.0001156663],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006846061,0.0003509677,0.9224019,0.0002256849,0.0001663431,0.0004013725,0.001561638,0.04862604,0.01941996],"genre_scores_gemma":[0.08142891,0.0003254456,0.8785304,0.0004120803,0.00004690278,0.0006197538,0.006970129,0.007443244,0.02422317],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02858166,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02496960565977006,"score_gpt":0.2409004731539503,"score_spread":0.2159308674941803,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}