{"id":"W3208243843","doi":"10.1109/rew53955.2021.00025","title":"Is BERT the New Silver Bullet? - An Empirical Investigation of Requirements Dependency Classification","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Transformer; Artificial intelligence; Silver bullet; Dependency (UML); Empirical research; Encoder; Machine learning; Data mining; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002916814,0.00006607838,0.00006936615,0.00004813232,0.00005124931,0.00008453932,0.0006006972,0.00004851801,0.0001532202],"category_scores_gemma":[0.0003596022,0.00004939722,0.00003109863,0.0005189574,0.00003119521,0.0003576537,0.0001674563,0.000106796,0.00005256172],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004145767,"about_ca_system_score_gemma":0.0003027879,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006810713,"about_ca_topic_score_gemma":0.00001401532,"domain_scores_codex":[0.9988067,0.0000757649,0.0001776402,0.000271498,0.0005196525,0.0001487288],"domain_scores_gemma":[0.9987211,0.0002329605,0.00003633343,0.0007327236,0.0001680407,0.0001088455],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000008656432,0.000187235,0.6305229,0.00006271112,0.00007931123,0.00002124618,0.008833674,0.0002385356,0.1388153,0.02636044,0.1392713,0.05559863],"study_design_scores_gemma":[0.0002950528,0.00008158195,0.7949411,0.00002171821,0.000006679174,0.00001344796,0.00007800482,0.04209727,0.1500658,0.00816124,0.004067906,0.0001701753],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3895782,0.00009745169,0.5886649,0.02048552,0.0002289845,0.0001495738,7.300795e-7,0.0001729512,0.0006216822],"genre_scores_gemma":[0.9536896,0.000008849619,0.04362322,0.0007781202,0.00005358556,0.000006952609,0.000005125734,0.000007140693,0.001827388],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5641114,"threshold_uncertainty_score":0.201436,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1038612302168992,"score_gpt":0.3512246052520377,"score_spread":0.2473633750351385,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}