{"id":"W4393617534","doi":"10.5281/zenodo.5732299","title":"Bash in the Wild: Language Usage, Code Smells, and Bugs - Dataset","year":2022,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Hate Speech and Cyberbullying Detection","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Code (set theory); Computer science; Programming language; Code smell; Software; Software quality; Set (abstract data type)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","scholarly_communication","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001533474,0.0001942121,0.0001685463,0.0003384988,0.001691031,0.001617433,0.003513016,0.00008770926,0.01819395],"category_scores_gemma":[0.0003104524,0.0001747962,0.00003187486,0.0008577742,0.0001027666,0.0003244349,0.003378516,0.0007960743,0.001845935],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009664587,"about_ca_system_score_gemma":0.000007619845,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000218039,"about_ca_topic_score_gemma":0.00001355778,"domain_scores_codex":[0.9972915,0.0008843649,0.0002423169,0.0006231756,0.0005883106,0.0003703349],"domain_scores_gemma":[0.9983903,0.00005664616,0.0001269573,0.00125465,0.00006562712,0.0001057963],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000009559174,0.00007091752,1.41588e-7,0.00004284618,0.00001169649,0.0001248752,0.0004826647,0.000006351202,0.00003244194,0.00014907,0.9759075,0.02316191],"study_design_scores_gemma":[0.0002489529,0.0001660243,0.00003460252,0.00001647439,0.00001011183,0.0002946047,0.0002249829,0.0001376376,0.0000138753,0.00003743012,0.9986215,0.0001938447],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0001915521,0.0001692373,0.0008192057,0.0006608703,0.000160035,0.0004395611,0.9953582,0.0002478805,0.00195341],"genre_scores_gemma":[0.0003748933,0.0003668002,0.0001305769,0.0007010495,0.00008855401,1.554032e-7,0.9978849,0.0002861609,0.000166899],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02296807,"threshold_uncertainty_score":0.9996086,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02404359875268159,"score_gpt":0.2531799358686012,"score_spread":0.2291363371159196,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}