{"id":"W2605988416","doi":"10.7287/peerj.preprints.2617v1","title":"Curating GitHub for engineered software projects","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software; Classifier (UML); Software engineering; Set (abstract data type); Software bug; Software metric; Skew; Software development; Data mining; Machine learning; Data science; Software quality; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007754814,0.001346037,0.0007814755,0.02440777,0.001587618,0.00314533,0.001772756,0.001152516,0.002047325],"category_scores_gemma":[0.05381469,0.0006623246,0.0009609798,0.01622475,0.001144796,0.003318577,0.006240573,0.001449212,0.002352564],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001507164,"about_ca_system_score_gemma":0.003676893,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01020841,"about_ca_topic_score_gemma":0.02341218,"domain_scores_codex":[0.9912901,0.001431474,0.001358905,0.002082267,0.003310984,0.0005262309],"domain_scores_gemma":[0.9504173,0.01120214,0.01447715,0.01135304,0.0106832,0.001867264],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004756016,0.0002515099,0.4143955,0.003244899,0.0003099061,0.002170129,0.01042539,0.003960328,0.01234407,0.008884588,0.1149625,0.4285756],"study_design_scores_gemma":[0.00009770742,0.0002963727,0.6335145,0.001229251,0.0002227424,0.002564922,0.007176461,0.04689293,0.0298673,0.01128937,0.2665604,0.0002880846],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.729637,0.00470881,0.1309038,0.002364744,0.00049387,0.002762731,0.07653728,0.03442331,0.01816846],"genre_scores_gemma":[0.4791713,0.001738093,0.3683485,0.0003865165,0.0001680774,0.001813983,0.1386996,0.002756261,0.006917635],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02440777,"threshold_uncertainty_score":0.04101187,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03083896895676235,"score_gpt":0.2727086505746935,"score_spread":0.2418696816179312,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}