{"id":"W2605988416","doi":"10.7287/peerj.preprints.2617v1","title":"Curating GitHub for engineered software projects","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software; Classifier (UML); Software engineering; Set (abstract data type); Software bug; Software metric; Skew; Software development; Data mining; Machine learning; Data science; Software quality; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002354393,0.0001031155,0.00009581003,0.0001096487,0.00007225919,0.0001297763,0.0006436786,0.00004297827,0.00001615992],"category_scores_gemma":[0.002850621,0.00006489465,0.00004818211,0.000253976,0.00001859568,0.0004264144,0.0001591364,0.00004713466,0.0000621338],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005046628,"about_ca_system_score_gemma":0.0001033364,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004117591,"about_ca_topic_score_gemma":0.000001280755,"domain_scores_codex":[0.9989592,0.00001034193,0.000125279,0.0003036549,0.0002209265,0.0003805677],"domain_scores_gemma":[0.9976552,0.001667266,0.00001847546,0.0004481663,0.0001222949,0.00008866555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001663964,0.0001018503,0.01680901,0.0002596351,0.00008401996,0.00002558391,0.001164742,0.0003160641,0.03152408,0.07933238,0.03604249,0.8343235],"study_design_scores_gemma":[0.01013966,0.001943858,0.06132499,0.001056937,0.00002373572,0.0002039135,0.0001015423,0.2033147,0.5540806,0.01794922,0.1454434,0.004417432],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008586569,0.00002829943,0.9889042,0.0007045455,0.0002411338,0.0003035305,0.000002849026,0.001153522,0.00007539224],"genre_scores_gemma":[0.4562629,0.000001361612,0.541567,0.00004534096,0.00009666096,0.0001340468,5.972458e-7,0.00001758716,0.001874467],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.829906,"threshold_uncertainty_score":0.3412665,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03083896895676235,"score_gpt":0.2727086505746935,"score_spread":0.2418696816179312,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}