{"id":"W2559885217","doi":"10.1007/s10664-017-9512-6","title":"Curating GitHub for engineered software projects","year":2017,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":345,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Software; Classifier (UML); Software engineering; Software development; Software bug; Software metric; Data mining; Machine learning; Data science; Software quality; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005722354,0.001572351,0.0007534124,0.006204288,0.00193514,0.003147876,0.001891491,0.001330189,0.01171539],"category_scores_gemma":[0.05672424,0.0009064901,0.001566149,0.00276721,0.001008918,0.004165006,0.008731061,0.002056815,0.008287753],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009637127,"about_ca_system_score_gemma":0.004522694,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004172427,"about_ca_topic_score_gemma":0.01070553,"domain_scores_codex":[0.99224,0.002226207,0.0004216188,0.0009854302,0.003583757,0.0005430517],"domain_scores_gemma":[0.9681357,0.00811609,0.002003833,0.01348971,0.007088271,0.001166333],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006327671,0.0004604532,0.02030483,0.002400512,0.0002289062,0.002096733,0.005882422,0.005422005,0.03039898,0.02430057,0.182771,0.7251008],"study_design_scores_gemma":[0.000294969,0.0008783297,0.0322842,0.002256264,0.0005327007,0.003878438,0.005613033,0.1172522,0.08057525,0.08575327,0.6703022,0.0003792718],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1867958,0.00200665,0.5090892,0.003538195,0.001292021,0.001622779,0.009279256,0.2114196,0.07495655],"genre_scores_gemma":[0.2968762,0.001337425,0.589691,0.0008908056,0.0002209033,0.00085102,0.02386589,0.04813353,0.03813322],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01171539,"threshold_uncertainty_score":0.03919184,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05729360034467611,"score_gpt":0.325474775312802,"score_spread":0.2681811749681259,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}