{"id":"W3197531090","doi":"","title":"A Data Extraction Algorithm From Open Source Software Project Repositories for Building Duration Estimation Models: Case Study of GitHub","year":2020,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Duration (music); Computer science; Estimation; Schedule; Software; Data extraction; Variable (mathematics); Work (physics); Data mining; Data science; Engineering; Systems engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003690327,0.0008180329,0.0006149978,0.003911467,0.0006973413,0.001404426,0.001035922,0.000884788,0.0008360274],"category_scores_gemma":[0.01809134,0.0004180313,0.001004119,0.004903295,0.0003359181,0.002206845,0.001313161,0.00103432,0.0008251431],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007860197,"about_ca_system_score_gemma":0.002224673,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008509718,"about_ca_topic_score_gemma":0.01223664,"domain_scores_codex":[0.9980655,0.0006106543,0.0003279278,0.0004492954,0.0004491549,0.00009741839],"domain_scores_gemma":[0.9879067,0.007281456,0.001183368,0.001295707,0.002103032,0.000229877],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002544787,0.000649824,0.08738345,0.0007033557,0.0001603738,0.0006563488,0.001632471,0.07316828,0.007975934,0.006882406,0.008192487,0.8123406],"study_design_scores_gemma":[0.00006129564,0.0003066387,0.04113799,0.0002375149,0.0001359705,0.000726831,0.001482741,0.9071767,0.01502729,0.009392053,0.02419985,0.0001152106],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1599371,0.0004600357,0.8291007,0.0006016981,0.00005275173,0.0006034415,0.004233442,0.003089292,0.001921559],"genre_scores_gemma":[0.2362129,0.000358796,0.7536469,0.00004858239,0.00002735078,0.0006939758,0.00774939,0.0001495627,0.001112613],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008509718,"threshold_uncertainty_score":0.01951659,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07751349942747296,"score_gpt":0.3566497924798842,"score_spread":0.2791362930524113,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}