{"id":"W2990774762","doi":"10.1109/models.2019.00-19","title":"Pitfalls Analyzer: Quality Control for Model-Driven Data Science Pipelines","year":2019,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Pipeline transport; Pipeline (software); Data science; Raw data; Quality (philosophy); Data mining; Software engineering; Programming language; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01806444,0.003566982,0.00137808,0.005972452,0.001169886,0.004554213,0.005411651,0.001628581,0.005573568],"category_scores_gemma":[0.07953137,0.002780932,0.002749533,0.002592509,0.0024878,0.007918619,0.005213135,0.004282668,0.001972089],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002915373,"about_ca_system_score_gemma":0.006842121,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01015487,"about_ca_topic_score_gemma":0.007580103,"domain_scores_codex":[0.982226,0.003514242,0.002360325,0.00314067,0.007817931,0.0009408363],"domain_scores_gemma":[0.9217159,0.03775695,0.01003023,0.01654674,0.01249008,0.001460116],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003898757,0.001163546,0.07033359,0.004624871,0.001051643,0.002019672,0.004309896,0.1192815,0.07983488,0.04533291,0.1127512,0.5553976],"study_design_scores_gemma":[0.0003203163,0.000429035,0.007796142,0.0003592045,0.0002100675,0.0005963737,0.0003330246,0.8260205,0.09904215,0.02427491,0.0402852,0.0003330878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01602535,0.0002696137,0.681366,0.0004161378,0.0001114679,0.0005469567,0.002193335,0.2979638,0.001107348],"genre_scores_gemma":[0.2407582,0.0003372345,0.7225288,0.0006239291,0.00007796269,0.0009575665,0.008706591,0.02420336,0.001806349],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01806444,"threshold_uncertainty_score":0.09553504,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07607110023299138,"score_gpt":0.3725011453655077,"score_spread":0.2964300451325164,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}