{"id":"W3085364681","doi":"10.14778/3415478.3415562","title":"Data collection and quality challenges for deep learning","year":2020,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":145,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Machine learning; Artificial intelligence; Feature engineering; Deep learning; Data collection; Big data; Data science; Software; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05710194,0.00112651,0.002032391,0.004169252,0.002830537,0.009194133,0.005085642,0.00294739,0.003655462],"category_scores_gemma":[0.2077572,0.00137614,0.00154539,0.007496398,0.00409996,0.009662429,0.008774134,0.008299758,0.002638957],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003892775,"about_ca_system_score_gemma":0.008029495,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005440305,"about_ca_topic_score_gemma":0.006050148,"domain_scores_codex":[0.9378331,0.02391522,0.005539044,0.005116389,0.0264243,0.001171815],"domain_scores_gemma":[0.7931804,0.1000525,0.009855264,0.03990825,0.0531601,0.003843497],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004928268,0.0004301,0.02933097,0.003740502,0.0004032459,0.0004381976,0.002049126,0.02005059,0.007714521,0.08695638,0.1157516,0.7326419],"study_design_scores_gemma":[0.0001610188,0.0004631849,0.02111007,0.003583948,0.000193628,0.001066379,0.003058152,0.102932,0.02128894,0.3995869,0.4462947,0.0002610633],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01913211,0.01141685,0.9078093,0.03980681,0.001741266,0.001254668,0.008394149,0.003763807,0.006681067],"genre_scores_gemma":[0.1118938,0.009625044,0.8412076,0.008286663,0.001994267,0.003711052,0.01633989,0.001847473,0.005094176],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05710194,"threshold_uncertainty_score":0.3019876,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1284582489283515,"score_gpt":0.3110977474383105,"score_spread":0.182639498509959,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}