{"id":"W4311557029","doi":"10.1371/journal.pcbi.1010718","title":"Eleven quick tips for data cleaning and feature engineering","year":2022,"lang":"en","type":"article","venue":"PLoS Computational Biology","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Feature engineering; Computer science; Feature (linguistics); Data science; Field (mathematics); Component (thermodynamics); Informatics; Preprocessor; Data pre-processing; Data mining; Key (lock); Health informatics; Artificial intelligence; Machine learning; Engineering; Health care; Deep learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07069039,0.004701122,0.00244778,0.008720366,0.003164738,0.009381328,0.006894579,0.008028211,0.02546266],"category_scores_gemma":[0.3643239,0.003468125,0.003246345,0.007360393,0.00516299,0.01432361,0.007608915,0.02077778,0.03342189],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002250155,"about_ca_system_score_gemma":0.007614656,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002714002,"about_ca_topic_score_gemma":0.006485593,"domain_scores_codex":[0.9273282,0.04024939,0.0120444,0.00412169,0.01466388,0.001592483],"domain_scores_gemma":[0.6245761,0.2243312,0.01774216,0.04176325,0.08612567,0.005461512],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003224788,0.0002549617,0.001889083,0.002600511,0.0001571357,0.000527959,0.001816634,0.001407497,0.002704622,0.02594306,0.5903683,0.3720078],"study_design_scores_gemma":[0.0003038386,0.0003911596,0.00336278,0.004296551,0.000126828,0.001245027,0.001649368,0.004941929,0.006550162,0.1417699,0.8349423,0.0004200349],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001521175,0.00477367,0.8670491,0.08522592,0.01001821,0.002115419,0.004154354,0.01891213,0.006230016],"genre_scores_gemma":[0.004199163,0.002459042,0.9561446,0.02164071,0.002470691,0.00241317,0.002328158,0.003075256,0.005269161],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.07069039,"threshold_uncertainty_score":0.373851,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06451839150280238,"score_gpt":0.2890776891224214,"score_spread":0.2245592976196191,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}