{"id":"W4381938374","doi":"10.2196/44310","title":"Normal Workflow and Key Strategies for Data Cleaning Toward Real-World Data: Viewpoint","year":2023,"lang":"en","type":"article","venue":"Interactive Journal of Medical Research","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Workflow; Data quality; Computer science; Data science; Data governance; Data mining; Process (computing); Data virtualization; Data set; Data management; Key (lock); Quality (philosophy); Data cleansing; Database; Engineering; Computer security; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03906247,0.001387445,0.00158442,0.006544611,0.003433891,0.0113084,0.006889096,0.003071117,0.002367851],"category_scores_gemma":[0.06047616,0.001096224,0.002288339,0.005949075,0.007840862,0.01526235,0.005010455,0.004408066,0.001569816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00439495,"about_ca_system_score_gemma":0.01492506,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01006223,"about_ca_topic_score_gemma":0.004240473,"domain_scores_codex":[0.9677614,0.01187704,0.003530647,0.00594394,0.009470595,0.001416277],"domain_scores_gemma":[0.9493442,0.01260823,0.003517903,0.01326985,0.01932754,0.001932317],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002669875,0.0002441327,0.006032431,0.001033585,0.0001335126,0.0005217578,0.002663766,0.0273294,0.002943904,0.7142825,0.01701924,0.2275287],"study_design_scores_gemma":[0.0000693087,0.000153786,0.001575093,0.000410433,0.00007102868,0.0009246128,0.001679946,0.1700975,0.006236986,0.7628905,0.05574917,0.0001416451],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"commentary","genre_scores_codex":[0.003040413,0.0008855615,0.9879771,0.004976511,0.0001805649,0.0002404178,0.0001117745,0.0005378168,0.002049898],"genre_scores_gemma":[0.1110122,0.002280611,0.8817626,0.001016664,0.0003834262,0.0005225279,0.000501318,0.0002398894,0.002280725],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.9609375,"threshold_uncertainty_score":0.2065846,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7111703779494515,"score_gpt":0.6382519355459901,"score_spread":0.07291844240346135,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}