{"id":"W4312438866","doi":"10.1007/978-3-031-16990-8_4","title":"Data Preprocessing","year":2022,"lang":"en","type":"book-chapter","venue":"International series in management science/operations research/International series in operations research & management science","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":16,"is_retracted":false,"has_abstract":false,"ca_institutions":"Public Health Ontario; York University","funders":"","keywords":"Preprocessor; Software portability; Computer science; Raw data; Data pre-processing; Context (archaeology); Usability; Data quality; Task (project management); Data mining; Precondition; Data processing; Database; Artificial intelligence; Human–computer interaction; Engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001792701,0.002636473,0.001377607,0.005773802,0.001234111,0.003747076,0.001575166,0.0006861616,0.1723719],"category_scores_gemma":[0.01090202,0.000890765,0.001679947,0.004735501,0.0005981683,0.001630117,0.002807796,0.001692261,0.1589651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007772816,"about_ca_system_score_gemma":0.003020909,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002962713,"about_ca_topic_score_gemma":0.003015012,"domain_scores_codex":[0.9982129,0.0001272064,0.0002684075,0.0006581526,0.000527529,0.0002057556],"domain_scores_gemma":[0.995216,0.000980365,0.0002025486,0.001798104,0.001634144,0.0001688801],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007323536,0.0001905789,0.002818365,0.001197967,0.0001041055,0.0002196993,0.0002498098,0.0007648173,0.0187521,0.006734389,0.5671775,0.4010583],"study_design_scores_gemma":[0.0001408209,0.0001800612,0.004730668,0.0002436755,0.0001257118,0.0003640348,0.0002777424,0.005553995,0.06020053,0.008497643,0.919596,0.00008911329],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.01180857,0.001529736,0.2825898,0.001713714,0.002778453,0.005793486,0.4053771,0.2090668,0.07934241],"genre_scores_gemma":[0.03475739,0.001470303,0.3771279,0.002001005,0.0007001741,0.007844126,0.4445906,0.02087293,0.1106356],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.1723719,"threshold_uncertainty_score":0.5766416,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4274617544225517,"score_gpt":0.5633610717969169,"score_spread":0.1358993173743652,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}