{"id":"W4386466998","doi":"10.23889/ijpds.v6i1.1757","title":"A scoping review of preprocessing methods for unstructured text data to assess data quality.","year":2022,"lang":"en","type":"review","venue":"PubMed","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba; George & Fay Yee Centre for Healthcare Innovation; Manitoba Health","funders":"Canada Research Chairs","keywords":"Computer science; Preprocessor; Punctuation; Stop words; Data quality; Data pre-processing; Information retrieval; Lexical analysis; Natural language processing; Artificial intelligence; Data mining","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.006500347,0.0003344141,0.001761029,0.00007683007,0.00008783537,0.0000358371,0.003555172,0.0003055129,0.00002681092],"category_scores_gemma":[0.0158987,0.0002693544,0.0002068036,0.0003482307,0.0001068472,0.000006782072,0.004318771,0.0002167198,4.678167e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002474509,"about_ca_system_score_gemma":0.0006429979,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007061113,"about_ca_topic_score_gemma":0.000007051492,"domain_scores_codex":[0.9960241,0.0008456449,0.001081145,0.001395566,0.0002295531,0.0004239347],"domain_scores_gemma":[0.9946309,0.0005412488,0.0007733675,0.003838775,0.00006651288,0.0001492513],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000006694865,0.00001425733,2.806201e-7,0.2339131,0.0001285387,2.988801e-7,0.000002219146,3.043831e-8,0.000003359141,0.000001958912,0.003014567,0.7629147],"study_design_scores_gemma":[0.0001006848,0.0000258241,0.000003289915,0.05630078,0.0005089734,0.00001500787,0.000008961293,0.000001949189,0.00002043449,0.000008080595,0.9427088,0.0002971494],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[1.900801e-7,0.9723616,0.02212806,0.00009019211,0.0003779559,0.003386381,0.001494935,0.0000224184,0.0001382394],"genre_scores_gemma":[1.558485e-7,0.8849846,0.1011599,0.0003202449,0.0002262153,0.003033465,0.01010174,0.00004533422,0.0001283505],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9396943,"threshold_uncertainty_score":0.9999759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6516293020376296,"score_gpt":0.5791308034657566,"score_spread":0.07249849857187296,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}