{"id":"W7008012785","doi":"","title":"Assessing the effect of preprocessing of clinical notes on classification tasks and similarity measures","year":2024,"lang":"en","type":"dissertation","venue":"Mspace (University of Manitoba)","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research","keywords":"Cosine similarity; Similarity (geometry); Preprocessor; Punctuation; Word (group theory); Pattern recognition (psychology); Support vector machine; Noise (video); Statistical hypothesis testing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0745717,0.002163463,0.00189273,0.00294003,0.001501318,0.004386435,0.001649507,0.002442298,0.002118751],"category_scores_gemma":[0.3384773,0.0006894506,0.003769197,0.00258246,0.001881866,0.004994425,0.002624356,0.002940564,0.001147552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001490013,"about_ca_system_score_gemma":0.002514797,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00178794,"about_ca_topic_score_gemma":0.001840498,"domain_scores_codex":[0.9375882,0.03517452,0.01016301,0.01069209,0.005284268,0.001097904],"domain_scores_gemma":[0.3931766,0.5554575,0.01939309,0.01706747,0.01246781,0.002437406],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.04575821,0.007133074,0.4067248,0.005591133,0.008164986,0.0003471196,0.00332945,0.04280278,0.02059567,0.001433457,0.007613319,0.4505059],"study_design_scores_gemma":[0.00446196,0.05127309,0.5376468,0.002214996,0.009317051,0.001227428,0.00337397,0.3043891,0.05923287,0.01397776,0.01196987,0.0009150949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9132585,0.003986781,0.07001564,0.00114075,0.0007680944,0.003273129,0.003199601,0.00128821,0.003069289],"genre_scores_gemma":[0.876793,0.0007383424,0.1122872,0.0005337024,0.0003086634,0.003325291,0.005179261,0.000294092,0.0005405131],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0745717,"threshold_uncertainty_score":0.3943775,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06716601192906164,"score_gpt":0.3638520434767867,"score_spread":0.2966860315477251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}