{"id":"W7008012785","doi":"","title":"Assessing the effect of preprocessing of clinical notes on classification tasks and similarity measures","year":2024,"lang":"en","type":"dissertation","venue":"Mspace (University of Manitoba)","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research","keywords":"Cosine similarity; Similarity (geometry); Preprocessor; Punctuation; Word (group theory); Pattern recognition (psychology); Support vector machine; Noise (video); Statistical hypothesis testing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001963354,0.0001774288,0.0005060222,0.0001859522,0.0002031823,0.00007444717,0.0007704687,0.0002917126,7.937933e-7],"category_scores_gemma":[0.0007958346,0.0001536465,0.000158853,0.0003081101,0.0001565698,0.0002610318,0.0001266116,0.0007298474,0.000002148167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003737109,"about_ca_system_score_gemma":0.0001751926,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001279706,"about_ca_topic_score_gemma":0.0146803,"domain_scores_codex":[0.9978478,0.0007088696,0.0002781312,0.0004831977,0.0005528464,0.0001291875],"domain_scores_gemma":[0.9967054,0.001589351,0.0008598837,0.0005567495,0.0002402762,0.00004834314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000321181,0.0001158871,0.6704667,0.01143907,0.0002501295,0.00001960578,0.00549768,0.0004255663,0.001085219,0.003196194,0.0003810586,0.3068017],"study_design_scores_gemma":[0.0002313144,0.0004026765,0.9533289,0.001324521,0.0001438461,0.000001529727,0.00371515,0.03951864,0.0008047934,0.0002245988,0.0001629595,0.0001410303],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9908379,0.0004653529,0.00499221,0.001437143,0.0004436274,0.0002892966,0.000004876782,0.00005758541,0.001471994],"genre_scores_gemma":[0.9977532,0.00004997782,0.002064203,0.000009058813,0.00003833783,3.892746e-7,0.00001953406,0.00001178016,0.00005348902],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3066607,"threshold_uncertainty_score":0.8191952,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06716601192906164,"score_gpt":0.3638520434767867,"score_spread":0.2966860315477251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}