{"id":"W4390586053","doi":"10.1016/j.jpi.2023.100358","title":"Use of n-grams and K-means clustering to classify data from free text bone marrow reports","year":2024,"lang":"en","type":"article","venue":"Journal of Pathology Informatics","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Cluster analysis; Artificial intelligence; Security token; Centroid; Bone marrow; Text mining; Natural language processing; Data mining; Pattern recognition (psychology); Pathology; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007657753,0.0007935453,0.0008328378,0.003369044,0.001211949,0.001724361,0.0008464738,0.0008348949,0.000470681],"category_scores_gemma":[0.02863537,0.0002996229,0.0009471533,0.001894022,0.0006555014,0.001877832,0.0009237697,0.0009524408,0.0004950489],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001630299,"about_ca_system_score_gemma":0.002340725,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01367731,"about_ca_topic_score_gemma":0.01623303,"domain_scores_codex":[0.9940495,0.002702877,0.0006415743,0.001101874,0.001281222,0.0002229607],"domain_scores_gemma":[0.982165,0.01180965,0.001229828,0.001359518,0.003227024,0.0002089582],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009623923,0.0005250075,0.08691532,0.0002962624,0.0004949079,0.0002615376,0.002668017,0.168557,0.01686437,0.006978521,0.003145816,0.7123308],"study_design_scores_gemma":[0.00001910326,0.0001725379,0.01863001,0.00004741676,0.00006845997,0.000137345,0.0006555969,0.9574269,0.01410311,0.006981645,0.001681507,0.00007625457],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3631058,0.0003436902,0.6308201,0.0003109447,0.00007772275,0.0004255995,0.00041549,0.001961307,0.002539322],"genre_scores_gemma":[0.5439874,0.0001073126,0.4543852,0.00006338941,0.00002254725,0.0001550317,0.0005556716,0.00008958115,0.000633852],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01367731,"threshold_uncertainty_score":0.04049855,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2526589685442207,"score_gpt":0.3780998978172646,"score_spread":0.1254409292730438,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}