{"id":"W4388089782","doi":"10.1109/isc257844.2023.10293563","title":"A New Semantic Similarity Scheme for More Accurate Identification in Medical Data","year":2023,"lang":"en","type":"article","venue":"","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Jaccard index; Computer science; Similarity (geometry); Semantic similarity; Context (archaeology); Identification (biology); Information retrieval; Cosine similarity; Data mining; Set (abstract data type); Data set; Dice; Key (lock); Similarity measure; Scheme (mathematics); Artificial intelligence; Pattern recognition (psychology); Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005108262,0.0004255859,0.0008598476,0.006142169,0.000806882,0.001785045,0.001271661,0.0008965484,0.00157562],"category_scores_gemma":[0.01864009,0.0001840869,0.0007745333,0.005265906,0.001082653,0.005660214,0.00278753,0.0009621868,0.0008907048],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001085527,"about_ca_system_score_gemma":0.001375351,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007011623,"about_ca_topic_score_gemma":0.0007277303,"domain_scores_codex":[0.9932625,0.001526301,0.001377525,0.0008733538,0.002768497,0.000191914],"domain_scores_gemma":[0.9929651,0.002150827,0.001092045,0.00146829,0.002036197,0.000287439],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000687808,0.0002975317,0.01467699,0.0009471559,0.000247405,0.0002931931,0.00174971,0.02172801,0.05129191,0.1214718,0.006015877,0.7805926],"study_design_scores_gemma":[0.0001523598,0.001918128,0.02283325,0.0005033801,0.0002768502,0.004272961,0.002066095,0.6016837,0.07708016,0.2101614,0.0786623,0.0003894084],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03351116,0.0005331786,0.962346,0.0003057722,0.0001347672,0.0002896721,0.0004746681,0.0005464589,0.001858333],"genre_scores_gemma":[0.2598319,0.0002757482,0.7372108,0.000133776,0.0001060727,0.0002751579,0.001085415,0.00007036157,0.001010687],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006142169,"threshold_uncertainty_score":0.02701545,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08753904631601674,"score_gpt":0.3982877170484738,"score_spread":0.3107486707324571,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}