{"id":"W3203751410","doi":"10.1093/database/baab062","title":"Classifying domain-specific text documents containing ambiguous keywords","year":2021,"lang":"en","type":"article","venue":"Database","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Eunice Kennedy Shriver National Institute of Child Health and Human Development","keywords":"Computer science; Set (abstract data type); Domain (mathematical analysis); Information retrieval; Artificial intelligence; Naive Bayes classifier; Overfitting; Identifier; Machine learning; Support vector machine; Ambiguity; Data mining; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00884917,0.001097483,0.001711689,0.03569186,0.001295389,0.004597286,0.00111232,0.001494988,0.006779252],"category_scores_gemma":[0.03963737,0.0003144428,0.001604624,0.02888635,0.0005510208,0.002879648,0.001137713,0.0005608382,0.003566945],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001118711,"about_ca_system_score_gemma":0.003639628,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001713207,"about_ca_topic_score_gemma":0.003663522,"domain_scores_codex":[0.9948366,0.001044474,0.00179659,0.0007999432,0.001363765,0.0001587067],"domain_scores_gemma":[0.9314572,0.05016379,0.006612737,0.002803601,0.0082093,0.0007533582],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001570384,0.0004481158,0.03772796,0.05686154,0.001250881,0.00480184,0.002007313,0.002822254,0.05080663,0.007982881,0.08699573,0.7467245],"study_design_scores_gemma":[0.0008329843,0.001552027,0.1028407,0.01341078,0.004321645,0.01338038,0.009624744,0.02251966,0.1147949,0.03825587,0.6779065,0.0005597386],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4128582,0.1167127,0.1547312,0.01113487,0.004636717,0.006734461,0.2357333,0.01308541,0.04437329],"genre_scores_gemma":[0.3090231,0.03294844,0.511683,0.002868721,0.001458604,0.002540709,0.1288488,0.0008441078,0.009784491],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03569186,"threshold_uncertainty_score":0.04679948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02704470169982052,"score_gpt":0.290375614861075,"score_spread":0.2633309131612545,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}