{"id":"W2933094324","doi":"10.2196/12704","title":"Development of a Consumer Health Vocabulary by Mining Health Forum Texts Based on Word Embedding: Semiautomatic Approach","year":2019,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Health Literacy and Information Accessibility","field":"Health Professions","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Word embedding; Computer science; Word (group theory); Vocabulary; Artificial intelligence; Natural language processing; Unified Medical Language System; Space (punctuation); Health informatics; Distributional semantics; Word2vec; Rank (graph theory); Supervised learning; Information retrieval; Embedding; Health care; Semantic similarity; Artificial neural network; Mathematics; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002109169,0.001465994,0.0009960885,0.007926452,0.0006406857,0.00150887,0.00144517,0.001401841,0.00345093],"category_scores_gemma":[0.007331644,0.0005486026,0.001439864,0.003458943,0.0007149357,0.0038481,0.002382552,0.001382854,0.003077767],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007833085,"about_ca_system_score_gemma":0.002292999,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004490307,"about_ca_topic_score_gemma":0.005489457,"domain_scores_codex":[0.9974794,0.0005052889,0.0004145588,0.001037789,0.0004294372,0.0001336275],"domain_scores_gemma":[0.9960064,0.001842972,0.0004503726,0.0003985552,0.001164791,0.0001368818],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000311358,0.0004657731,0.009358676,0.001526831,0.0001583157,0.000421566,0.001465805,0.01028074,0.0584849,0.005175151,0.01181493,0.900536],"study_design_scores_gemma":[0.0001937868,0.0005260513,0.01468228,0.0003024249,0.0002561061,0.001133367,0.003636829,0.8869113,0.04222125,0.0195707,0.03039453,0.0001714786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08646467,0.0008129262,0.8937359,0.0004877061,0.0001403383,0.001780355,0.00633911,0.007956635,0.002282299],"genre_scores_gemma":[0.1567998,0.0002718338,0.8230827,0.00009231635,0.00007494023,0.000938017,0.01658116,0.0003083723,0.001850894],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007926452,"threshold_uncertainty_score":0.01154453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0376482009304372,"score_gpt":0.4284522705373222,"score_spread":0.390804069606885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}