{"id":"W2933094324","doi":"10.2196/12704","title":"Development of a Consumer Health Vocabulary by Mining Health Forum Texts Based on Word Embedding: Semiautomatic Approach","year":2019,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Health Literacy and Information Accessibility","field":"Health Professions","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Word embedding; Computer science; Word (group theory); Vocabulary; Artificial intelligence; Natural language processing; Unified Medical Language System; Space (punctuation); Health informatics; Distributional semantics; Word2vec; Rank (graph theory); Supervised learning; Information retrieval; Embedding; Health care; Semantic similarity; Artificial neural network; Mathematics; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.007094003,0.000342327,0.001007028,0.0003159847,0.0007143707,0.00002329482,0.0005176743,0.0004516801,0.001150654],"category_scores_gemma":[0.0004654758,0.0002653749,0.00009616185,0.0005366083,0.0001088731,0.0007419957,0.0001855834,0.001368378,0.0005537001],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006549078,"about_ca_system_score_gemma":0.007749297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002343063,"about_ca_topic_score_gemma":0.000007598985,"domain_scores_codex":[0.9894964,0.0005613212,0.006718429,0.0002194245,0.00177148,0.001233013],"domain_scores_gemma":[0.994036,0.0008219096,0.00307714,0.0007095825,0.0002103884,0.001144974],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002490647,0.001108282,0.0355182,0.04821126,0.00006993156,5.092428e-7,0.2285836,0.00008523381,0.000001633623,0.0005538364,0.3541048,0.3315137],"study_design_scores_gemma":[0.00336407,0.0002849528,0.002325615,0.004965168,0.000003803409,0.000002673887,0.03934254,0.6271699,0.000009566639,0.00001679585,0.3221321,0.0003828132],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8687035,0.0003789392,0.08056661,0.0066676,0.001333325,0.01028567,0.0001903647,0.0006611748,0.03121282],"genre_scores_gemma":[0.5272141,0.00007071605,0.27817,0.1910878,0.000100777,0.0009821235,0.001618546,0.0000750883,0.000680848],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6270847,"threshold_uncertainty_score":0.9999799,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0376482009304372,"score_gpt":0.4284522705373222,"score_spread":0.390804069606885,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}