{"id":"W4404081729","doi":"10.2139/ssrn.5010083","title":"Infusing Clinical Knowledge into Language Models by Subword Optimisation and Embedding Initialisation","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Institute of Population and Public Health","funders":"","keywords":"Embedding; Computer science; Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003613042,0.001563815,0.001104418,0.00144186,0.0005826878,0.002568304,0.00182406,0.002423838,0.01045262],"category_scores_gemma":[0.02049168,0.001257887,0.002200619,0.001052433,0.001202242,0.004606794,0.003611714,0.004588908,0.004717924],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001201876,"about_ca_system_score_gemma":0.001892598,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003809844,"about_ca_topic_score_gemma":0.005177816,"domain_scores_codex":[0.9974868,0.001284617,0.0001956884,0.0005657272,0.0002674662,0.0001996284],"domain_scores_gemma":[0.9839599,0.01303731,0.0003942085,0.001203578,0.001184399,0.00022071],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001557895,0.0003709691,0.00255515,0.0007526111,0.0002765185,0.001010407,0.001548057,0.4590139,0.01122268,0.0504472,0.012942,0.4583026],"study_design_scores_gemma":[0.00004301246,0.00008576398,0.000146337,0.00005765246,0.00007551773,0.0001003201,0.0001203412,0.9041016,0.005093675,0.08687995,0.003265979,0.00002981495],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02573651,0.0003086793,0.96293,0.001299851,0.0002049544,0.0001550835,0.0007840257,0.004912273,0.003668683],"genre_scores_gemma":[0.5875461,0.000358654,0.3979658,0.0004409046,0.0002574733,0.0003391379,0.002774911,0.001877256,0.00843971],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01045262,"threshold_uncertainty_score":0.03496754,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02280427145758566,"score_gpt":0.3596451233423589,"score_spread":0.3368408518847733,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}