{"id":"W4289667473","doi":"10.1101/2022.08.02.502449","title":"Using language models and ontology topology to perform semantic mapping of traits between biomedical datasets","year":2022,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Future Earth","funders":"Medical Research Council; University of Bristol","keywords":"Computer science; Biobank; Ontology; Pairwise comparison; Matching (statistics); Natural language processing; Phenome; Trait; Information retrieval; Semantic similarity; Artificial intelligence; Code (set theory); Bioinformatics; Set (abstract data type); Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007666901,0.001421628,0.0007296358,0.005046733,0.001156688,0.003254823,0.001614378,0.00140295,0.002563491],"category_scores_gemma":[0.0316916,0.0005471589,0.002753462,0.002851943,0.0007948069,0.004640218,0.002755145,0.002156828,0.001481129],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002357782,"about_ca_system_score_gemma":0.002592172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01717879,"about_ca_topic_score_gemma":0.02368402,"domain_scores_codex":[0.9964082,0.001841581,0.0002808441,0.0009087764,0.0003871105,0.0001735019],"domain_scores_gemma":[0.9831092,0.01339207,0.0009168605,0.001202027,0.001044944,0.0003349555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001349979,0.0007730895,0.09132033,0.001646409,0.001683064,0.001207364,0.003250411,0.4306829,0.01095635,0.03644985,0.02750065,0.3931797],"study_design_scores_gemma":[0.00004593983,0.00007486583,0.003453972,0.00009886683,0.00008841142,0.0001506506,0.0004258829,0.9442293,0.002218728,0.04416337,0.005002931,0.00004709334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1969794,0.001833407,0.7575288,0.00380563,0.000411967,0.0004209116,0.01283974,0.02080711,0.00537304],"genre_scores_gemma":[0.5973712,0.0005844627,0.3747217,0.0006611201,0.0001238857,0.0006110764,0.0228052,0.001345383,0.00177596],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01717879,"threshold_uncertainty_score":0.04054695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03905520679226976,"score_gpt":0.2897994130132271,"score_spread":0.2507442062209574,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}