{"id":"W2520572539","doi":"10.1101/073460","title":"Semi-Automated Identification of Ontological Labels in the Biomedical Literature with goldi","year":2016,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Ontario; University of Toronto; University of Ottawa; Centre for Addiction and Mental Health","funders":"","keywords":"Computer science; Leverage (statistics); Identification (biology); Parsing; Scope (computer science); Ontology; Data science; Function (biology); Dependency grammar; Artificial intelligence; Information retrieval; Natural language processing; Epistemology; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009043236,0.0003817136,0.0004406621,0.0001511428,0.00005942409,0.00008161295,0.0008525677,0.001152906,0.000006443161],"category_scores_gemma":[0.000495113,0.0002222045,0.0001090583,0.0004442715,0.000574971,0.000005871491,0.0002897124,0.0005213856,0.000006948482],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004065774,"about_ca_system_score_gemma":0.0003031719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000009202071,"about_ca_topic_score_gemma":0.000003538121,"domain_scores_codex":[0.9974714,0.0003067902,0.0005974394,0.0007994531,0.000412573,0.000412372],"domain_scores_gemma":[0.9980044,0.00007212462,0.0004154708,0.00111682,0.0002757674,0.0001153571],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001055874,0.0002603936,0.005061126,0.0002340173,0.0001142994,0.00006604246,0.00002673886,0.000004002117,0.9922047,0.0001695566,0.00173644,0.00001713243],"study_design_scores_gemma":[0.002345747,0.0008941833,0.3279951,0.002074754,0.0001692766,8.607502e-7,0.00003240709,0.0003554466,0.6383369,0.00002101537,0.02645763,0.001316678],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9927664,0.002684474,0.002195233,0.000965123,0.0003518407,0.0004462918,0.0004130599,0.0001673179,0.00001029204],"genre_scores_gemma":[0.9980458,0.000263495,0.001050574,0.0001804407,0.0002687555,0.0001388133,0.000008028146,0.00003655994,0.000007553371],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3538677,"threshold_uncertainty_score":0.9061237,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01025260645618114,"score_gpt":0.2417548716346633,"score_spread":0.2315022651784822,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}