{"id":"W3099059156","doi":"10.1101/414136","title":"Increasing metadata coverage of SRA BioSample entries using deep learning based Named Entity Recognition","year":2018,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institutes of Health; Canadian Institute for Advanced Research","keywords":"Metadata; Computer science; Scalability; Classifier (UML); Information retrieval; Artificial neural network; Named-entity recognition; Artificial intelligence; Annotation; World Wide Web; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00169904,0.0006816049,0.0004999247,0.0019477,0.0004883001,0.0007577719,0.0009239528,0.000662204,0.001183555],"category_scores_gemma":[0.004488037,0.0001950393,0.0007462804,0.001096485,0.0003108451,0.001511081,0.001191918,0.0008581463,0.001508559],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004698685,"about_ca_system_score_gemma":0.0004458582,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005868197,"about_ca_topic_score_gemma":0.0099237,"domain_scores_codex":[0.9992442,0.0001258483,0.00007173814,0.0003355593,0.0001531127,0.00006961944],"domain_scores_gemma":[0.9973903,0.00101543,0.0003440925,0.0004265411,0.0007038008,0.0001198721],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001587532,0.0008083485,0.2074471,0.0008076757,0.0004395914,0.0009923147,0.0008828599,0.07371496,0.1833683,0.001753378,0.0405534,0.4876444],"study_design_scores_gemma":[0.00004625099,0.0003071506,0.05274376,0.0001186486,0.0001599257,0.000244671,0.0005085368,0.8239905,0.1038842,0.003028963,0.01486829,0.00009906119],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8640954,0.001089503,0.08513535,0.0008254865,0.0002440555,0.0001067724,0.02563822,0.01900766,0.003857527],"genre_scores_gemma":[0.7873525,0.0003152868,0.1390184,0.0004751327,0.0000854407,0.0001936154,0.06930444,0.0004821458,0.002773028],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.005868197,"threshold_uncertainty_score":0.01166809,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02594117714831545,"score_gpt":0.2487920017763017,"score_spread":0.2228508246279862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}