{"id":"W3099059156","doi":"10.1101/414136","title":"Increasing metadata coverage of SRA BioSample entries using deep learning based Named Entity Recognition","year":2018,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institutes of Health; Canadian Institute for Advanced Research","keywords":"Metadata; Computer science; Scalability; Classifier (UML); Information retrieval; Artificial neural network; Named-entity recognition; Artificial intelligence; Annotation; World Wide Web; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001165315,0.0004708487,0.0005998886,0.0002081477,0.0002151673,0.0001661386,0.0005046082,0.0008848998,0.00005326175],"category_scores_gemma":[0.00257725,0.0005015746,0.0002203732,0.0002807931,0.0004829794,0.00001875547,0.0007077565,0.00051355,0.000008399196],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000852766,"about_ca_system_score_gemma":0.000526078,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002330203,"about_ca_topic_score_gemma":0.000009502914,"domain_scores_codex":[0.9971192,0.000484787,0.0005941396,0.0009518016,0.0003550775,0.0004949482],"domain_scores_gemma":[0.997526,0.0001026472,0.0006932321,0.0009260395,0.0005644259,0.0001876292],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001855296,0.0001317122,0.01593962,0.0004167344,0.0002783042,0.00001175026,0.000005905875,0.0000870772,0.9828111,0.000005239179,0.00005218389,0.00007482625],"study_design_scores_gemma":[0.0008262423,0.0002412537,0.01354181,0.0005311603,0.0003049384,1.124697e-7,0.000009331209,0.002140795,0.9717444,0.000004963162,0.009839647,0.0008153166],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9313948,0.00167101,0.06548724,0.00003070199,0.0006152573,0.0002849003,0.0004020096,0.0001094596,0.000004641147],"genre_scores_gemma":[0.9599388,0.0003749788,0.03899354,0.00009754441,0.0004638929,0.00002783365,0.00002430885,0.00007679523,0.000002304409],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02854402,"threshold_uncertainty_score":0.9997436,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02594117714831545,"score_gpt":0.2487920017763017,"score_spread":0.2228508246279862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}