{"id":"W3159409148","doi":"10.1093/database/baab021","title":"Increasing metadata coverage of SRA BioSample entries using deep learning–based named entity recognition","year":2021,"lang":"en","type":"article","venue":"Database","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institutes of Health Research; National Institutes of Health; National Institute of General Medical Sciences; Canadian Institute for Advanced Research","keywords":"Metadata; Computer science; Named-entity recognition; Information retrieval; Artificial intelligence; Deep learning; Natural language processing; World Wide Web","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.001639373,0.0008448887,0.0005758508,0.002017519,0.0005313392,0.0008918552,0.0009472034,0.0006944997,0.001422287],"category_scores_gemma":[0.004771614,0.0002625716,0.0008830635,0.001398568,0.0003372351,0.001873527,0.001326718,0.001010975,0.001957434],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005611786,"about_ca_system_score_gemma":0.0005910556,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006576307,"about_ca_topic_score_gemma":0.01479981,"domain_scores_codex":[0.9992398,0.0001031236,0.00007126324,0.0003509001,0.0001668567,0.00006802417],"domain_scores_gemma":[0.9977678,0.0007982479,0.0002986657,0.0004402101,0.000597556,0.00009746701],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001290433,0.0008528191,0.1779504,0.0009631697,0.0004879589,0.000885198,0.0009414377,0.05530151,0.1799827,0.002470942,0.05355375,0.5253198],"study_design_scores_gemma":[0.00006721522,0.0003641372,0.05826201,0.0001869613,0.0002212935,0.0003728359,0.0006635878,0.7698882,0.1317046,0.005782157,0.03235852,0.0001285242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.7660538,0.001702836,0.1414429,0.0009786191,0.0003066777,0.0001763067,0.05096965,0.03177181,0.00659751],"genre_scores_gemma":[0.6188905,0.0005478928,0.2312455,0.0006409494,0.00009824872,0.0003237704,0.1431564,0.0009233395,0.004173242],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9983606,"threshold_uncertainty_score":0.01307607,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0404997178453738,"score_gpt":0.2904007959088666,"score_spread":0.2499010780634928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}