{"id":"W2213189003","doi":"","title":"Chemistry-specific Features and Heuristics for Developing a CRF-based Chemical Named Entity Recogniser","year":2013,"lang":"en","type":"article","venue":"Research Explorer (The University of Manchester)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Open Text (Canada)","funders":"Wellcome Trust","keywords":"Conditional random field; Named-entity recognition; Heuristics; Computer science; Artificial intelligence; Task (project management); Natural language processing; Training set; Set (abstract data type); Sequence labeling; Security token; Search engine indexing; Pattern recognition (psychology); Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002550424,0.001275195,0.0009797319,0.001448593,0.0005129221,0.001039029,0.001906829,0.001408004,0.006983737],"category_scores_gemma":[0.00515211,0.000668299,0.0009027188,0.001183604,0.0005206207,0.002109089,0.0008551898,0.001505135,0.006442994],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007109836,"about_ca_system_score_gemma":0.001493137,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005072605,"about_ca_topic_score_gemma":0.008851959,"domain_scores_codex":[0.9988781,0.0002712548,0.0001390513,0.0003960415,0.0002048661,0.00011071],"domain_scores_gemma":[0.996919,0.002016842,0.000162213,0.0003783222,0.0004348723,0.00008874029],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006450445,0.0004199424,0.003857867,0.001002484,0.0001952742,0.0006238284,0.000306034,0.1090672,0.07879466,0.005942003,0.02069181,0.7784538],"study_design_scores_gemma":[0.0001847569,0.0005052396,0.004058304,0.00009660976,0.0002164023,0.0009143968,0.0002264246,0.8400971,0.1107856,0.008888615,0.03385057,0.0001758798],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02597259,0.000552375,0.9379559,0.0002410061,0.0001317013,0.0005610115,0.003153462,0.02842262,0.003009263],"genre_scores_gemma":[0.1230519,0.0002658223,0.8639789,0.0002161939,0.00005880804,0.0005131043,0.008554676,0.0008327663,0.002527877],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006983737,"threshold_uncertainty_score":0.02336287,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06544131363573949,"score_gpt":0.2888142198452974,"score_spread":0.2233729062095579,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}