{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":1034,"total_is_capped":false,"direct_labels_cover":1,"predictions_cover":1034,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"b05057a49128","filters":{"topic":"Machine Learning in Bioinformatics"}},"results":[{"id":"W2144499799","doi":"10.1007/978-3-642-24797-2","title":"Supervised Sequence Labelling with Recurrent Neural Networks","year":2012,"lang":"en","type":"book","venue":"Studies in computational intelligence","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3084,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Labelling; Computer science; Sequence (biology); Artificial intelligence; Artificial neural network; Recurrent neural network; Machine learning; Pattern recognition (psychology); Biology; Genetics","authors":[{"name":"Alex Graves","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06960736389883783,"gpt":0.3540499779592106,"spread":0.2844426140603727,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008710814,0.0008331029,0.001048929,0.0008020209,0.0003735518,0.001000789,0.00205222,0.001281558,0.004902355],"category_scores_gemma":[0.00237199,0.0006664012,0.0008936253,0.001151949,0.0006684978,0.002151995,0.001167726,0.001701416,0.004505007],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006000907,"about_ca_system_score_gemma":0.0006113214,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001369086,"about_ca_topic_score_gemma":0.002485652,"domain_scores_codex":[0.9994009,0.0001472606,0.00004348423,0.0001973621,0.0001695365,0.00004145236],"domain_scores_gemma":[0.9986822,0.00056891,0.0001031432,0.0003597332,0.0002460688,0.00003996734],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001538759,0.00009229568,0.0002453903,0.0003042258,0.000053033,0.00009324346,0.00007535618,0.1062971,0.0174284,0.02326779,0.01661426,0.835375],"study_design_scores_gemma":[0.00001689863,0.0000551972,0.0001372341,0.00004551484,0.00001976559,0.0001010543,0.00001631922,0.9184147,0.01138494,0.06082559,0.008959322,0.00002347544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005266602,0.0007205511,0.9867594,0.000129311,0.0001293276,0.00005594507,0.0003199744,0.003150885,0.003468048],"genre_scores_gemma":[0.07940558,0.0009170712,0.9010434,0.0001927933,0.0001714793,0.0002224721,0.002906393,0.0006681189,0.01447265],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004902355,"threshold_uncertainty_score":0.01640004,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2162792752","doi":"10.1093/bioinformatics/btq249","title":"PSORTb 3.0: improved protein subcellular localization prediction with refined localization subcategories and predictive capabilities for all prokaryotes","year":2010,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2574,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; Simon Fraser University; Michael Smith Health Research BC; Cystic Fibrosis Foundation","keywords":"Proteome; Protein subcellular localization prediction; Archaea; Computer science; Subcellular localization; Precision and recall; Computational biology; Interface (matter); Proteomics; Software; Metagenomics; Biology; Bioinformatics; Data mining; Artificial intelligence; Bacteria; Cytoplasm; Genetics; Gene; Programming language","authors":[{"name":"Nancy Yu","is_ca":true},{"name":"James Wagner","is_ca":true},{"name":"Matthew R. Laird","is_ca":true},{"name":"Gabor Melli","is_ca":true},{"name":"Sébastien Rey","is_ca":true},{"name":"Raymond Lo","is_ca":true},{"name":"Phuong Dao","is_ca":true},{"name":"S. Cenk Şahinalp","is_ca":true},{"name":"Martin Ester","is_ca":true},{"name":"Leonard J. Foster","is_ca":true},{"name":"Fiona S. L. Brinkman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.004537457218943714,"gpt":0.2113114192828754,"spread":0.2067739620639316,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00230192,0.002018392,0.001519675,0.001095361,0.000555147,0.001442636,0.00143963,0.0006951142,0.004834733],"category_scores_gemma":[0.005074054,0.000781537,0.00153168,0.001200974,0.0002976581,0.001218013,0.001456018,0.001624487,0.004395199],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005243601,"about_ca_system_score_gemma":0.001009831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003062548,"about_ca_topic_score_gemma":0.004309738,"domain_scores_codex":[0.9991978,0.000168045,0.00007287906,0.0002385616,0.0002548622,0.00006780422],"domain_scores_gemma":[0.9984868,0.000557114,0.0001998914,0.0002085026,0.0004101473,0.000137578],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007678094,0.0007391932,0.0608428,0.00285007,0.001412427,0.001081672,0.0004568542,0.1062439,0.3123614,0.00339053,0.169241,0.3337022],"study_design_scores_gemma":[0.0005461889,0.0004893991,0.01969438,0.0001385185,0.0004012526,0.0009964992,0.00007256969,0.8062905,0.1322566,0.003538151,0.0353558,0.0002200911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2500788,0.001873041,0.5257825,0.001338273,0.0002125625,0.0003313535,0.03199696,0.1845153,0.003871135],"genre_scores_gemma":[0.2771847,0.0006646733,0.6211372,0.0004525877,0.00008731647,0.0005964051,0.08416393,0.0111011,0.004612062],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004834733,"threshold_uncertainty_score":0.01617384,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2117486996","doi":"10.1038/nmeth.2340","title":"A large-scale evaluation of computational protein function prediction","year":2013,"lang":"en","type":"article","venue":"Nature Methods","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1095,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"U.S. National Library of Medicine; National Institute of General Medical Sciences; National Human Genome Research Institute; Office of Science; Biotechnology and Biological Sciences Research Council; Natural Sciences and Engineering Research Council of Canada; National Institutes of Health; Directorate for Biological Sciences; Ministero dell’Istruzione, dell’Università e della Ricerca; Biological and Environmental Research; National Natural Science Foundation of China; University of Miami; Alexander von Humboldt-Stiftung; U.S. Department of Energy; National Science Foundation","keywords":"Annotation; Computer science; Protein function prediction; Function (biology); Protein function; Set (abstract data type); Scale (ratio); Computational biology; Machine learning; Artificial intelligence; Data mining; Biology; Genetics; Gene","authors":[{"name":"Predrag Radivojac","is_ca":false},{"name":"Wyatt T. Clark","is_ca":false},{"name":"Tal Oron","is_ca":false},{"name":"Alexandra M. Schnoes","is_ca":false},{"name":"Tobias Wittkop","is_ca":false},{"name":"Artem Sokolov","is_ca":false},{"name":"Kiley Graim","is_ca":false},{"name":"Christopher S. Funk","is_ca":false},{"name":"Karin Verspoor","is_ca":false},{"name":"Asa Ben‐Hur","is_ca":false},{"name":"Gaurav Pandey","is_ca":false},{"name":"Jeffrey M. Yunes","is_ca":false},{"name":"Ameet Talwalkar","is_ca":false},{"name":"Susanna Repo","is_ca":false},{"name":"Michael L Souza","is_ca":false},{"name":"Damiano Piovesan","is_ca":false},{"name":"Rita Casadio","is_ca":false},{"name":"Zheng Wang","is_ca":false},{"name":"Jianlin Cheng","is_ca":false},{"name":"Hai Fang","is_ca":false},{"name":"Julian Gough","is_ca":false},{"name":"Patrik Koskinen","is_ca":false},{"name":"Petri Törönen","is_ca":false},{"name":"Jussi Nokso-Koivisto","is_ca":false},{"name":"Liisa Holm","is_ca":false},{"name":"Domenico Cozzetto","is_ca":false},{"name":"Daniel Buchan","is_ca":false},{"name":"Kevin Bryson","is_ca":false},{"name":"David T. Jones","is_ca":false},{"name":"Bhakti Limaye","is_ca":false},{"name":"Harshal Inamdar","is_ca":false},{"name":"Avik Datta","is_ca":false},{"name":"Sunitha K Manjari","is_ca":false},{"name":"Rajendra Joshi","is_ca":false},{"name":"Meghana Chitale","is_ca":false},{"name":"Daisuke Kihara","is_ca":false},{"name":"Andreas Martin Lisewski","is_ca":false},{"name":"Serkan Erdin","is_ca":false},{"name":"Eric Venner","is_ca":false},{"name":"Olivier Lichtarge","is_ca":false},{"name":"Robert Rentzsch","is_ca":false},{"name":"Haixuan Yang","is_ca":false},{"name":"Alfonso E. Romero","is_ca":false},{"name":"Prajwal Bhat","is_ca":false},{"name":"Alberto Paccanaro","is_ca":false},{"name":"Tobias Hamp","is_ca":false},{"name":"Rebecca Kaßner","is_ca":false},{"name":"Stefan Seemayer","is_ca":false},{"name":"Esmeralda Vicedo","is_ca":false},{"name":"Christian Schaefer","is_ca":false},{"name":"Dominik Achten","is_ca":false},{"name":"Florian Auer","is_ca":false},{"name":"Ariane C. Boehm","is_ca":false},{"name":"Tatjana Braun","is_ca":false},{"name":"Maximilian Hecht","is_ca":false},{"name":"B. Mark Heron","is_ca":false},{"name":"Peter Hönigschmid","is_ca":false},{"name":"Thomas A. Hopf","is_ca":false},{"name":"Stefanie Kaufmann","is_ca":false},{"name":"Michael Kiening","is_ca":false},{"name":"Denis Krompaß","is_ca":false},{"name":"Cedric Landerer","is_ca":false},{"name":"Yannick Mahlich","is_ca":false},{"name":"Manfred Roos","is_ca":false},{"name":"Jari Björne","is_ca":false},{"name":"Tapio Salakoski","is_ca":false},{"name":"Andrew Wong","is_ca":true},{"name":"Hagit Shatkay","is_ca":true},{"name":"Fanny Gatzmann","is_ca":false},{"name":"I. Sommer","is_ca":false},{"name":"Mark N. Wass","is_ca":false},{"name":"Michael J.E. Sternberg","is_ca":false},{"name":"Nives Škunca","is_ca":false},{"name":"Fran Supek","is_ca":false},{"name":"Matko Bošnjak","is_ca":false},{"name":"Panče Panov","is_ca":false},{"name":"Sašo Džeroski","is_ca":false},{"name":"Tomislav Šmuc","is_ca":false},{"name":"Yiannis Kourmpetis","is_ca":false},{"name":"Aalt D. J. van Dijk","is_ca":false},{"name":"Cajo J. F. ter Braak","is_ca":false},{"name":"Yuanpeng Zhou","is_ca":false},{"name":"Qingtian Gong","is_ca":false},{"name":"Xinran Dong","is_ca":false},{"name":"Weidong Tian","is_ca":false},{"name":"Marco Falda","is_ca":false},{"name":"Paolo Fontana","is_ca":false},{"name":"Enrico Lavezzo","is_ca":false},{"name":"Barbara Di Camillo","is_ca":false},{"name":"Stefano Toppo","is_ca":false},{"name":"Liang Lan","is_ca":false},{"name":"Nemanja Djuric","is_ca":false},{"name":"Yuhong Guo","is_ca":false},{"name":"Slobodan Vučetić","is_ca":false},{"name":"Amos Bairoch","is_ca":false},{"name":"Michal Linial","is_ca":false},{"name":"Patricia C. Babbitt","is_ca":false},{"name":"Steven E. Brenner","is_ca":false},{"name":"Christine Orengo","is_ca":false},{"name":"Burkhard Rost","is_ca":false},{"name":"Sean D. Mooney","is_ca":false},{"name":"Iddo Friedberg","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01041951972596435,"gpt":0.3501743383184968,"spread":0.3397548185925325,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01944185,0.002887662,0.001613613,0.003430276,0.001663516,0.001990213,0.003676044,0.00207979,0.002066692],"category_scores_gemma":[0.0354337,0.0005061392,0.001365416,0.003975017,0.001339996,0.003071866,0.00230904,0.002273912,0.001352797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002134306,"about_ca_system_score_gemma":0.002427778,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008282436,"about_ca_topic_score_gemma":0.008398658,"domain_scores_codex":[0.9882948,0.005112411,0.0005684212,0.001801398,0.003766594,0.0004562286],"domain_scores_gemma":[0.9639375,0.02443147,0.0008233972,0.003861931,0.006175441,0.0007702428],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002936429,0.003014029,0.04868483,0.003528696,0.002474797,0.0004160765,0.0004247687,0.3673985,0.03027672,0.006846879,0.0455024,0.4884959],"study_design_scores_gemma":[0.0003317057,0.00143736,0.02010041,0.0001636932,0.0002126156,0.0002745228,0.0003243066,0.9346015,0.02531586,0.00514597,0.0119803,0.0001118768],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7601141,0.01824133,0.1767736,0.002355613,0.0007763591,0.001136948,0.01100346,0.01527446,0.01432418],"genre_scores_gemma":[0.7077769,0.003286543,0.2592014,0.0004707933,0.0001400301,0.0005741199,0.02569946,0.00102121,0.001829464],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9805582,"threshold_uncertainty_score":0.1028196,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3019119825","doi":"10.1371/journal.pone.0232391","title":"Machine learning using intrinsic genomic signatures for rapid classification of novel pathogens: COVID-19 case study","year":2020,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1034,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Genome; Computational biology; Genomics; Biology; Novel virus; Virus classification; Decision tree; Virus; Artificial intelligence; Computer science; Genetics; Gene","authors":[{"name":"Gurjit S. Randhawa","is_ca":true},{"name":"Maximillian P. M. Soltysiak","is_ca":true},{"name":"Hadi El Roz","is_ca":true},{"name":"Camila P. E. de Souza","is_ca":true},{"name":"Kathleen A. Hill","is_ca":true},{"name":"Lila Kari","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1024710124406709,"gpt":0.3021214621668543,"spread":0.1996504497261834,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001222297,0.0008158898,0.000657324,0.001632908,0.0006492207,0.001155731,0.0009114246,0.001401242,0.0006115481],"category_scores_gemma":[0.002940891,0.0001674241,0.0007077128,0.001261817,0.0005268591,0.000772387,0.0008673299,0.0009075536,0.0004290179],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005613634,"about_ca_system_score_gemma":0.0004914395,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003865005,"about_ca_topic_score_gemma":0.004251098,"domain_scores_codex":[0.9991365,0.0001790028,0.00009070349,0.0001989109,0.0002832829,0.0001116006],"domain_scores_gemma":[0.9987347,0.0005229437,0.0001191768,0.0001569987,0.0003255779,0.0001406519],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001578786,0.002298394,0.2356919,0.001011983,0.0003333416,0.01547209,0.001485936,0.1288087,0.07562566,0.004836201,0.02411467,0.5087424],"study_design_scores_gemma":[0.00007946054,0.0006224133,0.03545123,0.00005607979,0.00008988744,0.005023781,0.001262791,0.9058139,0.03789047,0.003718215,0.009915962,0.00007579957],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9109117,0.001783482,0.0805119,0.001202801,0.0001810237,0.0001914309,0.001604254,0.0006818204,0.002931461],"genre_scores_gemma":[0.8857659,0.0006143135,0.1073501,0.000174508,0.0001494779,0.0000756821,0.003903673,0.0000573662,0.001908852],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003865005,"threshold_uncertainty_score":0.007685006,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2154202919","doi":"10.1093/bioinformatics/bti057","title":"PSORTb v.2.0: Expanded prediction of bacterial protein subcellular localization and insights gained from comparative proteome analysis","year":2004,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":723,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"Simon Fraser University; Michael Smith Health Research BC","keywords":"Proteome; Computer science; n-gram; Subsequence; Support vector machine; Computational biology; MIT License; Subcellular localization; Biology; Artificial intelligence; Software; Bioinformatics; Genetics; Mathematics; Programming language; Language model","authors":[{"name":"Jennifer L. Gardy","is_ca":true},{"name":"Matthew R. Laird","is_ca":false},{"name":"Fen‐Ling Chen","is_ca":false},{"name":"Sébastien Rey","is_ca":false},{"name":"Calum J. Walsh","is_ca":false},{"name":"Martin Ester","is_ca":false},{"name":"Fiona S. L. Brinkman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01112130957213176,"gpt":0.2311252125935061,"spread":0.2200039030213743,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001166368,0.0009815398,0.0009439721,0.001478544,0.0002601959,0.000754475,0.0009998893,0.0004714882,0.005727987],"category_scores_gemma":[0.002935206,0.000483201,0.0006899817,0.001200774,0.000158103,0.0007130182,0.0007933745,0.0006167167,0.003732609],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002498478,"about_ca_system_score_gemma":0.0003918521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000865289,"about_ca_topic_score_gemma":0.0009131095,"domain_scores_codex":[0.9997168,0.00006561656,0.000025593,0.00007838855,0.00009099415,0.0000225043],"domain_scores_gemma":[0.9993617,0.0003039629,0.00008923634,0.00008446423,0.000113477,0.00004710117],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004074526,0.0003218019,0.01593729,0.003199235,0.0004598754,0.001066672,0.0003615284,0.04210499,0.3341269,0.005094295,0.151509,0.4417439],"study_design_scores_gemma":[0.0005745821,0.0006620919,0.03235639,0.0003989735,0.0003264474,0.003753781,0.00008648938,0.6430902,0.1781543,0.009226647,0.1311671,0.0002029159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1380973,0.002320569,0.6610056,0.000498088,0.0001391324,0.0002469906,0.03089832,0.1616791,0.005114842],"genre_scores_gemma":[0.1903609,0.001248868,0.7235301,0.0001860744,0.00006903095,0.0005025039,0.07147302,0.009394811,0.003234725],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005727987,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2160072419","doi":"10.1093/nar/gkg602","title":"PSORT-B: improving protein subcellular localization prediction for Gram-negative bacteria","year":2003,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":426,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Biology; Computational biology; Subcellular localization; Source code; Probabilistic logic; Protein subcellular localization prediction; Bacterial genome size; Protein sequencing; Genetics; Peptide sequence; Genome; Artificial intelligence; Computer science; Gene","authors":[{"name":"Jennifer L. Gardy","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02278033779832832,"gpt":0.3099796708341906,"spread":0.2871993330358623,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001415889,0.001969483,0.001144828,0.001609081,0.0006377532,0.000771838,0.001213907,0.001003221,0.001330033],"category_scores_gemma":[0.003792151,0.0005586951,0.0009598192,0.001693527,0.0002986971,0.001019655,0.001084205,0.001086366,0.002029329],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005036245,"about_ca_system_score_gemma":0.0009206881,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004613827,"about_ca_topic_score_gemma":0.005875292,"domain_scores_codex":[0.9992691,0.0001859598,0.00007164659,0.000200913,0.0002207456,0.00005171951],"domain_scores_gemma":[0.9992604,0.0002406868,0.0001376471,0.0001198918,0.0001684711,0.00007292666],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003739993,0.0007373653,0.02449633,0.001916074,0.0006418596,0.000946367,0.0002257053,0.147625,0.1242344,0.003220744,0.1369344,0.5552818],"study_design_scores_gemma":[0.0007015486,0.0005986162,0.01325341,0.00009375566,0.0001866283,0.0009486906,0.00007481257,0.89413,0.04450008,0.006692874,0.03870699,0.0001125775],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.433747,0.009096141,0.4310856,0.001423537,0.000394842,0.0005371674,0.02396289,0.09592244,0.003830465],"genre_scores_gemma":[0.245691,0.002467189,0.6943591,0.0002698198,0.00008151517,0.0003884977,0.05212468,0.001951625,0.002666687],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.004613827,"threshold_uncertainty_score":0.00917393,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2151790371","doi":"10.1093/bioinformatics/btg447","title":"Predicting subcellular localization of proteins using machine-learned classifiers","year":2004,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":334,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Subcellular localization; Computer science; Artificial intelligence; Protein subcellular localization prediction; Machine learning; Computational biology; Pattern recognition (psychology); Chemistry; Biology; Biochemistry; Cytoplasm; Gene","authors":[{"name":"Zhonghua Lu","is_ca":true},{"name":"Duane Szafron","is_ca":false},{"name":"Russell Greiner","is_ca":false},{"name":"P. Lu","is_ca":false},{"name":"David S. Wishart","is_ca":false},{"name":"Brett Poulin","is_ca":false},{"name":"John Anvik","is_ca":false},{"name":"Cam Macdonell","is_ca":false},{"name":"Roman Eisner","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01636032850905748,"gpt":0.2565539395148381,"spread":0.2401936110057807,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001605731,0.0009999998,0.0007692644,0.001827884,0.0005086212,0.001224818,0.0008635198,0.001301174,0.001374691],"category_scores_gemma":[0.005892015,0.0002012723,0.0005536355,0.001272891,0.0003884324,0.001383709,0.0003916012,0.001137074,0.001704004],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000727628,"about_ca_system_score_gemma":0.0009743246,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003008402,"about_ca_topic_score_gemma":0.002507197,"domain_scores_codex":[0.9992713,0.0001848781,0.00007172086,0.0001886365,0.000188052,0.00009529493],"domain_scores_gemma":[0.9945561,0.003391995,0.0004762525,0.0002372805,0.001192959,0.0001454284],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006753776,0.0007225715,0.1302864,0.0007647161,0.0002735602,0.0006887997,0.0001330424,0.3642166,0.02794119,0.002958906,0.0189751,0.4523638],"study_design_scores_gemma":[0.00003002292,0.00008338333,0.006273776,0.00005533452,0.00007263832,0.0001989223,0.00004289263,0.9678016,0.01913107,0.004038227,0.002253691,0.00001845828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5351475,0.003599474,0.44152,0.001714312,0.0002897501,0.0001707505,0.005740761,0.006227084,0.005590328],"genre_scores_gemma":[0.8100172,0.0009264064,0.1773298,0.0003020031,0.0001677151,0.0001253299,0.00926194,0.000101916,0.00176776],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003008402,"threshold_uncertainty_score":0.008492053,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2788724289","doi":"10.1073/pnas.1800256115","title":"Classification and interaction in random forests","year":2018,"lang":"en","type":"letter","venue":"Proceedings of the National Academy of Sciences","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":276,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada","keywords":"Random forest; Ask price; Computer science; Artificial intelligence; Machine learning; Classifier (UML); Medical diagnosis; Decision tree; Complaint; Data science; Medicine","authors":[{"name":"Danielle Denisko","is_ca":true},{"name":"Michael M. Hoffman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03080431165518348,"gpt":0.3208278370013131,"spread":0.2900235253461296,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009815196,0.0006233217,0.001385353,0.001048159,0.001132351,0.003109765,0.002250072,0.004795875,0.004654244],"category_scores_gemma":[0.05283957,0.0005976947,0.0009614452,0.00140446,0.002223762,0.004058985,0.001590256,0.00727936,0.004016242],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001991543,"about_ca_system_score_gemma":0.0008272901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002069901,"about_ca_topic_score_gemma":0.003436722,"domain_scores_codex":[0.9931937,0.004315357,0.0002582523,0.0009068516,0.001042552,0.0002834406],"domain_scores_gemma":[0.9719039,0.02329891,0.0007645729,0.001926525,0.001617158,0.0004889516],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006947138,0.0001677145,0.008192698,0.0003512961,0.0003603189,0.0006595941,0.0002229275,0.03966692,0.001014993,0.2618042,0.4105441,0.2763205],"study_design_scores_gemma":[0.0001847791,0.00005324099,0.0009551004,0.00007262261,0.00004304348,0.0003663148,0.00002939442,0.3634702,0.000853724,0.5947036,0.03922741,0.00004054841],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03056354,0.01110965,0.5582104,0.3569738,0.01638886,0.0001813702,0.001809765,0.002790133,0.02197241],"genre_scores_gemma":[0.6770667,0.005639602,0.1877737,0.06893822,0.03438232,0.0008952984,0.002629307,0.0007992788,0.02187558],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009815196,"threshold_uncertainty_score":0.05190837,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2045510682","doi":"10.1101/gr.147901","title":"Evaluation of Gene-Finding Programs on Mammalian Sequences","year":2001,"lang":"en","type":"article","venue":"Genome Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":272,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Pacific Centre for Reproductive Medicine; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Biology; Exon; Sequence (biology); Gene prediction; Computational biology; Gene; Sequence analysis; Genetics; Strengths and weaknesses; Function (biology); Genome","authors":[{"name":"Sanja Rogić","is_ca":true},{"name":"Alan K. Mackworth","is_ca":true},{"name":"B. F. Francis Ouellette","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1565077522533462,"gpt":0.4316221012825429,"spread":0.2751143490291966,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004540459,0.001369302,0.0008863034,0.002393091,0.0009225181,0.0009462368,0.002077331,0.0008875182,0.001781702],"category_scores_gemma":[0.008240609,0.0003506723,0.0007852648,0.002745455,0.0005153156,0.001119498,0.001066015,0.0009015515,0.0008214493],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001277996,"about_ca_system_score_gemma":0.001220258,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005789371,"about_ca_topic_score_gemma":0.007920252,"domain_scores_codex":[0.9971316,0.0008875215,0.0002246714,0.0007758513,0.0007821936,0.0001982108],"domain_scores_gemma":[0.9935819,0.004269875,0.0002442189,0.0004994373,0.001202358,0.0002022805],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005885565,0.002242717,0.04583666,0.002820028,0.001202963,0.0005903562,0.0005742642,0.1503692,0.09021167,0.005854615,0.04156478,0.6528472],"study_design_scores_gemma":[0.0005871656,0.001537142,0.03392073,0.0001387909,0.0003696797,0.0006658809,0.0003266939,0.760402,0.1631009,0.002561488,0.0362721,0.0001173716],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8908829,0.002793774,0.06156519,0.0003410432,0.000104745,0.0003430212,0.007765447,0.02897933,0.007224503],"genre_scores_gemma":[0.6493326,0.001457192,0.2810778,0.0003475337,0.00005115973,0.000590078,0.0597634,0.003280389,0.004099904],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005789371,"threshold_uncertainty_score":0.02401251,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2160115185","doi":"10.1371/journal.pgen.1002741","title":"Stratifying Type 2 Diabetes Cases by BMI Identifies Genetic Risk Variants in LAMA1 and Enrichment for Risk Variants in Lean Compared to Obese Cases","year":2012,"lang":"en","type":"article","venue":"PLoS Genetics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":267,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"National Cancer Institute; National Institute on Alcohol Abuse and Alcoholism; Medical Research Council; National Center for Research Resources; National Institute of Dental and Craniofacial Research; Novartis Pharma; National Institutes of Health; Fondation de France; National Heart, Lung, and Blood Institute; Vetenskapsrådet; Sanofi; European Commission; National Human Genome Research Institute; Wellcome Trust; Agence Nationale de la Recherche; Les Laboratories Pierre Fabre; Institut National de la Santé et de la Recherche Médicale; School of Medicine, Boston University; National Institute of Diabetes and Digestive and Kidney Diseases; Lunds Universitet; National Institute on Drug Abuse; Eli Lilly and Company","keywords":"Type 2 diabetes; Overweight; Obesity; Body mass index; Diabetes mellitus; Single-nucleotide polymorphism; Internal medicine; Genome-wide association study; Biology; Risk factor; Case-control study; Genetic association; Endocrinology; Medicine; Bioinformatics; Genetics; Genotype; Gene","authors":[{"name":"John R. B. Perry","is_ca":false},{"name":"Benjamin F. Voight","is_ca":false},{"name":"Loïc Yengo","is_ca":false},{"name":"Najaf Amin","is_ca":false},{"name":"Josée Dupuis","is_ca":false},{"name":"Martha Ganser","is_ca":false},{"name":"Harald Grallert","is_ca":false},{"name":"Pau Navarro","is_ca":false},{"name":"Man Li","is_ca":false},{"name":"Lu Qi","is_ca":false},{"name":"Valgerður Steinthórsdóttir","is_ca":false},{"name":"Robert A. Scott","is_ca":false},{"name":"Peter Almgren","is_ca":false},{"name":"Dan E. Arking","is_ca":false},{"name":"Yurii S. Aulchenko","is_ca":false},{"name":"Beverley Balkau","is_ca":false},{"name":"Rafn Benediktsson","is_ca":false},{"name":"Richard N. Bergman","is_ca":false},{"name":"Eric Boerwinkle","is_ca":false},{"name":"Lori L. Bonnycastle","is_ca":false},{"name":"Noël P. Burtt","is_ca":false},{"name":"Harry Campbell","is_ca":false},{"name":"G. Charpentier","is_ca":false},{"name":"Francis S. Collins","is_ca":false},{"name":"Christian Gieger","is_ca":false},{"name":"Todd J. Green","is_ca":false},{"name":"Samy Hadjadj","is_ca":false},{"name":"Andrew T. Hattersley","is_ca":false},{"name":"Christian Herder","is_ca":false},{"name":"Albert Hofman","is_ca":false},{"name":"Andrew D. Johnson","is_ca":false},{"name":"Anna Köttgen","is_ca":false},{"name":"Peter Kraft","is_ca":false},{"name":"Yann Labrune","is_ca":false},{"name":"Claudia Langenberg","is_ca":false},{"name":"Alisa K. Manning","is_ca":false},{"name":"Karen L. Mohlke","is_ca":false},{"name":"Andrew P. Morris","is_ca":false},{"name":"Ben A. Oostra","is_ca":false},{"name":"James S. Pankow","is_ca":false},{"name":"Ann-Kristin Petersen","is_ca":false},{"name":"Peter P. Pramstaller","is_ca":false},{"name":"Inga Prokopenko","is_ca":false},{"name":"Wolfgang Rathmann","is_ca":false},{"name":"W Rayner","is_ca":false},{"name":"Michael Roden","is_ca":false},{"name":"Igor Rudan","is_ca":false},{"name":"Denis Rybin","is_ca":false},{"name":"Laura J. Scott","is_ca":false},{"name":"Gunnar Sigurðsson","is_ca":false},{"name":"Robert Sladek","is_ca":true},{"name":"Guðmar Þorleifsson","is_ca":false},{"name":"Unnur Þorsteinsdóttir","is_ca":false},{"name":"Jaakko Tuomilehto","is_ca":false},{"name":"André G. Uitterlinden","is_ca":false},{"name":"Sidonie Vivequin","is_ca":false},{"name":"Michael N. Weedon","is_ca":false},{"name":"Alan F. Wright","is_ca":false},{"name":"Frank B. Hu","is_ca":false},{"name":"Thomas Illig","is_ca":false},{"name":"Linda Kao","is_ca":false},{"name":"James B. Meigs","is_ca":false},{"name":"James F. Wilson","is_ca":false},{"name":"Kāri Stefánsson","is_ca":false},{"name":"Cornelia M. van Duijn","is_ca":false},{"name":"David M. Altschuler","is_ca":false},{"name":"Andrew D. Morris","is_ca":false},{"name":"Michael Boehnke","is_ca":false},{"name":"Mark I. McCarthy","is_ca":false},{"name":"Philippe Froguel","is_ca":false},{"name":"Nicholas J. Wareham","is_ca":false},{"name":"Leif Groop","is_ca":false},{"name":"Timothy M. Frayling","is_ca":false},{"name":"Stéphane Cauchi","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02014445139635725,"gpt":0.2782651669371498,"spread":0.2581207155407926,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002302371,0.000520307,0.000663329,0.00150408,0.0006290699,0.001111318,0.0005493535,0.0007126405,0.001628688],"category_scores_gemma":[0.007071442,0.0004456148,0.001332604,0.001172374,0.0004142656,0.000282602,0.0006033462,0.0006955353,0.0002525315],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002310423,"about_ca_system_score_gemma":0.0002541559,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004280476,"about_ca_topic_score_gemma":0.005649316,"domain_scores_codex":[0.998082,0.0006843306,0.0001814868,0.0006226535,0.000223139,0.0002064851],"domain_scores_gemma":[0.9972246,0.001218256,0.0006832082,0.0004467375,0.0001770322,0.0002502832],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001026031,0.00004677651,0.9889423,0.00001965596,0.0005019075,0.0002464449,0.0001716402,0.0001201313,0.004964767,0.00004587085,0.0001095981,0.003804835],"study_design_scores_gemma":[0.00008059798,0.0001691283,0.996834,0.00001261558,0.0004406655,0.0005498042,0.0001775698,0.0007419059,0.0005959501,0.0001617868,0.0002235142,0.00001245582],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9988297,0.0003028959,0.0004581858,0.00004879684,0.000008604221,0.000008783686,0.0001221959,0.00001304481,0.0002079495],"genre_scores_gemma":[0.9990283,0.00007800246,0.0004951402,0.00003708533,0.00001117561,0.00001078965,0.0002084573,0.000007356315,0.0001236272],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004280476,"threshold_uncertainty_score":0.01217622,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3088578860","doi":"10.1016/j.cels.2020.08.016","title":"Fast and Flexible Protein Design Using Deep Graph Neural Networks","year":2020,"lang":"en","type":"article","venue":"Cell Systems","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":230,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds de recherche du Québec – Nature et technologies; Canadian Institutes of Health Research; Nvidia","keywords":"Computer science; In silico; Benchmark (surveying); Protein design; Graph; Artificial neural network; Constraint (computer-aided design); Algorithm; Protein structure prediction; Protein sequencing; Sequence (biology); Protein structure; Theoretical computer science; Artificial intelligence; Peptide sequence; Mathematics; Biology; Gene; Genetics","authors":[{"name":"Alexey Strokach","is_ca":true},{"name":"David Becerra","is_ca":true},{"name":"Carles Corbi‐Verge","is_ca":true},{"name":"Albert Perez‐Riba","is_ca":true},{"name":"Philip M. Kim","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01847110912025982,"gpt":0.2246390681826954,"spread":0.2061679590624356,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005454061,0.001192401,0.0008363337,0.0006189361,0.0004623666,0.0007407318,0.001187515,0.001165843,0.00400602],"category_scores_gemma":[0.001318302,0.0006637571,0.0009804631,0.000764244,0.0006440884,0.0009587822,0.0008988193,0.001343461,0.001125712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001497812,"about_ca_system_score_gemma":0.001243767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0047717,"about_ca_topic_score_gemma":0.01195139,"domain_scores_codex":[0.9996774,0.00007381273,0.00001388295,0.00009140246,0.0001019399,0.00004169071],"domain_scores_gemma":[0.9995373,0.0001992114,0.00005343039,0.00009170957,0.0000772174,0.00004120355],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001374914,0.0001119664,0.001133613,0.0002899445,0.00008389594,0.00013387,0.00005446695,0.8024563,0.01659422,0.02030407,0.008439207,0.1502609],"study_design_scores_gemma":[0.00002054463,0.00002763067,0.00006619366,0.000007500949,0.000005699829,0.00001393056,0.00000784422,0.9871193,0.001912711,0.009186118,0.001627815,0.000004694113],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09240169,0.001064184,0.8883299,0.0005920953,0.000127318,0.0001480053,0.0007558627,0.008006416,0.008574368],"genre_scores_gemma":[0.4268332,0.0005721434,0.5624157,0.0004446393,0.00003997991,0.000298473,0.00217558,0.0009570174,0.006263191],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0047717,"threshold_uncertainty_score":0.01340151,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2790921906","doi":"10.1093/bioinformatics/bty095","title":"Improved genomic island predictions with IslandPath-DIMOB","year":2018,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":223,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Computer science; Computational biology; Biology","authors":[{"name":"Claire Bertelli","is_ca":true},{"name":"Fiona S. L. Brinkman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.004793943520078781,"gpt":0.2206221854113326,"spread":0.2158282418912538,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00161759,0.002191327,0.001203005,0.004528672,0.0005193282,0.001606032,0.001419008,0.0009765616,0.004662714],"category_scores_gemma":[0.005876365,0.0006070168,0.001407677,0.002161152,0.0002041274,0.00161111,0.001603872,0.0008956212,0.003131031],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003756477,"about_ca_system_score_gemma":0.0008682085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003907816,"about_ca_topic_score_gemma":0.005947863,"domain_scores_codex":[0.9991863,0.0001586885,0.00005946111,0.0002816817,0.0002485176,0.00006539955],"domain_scores_gemma":[0.9979613,0.00116315,0.0001570555,0.0002193065,0.0003655179,0.0001337142],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002716044,0.0005581274,0.06714083,0.005521993,0.001218732,0.00094442,0.0005198577,0.06550548,0.06034166,0.003406497,0.1967572,0.5953692],"study_design_scores_gemma":[0.0003167163,0.000391138,0.01896826,0.0004444824,0.0003463456,0.0007860551,0.000262219,0.8308762,0.05137121,0.007369102,0.0886218,0.0002464635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2019352,0.006115608,0.4204929,0.0008787574,0.0006068297,0.0004453676,0.08051275,0.2801293,0.008883328],"genre_scores_gemma":[0.2349492,0.001264434,0.5420482,0.0004539798,0.0001628574,0.0003620368,0.209775,0.008212293,0.002772061],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004662714,"threshold_uncertainty_score":0.01559836,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2112905569","doi":"10.1093/nar/gkh380","title":"ConPred II: a consensus prediction method for obtaining transmembrane topology models with high reliability","year":2004,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":202,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Institute of Aging","funders":"Ministry of Education, Culture, Sports, Science and Technology","keywords":"Topology (electrical circuits); Network topology; Biology; Computer science; Reliability (semiconductor); Process (computing); Algorithm; Data mining; Mathematics; Physics; Computer network; Combinatorics; Power (physics); Operating system","authors":[{"name":"M. Arai","is_ca":false},{"name":"Hironori Mitsuke","is_ca":false},{"name":"Masami Ikeda","is_ca":false},{"name":"Jun-Xiong Xia","is_ca":false},{"name":"Toru Kikuchi","is_ca":false},{"name":"Masanobu Satake","is_ca":true},{"name":"Toshio Shimizu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03156165825524426,"gpt":0.349017576479136,"spread":0.3174559182238917,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00351841,0.003135092,0.002538212,0.004187013,0.001629104,0.001642624,0.0031803,0.001789312,0.0143748],"category_scores_gemma":[0.007461053,0.001563171,0.002092101,0.002794965,0.0005035167,0.002240285,0.001873608,0.002539227,0.01254626],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007341222,"about_ca_system_score_gemma":0.002077437,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001717191,"about_ca_topic_score_gemma":0.002563073,"domain_scores_codex":[0.9978428,0.0005292494,0.0001975655,0.0007001153,0.0005789974,0.0001512812],"domain_scores_gemma":[0.9976268,0.0007933179,0.0003282154,0.0003222618,0.0007748314,0.0001545482],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002479724,0.0005247114,0.01130279,0.00581944,0.001622002,0.002077817,0.0007532082,0.0505636,0.145596,0.01082719,0.3241031,0.4443304],"study_design_scores_gemma":[0.0004572737,0.0003095714,0.005466561,0.0003489137,0.0004230171,0.002041306,0.000270936,0.8037885,0.07735112,0.01995057,0.08924203,0.0003502085],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02230956,0.00160963,0.8973244,0.0003164907,0.0003051792,0.0004484664,0.01065239,0.06370991,0.003323894],"genre_scores_gemma":[0.09893947,0.001043707,0.8439563,0.0002562944,0.000111751,0.001227612,0.04386408,0.006824323,0.003776472],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0143748,"threshold_uncertainty_score":0.04808849,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2366093519","doi":"10.1093/nar/gkw409","title":"MoRFchibi SYSTEM: software tools for the identification of MoRFs in protein sequences","year":2016,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":182,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia; Canada's Michael Smith Genome Sciences Centre","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; Genome British Columbia; Michael Smith Health Research BC; Genome Canada","keywords":"Biology; Software; Web server; Identification (biology); Computational biology; Computer science; Bioinformatics; The Internet; World Wide Web; Operating system","authors":[{"name":"Nawar Malhis","is_ca":true},{"name":"Matthew Jacobson","is_ca":true},{"name":"Jörg Gsponer","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04208792933643683,"gpt":0.3482737489374061,"spread":0.3061858196009692,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001054271,0.001684671,0.001223742,0.001628063,0.0008976044,0.001142253,0.002700764,0.001037347,0.02757133],"category_scores_gemma":[0.002922231,0.0009663206,0.001282152,0.001254215,0.0003039165,0.001234724,0.00171704,0.002055351,0.01798599],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007448923,"about_ca_system_score_gemma":0.002064417,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004677079,"about_ca_topic_score_gemma":0.005268144,"domain_scores_codex":[0.9995795,0.00006501286,0.00003305455,0.0001074965,0.0001486382,0.0000663621],"domain_scores_gemma":[0.9991731,0.0003806736,0.0001026313,0.0001275684,0.0001415272,0.00007450645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001424639,0.0002122553,0.00961118,0.002075886,0.0005662509,0.0005793939,0.0003376795,0.01237121,0.02801769,0.006972588,0.7369103,0.2009209],"study_design_scores_gemma":[0.0005386999,0.0003074048,0.01685942,0.0004199333,0.0003294749,0.001478875,0.0002164691,0.4525374,0.08532936,0.01839205,0.4232235,0.0003675257],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"software","genre_scores_codex":[0.01755102,0.001271381,0.2372286,0.0003399215,0.0002208299,0.0003469083,0.05664859,0.6790986,0.0072941],"genre_scores_gemma":[0.1180605,0.001734344,0.5371334,0.001063028,0.0001593452,0.00228913,0.2629721,0.06028694,0.0163012],"genre_candidate":"software","genre_consensus":"software","teacher_disagreement_score":0.02757133,"threshold_uncertainty_score":0.09223527,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2163449716","doi":"10.1002/jcc.20918","title":"Prediction of protein structural class using novel evolutionary collocation‐based sequence representation","year":2008,"lang":"en","type":"article","venue":"Journal of Computational Chemistry","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":172,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Classifier (UML); Artificial intelligence; Pattern recognition (psychology); Representation (politics); Support vector machine; Sequence (biology); Protein methods; Class (philosophy); Pseudo amino acid composition; Data mining; Machine learning; Sequence analysis; Amino acid; Biology","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true},{"name":"Jishou Ruan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03889561558284672,"gpt":0.2932569067144774,"spread":0.2543612911316307,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004999769,0.0005627737,0.0004857183,0.001680474,0.0002570954,0.0006239774,0.0005190876,0.0006361994,0.001235378],"category_scores_gemma":[0.001609569,0.0001255006,0.000428007,0.001279306,0.0001972713,0.0008998257,0.0004719086,0.0005633772,0.000830337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003281807,"about_ca_system_score_gemma":0.0006045282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001273273,"about_ca_topic_score_gemma":0.001791028,"domain_scores_codex":[0.9996718,0.00005456611,0.00002377895,0.00009574222,0.0001200208,0.00003406372],"domain_scores_gemma":[0.9994304,0.0001616398,0.0001226023,0.00009131255,0.000158746,0.00003532449],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009661449,0.0007224789,0.03168133,0.0003902462,0.0001426026,0.0003504074,0.0001978186,0.123753,0.1918097,0.006877956,0.011374,0.6317343],"study_design_scores_gemma":[0.00002498219,0.0001445016,0.006190292,0.00002191594,0.00002659325,0.0001902348,0.0000420299,0.9707134,0.01702215,0.002943636,0.002656188,0.00002398293],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3987942,0.0008129903,0.5884517,0.0002539732,0.00008950372,0.0001609266,0.003182846,0.004941437,0.003312483],"genre_scores_gemma":[0.7572429,0.0003789198,0.2329514,0.0001005913,0.00004505406,0.000146391,0.007639181,0.0001480024,0.001347636],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001680474,"threshold_uncertainty_score":0.004132688,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2036790151","doi":"10.1038/nrmicro1494","title":"Methods for predicting bacterial protein subcellular localization","year":2006,"lang":"en","type":"review","venue":"Nature Reviews Microbiology","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":170,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University; University of British Columbia","funders":"","keywords":"Subcellular localization; Protein subcellular localization prediction; Biology; Computational biology; Identification (biology); Bacterial protein; Drug target; Annotation; Genome; Bioinformatics; Genetics; Gene; Biochemistry","authors":[{"name":"Jennifer L. Gardy","is_ca":true},{"name":"Fiona S. L. Brinkman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01785802078933187,"gpt":0.3771686864157439,"spread":0.359310665626412,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003198999,0.002130945,0.003191703,0.002628296,0.0003626051,0.001551334,0.004471756,0.001876545,0.002249005],"category_scores_gemma":[0.005426856,0.00111511,0.0008130299,0.004395507,0.001570927,0.002887451,0.001354157,0.003149958,0.003975864],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001019695,"about_ca_system_score_gemma":0.001157168,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002631555,"about_ca_topic_score_gemma":0.002777521,"domain_scores_codex":[0.9989127,0.0001820551,0.00009189416,0.0002330524,0.0005403181,0.00004000643],"domain_scores_gemma":[0.9963028,0.002219023,0.0002010116,0.0002499505,0.0009508305,0.00007634358],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006637835,0.00007636467,0.0003341626,0.003376301,0.0001446648,0.00004703628,0.00002336267,0.004981369,0.004480691,0.006606593,0.0218043,0.9580588],"study_design_scores_gemma":[0.0001133477,0.0002012486,0.002296056,0.003025818,0.0003909464,0.001555405,0.00007777056,0.05098207,0.02606245,0.06840061,0.8466467,0.0002476397],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0009125149,0.7952554,0.1964325,0.001821572,0.001570708,0.00006797639,0.0004354932,0.0008108068,0.002693046],"genre_scores_gemma":[0.006491048,0.8104168,0.1760739,0.001046667,0.001267341,0.0001451845,0.0010174,0.000151341,0.003390245],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.004471756,"threshold_uncertainty_score":0.01691812,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1986913015","doi":"10.1016/j.bbrc.2006.07.141","title":"Classifier ensembles for protein structural class prediction with varying homology","year":2006,"lang":"en","type":"article","venue":"Biochemical and Biophysical Research Communications","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":164,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Pattern recognition (psychology); Classifier (UML); Artificial intelligence; Computer science; Support vector machine; Homology (biology); Protein structure prediction; Computational biology; Mathematics; Algorithm; Protein structure; Biology; Genetics; Amino acid","authors":[{"name":"Kanaka Durga Kedarisetti","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true},{"name":"Scott Dick","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03527510867903335,"gpt":0.3373514513869992,"spread":0.3020763427079659,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004457245,0.0006844244,0.001549975,0.001646786,0.0008829125,0.0008059109,0.001099012,0.001310213,0.001224443],"category_scores_gemma":[0.0117844,0.0004382753,0.0008745656,0.0009718219,0.0002590387,0.001431055,0.001247854,0.001798839,0.00073751],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000580314,"about_ca_system_score_gemma":0.0008435819,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001904288,"about_ca_topic_score_gemma":0.003021902,"domain_scores_codex":[0.9982363,0.0006664396,0.0001341647,0.0002677997,0.0005391603,0.0001561387],"domain_scores_gemma":[0.991652,0.005106164,0.0002442025,0.0008841969,0.001887848,0.0002255484],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00111671,0.0004924308,0.01105656,0.0001016372,0.0003920499,0.000142303,0.000132311,0.2867505,0.009574015,0.002680229,0.009732457,0.6778287],"study_design_scores_gemma":[0.0000188094,0.00006200775,0.000691215,0.00000699257,0.00004155772,0.0000394466,0.00001318622,0.9938425,0.00255474,0.00231734,0.0004046271,0.000007580922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4224299,0.001620135,0.5677806,0.0006395223,0.0003535816,0.0001767993,0.0008829583,0.003219445,0.002897193],"genre_scores_gemma":[0.882969,0.000333711,0.1126036,0.000208545,0.000191704,0.000198783,0.001761557,0.0001679762,0.001565155],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004457245,"threshold_uncertainty_score":0.02357244,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2151031196","doi":"10.1093/bioinformatics/btr525","title":"Computational prediction of eukaryotic phosphorylation sites","year":2011,"lang":"en","type":"review","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":157,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Phosphorylation; Kinase; Computational biology; Protein phosphorylation; Mechanism (biology); Resource (disambiguation); Sequence (biology); Posttranslational modification; Biology; Computer science; Bioinformatics; Protein kinase A; Biochemistry; Enzyme","authors":[{"name":"Brett Trost","is_ca":true},{"name":"Anthony Kusalik","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02921289647721682,"gpt":0.2834585629456036,"spread":0.2542456664683868,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007379685,0.000847286,0.001260961,0.001357627,0.0001783053,0.0008119631,0.001434418,0.0005914997,0.001586622],"category_scores_gemma":[0.002056893,0.0004053881,0.0007722619,0.001795266,0.000252242,0.0008208455,0.0004646239,0.0007975146,0.001716157],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004090021,"about_ca_system_score_gemma":0.0006348717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001637943,"about_ca_topic_score_gemma":0.001666649,"domain_scores_codex":[0.9997804,0.00005068988,0.00002074389,0.00006922238,0.00006474974,0.00001426508],"domain_scores_gemma":[0.9990919,0.0006210749,0.0000781042,0.00003959147,0.0001468061,0.00002253807],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002109661,0.0001058464,0.005078196,0.005779924,0.0004122373,0.0002166572,0.00003417298,0.2314672,0.005165766,0.009814816,0.0203312,0.7213829],"study_design_scores_gemma":[0.0001077283,0.0001972657,0.006503695,0.001729787,0.0004404461,0.001416962,0.00006323792,0.7983738,0.02408459,0.04247134,0.12449,0.0001212533],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.05318824,0.4507411,0.4618935,0.002923975,0.0008399728,0.0001441248,0.00575299,0.004129143,0.020387],"genre_scores_gemma":[0.216362,0.3676453,0.3922625,0.000770142,0.0005633614,0.0002635411,0.01469263,0.0004415436,0.006998911],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.001637943,"threshold_uncertainty_score":0.005307734,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1996768014","doi":"10.1186/1472-6807-7-25","title":"Prediction of flexible/rigid regions from protein sequences using k-spaced amino acid pairs","year":2007,"lang":"en","type":"article","venue":"BMC Structural Biology","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":145,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; National Natural Science Foundation of China","keywords":"Curse of dimensionality; Pattern recognition (psychology); Feature (linguistics); Protein methods; Computer science; Representation (politics); Threading (protein sequence); Feature selection; Protein structure prediction; Artificial intelligence; Support vector machine; Protein structure; Pseudo amino acid composition; Sequence (biology); Protein tertiary structure; Algorithm; Peptide sequence; Amino acid; Biology; Biochemistry","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true},{"name":"Jishou Ruan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03267685854606961,"gpt":0.2933025375500183,"spread":0.2606256790039487,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005105795,0.0007091667,0.0006458255,0.001935617,0.000309866,0.0003544495,0.0004363485,0.0006527929,0.0008860395],"category_scores_gemma":[0.001376099,0.000157793,0.0007128939,0.001114832,0.0002396869,0.0005119345,0.0002830275,0.0005312752,0.0007849011],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003281163,"about_ca_system_score_gemma":0.0005420333,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001934967,"about_ca_topic_score_gemma":0.00166516,"domain_scores_codex":[0.9996212,0.0000550468,0.00004012084,0.000154401,0.00008517446,0.00004400633],"domain_scores_gemma":[0.9990332,0.0003492772,0.0002839501,0.0000722918,0.0001949633,0.00006630313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002337813,0.0009173458,0.09633837,0.0008468933,0.0003203707,0.001311208,0.000200423,0.2203332,0.1903574,0.00149359,0.007640449,0.4779029],"study_design_scores_gemma":[0.00004154289,0.0001633202,0.01760275,0.00003251191,0.00005296676,0.0004870198,0.0000538915,0.961193,0.01797155,0.001071476,0.00130532,0.00002467107],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7503201,0.00114872,0.242075,0.000143237,0.00005102742,0.0001291376,0.002762168,0.002152026,0.001218616],"genre_scores_gemma":[0.8371077,0.0002765503,0.1560417,0.00005218127,0.00003721635,0.00009298813,0.005838825,0.00006324872,0.0004896208],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.001935617,"threshold_uncertainty_score":0.003847361,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1997967533","doi":"10.1021/pr900665y","title":"SherLoc2: A High-Accuracy Hybrid Method for Predicting Subcellular Localization of Proteins","year":2009,"lang":"en","type":"article","venue":"Journal of Proteome Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":143,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Gene ontology; Subcellular localization; Data mining; Protein sequencing; Sequence (biology); Protein subcellular localization prediction; Feature (linguistics); Computational biology; Artificial intelligence; Gene; Biology; Peptide sequence; Genetics; Gene expression","authors":[{"name":"Sebastian Briesemeister","is_ca":true},{"name":"Torsten Blum","is_ca":true},{"name":"Scott T. Brady","is_ca":true},{"name":"Yin Pak Lam","is_ca":true},{"name":"Oliver Kohlbacher","is_ca":true},{"name":"Hagit Shatkay","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02805623053856246,"gpt":0.3845717248430646,"spread":0.3565154943045022,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002408064,0.002118599,0.001390887,0.004393671,0.0007778258,0.001546609,0.002495031,0.002176379,0.00579163],"category_scores_gemma":[0.004726007,0.0005588779,0.001316997,0.00229676,0.0003715655,0.001565672,0.001862663,0.001303414,0.005239043],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006039073,"about_ca_system_score_gemma":0.0008350326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002495441,"about_ca_topic_score_gemma":0.005727827,"domain_scores_codex":[0.998449,0.0003438007,0.00009751895,0.0003442687,0.0006328009,0.0001325832],"domain_scores_gemma":[0.9972417,0.001255692,0.0002305717,0.0003659657,0.0007549455,0.0001511071],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001599717,0.0003657543,0.01554571,0.0006864005,0.0009307648,0.0005424052,0.0001870524,0.07248852,0.04658982,0.002855494,0.0583665,0.7998419],"study_design_scores_gemma":[0.00009680627,0.0001314048,0.004327288,0.00003093914,0.0001029982,0.0005783569,0.00005199254,0.9608335,0.02132663,0.00326536,0.00914793,0.0001067742],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05548036,0.001288009,0.8882227,0.0002218326,0.0001941763,0.0001524782,0.003649223,0.04792008,0.002871147],"genre_scores_gemma":[0.2589467,0.000427134,0.7180033,0.0003097938,0.0001391295,0.0003949656,0.01040312,0.001924049,0.009451765],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00579163,"threshold_uncertainty_score":0.01937497,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2145023212","doi":"10.1093/bioinformatics/btr657","title":"Prediction and analysis of nucleotide-binding residues using sequence and sequence-derived structural descriptors","year":2011,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":137,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Killam Trusts","keywords":"Sequence (biology); Computational biology; Sequence alignment; Peptide sequence; Sequence analysis; Nucleotide; Multiple sequence alignment; Protein sequencing; Binding site; Sequence logo; Sequence motif; Consensus sequence; Conserved sequence; Biology; Biochemistry; Gene","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Marcin J. Mizianty","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05324823264434206,"gpt":0.2742169801333416,"spread":0.2209687474889996,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000694311,0.0008379219,0.0007504874,0.001699018,0.0002991186,0.0005180706,0.000573416,0.0005378157,0.002573912],"category_scores_gemma":[0.001819941,0.0002141161,0.0005485232,0.00150911,0.0002283647,0.0005035414,0.0003952399,0.0006487395,0.001703318],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005088843,"about_ca_system_score_gemma":0.0008253098,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00256924,"about_ca_topic_score_gemma":0.004739906,"domain_scores_codex":[0.9997222,0.00005399076,0.00001810464,0.00008743862,0.00008820404,0.00003009862],"domain_scores_gemma":[0.9992232,0.0003440786,0.000141522,0.00005643151,0.0001519667,0.00008272071],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002004312,0.001149415,0.1426285,0.001299958,0.0004125331,0.001166211,0.0001495887,0.4579119,0.1423053,0.003012318,0.01946664,0.2284933],"study_design_scores_gemma":[0.00006053433,0.0002565118,0.01783865,0.00004445552,0.00011381,0.0003741801,0.00004734978,0.9467272,0.02793917,0.002377656,0.004186092,0.00003438977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.7725428,0.001713235,0.1959145,0.0003600215,0.00006279009,0.000212617,0.02063572,0.005223982,0.003334176],"genre_scores_gemma":[0.8655009,0.0007592326,0.0852695,0.0001036005,0.00004528372,0.0001504325,0.04637469,0.0003058731,0.001490436],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.002573912,"threshold_uncertainty_score":0.008610606,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2118769225","doi":"10.1093/nar/gki027","title":"PSORTdb: a protein subcellular localization database for bacteria","year":2004,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":135,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada; Genome British Columbia; Genome Canada; Genome Prairie; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; Michael Smith Health Research BC; National Science Foundation","keywords":"Biology; Identification (biology); Annotation; Database; Bacterial genome size; Computational biology; Function (biology); Protein function; Genome; Database search engine; Computer science; Information retrieval; Bioinformatics; Search engine; Data mining; Genetics; Gene","authors":[{"name":"Sébastien Rey","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03065818731062588,"gpt":0.3387933180577546,"spread":0.3081351307471287,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005077466,0.001497607,0.001688759,0.003077742,0.0008431411,0.001516727,0.00175839,0.001157887,0.01243724],"category_scores_gemma":[0.001449951,0.0006319709,0.0009490086,0.003740744,0.0002506583,0.001631725,0.001611404,0.001460343,0.02359797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006700861,"about_ca_system_score_gemma":0.001774822,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002241847,"about_ca_topic_score_gemma":0.002355011,"domain_scores_codex":[0.9996761,0.00003869894,0.00006495926,0.00008406923,0.00009108883,0.00004510561],"domain_scores_gemma":[0.9995624,0.00005583759,0.00008703616,0.0000880732,0.0001054127,0.0001011645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002038852,0.0002410943,0.003976204,0.005493015,0.0002304172,0.0008450525,0.0002954113,0.00306969,0.0713085,0.009051722,0.7988518,0.1045982],"study_design_scores_gemma":[0.0004302207,0.0001872802,0.007155327,0.000329354,0.0001509747,0.001433004,0.0001304719,0.006457821,0.01478557,0.009679311,0.9591544,0.0001063724],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"software","genre_scores_codex":[0.01624337,0.007596121,0.05816744,0.000890872,0.0003367653,0.0003163157,0.8516898,0.05111445,0.01364488],"genre_scores_gemma":[0.01141179,0.002305439,0.03191081,0.0002068096,0.00003969476,0.0002343601,0.9503447,0.001267802,0.002278541],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.01243724,"threshold_uncertainty_score":0.04160666,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2164620224","doi":"10.1093/bioinformatics/btm115","title":"SherLoc: high-accuracy prediction of protein subcellular localization by integrating text and protein sequence data","year":2007,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":127,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Sequence (biology); Protein sequencing; Artificial intelligence; Computational biology; Protein subcellular localization prediction; Data mining; Pattern recognition (psychology); Peptide sequence; Biology; Genetics; Gene","authors":[{"name":"Hagit Shatkay","is_ca":true},{"name":"Annette Höglund","is_ca":true},{"name":"Scott T. Brady","is_ca":true},{"name":"Torsten Blum","is_ca":true},{"name":"Pierre Dönnes","is_ca":true},{"name":"Oliver Kohlbacher","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01554791924248027,"gpt":0.2587434904356499,"spread":0.2431955711931696,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001772015,0.001758799,0.0009770418,0.005406751,0.0005387202,0.001142384,0.001440819,0.001533326,0.003545513],"category_scores_gemma":[0.006913758,0.0003171204,0.0007025228,0.003367528,0.0003888097,0.002176547,0.001348807,0.0008123473,0.004256485],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005050951,"about_ca_system_score_gemma":0.0007353626,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001802921,"about_ca_topic_score_gemma":0.003486579,"domain_scores_codex":[0.9992455,0.0001821913,0.0000674243,0.0002309927,0.0002206871,0.00005312826],"domain_scores_gemma":[0.9950516,0.00267356,0.0006706187,0.0006248266,0.0007522199,0.0002272328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002418383,0.0006540655,0.06667293,0.002154649,0.0006075591,0.001300796,0.000273539,0.08322464,0.04676523,0.002449909,0.1727264,0.620752],"study_design_scores_gemma":[0.0002471506,0.0003644378,0.0162793,0.00008558876,0.00013358,0.001008239,0.0001248814,0.9176593,0.03986145,0.006826465,0.01730588,0.0001038025],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3314142,0.004545487,0.416853,0.001622726,0.000380685,0.0003560547,0.08941814,0.1512052,0.00420453],"genre_scores_gemma":[0.4526826,0.001091543,0.4055614,0.0003850624,0.0002640551,0.000383117,0.133903,0.001559681,0.004169591],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005406751,"threshold_uncertainty_score":0.01186097,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2021901733","doi":"10.1016/s1672-0229(07)60022-9","title":"FragAnchor: A Large-Scale Predictor of Glycosylphosphatidylinositol Anchors in Eukaryote Protein Sequences by Qualitative Scoring","year":2007,"lang":"en","type":"article","venue":"Genomics Proteomics & Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":125,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University; Université du Québec à Montréal","funders":"National Center for Research Resources; National Institutes of Health","keywords":"Eukaryote; UniProt; Proteome; Computational biology; Hidden Markov model; Protein sequencing; Human proteome project; Computer science; Noise (video); Scale (ratio); Biology; Artificial intelligence; Proteomics; Bioinformatics; Machine learning; Pattern recognition (psychology); Genetics; Peptide sequence; Gene; Genome","authors":[{"name":"Guylaine Poisson","is_ca":false},{"name":"Cédric Chauve","is_ca":true},{"name":"Xin Chen","is_ca":false},{"name":"Anne Bergeron","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.00645521733399634,"gpt":0.2699390928318512,"spread":0.2634838754978549,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008510327,0.0009823359,0.0005828678,0.001802624,0.0003824855,0.0005489627,0.0007172245,0.000731648,0.002596079],"category_scores_gemma":[0.002086032,0.0002740021,0.0005427522,0.000966816,0.000284169,0.0009692555,0.0008682006,0.0005500655,0.001116183],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004958119,"about_ca_system_score_gemma":0.0006711019,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002442277,"about_ca_topic_score_gemma":0.003625623,"domain_scores_codex":[0.9997209,0.00007490133,0.00001703748,0.00007440056,0.00007487862,0.00003792074],"domain_scores_gemma":[0.9992989,0.0003354909,0.0001144745,0.00006591294,0.0001053073,0.00007986868],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005151629,0.0005323639,0.09771536,0.001164188,0.0005868856,0.0009100462,0.0001945828,0.1327239,0.2245292,0.003913006,0.03756505,0.4950137],"study_design_scores_gemma":[0.0001551355,0.0003401229,0.01960721,0.00003899022,0.0000882746,0.0004792708,0.00005517028,0.9204985,0.04939514,0.003586743,0.005672706,0.00008264958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5935416,0.001851823,0.3520387,0.0004003894,0.00007640114,0.0001859162,0.01696945,0.03277822,0.00215748],"genre_scores_gemma":[0.7131921,0.0004618023,0.2594465,0.0001602406,0.00003723229,0.0001745047,0.02359353,0.0006960868,0.00223802],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.002596079,"threshold_uncertainty_score":0.008684754,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2040630248","doi":"10.1371/journal.pcbi.0010066","title":"Refining Protein Subcellular Localization","year":2005,"lang":"en","type":"article","venue":"PLoS Computational Biology","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":123,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"Canadian Institutes of Health Research; Genome Canada","keywords":"Proteome; Computational biology; Subcellular localization; Protein subcellular localization prediction; Protein–protein interaction; Biology; Protein targeting; Function (biology); Organelle; Proteomics; Computer science; Bioinformatics; Membrane protein; Cell biology; Biochemistry; Gene; Cytoplasm","authors":[{"name":"Michelle S. Scott","is_ca":true},{"name":"Sara J Calafell","is_ca":true},{"name":"David Y. Thomas","is_ca":true},{"name":"Michael Hallett","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009820439929410696,"gpt":0.2489673547677503,"spread":0.2391469148383396,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001323699,0.0009113129,0.001057115,0.001430747,0.0006547615,0.00120306,0.0007362471,0.0007841499,0.002706871],"category_scores_gemma":[0.005050863,0.0004769826,0.0009762778,0.001727946,0.0004482173,0.001721136,0.001238027,0.00121969,0.001721398],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008628996,"about_ca_system_score_gemma":0.001367386,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005252529,"about_ca_topic_score_gemma":0.00744433,"domain_scores_codex":[0.9994488,0.000139495,0.00003240439,0.0001829954,0.0001436286,0.00005254528],"domain_scores_gemma":[0.9987213,0.0005260204,0.0001432905,0.0002408297,0.0003147884,0.00005368212],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001922262,0.0002398321,0.05222981,0.001478131,0.0002806171,0.0008031863,0.0003367427,0.439204,0.1663334,0.02328291,0.01299252,0.3008966],"study_design_scores_gemma":[0.00004704625,0.00009138055,0.005248797,0.00003899182,0.00007177126,0.0002624943,0.00008701865,0.9354259,0.03742516,0.01384738,0.00741538,0.00003874636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2155828,0.001083245,0.7684931,0.0004738719,0.00008183046,0.00008332625,0.003179234,0.00784711,0.003175532],"genre_scores_gemma":[0.7227254,0.0009415562,0.2659424,0.0001616854,0.00002856138,0.0001230828,0.007576855,0.0007276904,0.001772723],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005252529,"threshold_uncertainty_score":0.01044393,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3157437194","doi":"10.1002/cpz1.113","title":"Learned Embeddings from Deep Learning to Visualize and Predict Protein Sets","year":2021,"lang":"en","type":"article","venue":"Current Protocols","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":123,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Vector Institute; University of Toronto","funders":"Deutsche Forschungsgemeinschaft","keywords":"Computer science; Workflow; Artificial intelligence; Pipeline (software); ENCODE; Classifier (UML); Protocol (science); Machine learning; Inference; Natural language processing; Programming language; Biology","authors":[{"name":"Christian Dallago","is_ca":false},{"name":"Konstantin Schütze","is_ca":false},{"name":"Michael Heinzinger","is_ca":false},{"name":"Tobias Olenyi","is_ca":false},{"name":"Maria Littmann","is_ca":false},{"name":"Amy X. Lu","is_ca":true},{"name":"Kevin Yang","is_ca":false},{"name":"Seonwoo Min","is_ca":false},{"name":"Sungroh Yoon","is_ca":false},{"name":"James T. Morton","is_ca":false},{"name":"Burkhard Rost","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02184703544408889,"gpt":0.3633679176334731,"spread":0.3415208821893843,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007139348,0.001393491,0.0005751596,0.001434029,0.000428728,0.001676405,0.001623043,0.001057479,0.01053414],"category_scores_gemma":[0.003347323,0.0007552081,0.001050729,0.001178894,0.0005732384,0.002408264,0.001976218,0.002568513,0.00553411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001191633,"about_ca_system_score_gemma":0.001176305,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003472286,"about_ca_topic_score_gemma":0.006103168,"domain_scores_codex":[0.9995579,0.00008046791,0.00002886264,0.0001419482,0.0001429519,0.00004787166],"domain_scores_gemma":[0.9992213,0.0002591516,0.0000747583,0.0001869602,0.0002025011,0.00005523251],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004135763,0.0003484861,0.003618317,0.000655591,0.0001587293,0.0002833323,0.0002347306,0.2534106,0.0445236,0.04580276,0.07313441,0.5774159],"study_design_scores_gemma":[0.00002482113,0.00005734394,0.0004442754,0.00005436307,0.00001760961,0.00006367943,0.0000494412,0.9251221,0.02157043,0.03724796,0.01531694,0.00003102109],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03358988,0.000866631,0.9234171,0.0008367017,0.0002698507,0.0001940244,0.007308989,0.02668626,0.006830618],"genre_scores_gemma":[0.2101495,0.001284682,0.7468584,0.0004410359,0.00006985745,0.0006188597,0.02615808,0.002953771,0.0114658],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01053414,"threshold_uncertainty_score":0.03524017,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2104873783","doi":"10.1093/bioinformatics/btm475","title":"PFRES: protein fold classification by using evolutionary information and predicted secondary structure","year":2007,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":122,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Protein tertiary structure; Protein secondary structure; Computer science; Pseudo amino acid composition; Similarity (geometry); Classifier (UML); Pattern recognition (psychology); Protein Data Bank (RCSB PDB); Fold (higher-order function); Protein superfamily; Protein structure prediction; Representation (politics); Feature vector; Structural Classification of Proteins database; Protein structure; Sequence (biology); Structural alignment; Data mining; Algorithm; Artificial intelligence; Sequence alignment; Biology; Peptide sequence; Amino acid; Genetics","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006359821080226958,"gpt":0.233839707426869,"spread":0.2274798863466421,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001762268,0.002271978,0.0009641612,0.004458984,0.0005113001,0.0007400129,0.001837098,0.0008691765,0.009428702],"category_scores_gemma":[0.004529552,0.0003460553,0.000867346,0.001803791,0.0003854776,0.001712441,0.0008233617,0.0006871868,0.008114785],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004282333,"about_ca_system_score_gemma":0.0006279136,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001019288,"about_ca_topic_score_gemma":0.001064692,"domain_scores_codex":[0.9989383,0.0001826138,0.00008272678,0.0002290374,0.0004907784,0.00007662265],"domain_scores_gemma":[0.9988106,0.0004344647,0.0002461939,0.0001730959,0.0002654665,0.00007002835],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001316377,0.0002658779,0.017256,0.001452574,0.0002474306,0.0005319269,0.0001342109,0.02348853,0.03655168,0.003562001,0.1166196,0.7985737],"study_design_scores_gemma":[0.0001741587,0.00058609,0.01742295,0.0002586412,0.0001008752,0.002520771,0.000116313,0.8369206,0.0848482,0.009892185,0.0469894,0.0001698311],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08022684,0.002194039,0.6666846,0.0005556273,0.0002571969,0.0005764412,0.03260345,0.2111149,0.005786968],"genre_scores_gemma":[0.2992299,0.001020059,0.6056753,0.0001601769,0.0002349738,0.0006459322,0.08367693,0.003446512,0.005910262],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009428702,"threshold_uncertainty_score":0.03154218,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2059926522","doi":"10.1016/j.bbrc.2007.02.040","title":"Prediction of protein crystallization using collocation of amino acid pairs","year":2007,"lang":"en","type":"article","venue":"Biochemical and Biophysical Research Communications","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":118,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"In silico; Protein crystallization; Naive Bayes classifier; Computer science; Crystallization; Classifier (UML); Sequence (biology); Protein Data Bank (RCSB PDB); Artificial intelligence; Threading (protein sequence); Protein sequencing; Peptide sequence; Computational biology; Pattern recognition (psychology); Protein structure; Algorithm; Chemistry; Biology; Biochemistry; Support vector machine","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true},{"name":"Mandana Rahbari","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07206198932686944,"gpt":0.3613267705409309,"spread":0.2892647812140614,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009089924,0.0009109952,0.001261353,0.002748121,0.000760539,0.0007250766,0.0006401672,0.0009992189,0.001080809],"category_scores_gemma":[0.003553323,0.0005412329,0.001088908,0.00204612,0.0004381561,0.001152967,0.0007669064,0.0009558895,0.0005584687],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005662485,"about_ca_system_score_gemma":0.001068942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003449685,"about_ca_topic_score_gemma":0.005891742,"domain_scores_codex":[0.9994909,0.0001340015,0.00005567535,0.0001434528,0.0001311874,0.00004484319],"domain_scores_gemma":[0.9975647,0.001061204,0.0003824033,0.0003134478,0.0004997288,0.0001785236],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006448364,0.00082707,0.0993347,0.0009030786,0.000951724,0.00247569,0.0003238615,0.4541868,0.1705949,0.007313057,0.0126178,0.244023],"study_design_scores_gemma":[0.00006202198,0.00008379513,0.003603573,0.00001138381,0.00006025628,0.0001675589,0.00002672217,0.9805619,0.01233852,0.002502577,0.0005667973,0.00001479786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.767209,0.001076594,0.2250864,0.0002301878,0.0001225592,0.000157879,0.001817985,0.002764473,0.001534953],"genre_scores_gemma":[0.9299107,0.000231969,0.06724901,0.00003806836,0.00002396636,0.0000503117,0.002076172,0.0001283865,0.0002913859],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003449685,"threshold_uncertainty_score":0.006859183,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1972434016","doi":"10.1093/nar/gkn255","title":"PROTEUS2: a web server for comprehensive protein structure prediction and structure-based annotation","year":2008,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":116,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta; National Institute for Nanotechnology","funders":"Alberta Prion Research Institute; Natural Sciences and Engineering Research Council of Canada; Genome Alberta","keywords":"Protein secondary structure; Web server; Threading (protein sequence); Biology; Computational biology; Annotation; Protein structure prediction; Protein structure database; Homology modeling; Computer science; Hidden Markov model; Pipeline (software); Sequence alignment; Protein tertiary structure; Protein structure; Artificial intelligence; Bioinformatics; Peptide sequence; Sequence database; Genetics; The Internet","authors":[{"name":"Scott Montgomerie","is_ca":true},{"name":"J. A. Cruz","is_ca":true},{"name":"S. Shrivastava","is_ca":true},{"name":"D. Arndt","is_ca":true},{"name":"Mark Berjanskii","is_ca":true},{"name":"David S. Wishart","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03240950971587583,"gpt":0.3212105942469039,"spread":0.288801084531028,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001099451,0.003030751,0.001898285,0.003087409,0.001214041,0.001618795,0.002010121,0.0008562665,0.01611096],"category_scores_gemma":[0.001560719,0.001415344,0.001176028,0.002701179,0.0003507043,0.001344969,0.00200911,0.002065197,0.03608509],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008118906,"about_ca_system_score_gemma":0.002270573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002550483,"about_ca_topic_score_gemma":0.002570992,"domain_scores_codex":[0.9993069,0.0001246215,0.00006369668,0.0001564165,0.0002451583,0.0001031862],"domain_scores_gemma":[0.9995721,0.0000711045,0.00006504027,0.00005725516,0.0001388151,0.00009564086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001893398,0.0002269612,0.002852444,0.002944086,0.0003141651,0.0006267269,0.0002682123,0.003251027,0.07724613,0.00487331,0.8020537,0.1034499],"study_design_scores_gemma":[0.0006595769,0.0003147057,0.009600274,0.0006215498,0.0002310171,0.002093678,0.0001821457,0.0481091,0.06947878,0.01623647,0.8520852,0.0003875981],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"software","genre_scores_codex":[0.02972221,0.008201855,0.2656337,0.0008221355,0.0006457962,0.001151864,0.3131867,0.359791,0.02084467],"genre_scores_gemma":[0.03947291,0.002778951,0.2146432,0.000521436,0.0001498795,0.00134988,0.7084507,0.01898743,0.01364555],"genre_candidate":"software","genre_consensus":"software","teacher_disagreement_score":0.01611096,"threshold_uncertainty_score":0.05389655,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2012352014","doi":"10.1101/gr.2650004","title":"Predicting Subcellular Localization via Protein Motif Co-Occurrence","year":2004,"lang":"en","type":"article","venue":"Genome Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":114,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"Canadian Institutes of Health Research; Genome Canada","keywords":"Subcellular localization; Biology; Protein subcellular localization prediction; Computational biology; Protein Sorting Signals; Membrane protein; Organelle; Green fluorescent protein; Bioinformatics; Cell biology; Cytoplasm; Biochemistry; Gene; Peptide sequence; Signal peptide; Membrane","authors":[{"name":"Michelle S. Scott","is_ca":true},{"name":"David Y. Thomas","is_ca":true},{"name":"Michael Hallett","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02705706486612943,"gpt":0.337070739419401,"spread":0.3100136745532716,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009403304,0.00048524,0.0005967214,0.002718962,0.0002803777,0.0004458642,0.000398372,0.0005705807,0.001176625],"category_scores_gemma":[0.003148808,0.0002329046,0.0005257025,0.001745223,0.000257027,0.0007612098,0.0004316422,0.0005698538,0.001117382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003583329,"about_ca_system_score_gemma":0.0005503275,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002830203,"about_ca_topic_score_gemma":0.004097644,"domain_scores_codex":[0.9995066,0.0001243015,0.0000390274,0.0001408396,0.0001379025,0.00005135728],"domain_scores_gemma":[0.9980158,0.001153881,0.000362869,0.0001107435,0.0002612748,0.00009544679],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002443559,0.0005251564,0.258823,0.0009823502,0.0005778181,0.001005844,0.000196631,0.2381268,0.1036281,0.004847045,0.009552156,0.3792915],"study_design_scores_gemma":[0.00003271604,0.000099693,0.02147644,0.00002292716,0.000061557,0.0004980967,0.00003399665,0.9577944,0.01347177,0.004649636,0.001832887,0.00002586339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5191103,0.00115293,0.4680123,0.0002933588,0.0000319863,0.00009702769,0.00476484,0.004802874,0.001734396],"genre_scores_gemma":[0.8658625,0.000377058,0.1263021,0.00004849457,0.00002437416,0.0001087081,0.006496864,0.0001008774,0.0006790838],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002830203,"threshold_uncertainty_score":0.005627453,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2069116068","doi":"10.1145/507515.507523","title":"KDD Cup 2001 report","year":2002,"lang":"en","type":"article","venue":"ACM SIGKDD Explorations Newsletter","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":110,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Canadian Imperial Bank of Commerce (Canada)","funders":"National Science Foundation","keywords":"Computer science; Artificial intelligence","authors":[{"name":"Jie Cheng","is_ca":true},{"name":"Christos Hatzis","is_ca":false},{"name":"Hisashi Hayashi","is_ca":false},{"name":"Mark-A. Krogel","is_ca":false},{"name":"Shinichi Morishita","is_ca":false},{"name":"David Page","is_ca":false},{"name":"Jun Sese","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03648866402031892,"gpt":0.2723488489952492,"spread":0.2358601849749303,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01427799,0.004349593,0.003919384,0.01407465,0.003423423,0.01518375,0.007351962,0.005119074,0.06238117],"category_scores_gemma":[0.04080032,0.001366317,0.002700991,0.01372903,0.0007897636,0.009432475,0.005072314,0.004595072,0.1541951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004508577,"about_ca_system_score_gemma":0.01078803,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02085322,"about_ca_topic_score_gemma":0.01241318,"domain_scores_codex":[0.9832467,0.002093966,0.001680836,0.001879014,0.01016904,0.0009303888],"domain_scores_gemma":[0.9593453,0.003876726,0.001049142,0.007337234,0.02550213,0.002889462],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004626579,0.00004531195,0.0003155106,0.000190382,0.00002777871,0.00002789148,0.000009346163,0.000566337,0.00005136554,0.001528973,0.9649695,0.03222144],"study_design_scores_gemma":[0.00008870962,0.00003425711,0.001162278,0.0001766385,0.00004481963,0.0001362808,0.0000494795,0.004788816,0.0007265691,0.004626045,0.9881252,0.00004091719],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.0058059,0.03123854,0.07683369,0.04617757,0.02568327,0.002634465,0.5110664,0.06140276,0.2391573],"genre_scores_gemma":[0.01152322,0.01119728,0.0411831,0.003767368,0.002392936,0.001354031,0.8502707,0.002889219,0.07542215],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.06238117,"threshold_uncertainty_score":0.2086858,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2887711510","doi":"10.1038/s41598-019-40561-2","title":"Interpretable genotype-to-phenotype classifiers with performance guarantees","year":2019,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":108,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada; University of Waterloo; Canada Research Chairs; Government of Canada; Compute Canada; Université Laval","keywords":"Interpretability; Machine learning; Computer science; Artificial intelligence; Scalability; Curse of dimensionality; Generalization; Turnkey; Mathematics","authors":[{"name":"Alexandre Drouin","is_ca":true},{"name":"Gaël Letarte","is_ca":true},{"name":"Frédéric Raymond","is_ca":true},{"name":"Mario Marchand","is_ca":true},{"name":"Jacques Corbeil","is_ca":true},{"name":"François Laviolette","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.003900188324957552,"gpt":0.2139763805367878,"spread":0.2100761922118303,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007968355,0.001967855,0.002431928,0.001437444,0.000892167,0.003790462,0.00341281,0.004263637,0.002955381],"category_scores_gemma":[0.03698713,0.000881549,0.001496591,0.001220757,0.001475566,0.004428925,0.00332221,0.005239045,0.00281449],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002046863,"about_ca_system_score_gemma":0.002347538,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001973063,"about_ca_topic_score_gemma":0.001846411,"domain_scores_codex":[0.9951681,0.001334129,0.000487072,0.0009850576,0.001555965,0.0004696657],"domain_scores_gemma":[0.9801878,0.01176166,0.001121235,0.004176225,0.002393009,0.0003600279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007742448,0.0004295537,0.006424355,0.0002278623,0.0001602163,0.000382297,0.0002900815,0.6634042,0.006775297,0.02736432,0.01071705,0.2830504],"study_design_scores_gemma":[0.00003757203,0.00006307631,0.0003729139,0.00001876577,0.00001854988,0.00006214828,0.00003036608,0.9577817,0.001821232,0.03913528,0.00064624,0.00001203514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06140827,0.001161576,0.9252481,0.002808758,0.0001718906,0.0001992714,0.0009445092,0.004500612,0.003557093],"genre_scores_gemma":[0.6883439,0.0007093727,0.2994263,0.001374503,0.0004454399,0.000608346,0.004219433,0.0005470584,0.004325605],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007968355,"threshold_uncertainty_score":0.0421412,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2910654431","doi":"10.1016/j.gpb.2018.08.004","title":"Integration of A Deep Learning Classifier with A Random Forest Approach for Predicting Malonylation Sites","year":2018,"lang":"en","type":"article","venue":"Genomics Proteomics & Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":107,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"National Natural Science Foundation of China; Natural Science Foundation of Shandong Province; Qingdao Postdoctoral Science Foundation","keywords":"Classifier (UML); Computer science; Random forest; Artificial intelligence; Deep learning; Machine learning; Training set; Pattern recognition (psychology); End-to-end principle","authors":[{"name":"Zhen Chen","is_ca":false},{"name":"Ningning He","is_ca":false},{"name":"Yu Huang","is_ca":false},{"name":"Wen Qin","is_ca":true},{"name":"Xuhan Liu","is_ca":false},{"name":"Lei Li","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.009638930436779197,"gpt":0.2327989324049696,"spread":0.2231600019681904,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001221939,0.001070885,0.0008246224,0.001124793,0.0002962216,0.0004664753,0.0009639532,0.001149134,0.001410472],"category_scores_gemma":[0.001485604,0.0003082099,0.0008017241,0.0006573046,0.0001802247,0.001077634,0.0005476713,0.001068831,0.0008012891],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004992673,"about_ca_system_score_gemma":0.0008891252,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004329004,"about_ca_topic_score_gemma":0.005237053,"domain_scores_codex":[0.9996295,0.00006690897,0.00002974089,0.0001091732,0.00007724227,0.00008741593],"domain_scores_gemma":[0.9992862,0.0002748182,0.00005069682,0.00004816882,0.0002942886,0.00004582283],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005364587,0.0005435741,0.006806948,0.0001355895,0.0001813573,0.0002670948,0.00003506908,0.2203388,0.04151992,0.001513425,0.005153526,0.7229683],"study_design_scores_gemma":[0.000008487651,0.00005907938,0.0003386808,0.000004721684,0.00001778068,0.00002266011,0.000003709476,0.9941046,0.004733441,0.0004510943,0.0002492929,0.000006565977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1060486,0.000653915,0.8857809,0.0002345912,0.0001329045,0.0001333008,0.0004773457,0.005282813,0.001255564],"genre_scores_gemma":[0.6430444,0.0003059727,0.3519935,0.0002561233,0.00008622068,0.0001998049,0.001608574,0.0001096809,0.002395671],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004329004,"threshold_uncertainty_score":0.008607626,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2150408357","doi":"10.1093/nar/gkh485","title":"Proteome Analyst: custom predictions with explanations in a web-based tool for high-throughput proteome annotations","year":2004,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":100,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Proteome; Gene ontology; Classifier (UML); Naive Bayes classifier; Computer science; Biology; Human proteome project; Bayes' theorem; Function (biology); Machine learning; Protein function; Artificial intelligence; Protein function prediction; Computational biology; Bioinformatics; Gene; Proteomics; Bayesian probability; Genetics; Support vector machine; Gene expression","authors":[{"name":"Duane Szafron","is_ca":true},{"name":"P. Lu","is_ca":true},{"name":"Russell Greiner","is_ca":true},{"name":"David S. Wishart","is_ca":true},{"name":"Brett Poulin","is_ca":true},{"name":"Roman Eisner","is_ca":true},{"name":"Zhiyong Lu","is_ca":true},{"name":"John Anvik","is_ca":true},{"name":"Cam Macdonell","is_ca":true},{"name":"Alona Fyshe","is_ca":true},{"name":"David Meeuwis","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02065510645228343,"gpt":0.3270776948286184,"spread":0.306422588376335,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002689346,0.002625271,0.000723227,0.002261294,0.0005909337,0.001550094,0.002884675,0.002117124,0.04132002],"category_scores_gemma":[0.009795385,0.001436249,0.001567198,0.001357093,0.0004914353,0.003302589,0.002133062,0.002030793,0.01353179],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008527251,"about_ca_system_score_gemma":0.001285451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002311546,"about_ca_topic_score_gemma":0.003356237,"domain_scores_codex":[0.9991112,0.0001776011,0.0001028947,0.0001945027,0.0003540201,0.00005971053],"domain_scores_gemma":[0.9927657,0.005249213,0.0004741054,0.0007959855,0.000538181,0.0001768848],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002223543,0.0005912882,0.009890618,0.002475045,0.0003313418,0.002561414,0.001082284,0.02756136,0.02689051,0.02167264,0.508059,0.396661],"study_design_scores_gemma":[0.000940495,0.0001914116,0.005896262,0.0004758934,0.0002114997,0.001596675,0.0003026685,0.5937495,0.07465773,0.05536639,0.2661898,0.0004217155],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"software","genre_scores_codex":[0.004116299,0.0001147863,0.4577691,0.0004210507,0.00008909657,0.0002504849,0.01471376,0.5206503,0.001875192],"genre_scores_gemma":[0.07203792,0.0005518846,0.830804,0.0008915685,0.0001317633,0.001220897,0.05303964,0.0323947,0.008927695],"genre_candidate":"software","genre_consensus":"software","teacher_disagreement_score":0.04132002,"threshold_uncertainty_score":0.1382293,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2148271643","doi":"10.1093/bioinformatics/btv060","title":"Computational identification of MoRFs in protein sequences","year":2015,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":99,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; Genome Canada","keywords":"Computer science; Support vector machine; Identification (biology); Sequence (biology); Artificial intelligence; Computational biology; Data mining; Set (abstract data type); Machine learning; Pattern recognition (psychology); Biology; Genetics","authors":[{"name":"Nawar Malhis","is_ca":true},{"name":"Jörg Gsponer","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01588503249904103,"gpt":0.2752276049529452,"spread":0.2593425724539041,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001026727,0.0008362245,0.001058475,0.001815945,0.000488129,0.0006394546,0.001209546,0.0009148263,0.002092167],"category_scores_gemma":[0.003998599,0.0003181357,0.0005975591,0.001362178,0.0003943637,0.001088666,0.0007414183,0.000853813,0.0007231957],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006756626,"about_ca_system_score_gemma":0.001399618,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004242516,"about_ca_topic_score_gemma":0.005937238,"domain_scores_codex":[0.9995797,0.00008469967,0.00002523467,0.0001321349,0.0001303046,0.00004781649],"domain_scores_gemma":[0.9978234,0.001386467,0.0002232407,0.0001475599,0.0002757935,0.0001435745],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008525687,0.0003622669,0.02954369,0.0006587738,0.0002199257,0.0004074811,0.00007990881,0.7600415,0.008196912,0.003346607,0.01270918,0.1835813],"study_design_scores_gemma":[0.00001080452,0.00003034285,0.0007392006,0.000005039315,0.000006432014,0.00003127104,0.000009251903,0.9970336,0.0008836264,0.0009864133,0.0002608873,0.00000306373],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7194423,0.002254302,0.2567613,0.0008333416,0.0001204975,0.0002219871,0.003997831,0.01254426,0.00382435],"genre_scores_gemma":[0.713502,0.0005899885,0.2706887,0.0002319986,0.0001042991,0.0002912724,0.01222129,0.0003929282,0.001977438],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004242516,"threshold_uncertainty_score":0.008435607,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2076460489","doi":"10.1186/1471-2105-10-414","title":"Modular prediction of protein structural classes from sequences of twilight-zone identity with predicting sequences","year":2009,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":96,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Artificial intelligence; Support vector machine; Modular design; Similarity (geometry); Feature selection; Feature (linguistics); Pattern recognition (psychology); Class (philosophy); Sequence (biology); Contrast (vision); Data mining; Machine learning; Biology; Image (mathematics)","authors":[{"name":"Marcin J. Mizianty","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01020080503328411,"gpt":0.2450921923098636,"spread":0.2348913872765795,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006842242,0.0007177558,0.0004384232,0.001337361,0.0002710791,0.0005562503,0.0006225752,0.0004578147,0.001415195],"category_scores_gemma":[0.00158909,0.0001217889,0.0006609767,0.0007612478,0.0002071482,0.0005808901,0.0005053002,0.0007411414,0.0009402703],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003408004,"about_ca_system_score_gemma":0.0007477426,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001794824,"about_ca_topic_score_gemma":0.002832456,"domain_scores_codex":[0.9996303,0.00004640136,0.00002503166,0.0001223527,0.0001269734,0.00004883596],"domain_scores_gemma":[0.9989783,0.0003008709,0.0002262055,0.0001134323,0.0002976964,0.00008356666],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001225787,0.0006638173,0.1554181,0.0003696885,0.000221235,0.0002547988,0.0001340525,0.08986264,0.0982464,0.001359177,0.01152406,0.6407202],"study_design_scores_gemma":[0.00004179334,0.0003963003,0.03597361,0.00004236151,0.00006981428,0.0002999974,0.00008138883,0.9179043,0.03967698,0.00211754,0.003369877,0.00002615425],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7948865,0.0008885558,0.1947156,0.0002659296,0.00006936587,0.000179908,0.002471041,0.003818751,0.002704369],"genre_scores_gemma":[0.8877664,0.0002118871,0.1034728,0.00009911836,0.00005609793,0.0001305548,0.006867981,0.00008492981,0.001310283],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.001794824,"threshold_uncertainty_score":0.004734278,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2132581719","doi":"10.1002/jcc.21053","title":"Prediction of integral membrane protein type by collocated hydrophobic amino acid pairs","year":2008,"lang":"en","type":"article","venue":"Journal of Computational Chemistry","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":89,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Feature selection; Computer science; Classifier (UML); Benchmark (surveying); Transmembrane protein; Feature (linguistics); Transmembrane domain; Integral membrane protein; Sequence (biology); Artificial intelligence; Pattern recognition (psychology); Membrane protein; Algorithm; Biological system; Chemistry; Amino acid; Membrane; Biology; Biochemistry","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Yingfu Jiang","is_ca":true},{"name":"Li Du","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.008975504715840538,"gpt":0.2198560894298847,"spread":0.2108805847140442,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005915488,0.0007707547,0.0007310243,0.001812658,0.0003710793,0.0007565319,0.001076844,0.0008160275,0.001043829],"category_scores_gemma":[0.001860793,0.0002702601,0.00117949,0.001036372,0.0002442209,0.0008656502,0.000573266,0.0007150549,0.0007084294],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004591993,"about_ca_system_score_gemma":0.0007904334,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002514143,"about_ca_topic_score_gemma":0.002672057,"domain_scores_codex":[0.999724,0.00004272861,0.00002179084,0.00008840639,0.00008530703,0.00003770053],"domain_scores_gemma":[0.9991876,0.0003310197,0.0001271183,0.0000851791,0.0001946612,0.00007440856],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00160264,0.0006932634,0.1204137,0.0004283172,0.0003921225,0.00124343,0.0001719609,0.4458838,0.0457117,0.005304254,0.01078545,0.3673693],"study_design_scores_gemma":[0.000007041053,0.00003834923,0.001977654,0.000005341324,0.00001218603,0.0000775895,0.00001349401,0.9946979,0.001861225,0.001070536,0.0002308863,0.000007738542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5224594,0.0006673628,0.4687003,0.0003511239,0.0001047739,0.0001675844,0.002030554,0.003406042,0.002112941],"genre_scores_gemma":[0.8761757,0.00019797,0.1200513,0.0001035764,0.00003205618,0.00009937775,0.002383215,0.00009921886,0.0008575286],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002514143,"threshold_uncertainty_score":0.004998982,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2098676247","doi":"10.1093/nar/gkq1093","title":"PSORTdb--an expanded, auto-updated, user-friendly protein subcellular localization database for Bacteria and Archaea","year":2010,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":89,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Biology; Archaea; Bacteria; Database; Bacterial protein; Computational biology; Genetics; Computer science","authors":[{"name":"Nancy Yu","is_ca":true},{"name":"Matthew R. Laird","is_ca":true},{"name":"Careni Spencer","is_ca":true},{"name":"Fiona S. L. Brinkman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01996824358640841,"gpt":0.3300858543855556,"spread":0.3101176107991472,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001067865,0.001503525,0.001491704,0.002643042,0.0007611408,0.001930119,0.001987653,0.0009356698,0.009188908],"category_scores_gemma":[0.002333979,0.000829575,0.001283062,0.002988512,0.0002949971,0.002295111,0.002222097,0.001430326,0.01728842],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005058998,"about_ca_system_score_gemma":0.0017194,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001443511,"about_ca_topic_score_gemma":0.001544673,"domain_scores_codex":[0.9994641,0.00006990956,0.00009532168,0.0001458456,0.0001702008,0.00005458933],"domain_scores_gemma":[0.9990634,0.000152619,0.0001695523,0.0002412945,0.0002170633,0.0001560143],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.003860606,0.0003090762,0.01356005,0.009240792,0.0005904256,0.001463758,0.0005166789,0.005724353,0.1480461,0.01009858,0.5372591,0.2693304],"study_design_scores_gemma":[0.0003557982,0.0002796975,0.01025889,0.0006111195,0.0003058281,0.002836145,0.000217121,0.01776688,0.0556336,0.008315225,0.9032362,0.000183546],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"software","genre_scores_codex":[0.04146274,0.01008412,0.2155268,0.0008670209,0.00073534,0.000361406,0.5144919,0.202398,0.01407268],"genre_scores_gemma":[0.05087049,0.003578439,0.1119754,0.0003880128,0.0001023682,0.0003480864,0.8217097,0.006920855,0.004106681],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.009188908,"threshold_uncertainty_score":0.03074002,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2512143001","doi":"10.1186/s12859-016-1115-5","title":"Predicting essential proteins based on subcellular localization, orthology and PPI networks","year":2016,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":88,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Saskatchewan","funders":"National Natural Science Foundation of China","keywords":"Subcellular localization; Computational biology; DNA microarray; Protein subcellular localization prediction; Biology; Computer science; Genetics; Gene; Gene expression","authors":[{"name":"Gaoshi Li","is_ca":false},{"name":"Min Li","is_ca":false},{"name":"Jianxin Wang","is_ca":false},{"name":"Jingli Wu","is_ca":false},{"name":"Fang‐Xiang Wu","is_ca":true},{"name":"Yi Pan","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.00529630127268006,"gpt":0.2175242514484297,"spread":0.2122279501757496,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005434365,0.0008296157,0.0005953614,0.002626274,0.0002769056,0.0004269866,0.000420254,0.0003812287,0.0009166574],"category_scores_gemma":[0.002091159,0.0001596619,0.0006143677,0.001355953,0.0002482535,0.0007270367,0.0004531883,0.0004042171,0.0003407276],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003568228,"about_ca_system_score_gemma":0.0005358446,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001295808,"about_ca_topic_score_gemma":0.001739669,"domain_scores_codex":[0.9996265,0.00007819936,0.00003675821,0.0001094282,0.000121641,0.00002742481],"domain_scores_gemma":[0.9989694,0.0005166442,0.0002122874,0.00006172563,0.0001685614,0.00007137364],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001315193,0.0005925085,0.220141,0.001934161,0.0008127948,0.002061316,0.0002252725,0.2292921,0.1597647,0.008161784,0.01106694,0.3646323],"study_design_scores_gemma":[0.00003164044,0.00009685208,0.02970038,0.00003071105,0.0001443353,0.0009083676,0.00006248632,0.9394454,0.02122538,0.006021467,0.002300118,0.00003279172],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5338508,0.002697301,0.4533421,0.0003538517,0.000058839,0.0001316825,0.003921745,0.002384238,0.003259347],"genre_scores_gemma":[0.8717709,0.0008379729,0.1216105,0.0000542128,0.00003402073,0.00006612365,0.005024604,0.00007426389,0.0005273764],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002626274,"threshold_uncertainty_score":0.00306654,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2104332903","doi":"10.1016/j.bbrc.2007.03.164","title":"Prediction of protein structural class for the twilight zone sequences","year":2007,"lang":"en","type":"article","venue":"Biochemical and Biophysical Research Communications","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":87,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Representation (politics); Computer science; In silico; Classifier (UML); Artificial intelligence; Protein structure prediction; Pattern recognition (psychology); Protein structure; Biology","authors":[{"name":"Lukasz Kurgan","is_ca":true},{"name":"Ke Chen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06387065908451027,"gpt":0.3770602085118304,"spread":0.3131895494273201,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001676184,0.0004249446,0.0004756378,0.001745312,0.0005578814,0.0004215225,0.0003778426,0.0004801919,0.002929665],"category_scores_gemma":[0.0006065507,0.0001362828,0.0007095477,0.000558596,0.0001858921,0.0002824848,0.0001681686,0.0006579837,0.001559267],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004967727,"about_ca_system_score_gemma":0.000415739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004386523,"about_ca_topic_score_gemma":0.005522213,"domain_scores_codex":[0.9999236,0.000007066187,0.000004591211,0.00002922291,0.00001710866,0.00001825915],"domain_scores_gemma":[0.9995722,0.0001135969,0.00009198956,0.00003648318,0.00008510549,0.0001005843],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004169946,0.0003987858,0.0757567,0.0003891429,0.0001388531,0.0008995926,0.0002520268,0.01940545,0.737269,0.002929701,0.006758411,0.1516324],"study_design_scores_gemma":[0.0002912613,0.0008614772,0.1979001,0.00006226816,0.0002340504,0.001438098,0.0003572599,0.6041389,0.1794883,0.004744072,0.01039778,0.00008647577],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9688488,0.0003308531,0.02516736,0.0001063431,0.00002687651,0.00005387068,0.002825789,0.001280756,0.001359325],"genre_scores_gemma":[0.9775404,0.0001397847,0.014679,0.00002810342,0.00002039226,0.00002908747,0.006189083,0.0001464082,0.00122766],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004386523,"threshold_uncertainty_score":0.009800673,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2962655813","doi":"10.1093/bioinformatics/btz337","title":"Prediction of mRNA subcellular localization using deep recurrent neural networks","year":2019,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":86,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Montreal Clinical Research Institute; McGill University","funders":"Institut de Valorisation des Données","keywords":"Computational biology; Sequence (biology); Computer science; Identification (biology); RNA; Subcellular localization; Variety (cybernetics); Messenger RNA; Biology; Set (abstract data type); Deep learning; RNA-binding protein; Artificial neural network; Artificial intelligence; Gene; Genetics","authors":[{"name":"Zichao Yan","is_ca":true},{"name":"Éric Lécuyer","is_ca":true},{"name":"Mathieu Blanchette","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01212322715081878,"gpt":0.2315241812715117,"spread":0.2194009541206929,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003727928,0.0008370242,0.000469627,0.0006033487,0.0001727825,0.0005552663,0.0007384542,0.0005855793,0.001440438],"category_scores_gemma":[0.001164267,0.0002975721,0.000432448,0.0004753293,0.0002692344,0.0005182988,0.0003690749,0.0008988288,0.0009705163],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008321742,"about_ca_system_score_gemma":0.0005785446,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006031896,"about_ca_topic_score_gemma":0.008377201,"domain_scores_codex":[0.999853,0.00002209469,0.000007199414,0.00006048161,0.00003052201,0.00002670046],"domain_scores_gemma":[0.9995516,0.0001934392,0.00007420644,0.00003228538,0.0001217832,0.00002669524],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005099013,0.0001539732,0.007303181,0.0002978133,0.0001063858,0.0003406011,0.00004564533,0.7865116,0.04999255,0.004663385,0.007656909,0.142418],"study_design_scores_gemma":[0.000005376842,0.00001266019,0.000370827,0.000005434814,0.000006380824,0.00001472288,0.000003707964,0.9928202,0.005260788,0.0012626,0.0002340179,0.000003326541],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2756405,0.00252247,0.7018111,0.0007476914,0.0001183981,0.00005873387,0.003960232,0.01113694,0.004004026],"genre_scores_gemma":[0.897635,0.0007645207,0.08994395,0.0001760483,0.0000484029,0.00006790537,0.006307368,0.0001944456,0.004862232],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006031896,"threshold_uncertainty_score":0.01199359,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2113944588","doi":"10.1093/bioinformatics/btn433","title":"Accurate sequence-based prediction of catalytic residues","year":2008,"lang":"en","type":"article","venue":"Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":85,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Nankai University","keywords":"Support vector machine; Curse of dimensionality; Sequence (biology); Computer science; Classifier (UML); Protein structure prediction; Feature selection; Artificial intelligence; Biological system; Data mining; Pattern recognition (psychology); Protein structure; Chemistry; Biology; Biochemistry","authors":[{"name":"Tuo Zhang","is_ca":true},{"name":"Hua Zhang","is_ca":true},{"name":"Ke Chen","is_ca":true},{"name":"Shiyi Shen","is_ca":true},{"name":"Jishou Ruan","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02821141345171289,"gpt":0.2624030186154391,"spread":0.2341916051637262,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012644,0.0009282973,0.0007739869,0.001279155,0.0002663598,0.0008248955,0.0006876886,0.000984329,0.00127031],"category_scores_gemma":[0.003775498,0.0002646203,0.0004586986,0.0008338963,0.0002762785,0.001141909,0.0004813195,0.0009375165,0.00190982],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003408814,"about_ca_system_score_gemma":0.0005245066,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001001528,"about_ca_topic_score_gemma":0.0009817609,"domain_scores_codex":[0.99914,0.0001581204,0.00008155651,0.000251814,0.0003145788,0.00005385392],"domain_scores_gemma":[0.9973908,0.001025018,0.000513437,0.0002741356,0.000697553,0.00009906089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008899393,0.0004321153,0.09294408,0.001790129,0.0002540501,0.0008233608,0.00009134696,0.1870845,0.2226762,0.0020636,0.01197492,0.4789757],"study_design_scores_gemma":[0.00004775807,0.0002276023,0.02225304,0.00008687928,0.00008470315,0.0008967205,0.00004387323,0.8575967,0.1083416,0.003860252,0.006517851,0.00004292096],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4861636,0.008682016,0.4890255,0.0005475204,0.0001817484,0.0001137328,0.004077362,0.006864098,0.004344356],"genre_scores_gemma":[0.8359466,0.001578549,0.1518073,0.0001607575,0.0001211217,0.00006311066,0.008609102,0.0001872746,0.001526286],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.001279155,"threshold_uncertainty_score":0.006686807,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2152856869","doi":"10.1186/1477-5956-9-s1-s4","title":"ATPsite: sequence-based prediction of ATP-binding residues","year":2011,"lang":"en","type":"article","venue":"Proteome Science","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":81,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Killam Trusts","keywords":"Support vector machine; Dihedral angle; Classifier (UML); Annotation; Computer science; Machine learning; Artificial intelligence; Matthews correlation coefficient; Sequence (biology); Sequence alignment; Computational biology; Data mining; Bioinformatics; Peptide sequence; Biology; Chemistry; Biochemistry; Gene","authors":[{"name":"Ke Chen","is_ca":true},{"name":"Marcin J. Mizianty","is_ca":true},{"name":"Lukasz Kurgan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03824075508694257,"gpt":0.2718583389464778,"spread":0.2336175838595352,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000717738,0.0007652919,0.0007250134,0.001757716,0.000276233,0.0005572676,0.000687269,0.0007454057,0.003243256],"category_scores_gemma":[0.002708156,0.0001766966,0.0004040874,0.001272928,0.0001820745,0.0005940481,0.0004630831,0.000631313,0.001626721],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002550952,"about_ca_system_score_gemma":0.0007399187,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001025334,"about_ca_topic_score_gemma":0.001209699,"domain_scores_codex":[0.9995843,0.00008030773,0.00003820014,0.0001098729,0.0001476574,0.00003961929],"domain_scores_gemma":[0.9988478,0.0004838385,0.0002155126,0.00006340204,0.0002731308,0.0001162431],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002411149,0.001075875,0.1113477,0.001671854,0.0004337225,0.0007029325,0.00009219641,0.1210836,0.07302421,0.003383446,0.04500187,0.6397714],"study_design_scores_gemma":[0.0001818406,0.0007250553,0.02279398,0.0000898435,0.0001229541,0.001199181,0.00005253496,0.9279488,0.03331295,0.00487275,0.008648394,0.00005180114],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5255632,0.005363471,0.4218576,0.0007814197,0.000349871,0.0004183979,0.01603948,0.02356679,0.006059838],"genre_scores_gemma":[0.8219122,0.0009535414,0.1569802,0.0001956645,0.000165162,0.0002853252,0.01713568,0.0002465272,0.002125776],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.003243256,"threshold_uncertainty_score":0.01084983,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2085102470","doi":"10.1186/1471-2164-15-50","title":"Prediction of bacterial type IV secreted effectors by C-terminal features","year":2014,"lang":"en","type":"article","venue":"BMC Genomics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":80,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Calgary","funders":"National Institutes of Health; National Natural Science Foundation of China","keywords":"Effector; Biology; Computational biology; Secretion; DNA microarray; Gene; Genetics; Gene expression; Cell biology; Biochemistry","authors":[{"name":"Yejun Wang","is_ca":false},{"name":"Xiaowei Wei","is_ca":false},{"name":"Hongxia Bao","is_ca":false},{"name":"Shu‐Lin Liu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006200473368906618,"gpt":0.2176059170049314,"spread":0.2114054436360248,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007030915,0.00137398,0.0005687437,0.001141801,0.0003260641,0.0007259716,0.0004455118,0.0009553605,0.001093047],"category_scores_gemma":[0.001079487,0.0002235805,0.001636178,0.0005495315,0.0001894963,0.0004899667,0.0003890403,0.0005697325,0.0008347247],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004370889,"about_ca_system_score_gemma":0.0006411297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001896025,"about_ca_topic_score_gemma":0.001706715,"domain_scores_codex":[0.9996828,0.00005024448,0.00003210554,0.0001108027,0.00007670697,0.00004737521],"domain_scores_gemma":[0.9992668,0.0002608878,0.0001410051,0.00004690871,0.0002087674,0.00007555387],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003513737,0.00106869,0.2755494,0.001396313,0.0006808605,0.002156274,0.0001730862,0.2769214,0.2862234,0.001021078,0.00741453,0.1438812],"study_design_scores_gemma":[0.00004122835,0.0003976341,0.02335875,0.00005120197,0.0001685997,0.0004675601,0.00006627873,0.9386659,0.03449278,0.0004542492,0.001799679,0.00003610392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9415634,0.001413928,0.04974208,0.0001550801,0.00005122472,0.0001250615,0.003242887,0.001927773,0.001778641],"genre_scores_gemma":[0.9398333,0.00035905,0.04933202,0.00009978439,0.00001878349,0.00007074072,0.009419871,0.0001148651,0.0007515686],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.001896025,"threshold_uncertainty_score":0.003769994,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1496257230","doi":"10.1186/s12859-015-0586-0","title":"ProtDCal: A program to compute general-purpose-numerical descriptors for sequences and 3D-structures of proteins","year":2015,"lang":"en","type":"article","venue":"BMC Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":80,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada; Universidad de San Buenaventura","keywords":"Computer science; Data mining; Sequence (biology); Java; Feature (linguistics); Sequence alignment; Software; Key (lock); Protein function prediction; Artificial intelligence; Theoretical computer science; Machine learning; Programming language; Protein function; Peptide sequence","authors":[{"name":"Yasser B. Ruiz‐Blanco","is_ca":true},{"name":"Waldo Paz","is_ca":false},{"name":"James R. Green","is_ca":true},{"name":"Yovani Marrero‐Ponce","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03172182749760137,"gpt":0.3074146386683912,"spread":0.2756928111707899,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000921294,0.00197396,0.001180804,0.001577127,0.0006821202,0.001611743,0.003393234,0.0008122323,0.01924598],"category_scores_gemma":[0.002821401,0.0008704077,0.001480144,0.001314862,0.0007749814,0.001300351,0.00181157,0.001913603,0.007015446],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009782778,"about_ca_system_score_gemma":0.002075069,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003987206,"about_ca_topic_score_gemma":0.004626581,"domain_scores_codex":[0.9995553,0.00005726512,0.00004025893,0.0001173375,0.0001846722,0.00004516701],"domain_scores_gemma":[0.9989989,0.0005514036,0.00009885585,0.0001161388,0.0001554156,0.00007927413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002949845,0.0007214478,0.01411512,0.00500753,0.0008516621,0.002281714,0.001183389,0.1249902,0.06044621,0.03737951,0.4020338,0.3480395],"study_design_scores_gemma":[0.0008164975,0.0002326472,0.003520774,0.0001695486,0.0001250432,0.0009654064,0.0001664942,0.7871431,0.04317873,0.01579446,0.147713,0.0001742181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.03212856,0.0005221481,0.602225,0.0003423773,0.0001707876,0.0005495641,0.02817761,0.3256285,0.01025536],"genre_scores_gemma":[0.1552056,0.0009853867,0.7140734,0.0005010827,0.00007174152,0.003373537,0.07688837,0.03901942,0.009881601],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.01924598,"threshold_uncertainty_score":0.06438422,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2143507938","doi":"10.1145/956750.956800","title":"Frequent-subsequence-based prediction of outer membrane proteins","year":2003,"lang":"en","type":"article","venue":"","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":78,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Bacterial outer membrane; Subsequence; Support vector machine; Computational biology; Classifier (UML); Bacteria; Membrane protein; Computer science; Artificial intelligence; Biology; Bioinformatics; Machine learning; Biochemistry; Membrane; Mathematics; Gene; Genetics","authors":[{"name":"Rong She","is_ca":true},{"name":"Fei Chen","is_ca":true},{"name":"Ke Wang","is_ca":true},{"name":"Martin Ester","is_ca":true},{"name":"Jennifer L. Gardy","is_ca":true},{"name":"Fiona S. L. Brinkman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01055053056386303,"gpt":0.2280865789526886,"spread":0.2175360483888255,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007570116,0.0002688479,0.0004787739,0.002157217,0.0002720866,0.0003716594,0.0005779135,0.0005645797,0.000428934],"category_scores_gemma":[0.004499279,0.0001444457,0.0004412149,0.001034336,0.0002236632,0.0005768811,0.0002412107,0.0003186296,0.0003288575],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002093,"about_ca_system_score_gemma":0.0003182085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001454188,"about_ca_topic_score_gemma":0.001453393,"domain_scores_codex":[0.9994749,0.0001367349,0.00007114391,0.00008282041,0.0001876317,0.00004668593],"domain_scores_gemma":[0.9967465,0.001516326,0.0006229189,0.0001913343,0.0007881195,0.0001348238],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001301809,0.0007014431,0.1390378,0.0006033724,0.0003096133,0.002296127,0.0002360176,0.1218702,0.1006946,0.003714238,0.004928889,0.624306],"study_design_scores_gemma":[0.00005214146,0.0002828505,0.03727489,0.00003553572,0.00006727025,0.00224059,0.0001005677,0.9238582,0.02811196,0.005562586,0.00238075,0.00003279835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8016004,0.001163159,0.1942507,0.0002964407,0.00008379407,0.00005836394,0.001039818,0.0007101436,0.0007971029],"genre_scores_gemma":[0.8900455,0.0002969351,0.1072872,0.00002639438,0.00004452462,0.0000264901,0.001773898,0.00002083134,0.0004782209],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002157217,"threshold_uncertainty_score":0.004003465,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2065795346","doi":"10.1093/nar/gkn619","title":"Protein networks markedly improve prediction of subcellular localization in multiple eukaryotic species","year":2008,"lang":"en","type":"article","venue":"Nucleic Acids Research","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":77,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Korea Advanced Institute of Science and Technology; National Chiao Tung University; National Institute of General Medical Sciences; McGill University","keywords":"Biology; Subcellular localization; Protein subcellular localization prediction; Computational biology; Genetics; Evolutionary biology; Gene","authors":[{"name":"KiYoung Lee","is_ca":false},{"name":"Han‐Yu Chuang","is_ca":false},{"name":"Andreas Beyer","is_ca":false},{"name":"Min Kyung Sung","is_ca":false},{"name":"Won‐Ki Huh","is_ca":false},{"name":"Bong‐Hee Lee","is_ca":false},{"name":"Trey Ideker","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02705196244913349,"gpt":0.2706103783399174,"spread":0.2435584158907839,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008989974,0.001000957,0.000664409,0.002107412,0.0005101155,0.0007911179,0.0004489868,0.0007459966,0.0009113351],"category_scores_gemma":[0.003089293,0.0003808322,0.0005617052,0.0009608571,0.0003181251,0.001364521,0.0008107036,0.0007228702,0.0005487247],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007585204,"about_ca_system_score_gemma":0.0003444945,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006504895,"about_ca_topic_score_gemma":0.007850714,"domain_scores_codex":[0.9996031,0.0001109806,0.00002033739,0.000154431,0.00007842015,0.00003273771],"domain_scores_gemma":[0.9986684,0.000796578,0.0002107787,0.0001275782,0.0001400951,0.00005652305],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006421605,0.0001925565,0.06145479,0.0001916504,0.00041649,0.0003283694,0.0001299002,0.7654098,0.01202959,0.004097539,0.003561584,0.1515456],"study_design_scores_gemma":[0.000008167505,0.00001815794,0.004328523,0.000009567403,0.00003366997,0.00005505892,0.00001473932,0.9890345,0.001533403,0.004120959,0.0008359575,0.000007335758],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6482806,0.003622139,0.3362479,0.0008645438,0.00008779705,0.00007271789,0.002121083,0.003783789,0.004919414],"genre_scores_gemma":[0.9283865,0.0008143542,0.06683093,0.00007658512,0.00004143148,0.00003059514,0.002303554,0.0001392453,0.001376859],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006504895,"threshold_uncertainty_score":0.01293409,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4385373960","doi":"10.1093/bib/bbad289","title":"Artificial intelligence-aided protein engineering: from topological data analysis to deep protein language models","year":2023,"lang":"en","type":"review","venue":"Briefings in Bioinformatics","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":77,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"National Institute of General Medical Sciences; Bristol-Myers Squibb Canada; Michigan State University Foundation; Nuclear Safety and Security Commission; Bristol-Myers Squibb; National Aeronautics and Space Administration; National Institutes of Health; National Science Foundation; Michigan Economic Development Corporation; National Institute of Allergy and Infectious Diseases; Pfizer","keywords":"Computer science; Artificial intelligence; Topological data analysis; Protein engineering; Natural language processing; Computational biology; Biology; Algorithm","authors":[{"name":"Yuchi Qiu","is_ca":false},{"name":"Guo‐Wei Wei","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08793799256485617,"gpt":0.3524052727944185,"spread":0.2644672802295623,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007436334,0.0009120913,0.001144087,0.00187125,0.00016079,0.001256672,0.0009992155,0.0008413158,0.001950429],"category_scores_gemma":[0.001368099,0.0003902685,0.000728522,0.002294444,0.0005658344,0.001872399,0.0008027251,0.001763023,0.001424729],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005652983,"about_ca_system_score_gemma":0.0009136552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009474665,"about_ca_topic_score_gemma":0.001092422,"domain_scores_codex":[0.9997858,0.00005140922,0.00002309306,0.00004527356,0.00007882487,0.00001557184],"domain_scores_gemma":[0.9995577,0.0003068497,0.00003340074,0.00001682485,0.00007016646,0.00001502811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00002317143,0.00005023435,0.0002132126,0.01003546,0.0001261274,0.0001097203,0.00006267573,0.003287839,0.001844703,0.02525062,0.0146375,0.9443587],"study_design_scores_gemma":[0.00002073632,0.0001335717,0.0007429011,0.005124612,0.0002118743,0.001015108,0.00008395471,0.009004879,0.003863022,0.04139035,0.9383099,0.00009906935],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0004735503,0.9789027,0.01643542,0.0008988417,0.0002969351,0.00001949605,0.00009054628,0.0001132566,0.002769202],"genre_scores_gemma":[0.003221398,0.9865887,0.008710716,0.0002507852,0.0002103065,0.00002648275,0.0001405272,0.0000162224,0.0008349867],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.001950429,"threshold_uncertainty_score":0.006524861,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3083386021","doi":"10.1101/2020.09.04.283929","title":"Self-Supervised Contrastive Learning of Protein Representations By Mutual Information Maximization","year":2020,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":77,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Vector Institute; Microsoft Research","keywords":"Maximization; Computer science; Mutual information; Embedding; Artificial intelligence; Hyperparameter; Machine learning; Autoregressive model; Supervised learning; Contrast (vision); Mathematics; Artificial neural network","authors":[{"name":"Amy X. Lu","is_ca":false},{"name":"Haoran Zhang","is_ca":true},{"name":"Marzyeh Ghassemi","is_ca":true},{"name":"Alan M. Moses","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.006008598428195527,"gpt":0.2124216015829576,"spread":0.2064130031547621,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003159846,0.001133337,0.0009714942,0.0008730881,0.0003096903,0.0009124902,0.001750964,0.001430266,0.001366701],"category_scores_gemma":[0.008539037,0.0005770348,0.0008256492,0.0005691512,0.001438647,0.002603747,0.001934282,0.002555206,0.0006461552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001005153,"about_ca_system_score_gemma":0.0006615449,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001078836,"about_ca_topic_score_gemma":0.001721998,"domain_scores_codex":[0.9988355,0.0005771745,0.00004308911,0.0003187172,0.0001551657,0.00007033678],"domain_scores_gemma":[0.9958935,0.002345396,0.0004868245,0.000661278,0.0004705142,0.0001424282],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004196092,0.0003378007,0.003021737,0.0001808852,0.0002056272,0.00009682365,0.0001403373,0.8033014,0.02317915,0.01298592,0.004549006,0.1515817],"study_design_scores_gemma":[0.000004876087,0.0000249706,0.0001035911,0.00000321314,0.000003044148,0.000008938337,0.00000291423,0.9945164,0.002566768,0.002658461,0.0001031858,0.000003587157],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09388607,0.0002915922,0.9022034,0.0004281815,0.00004310006,0.00007635884,0.0001812703,0.001644478,0.001245494],"genre_scores_gemma":[0.7554861,0.0001601934,0.2398486,0.0003168168,0.00008519956,0.000175701,0.0008228024,0.0003548664,0.002749655],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003159846,"threshold_uncertainty_score":0.01671112,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2074156599","doi":"10.1142/9789812776136_0058","title":"EPILOC: A (WORKING) TEXT-BASED SYSTEM FOR PREDICTING PROTEIN SUBCELLULAR LOCATION","year":2007,"lang":"en","type":"article","venue":"","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":74,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Artificial intelligence","authors":[{"name":"Scott T. Brady","is_ca":true},{"name":"Hagit Shatkay","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.00840131759235885,"gpt":0.2408423958874599,"spread":0.2324410782951011,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001428055,0.001418799,0.000837522,0.003900977,0.0005439436,0.001026588,0.001905611,0.001633014,0.01390088],"category_scores_gemma":[0.006683664,0.0003582867,0.000489218,0.001791805,0.0003042168,0.003556576,0.001435335,0.0006496322,0.01102758],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000503481,"about_ca_system_score_gemma":0.0005725559,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001337845,"about_ca_topic_score_gemma":0.001597109,"domain_scores_codex":[0.9994043,0.0001093395,0.00006550465,0.0002463135,0.000138334,0.00003616365],"domain_scores_gemma":[0.9961621,0.00182625,0.0003881508,0.0005700476,0.0007864819,0.0002669587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002901646,0.0004145176,0.007581891,0.001186388,0.0001521001,0.0009072851,0.0003471224,0.007344781,0.05890078,0.003792528,0.2248924,0.6915786],"study_design_scores_gemma":[0.000372151,0.001003759,0.01329862,0.0002049236,0.0002105259,0.001922201,0.0003647408,0.68564,0.1415402,0.01559488,0.1396109,0.0002370236],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06635587,0.001109898,0.4814332,0.001088301,0.0004614707,0.0005236588,0.05215211,0.3885042,0.008371226],"genre_scores_gemma":[0.2257006,0.0006476839,0.6535019,0.001042064,0.0003850118,0.0007415008,0.09751274,0.004517416,0.01595101],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01390088,"threshold_uncertainty_score":0.04650307,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}