{"id":"W4293519117","doi":"10.1109/cibcb55180.2022.9863026","title":"TooT-BERT-M: Discriminating Membrane Proteins from Non-Membrane Proteins using a BERT Representation of Protein Primary Sequences","year":2022,"lang":"en","type":"article","venue":"","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Membrane protein; Membrane; Representation (politics); Vesicle-associated membrane protein 8; Cell membrane; Encoder; Chemistry; Computer science; Computational biology; Biophysics; Biochemistry; Artificial intelligence; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004020887,0.0007269599,0.0003663275,0.0008353664,0.0002683836,0.0007388812,0.0005544131,0.0006232095,0.001640253],"category_scores_gemma":[0.001452286,0.0001727461,0.0005299324,0.0004657377,0.0003036535,0.001260569,0.0007498072,0.0007712509,0.001101197],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000360049,"about_ca_system_score_gemma":0.000480005,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001278998,"about_ca_topic_score_gemma":0.002227259,"domain_scores_codex":[0.9998387,0.00003288162,0.000008390096,0.00003895609,0.00004378748,0.00003735358],"domain_scores_gemma":[0.9996246,0.0001185636,0.00006017687,0.00006917438,0.00008095252,0.00004653742],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001349587,0.0003544118,0.01614039,0.0003477846,0.0001425696,0.0006876626,0.0002968539,0.0500335,0.3053655,0.01840932,0.01487434,0.5919982],"study_design_scores_gemma":[0.000033174,0.0003196853,0.004774432,0.00004058946,0.00005229171,0.0008685796,0.0002029148,0.8861228,0.08514275,0.0149536,0.007436723,0.00005245608],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2872622,0.0008714867,0.6978573,0.0003649361,0.0001719625,0.0001183566,0.001203109,0.007207832,0.00494275],"genre_scores_gemma":[0.8062891,0.0004896318,0.185346,0.000175476,0.000044171,0.00007263513,0.003040556,0.0003931662,0.004149155],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001640253,"threshold_uncertainty_score":0.005487204,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01573214362521723,"score_gpt":0.2700129842050316,"score_spread":0.2542808405798144,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}