{"id":"W4317911177","doi":"10.1142/s021969132250045x","title":"Calibrating the classifier for protein family prediction with protein sequence using machine learning techniques: An empirical investigation","year":2023,"lang":"en","type":"article","venue":"International Journal of Wavelets Multiresolution and Information Processing","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Classifier (UML); Feature selection; Random forest; Artificial intelligence; Protein family; Computer science; Protein sequencing; Machine learning; Pattern recognition (psychology); Protein structure prediction; Computational biology; Data mining; Protein structure; Peptide sequence; Gene; Biology; Genetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007904071,0.0007230703,0.0007609947,0.002177471,0.0005461768,0.0009180474,0.0008212673,0.001141503,0.0009074805],"category_scores_gemma":[0.02728823,0.0001909173,0.0007443052,0.002041845,0.0005282726,0.001801351,0.0004296582,0.001229687,0.0005664536],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006302038,"about_ca_system_score_gemma":0.0006607138,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002780549,"about_ca_topic_score_gemma":0.001539326,"domain_scores_codex":[0.9966983,0.001397011,0.0002163767,0.000469933,0.001013677,0.0002045519],"domain_scores_gemma":[0.9663322,0.02790678,0.0008995903,0.001625892,0.003027037,0.0002085311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005901387,0.001038982,0.2352367,0.0004239028,0.0003521638,0.000419927,0.0004601798,0.1703414,0.003841979,0.002960107,0.005543977,0.5787904],"study_design_scores_gemma":[0.00002250345,0.000507444,0.05134539,0.0001185739,0.0001167125,0.0006358064,0.0005655969,0.9353369,0.005682434,0.002903476,0.00272286,0.00004226952],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8115396,0.00438336,0.1788578,0.0006564152,0.0001175966,0.0001987251,0.0005553497,0.0003810691,0.003310015],"genre_scores_gemma":[0.9428831,0.0008226587,0.05420007,0.0000561237,0.00003807801,0.0001044847,0.0009863467,0.00003667199,0.0008725565],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007904071,"threshold_uncertainty_score":0.04180121,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03711071404583471,"score_gpt":0.3202827510113445,"score_spread":0.2831720369655097,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}