{"id":"W4391380207","doi":"10.1101/2024.01.28.577662","title":"Predicting protein functions using positive-unlabeled ranking with ontology-based priors","year":2024,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canada Research Chairs; University of Toronto; University of New Brunswick","funders":"King Abdullah University of Science and Technology","keywords":"Computer science; Prior probability; Classifier (UML); Benchmark (surveying); Ranking (information retrieval); Artificial intelligence; Machine learning; Gene ontology; Ontology; Data mining; Function (biology); Pattern recognition (psychology); Bayesian probability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003604189,0.001475843,0.001626534,0.003277693,0.0008068783,0.00183157,0.002347184,0.002048175,0.00158919],"category_scores_gemma":[0.009547578,0.0004311425,0.001055981,0.001767518,0.001062985,0.002538239,0.00145884,0.00180255,0.001304728],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001480201,"about_ca_system_score_gemma":0.001364921,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004862966,"about_ca_topic_score_gemma":0.009193199,"domain_scores_codex":[0.9973036,0.0010089,0.00009863991,0.000749282,0.0006367861,0.0002027975],"domain_scores_gemma":[0.9936645,0.003344068,0.0005523309,0.001187601,0.0009415728,0.000309915],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001082707,0.001423012,0.02497775,0.0003729975,0.0002434459,0.00028816,0.0001306887,0.6352001,0.01618152,0.02098653,0.02865088,0.2704622],"study_design_scores_gemma":[0.00001736358,0.00003650016,0.000850259,0.000009759351,0.00001068187,0.00003725753,0.00001485924,0.9805986,0.002030492,0.01556637,0.0008149943,0.00001280476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2563955,0.001680696,0.7237408,0.001403268,0.0001546158,0.0001893653,0.004194886,0.006395703,0.005845101],"genre_scores_gemma":[0.7937521,0.0002735253,0.1916007,0.000459121,0.0002290432,0.000136042,0.009861009,0.0004254343,0.003262996],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004862966,"threshold_uncertainty_score":0.01906097,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007545592311798487,"score_gpt":0.2208447805446984,"score_spread":0.2132991882328999,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}