{"id":"W2155675444","doi":"10.1186/1471-2164-15-1091","title":"Categorizer: a tool to categorize genes into user-defined biological groups based on semantic similarity","year":2014,"lang":"en","type":"article","venue":"BMC Genomics","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"National Research Foundation of Korea; Natural Sciences and Engineering Research Council of Canada; Chung-Ang University; National Research Foundation","keywords":"Categorization; Semantic similarity; Similarity (geometry); Context (archaeology); Computer science; Gene ontology; Ontology; Semantics (computer science); Biological data; Information retrieval; Gene; Computational biology; Natural language processing; Biology; Artificial intelligence; Bioinformatics; Genetics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004972751,0.0003381916,0.0003201044,0.00007043749,0.0001763245,0.00008206366,0.0004471452,0.0003745534,0.00001936009],"category_scores_gemma":[0.0001092026,0.0002985718,0.0001796008,0.0001195561,0.00006895125,0.000003901427,0.0002398081,0.0001596607,0.0001373836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006025355,"about_ca_system_score_gemma":0.000136085,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000050693,"about_ca_topic_score_gemma":0.000196725,"domain_scores_codex":[0.998318,0.00009126623,0.0004399481,0.000530775,0.0001337379,0.0004862871],"domain_scores_gemma":[0.9987516,0.00005430141,0.0001248656,0.0007893887,0.0000760831,0.0002037877],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002270676,0.0007653135,0.02794939,0.0003411266,0.0002227553,0.000009170737,0.0005430307,0.07868757,0.8090745,0.02955437,0.01693531,0.03364682],"study_design_scores_gemma":[0.007737087,0.006766542,0.01854527,0.00006554321,0.0001959888,0.00004400406,0.0003330733,0.2077378,0.1636027,0.01997091,0.5703206,0.004680506],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8141608,0.0001526916,0.1838529,0.0003121147,0.0004081856,0.000451176,0.0000223854,0.00003854789,0.0006011695],"genre_scores_gemma":[0.9703366,0.00006155521,0.02556536,0.002934861,0.000620786,0.00005113743,0.0002515309,0.0000421867,0.0001359916],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6454718,"threshold_uncertainty_score":0.9999467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01544071469148741,"score_gpt":0.2287537617253164,"score_spread":0.213313047033829,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}