{"id":"W2111255192","doi":"","title":"Applying Probabilistic Thematic Clustering for Classification in the TREC 2005 Genomics Track","year":2005,"lang":"en","type":"article","venue":"","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Artificial intelligence; Feature selection; Classifier (UML); Cluster analysis; Categorization; Naive Bayes classifier; Test set; Machine learning; Pattern recognition (psychology); Data mining; Natural language processing; Support vector machine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004163181,0.00008156052,0.00008557394,0.00002026737,0.00005172579,0.00002207995,0.0001906745,0.0001027851,0.000008487581],"category_scores_gemma":[0.000156383,0.00005236296,0.00004514717,0.00003929158,0.00005812281,0.000001679564,0.00003223645,0.00005411514,0.000005800438],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000206916,"about_ca_system_score_gemma":0.00002870475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000003249222,"about_ca_topic_score_gemma":0.0002915181,"domain_scores_codex":[0.9993494,0.00003223698,0.0001874479,0.0001882425,0.00005790589,0.0001847745],"domain_scores_gemma":[0.9996457,0.00006566936,0.0000437202,0.0002076898,0.00001422575,0.00002296039],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008276072,0.0002116651,0.0001586998,0.000122016,0.00002608194,4.774962e-7,0.0007081803,0.0005780101,0.1497093,0.0007791268,0.005832439,0.8417913],"study_design_scores_gemma":[0.002383894,0.0004839339,0.004151703,0.00006911525,0.00006275397,0.00005836393,0.007030721,0.1813063,0.01437316,0.002342661,0.7870707,0.000666711],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7428231,0.001348812,0.2331319,0.008557424,0.0001883855,0.0032992,0.00001772472,0.00008010155,0.01055338],"genre_scores_gemma":[0.9742374,0.0000453011,0.02367881,0.0005939296,0.000227415,0.0006061945,0.00003574635,0.00001022481,0.0005649772],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8411245,"threshold_uncertainty_score":0.21353,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05130768496816875,"score_gpt":0.302721482534822,"score_spread":0.2514137975666533,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}