{"id":"W4378783088","doi":"10.1101/gr.277395.122","title":"Genealogical inference and more flexible sequence clustering using iterative-PopPUNK","year":2023,"lang":"en","type":"article","venue":"Genome Research","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"European and Developing Countries Clinical Trials Partnership; Youth Innovation Promotion Association; National Key Research and Development Program of China; Shanghai Rising-Star Program; Youth Innovation Promotion Association of the Chinese Academy of Sciences; Department for International Development; Medical Research Council; Chinese Academy of Sciences; Medical Research Council Canada; National Natural Science Foundation of China; European Commission","keywords":"Biology; Inference; Cluster analysis; Genome; Annotation; Population; Iterative method; Computational biology; Bacterial genome size; Data mining; Computer science; Genetics; Artificial intelligence; Algorithm; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005665815,0.0001197569,0.0001277367,0.0001111435,0.0003254262,0.00006975276,0.0001957942,0.00009295747,0.00001391111],"category_scores_gemma":[0.000104755,0.0001092321,0.00003180267,0.0003014222,0.000256563,0.000001030368,0.0008174056,0.0001396477,0.00001790606],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002256042,"about_ca_system_score_gemma":0.00008669882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005309279,"about_ca_topic_score_gemma":0.00001747782,"domain_scores_codex":[0.9986511,0.00009754499,0.0001397382,0.0004146015,0.0002025097,0.0004945057],"domain_scores_gemma":[0.9994177,0.00004194936,0.00002141699,0.0002574302,0.0001516249,0.0001098735],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.00001995504,0.000007596987,0.003965311,0.0000257322,0.00002523299,0.00001163539,0.0002906831,0.001051941,0.9928659,0.00005815904,0.00004256018,0.001635305],"study_design_scores_gemma":[0.002914865,0.002963463,0.4299351,0.0001591442,0.00005464444,0.0002880642,0.005047448,0.05611363,0.3126405,0.005488644,0.18176,0.002634458],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9969462,0.001564982,0.0005421365,0.000185738,0.00004333958,0.0001884345,0.00004746611,0.00000708275,0.0004746358],"genre_scores_gemma":[0.9956935,0.002050716,0.001406374,0.00004974874,0.0001820333,0.00002851676,0.00003721276,0.00001712834,0.0005347462],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6802253,"threshold_uncertainty_score":0.4454356,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2548564996911687,"score_gpt":0.4422091171988944,"score_spread":0.1873526175077257,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}