{"id":"W3103745584","doi":"10.21203/rs.3.rs-106139/v1","title":"Unsupervised explainable artificial intelligence for molecular evolutionary studies of over forty thousand SARS-CoV-2 genomes","year":2020,"lang":"en","type":"preprint","venue":"Research Square (Research Square)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Research Organization of Information and Systems; Institute of Genetics; Japan Agency for Medical Research and Development","keywords":"Genome; Clade; Unsupervised learning; Cluster analysis; Rand index; Artificial intelligence; Computational biology; Coronavirus disease 2019 (COVID-19); Big data; Computer science; Evolutionary biology; Biology; Phylogenetics; Genetics; Data mining; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.005191866,0.0006917303,0.001105885,0.0007228573,0.0008402593,0.000166163,0.001712443,0.0007303776,0.00002515217],"category_scores_gemma":[0.003280077,0.0006712829,0.0007148643,0.0008440278,0.001696621,0.000006961552,0.005790486,0.001509712,0.00002177504],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003113874,"about_ca_system_score_gemma":0.001761324,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003399538,"about_ca_topic_score_gemma":0.000164763,"domain_scores_codex":[0.9911726,0.001424553,0.001148261,0.002049366,0.002193958,0.002011222],"domain_scores_gemma":[0.9917337,0.0009414196,0.0002308208,0.001668743,0.005119452,0.0003058797],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00145032,0.000514323,0.001020905,0.00519814,0.001419731,0.00005590704,0.001645361,0.001143133,0.9636273,0.003953393,0.0140969,0.0058746],"study_design_scores_gemma":[0.000645304,0.003880704,0.001024182,0.0006744899,0.00008377661,0.000006242412,0.006997168,0.002124632,0.8366466,0.1193321,0.02753377,0.001050994],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9054113,0.07999179,0.004731241,0.001686336,0.0003989807,0.00578497,0.00139927,0.00002701487,0.0005691208],"genre_scores_gemma":[0.9761264,0.01662551,0.003434515,0.00007183483,0.0009067116,0.002043791,0.0004817724,0.0001637775,0.0001456703],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1269807,"threshold_uncertainty_score":0.9995738,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2364885434751194,"score_gpt":0.4606954340693958,"score_spread":0.2242068905942764,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}