{"id":"W3103745584","doi":"10.21203/rs.3.rs-106139/v1","title":"Unsupervised explainable artificial intelligence for molecular evolutionary studies of over forty thousand SARS-CoV-2 genomes","year":2020,"lang":"en","type":"preprint","venue":"Research Square (Research Square)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Research Organization of Information and Systems; Institute of Genetics; Japan Agency for Medical Research and Development","keywords":"Genome; Clade; Unsupervised learning; Cluster analysis; Rand index; Artificial intelligence; Computational biology; Coronavirus disease 2019 (COVID-19); Big data; Computer science; Evolutionary biology; Biology; Phylogenetics; Genetics; Data mining; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006086662,0.0003197189,0.0002792266,0.0009907938,0.0002901097,0.0004550436,0.0003796901,0.0004200834,0.000847979],"category_scores_gemma":[0.002316003,0.0001833566,0.0007446209,0.0006052026,0.0004126795,0.0005184,0.0005509722,0.0006098514,0.0001101649],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006856024,"about_ca_system_score_gemma":0.000557198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002251316,"about_ca_topic_score_gemma":0.002528256,"domain_scores_codex":[0.9998479,0.00005807838,0.000008919307,0.00003394425,0.00003924683,0.00001195444],"domain_scores_gemma":[0.9991997,0.0005277457,0.00009225059,0.00007191022,0.00007700082,0.00003153228],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001894069,0.0001207536,0.01887438,0.000240522,0.0001912947,0.0002042516,0.0002751745,0.8015921,0.03614515,0.02571447,0.001208572,0.115244],"study_design_scores_gemma":[0.000004211542,0.00001121742,0.0015302,0.000002122322,0.000004011399,0.00001033121,0.00001266625,0.9886156,0.001032253,0.008549383,0.0002246145,0.00000343031],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3205236,0.0002627116,0.6756384,0.000400192,0.00002838044,0.00006569403,0.0005387369,0.0010657,0.001476489],"genre_scores_gemma":[0.7577945,0.0001201757,0.2404664,0.00004859152,0.00002697486,0.0001108138,0.0006736187,0.00006754568,0.0006912129],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002251316,"threshold_uncertainty_score":0.004974425,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2364885434751194,"score_gpt":0.4606954340693958,"score_spread":0.2242068905942764,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}