{"id":"W3126098228","doi":"10.21203/rs.3.rs-91227/v1","title":"Unsupervised explainable AI for molecular evolutionary study of forty thousand SARS-CoV-2 genomes","year":2020,"lang":"en","type":"preprint","venue":"Research Square (Research Square)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Research Organization of Information and Systems; Institute of Genetics; Japan Agency for Medical Research and Development","keywords":"Genome; Clade; Unsupervised learning; Cluster analysis; Computational biology; Evolutionary biology; Rand index; Biology; Artificial intelligence; Computer science; Phylogenetics; Genetics; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.004953607,0.0007355236,0.001117256,0.0008389277,0.0009166646,0.0002158247,0.002136696,0.000772037,0.00002480999],"category_scores_gemma":[0.00153835,0.0007290463,0.000622705,0.0009044342,0.0007744427,0.000006504416,0.006349387,0.001955198,0.00002051793],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002694334,"about_ca_system_score_gemma":0.002064542,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001169027,"about_ca_topic_score_gemma":0.0003312465,"domain_scores_codex":[0.989606,0.002151727,0.0010379,0.002354235,0.002674214,0.00217594],"domain_scores_gemma":[0.9916146,0.0005317067,0.0002067132,0.002269452,0.004980541,0.0003969891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001980023,0.002387112,0.009484845,0.00357401,0.001518598,0.0001060075,0.002572917,0.001271795,0.9386859,0.0004647219,0.03563976,0.002314302],"study_design_scores_gemma":[0.01155763,0.03436393,0.01974821,0.0008945291,0.0002997528,0.000023114,0.02647371,0.004136072,0.6950682,0.03606619,0.16805,0.003318646],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9679289,0.01840959,0.001077625,0.001275857,0.0002253971,0.009263976,0.0009934048,0.00002504303,0.0008001426],"genre_scores_gemma":[0.9903285,0.003019597,0.001178339,0.00009359371,0.000699904,0.003559612,0.0006474378,0.0002129694,0.0002600759],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2436177,"threshold_uncertainty_score":0.9995161,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1253952233724103,"score_gpt":0.4252861608709842,"score_spread":0.2998909374985739,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}