{"id":"W3126098228","doi":"10.21203/rs.3.rs-91227/v1","title":"Unsupervised explainable AI for molecular evolutionary study of forty thousand SARS-CoV-2 genomes","year":2020,"lang":"en","type":"preprint","venue":"Research Square (Research Square)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Research Organization of Information and Systems; Institute of Genetics; Japan Agency for Medical Research and Development","keywords":"Genome; Clade; Unsupervised learning; Cluster analysis; Computational biology; Evolutionary biology; Rand index; Biology; Artificial intelligence; Computer science; Phylogenetics; Genetics; Gene","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005020124,0.0003438327,0.0002889592,0.001175496,0.0003373491,0.0004105087,0.0004628795,0.0004776496,0.001186972],"category_scores_gemma":[0.001943679,0.0001723644,0.0007978118,0.0007132174,0.0003594673,0.0004884271,0.0005008326,0.0005507342,0.0001506918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006779885,"about_ca_system_score_gemma":0.0005806471,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003629465,"about_ca_topic_score_gemma":0.003322881,"domain_scores_codex":[0.9998471,0.00004834081,0.000008430911,0.00004123515,0.00003679112,0.00001802836],"domain_scores_gemma":[0.9992307,0.0004587653,0.00009148291,0.00007553457,0.0001016851,0.00004173787],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003116958,0.0001565289,0.0277807,0.0002569478,0.0002032233,0.000269074,0.0002964809,0.7729954,0.0522774,0.01664306,0.001607908,0.1272015],"study_design_scores_gemma":[0.000004628871,0.00001213684,0.001916937,0.000001500256,0.000003585596,0.00001133634,0.00001253817,0.991327,0.001218926,0.005291669,0.0001958054,0.000003881919],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3823059,0.0002312778,0.6133119,0.0003006684,0.00003097537,0.00006552491,0.000834168,0.001507741,0.001411773],"genre_scores_gemma":[0.8111331,0.00008677325,0.186553,0.00003904535,0.00002509528,0.0001044606,0.001070823,0.00008849533,0.0008990916],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003629465,"threshold_uncertainty_score":0.007216692,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1253952233724103,"score_gpt":0.4252861608709842,"score_spread":0.2998909374985739,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}