{"id":"W4311710514","doi":"10.26434/chemrxiv-2022-fh0t2","title":"An Unsupervised Machine Learning Workflow for Assigning and Predicting Generality in Asymmetric Catalysis","year":2022,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; University of British Columbia; Mitacs","keywords":"Generality; Workflow; Catalysis; Computer science; Identification (biology); Chemistry; Machine learning; Artificial intelligence; Organic chemistry; Database","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004096839,0.001831299,0.001267073,0.00277572,0.001259356,0.00239888,0.002777328,0.001290156,0.003884974],"category_scores_gemma":[0.01053663,0.0007578355,0.002205033,0.001870387,0.0008672543,0.001331317,0.001658521,0.002890491,0.00271765],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001675358,"about_ca_system_score_gemma":0.00418107,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006975039,"about_ca_topic_score_gemma":0.01301081,"domain_scores_codex":[0.9976986,0.0004340977,0.0002627096,0.0008918314,0.0005755266,0.0001371852],"domain_scores_gemma":[0.9954906,0.002098644,0.0004085249,0.0007991887,0.001059458,0.0001436013],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004620359,0.0006674965,0.01494101,0.0007573077,0.0004381243,0.0005112803,0.0005941133,0.2412847,0.03972997,0.02361229,0.02173596,0.6552657],"study_design_scores_gemma":[0.00004865306,0.00007451927,0.001703074,0.00004075316,0.00004651042,0.00009742381,0.00006587022,0.9268901,0.02608313,0.0363895,0.008499705,0.00006065664],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009399232,0.00009539562,0.9679409,0.0001690663,0.0000350025,0.0002724458,0.002615087,0.01798615,0.001486571],"genre_scores_gemma":[0.06093022,0.00008312208,0.9316835,0.0001287904,0.00002433052,0.0006673083,0.004564552,0.0006828628,0.001235228],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006975039,"threshold_uncertainty_score":0.02166641,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02551712177798919,"score_gpt":0.2916732777207035,"score_spread":0.2661561559427143,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}