{"id":"W3117879109","doi":"10.1039/d0qo01636e","title":"Data augmentation and transfer learning strategies for reaction prediction in low chemical data regimes","year":2021,"lang":"en","type":"article","venue":"Organic Chemistry Frontiers","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":71,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Natural Science Foundation of China","keywords":"Chemistry; Drug discovery; Transfer of learning; Biochemical engineering; Data science; Artificial intelligence; Computer science; Biochemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00157189,0.0009662592,0.0008168669,0.0008508526,0.0004181973,0.0007175896,0.001769802,0.00110798,0.002320368],"category_scores_gemma":[0.005415009,0.0004054004,0.0008161009,0.0009159423,0.0008816015,0.002376597,0.001664081,0.002359737,0.000826447],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006852112,"about_ca_system_score_gemma":0.001036596,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002789831,"about_ca_topic_score_gemma":0.002679442,"domain_scores_codex":[0.9996626,0.00009874612,0.00002800003,0.0000857208,0.00008722221,0.00003768265],"domain_scores_gemma":[0.9978254,0.001292575,0.0001568061,0.000309229,0.0003362593,0.0000797579],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003125207,0.0004542484,0.002741657,0.0001565836,0.00008205182,0.0001530723,0.0001198326,0.6357798,0.01238229,0.01263029,0.004082112,0.3311056],"study_design_scores_gemma":[0.000003101468,0.00001631111,0.00006801486,0.000002677496,0.000003130393,0.000005186379,0.000002774388,0.9949951,0.001396391,0.003303837,0.0002001655,0.00000323187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06553064,0.0009133788,0.9279392,0.000752538,0.0001491414,0.00009877898,0.000324734,0.002137227,0.002154191],"genre_scores_gemma":[0.7692428,0.000721187,0.2229794,0.0003274422,0.0002100764,0.0003966536,0.001230123,0.0002000999,0.004692222],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002789831,"threshold_uncertainty_score":0.00831306,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01940198101854031,"score_gpt":0.2656183214908371,"score_spread":0.2462163404722968,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}