{"id":"W4362700002","doi":"10.1038/s41524-023-01012-9","title":"A critical examination of robustness and generalizability of machine learning prediction of materials properties","year":2023,"lang":"en","type":"article","venue":"npj Computational Materials","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":97,"is_retracted":false,"has_abstract":true,"ca_institutions":"Natural Resources Canada; University of Toronto; University of New Brunswick","funders":"Office of Energy Research and Development; Natural Resources Canada","keywords":"Generalizability theory; Robustness (evolution); Computer science; Machine learning; Artificial intelligence; Benchmark (surveying); Artificial neural network; Feature engineering; Data mining; Feature vector; Test data; Deep learning; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05267985,0.001597245,0.001608911,0.002268977,0.001118216,0.002719887,0.002668111,0.00220394,0.001533067],"category_scores_gemma":[0.1589807,0.0006246736,0.001352119,0.001277319,0.003585346,0.003911415,0.003159971,0.003595351,0.0005336298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001382109,"about_ca_system_score_gemma":0.001098537,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007793903,"about_ca_topic_score_gemma":0.003224839,"domain_scores_codex":[0.9821921,0.0113499,0.0008019372,0.002977889,0.002091327,0.0005868685],"domain_scores_gemma":[0.8646557,0.09100943,0.004659483,0.03301675,0.00579389,0.0008646783],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001020334,0.0002136724,0.04923753,0.0002657168,0.0009492263,0.0002188371,0.000332299,0.8667845,0.005590465,0.01109402,0.003923167,0.06037023],"study_design_scores_gemma":[0.00002945036,0.000191621,0.007340874,0.00005095034,0.00007195339,0.000054065,0.0001090582,0.974477,0.004412392,0.01246198,0.0007663991,0.00003420037],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7961022,0.002876446,0.180039,0.00649697,0.0003091321,0.0002475785,0.001457388,0.002703904,0.009767357],"genre_scores_gemma":[0.9897388,0.00015459,0.008344651,0.0002460119,0.00008593554,0.00004449222,0.0008679509,0.0001566625,0.0003610862],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9473202,"threshold_uncertainty_score":0.278601,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03174563863098726,"score_gpt":0.2746000237751183,"score_spread":0.2428543851441311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}