{"id":"W4307415438","doi":"10.48550/arxiv.2210.13597","title":"A critical examination of robustness and generalizability of machine learning prediction of materials properties","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Natural Resources Canada; University of Toronto","funders":"","keywords":"Generalizability theory; Robustness (evolution); Computer science; Machine learning; Artificial intelligence; Benchmark (surveying); Artificial neural network; Feature engineering; Data mining; Feature vector; Test data; Deep learning; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03705357,0.002193397,0.001764105,0.002266721,0.001024381,0.002990273,0.002377211,0.002210452,0.001432196],"category_scores_gemma":[0.1062084,0.0006126384,0.001581653,0.001896939,0.003115576,0.005039641,0.003058554,0.004416499,0.001114558],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001510281,"about_ca_system_score_gemma":0.001262831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008414689,"about_ca_topic_score_gemma":0.004661741,"domain_scores_codex":[0.9849254,0.007527758,0.0008276539,0.003469403,0.002678963,0.0005707169],"domain_scores_gemma":[0.934517,0.03959075,0.002431304,0.01956216,0.003336934,0.0005617983],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001171036,0.0002799142,0.07135481,0.0005446206,0.001226744,0.0002342805,0.0003609012,0.7509704,0.007027042,0.009061138,0.01224799,0.1455212],"study_design_scores_gemma":[0.0000328786,0.0003068467,0.01286742,0.00009922479,0.00009109894,0.0001128553,0.0001789111,0.955045,0.009190285,0.01916153,0.00285933,0.00005464512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6948127,0.01095446,0.2532394,0.01285512,0.000751231,0.0003058531,0.005680013,0.007591102,0.01381013],"genre_scores_gemma":[0.9732726,0.0007674795,0.01937424,0.0005902078,0.00020128,0.00007413098,0.004414165,0.0003419751,0.0009638843],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9629464,"threshold_uncertainty_score":0.1959603,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06758512803701792,"score_gpt":0.2070397022359092,"score_spread":0.1394545741988913,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}