{"id":"W4402930836","doi":"10.26434/chemrxiv-2024-ctdm3","title":"Crash Testing Machine Learning Force Fields for Molecules, Materials, and Interfaces: Model Analysis in the TEA Challenge 2023","year":2024,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Vector Institute","funders":"","keywords":"Crash; Computer science; Nanotechnology; Materials science; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002998238,0.001202296,0.0008819237,0.0005691232,0.0009989034,0.0009139224,0.002631042,0.001712247,0.005940143],"category_scores_gemma":[0.01201084,0.0004353282,0.001224611,0.0006284365,0.0006892308,0.001415587,0.001892913,0.002364329,0.001459395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008195239,"about_ca_system_score_gemma":0.001039518,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00987591,"about_ca_topic_score_gemma":0.008900554,"domain_scores_codex":[0.9989461,0.0003843976,0.00005114569,0.0001574296,0.0003347887,0.0001262555],"domain_scores_gemma":[0.9964341,0.00212602,0.00009386698,0.0006264068,0.000583099,0.0001365116],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001188157,0.0009340883,0.01607617,0.0009399435,0.0003492467,0.0004155936,0.0004266521,0.784155,0.004851895,0.04034195,0.09612439,0.05419702],"study_design_scores_gemma":[0.0001300542,0.0002375836,0.001892878,0.00005226905,0.00001922678,0.00005845895,0.0001098651,0.9721501,0.005209043,0.007568101,0.01254339,0.00002892543],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8039458,0.002244796,0.1006155,0.004487799,0.001160563,0.0004971334,0.03772197,0.01313862,0.03618787],"genre_scores_gemma":[0.8333944,0.0006945923,0.09799706,0.0008608102,0.0001847071,0.0008619553,0.05543518,0.00312653,0.007444816],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00987591,"threshold_uncertainty_score":0.01987177,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03982651992374889,"score_gpt":0.2947916084681235,"score_spread":0.2549650885443746,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}