{"id":"W4416054943","doi":"10.48550/arxiv.2510.18033","title":"Benchmarking 34 OpenKIM Nickel Potentials with an Emphasis on Surfaces and Extended Defects","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Benchmarking; Interatomic potential; Suite; Nickel; Ab initio; Lattice (music); Surface (topology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","insufficient_payload"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.005877037,0.001725815,0.002095229,0.0005738175,0.001799624,0.002461573,0.003271585,0.0007858013,0.003044501],"category_scores_gemma":[0.001023266,0.001491442,0.0002264404,0.0008117695,0.00145334,0.001218219,0.003711129,0.001256626,0.0004573725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003004626,"about_ca_system_score_gemma":0.000978483,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002133638,"about_ca_topic_score_gemma":0.0004266922,"domain_scores_codex":[0.987325,0.002621781,0.001693577,0.00479971,0.001679554,0.001880337],"domain_scores_gemma":[0.9931688,0.0008326346,0.001736784,0.003069298,0.0004725021,0.0007199578],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.001909056,0.001138078,0.3884757,0.002099456,0.0002340103,0.0003140153,0.002565355,0.05009405,0.5462645,0.000897076,0.0001439315,0.005864802],"study_design_scores_gemma":[0.002802559,0.004164149,0.785884,0.005739428,0.0007591534,0.0001393838,0.0008226209,0.01445133,0.1789863,0.001075777,0.0012514,0.003924008],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9878851,0.0004432032,0.001121599,0.0005215626,0.004745181,0.00196788,0.0001851665,0.0003276569,0.002802663],"genre_scores_gemma":[0.9819943,0.0006479006,0.01405897,0.0006301088,0.0004966711,0.0001731501,0.00007580205,0.0001095664,0.001813564],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3974082,"threshold_uncertainty_score":0.9995488,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03071310445735917,"score_gpt":0.3041654009281402,"score_spread":0.273452296470781,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}