{"id":"W4396706898","doi":"10.1038/s41524-024-01259-w","title":"JARVIS-Leaderboard: a large scale benchmark of materials design methods","year":2024,"lang":"en","type":"article","venue":"npj Computational Materials","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; University of New Brunswick","funders":"National Institute of Standards and Technology; U.S. Department of Energy; National Science Foundation","keywords":"Benchmarking; Benchmark (surveying); Computer science; NIST; Computer engineering; Set (abstract data type); Theoretical computer science; Data science; Data mining; Computational science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01742776,0.004251007,0.002228146,0.004600065,0.001871522,0.003498679,0.0103835,0.003983136,0.01316456],"category_scores_gemma":[0.03368291,0.001459108,0.003254418,0.004550339,0.001966215,0.003963131,0.005940314,0.003671037,0.01002787],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001998784,"about_ca_system_score_gemma":0.003844565,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004813769,"about_ca_topic_score_gemma":0.006120083,"domain_scores_codex":[0.9870273,0.00435355,0.000876341,0.001761215,0.005268761,0.0007128697],"domain_scores_gemma":[0.9795813,0.008298005,0.000841997,0.005516059,0.004553382,0.001209351],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001599486,0.001626396,0.008893663,0.007909165,0.000998312,0.0004569123,0.0004573828,0.1566191,0.0106755,0.02185748,0.670683,0.1182236],"study_design_scores_gemma":[0.001518535,0.00123347,0.007705071,0.001136993,0.0002343846,0.0003239787,0.0003615467,0.539113,0.03303408,0.04156412,0.3734092,0.0003655933],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.1449928,0.01398639,0.2412772,0.005598412,0.005516992,0.00277832,0.1716832,0.3417428,0.07242385],"genre_scores_gemma":[0.1781672,0.002284832,0.3256934,0.001875969,0.000463048,0.003513115,0.4205473,0.0579331,0.009521931],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01742776,"threshold_uncertainty_score":0.09216791,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02412037443777965,"score_gpt":0.3478499076356072,"score_spread":0.3237295331978276,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}