{"id":"W7083179284","doi":"10.18434/mds2-3172","title":"JARVIS-Leaderboard: Large Scale Benchmark of Materials Design Methods","year":2024,"lang":"en","type":"dataset","venue":"National Institute of Standards and Technology (NIST)","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Scale (ratio); Benchmark (surveying); Design methods; Stability (learning theory); Context (archaeology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003409283,0.005842265,0.002135937,0.005438958,0.001483197,0.003002934,0.007608127,0.004933755,0.03674486],"category_scores_gemma":[0.01362109,0.001339486,0.003632242,0.005733391,0.0008125048,0.001887467,0.002516295,0.002925765,0.04992159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002370308,"about_ca_system_score_gemma":0.004041955,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02842815,"about_ca_topic_score_gemma":0.06234431,"domain_scores_codex":[0.9966857,0.0008372823,0.0003389094,0.000762927,0.00106136,0.000313933],"domain_scores_gemma":[0.9933444,0.002826441,0.0003742396,0.001686764,0.001369632,0.0003985825],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001586935,0.0001146479,0.000805169,0.001139264,0.0001041133,0.00002922796,0.00002114015,0.003238719,0.0003002462,0.000979139,0.9871222,0.005987369],"study_design_scores_gemma":[0.001885116,0.0001496232,0.004900231,0.0006412657,0.0001874628,0.0001200076,0.0001411835,0.017031,0.003487615,0.009972788,0.961354,0.0001296032],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0008584345,0.0002837388,0.001169279,0.0001740522,0.00008992953,0.0000734798,0.9901416,0.005391346,0.00181819],"genre_scores_gemma":[0.00093079,0.00009088498,0.003108202,0.00008189229,0.000009679089,0.0001611702,0.9946036,0.0003563045,0.0006574428],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.03674486,"threshold_uncertainty_score":0.1229238,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01982016710292146,"score_gpt":0.3245723505298443,"score_spread":0.3047521834269228,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}