{"id":"W3202820157","doi":"","title":"An Extensible Benchmark Suite for Learning to Simulate Physical Systems","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Model Reduction and Neural Networks","field":"Physics and Astronomy","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Benchmark (surveying); Suite; Kernel (algebra); Physical system; Complement (music); Machine learning; Stability (learning theory); Distributed computing; Set (abstract data type); Artificial intelligence; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003751675,0.002949176,0.001162241,0.002065044,0.0007321161,0.001608143,0.005357827,0.002163303,0.006383722],"category_scores_gemma":[0.01147291,0.0007329607,0.001819849,0.003084515,0.0008985419,0.001575796,0.001830086,0.002507302,0.002238295],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001441813,"about_ca_system_score_gemma":0.00296387,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01272809,"about_ca_topic_score_gemma":0.01524098,"domain_scores_codex":[0.9981006,0.0005966217,0.0002528902,0.0002049813,0.0006613861,0.0001836281],"domain_scores_gemma":[0.9942989,0.003063177,0.0002835291,0.0009461204,0.001156658,0.0002515915],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003473956,0.0007752243,0.005996846,0.002414576,0.0004216478,0.0003651445,0.0001060351,0.8122094,0.002990825,0.0217228,0.09971171,0.05293826],"study_design_scores_gemma":[0.0002047122,0.0001673383,0.001298211,0.000114865,0.00004046614,0.0001185974,0.0000509107,0.9550157,0.003641435,0.0124635,0.02684359,0.00004070612],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.266229,0.008869497,0.4267595,0.004084356,0.002164651,0.003180909,0.1405583,0.07389199,0.0742619],"genre_scores_gemma":[0.3402987,0.003637232,0.4498399,0.001088418,0.0002721258,0.004503365,0.1804871,0.007448524,0.01242457],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01272809,"threshold_uncertainty_score":0.02530801,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01692231316719438,"score_gpt":0.2951060489753647,"score_spread":0.2781837358081703,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}