{"id":"W6979317629","doi":"","title":"Benchmark control problems in nonequilibrium statistical mechanics","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Advanced Thermodynamics and Statistical Mechanics","field":"Physics and Astronomy","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Basic Energy Sciences; National Energy Research Scientific Computing Center; Lawrence Berkeley National Laboratory; Directorate for Mathematical and Physical Sciences; U.S. Department of Energy; Office of Science; National Science Foundation","keywords":"Benchmark (surveying); Python (programming language); Statistical mechanics; Non-equilibrium thermodynamics; Set (abstract data type); Implementation; Simple (philosophy); Statistical hypothesis testing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001532789,0.001008951,0.0007595742,0.0008268105,0.0009351101,0.0008826299,0.001889522,0.001288483,0.005368739],"category_scores_gemma":[0.007481321,0.0002956788,0.0007190082,0.001148081,0.000928441,0.0009220577,0.001014488,0.001322866,0.0004869964],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001446068,"about_ca_system_score_gemma":0.001725307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01091697,"about_ca_topic_score_gemma":0.008999481,"domain_scores_codex":[0.9990185,0.0003042961,0.00004655527,0.0001426792,0.0003498515,0.0001379512],"domain_scores_gemma":[0.9950003,0.003533245,0.0001960463,0.0004384812,0.0006179325,0.0002139009],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002192113,0.0003887669,0.003191563,0.0004831536,0.0001009875,0.0001392252,0.00007852426,0.8805245,0.001468072,0.07505521,0.02318875,0.01516208],"study_design_scores_gemma":[0.00009223392,0.00006967103,0.0008011493,0.00002730262,0.00001229057,0.00003208404,0.00003728048,0.9620988,0.002434108,0.02895097,0.005427639,0.0000163397],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6458203,0.004602463,0.2392273,0.003966693,0.0007742396,0.0006184487,0.01666762,0.003828672,0.0844942],"genre_scores_gemma":[0.865355,0.0008450765,0.1112622,0.0005001552,0.0001189254,0.0007505764,0.01098083,0.0007569381,0.009430265],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01091697,"threshold_uncertainty_score":0.02170688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00856512414337322,"score_gpt":0.2539368964938848,"score_spread":0.2453717723505116,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}