{"id":"W6979317629","doi":"","title":"Benchmark control problems in nonequilibrium statistical mechanics","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Advanced Thermodynamics and Statistical Mechanics","field":"Physics and Astronomy","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Basic Energy Sciences; National Energy Research Scientific Computing Center; Lawrence Berkeley National Laboratory; Directorate for Mathematical and Physical Sciences; U.S. Department of Energy; Office of Science; National Science Foundation","keywords":"Benchmark (surveying); Python (programming language); Statistical mechanics; Non-equilibrium thermodynamics; Set (abstract data type); Implementation; Simple (philosophy); Statistical hypothesis testing","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001510086,0.0001921434,0.0003047499,0.00006835371,0.00006656336,0.00002518823,0.0001895571,0.00005892066,0.0003824974],"category_scores_gemma":[0.00004101009,0.0001860522,0.00005421061,0.0002336025,0.00003024747,0.00008348523,0.00007461511,0.0002862236,0.00005458476],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004657408,"about_ca_system_score_gemma":0.00007408913,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007612305,"about_ca_topic_score_gemma":0.0000216001,"domain_scores_codex":[0.998708,0.00005465838,0.0003596171,0.0003476064,0.0001186245,0.0004115478],"domain_scores_gemma":[0.99928,0.0002472865,0.0000650962,0.0002618716,0.00005516845,0.00009058384],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002443937,0.0001569494,0.06559284,0.00002167597,0.00004523105,0.000005562346,0.0000258147,0.0009043652,0.00209769,0.9267943,0.00006512299,0.004265997],"study_design_scores_gemma":[0.001977353,0.00009678638,0.02416221,0.00008399341,0.00004905561,3.531842e-7,0.00009952729,0.2917684,0.0002038687,0.679563,0.001629599,0.000365901],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1263582,0.00003584979,0.8668303,0.0001729326,0.0002575346,0.0003107504,0.0002144983,0.00002740417,0.005792541],"genre_scores_gemma":[0.9975365,0.000003211759,0.001781932,0.000149883,0.00005438026,0.00006648438,0.00006808442,0.00001966117,0.0003198748],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8711783,"threshold_uncertainty_score":0.7586989,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00856512414337322,"score_gpt":0.2539368964938848,"score_spread":0.2453717723505116,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}