{"id":"W7117242258","doi":"10.48550/arxiv.2512.19799","title":"PhysMaster: Building an Autonomous AI Physicist for Theoretical and Computational Physics Research","year":2025,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Limiting; Computation; Reliability (semiconductor); Computational model; Uncertainty quantification; Open research","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002408001,0.0006630724,0.0004985475,0.0007660613,0.0009824412,0.001888103,0.00234126,0.001659834,0.008859397],"category_scores_gemma":[0.006544543,0.0005613634,0.0006252332,0.0006014996,0.001717927,0.003263389,0.004962742,0.001985142,0.004304172],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001140649,"about_ca_system_score_gemma":0.00495672,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002395674,"about_ca_topic_score_gemma":0.003924803,"domain_scores_codex":[0.9989495,0.0003259813,0.00005575745,0.0002518827,0.0002999343,0.0001170043],"domain_scores_gemma":[0.997572,0.0008623538,0.0001848166,0.0005766643,0.0003019712,0.0005022289],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001823767,0.001775202,0.02436729,0.001683099,0.0003803325,0.0008599991,0.002355235,0.1866778,0.05728635,0.1515412,0.1067245,0.4645252],"study_design_scores_gemma":[0.0003264075,0.0004747119,0.001722368,0.00009992535,0.00007828098,0.0002315431,0.0003886178,0.796035,0.01635882,0.06700704,0.1172052,0.00007202227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1459874,0.0009282042,0.7415597,0.007653882,0.0008500619,0.001362927,0.001472543,0.0410508,0.05913435],"genre_scores_gemma":[0.3619505,0.0003907975,0.6135821,0.001575958,0.0001548807,0.0008292308,0.001714952,0.001045742,0.0187559],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008859397,"threshold_uncertainty_score":0.02963763,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2897261005411493,"score_gpt":0.3592761273002468,"score_spread":0.06955002675909755,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}