{"id":"W4414021351","doi":"10.22331/q-2025-09-05-1848","title":"A Theory of Direct Randomized Benchmarking","year":2025,"lang":"en","type":"article","venue":"Quantum","topic":"Quantum Computing Algorithms and Architecture","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Office of Science; Office of the Director of National Intelligence; Advanced Scientific Computing Research; U.S. Department of Energy","keywords":"Benchmarking; Randomized controlled trial; Computer science; Business; Medicine; Internal medicine; Marketing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001399534,0.0001209348,0.0004233239,0.0001794098,0.00009361164,0.0000478738,0.0006441097,0.00004328552,0.000009556525],"category_scores_gemma":[0.0002144455,0.00009071093,0.0001892634,0.0004829816,0.0001123191,0.00006298982,0.0002529577,0.000126231,0.000004225226],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000009628086,"about_ca_system_score_gemma":0.00007202634,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001804397,"about_ca_topic_score_gemma":5.41658e-7,"domain_scores_codex":[0.998708,0.0003461963,0.0003025972,0.000278575,0.0001572767,0.0002073926],"domain_scores_gemma":[0.9981588,0.001192461,0.0001131966,0.0004517114,0.00005138569,0.00003251491],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004729403,0.00004002306,0.00003040383,0.0000326837,0.00005962007,0.000005267264,0.000567486,0.0004105734,0.0002009761,0.9177336,0.000384989,0.0800614],"study_design_scores_gemma":[0.009853243,0.00003565305,0.0002081111,0.0001481608,0.00001628219,0.000003264123,0.00001102257,0.761151,0.001228822,0.225604,0.00161983,0.0001205879],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05179518,0.001456182,0.9292515,0.0005846533,0.00106846,0.0002096375,0.000001246174,0.0001883758,0.01544478],"genre_scores_gemma":[0.9846893,0.00003751783,0.01468028,0.0002366212,0.00005418517,0.000009412883,7.516112e-7,0.00000491922,0.0002870319],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9328941,"threshold_uncertainty_score":0.3699084,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.006763334639812611,"score_gpt":0.235311112894705,"score_spread":0.2285477782548924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}