{"id":"W4417299029","doi":"10.48550/arxiv.2503.15772","title":"Detecting LLM-Generated Peer Reviews","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Catastrophic Loss Reduction; National Science Foundation","keywords":"Digital watermarking; Watermark; Covert; Implementation; Embedding; Statistical power; Statistical hypothesis testing; Word error rate; Type I and type II errors","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001214115,0.0003337142,0.0004447083,0.0001958021,0.0002447569,0.0003047364,0.001440746,0.0003524041,0.00003062996],"category_scores_gemma":[0.0006800984,0.0003191537,0.0002324811,0.0005904894,0.00002307012,0.0002196588,0.001509364,0.001017088,0.0003399012],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000126329,"about_ca_system_score_gemma":0.0001706774,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002840824,"about_ca_topic_score_gemma":0.00006765612,"domain_scores_codex":[0.9976512,0.0002674226,0.0005036991,0.000928935,0.0003105656,0.0003382156],"domain_scores_gemma":[0.9977946,0.00009458382,0.0003148644,0.001429769,0.0002726284,0.00009350832],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004803744,0.0004655201,0.1686188,0.002277188,0.000531942,0.0001373381,0.0071469,0.004503156,0.02582674,0.003351774,0.08380687,0.7032857],"study_design_scores_gemma":[0.0008586444,0.0002858338,0.05165118,0.002549321,0.0002154205,0.00004453703,0.0000363834,0.05288332,0.08841468,0.004960321,0.7953983,0.002702055],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5889059,0.006744228,0.3704564,0.00351814,0.01790596,0.001155669,0.00001267747,0.001555003,0.009746057],"genre_scores_gemma":[0.9559457,0.0008355785,0.01860086,0.001978443,0.001518436,0.0002149664,0.00003686438,0.00003656059,0.02083262],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7115914,"threshold_uncertainty_score":0.999926,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09076130526028045,"score_gpt":0.3050084791668169,"score_spread":0.2142471739065364,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}