{"id":"W4416018011","doi":"10.1145/3746252.3761506","title":"RottenReviews: Benchmarking Review Quality with Human and LLM-Based Judgments","year":2025,"lang":"","type":"article","venue":"","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Systems, Applications & Products in Data Processing (Canada)","funders":"","keywords":"Benchmarking; Set (abstract data type); Quality (philosophy); Benchmark (surveying); Empirical research; Quality assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.003725224,0.0005466373,0.001267982,0.000197649,0.0007843442,0.0005458919,0.00101903,0.000140965,0.0001452738],"category_scores_gemma":[0.0000896719,0.0004042608,0.0001915819,0.001239371,0.0001832418,0.0003896616,0.0003998549,0.0003363436,0.00003482945],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001341592,"about_ca_system_score_gemma":0.0003194031,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006600766,"about_ca_topic_score_gemma":0.0001389042,"domain_scores_codex":[0.9949026,0.0009897967,0.001448372,0.001350143,0.0006611664,0.0006478998],"domain_scores_gemma":[0.9971532,0.0002314085,0.0005540013,0.001609167,0.0002184143,0.0002338049],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004537572,0.001515632,0.07414647,0.08190773,0.0006551455,0.00008108067,0.001488556,0.00002183703,0.0007217244,0.162004,0.07960799,0.5978045],"study_design_scores_gemma":[0.006652111,0.002601387,0.02931685,0.4191024,0.0007528071,0.00006239846,0.0002431809,0.02844597,0.001397481,0.0007846716,0.5062974,0.004343297],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.006220417,0.4923862,0.3907657,0.01524761,0.002087987,0.004569747,0.000009478807,0.0003828943,0.08832992],"genre_scores_gemma":[0.8730111,0.03676269,0.02815381,0.04267633,0.000361742,0.0004574934,0.0000265597,0.00005758933,0.01849269],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8667907,"threshold_uncertainty_score":0.9998409,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03976771493929884,"score_gpt":0.3598475970057127,"score_spread":0.3200798820664139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}