{"id":"W4416018011","doi":"10.1145/3746252.3761506","title":"RottenReviews: Benchmarking Review Quality with Human and LLM-Based Judgments","year":2025,"lang":"","type":"article","venue":"","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Systems, Applications & Products in Data Processing (Canada)","funders":"","keywords":"Benchmarking; Set (abstract data type); Quality (philosophy); Benchmark (surveying); Empirical research; Quality assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05750069,0.001562137,0.001487542,0.009984697,0.001599218,0.005566191,0.002414966,0.002223823,0.004010484],"category_scores_gemma":[0.263306,0.0006419008,0.001166421,0.005486006,0.001397451,0.003639751,0.003994283,0.001557646,0.003390465],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002793441,"about_ca_system_score_gemma":0.006215796,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004776004,"about_ca_topic_score_gemma":0.01287909,"domain_scores_codex":[0.930434,0.04020073,0.007680268,0.007798336,0.01290473,0.000981917],"domain_scores_gemma":[0.7139822,0.1687334,0.02812117,0.03096074,0.05228267,0.005919861],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003288212,0.001030979,0.1725972,0.01390124,0.002704902,0.0005528879,0.007260589,0.04584079,0.02010053,0.01108713,0.2257819,0.4958536],"study_design_scores_gemma":[0.001376794,0.00328335,0.160792,0.002632339,0.0009394618,0.001467281,0.004029944,0.4950789,0.05164225,0.0331124,0.24449,0.001155273],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5372763,0.0263231,0.2683257,0.006163268,0.003758921,0.00543877,0.04705058,0.06858158,0.03708185],"genre_scores_gemma":[0.7400916,0.001716194,0.2054248,0.001341016,0.0006729563,0.002074174,0.03887662,0.003196222,0.006606458],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9424993,"threshold_uncertainty_score":0.3040963,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03976771493929884,"score_gpt":0.3598475970057127,"score_spread":0.3200798820664139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}