{"id":"W4412396369","doi":"10.1145/3726302.3730305","title":"Benchmarking LLM-based Relevance Judgment Methods","year":2025,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Benchmarking; Relevance (law); Computer science; Data science; Artificial intelligence; Political science; Business","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02235295,0.002434318,0.001340447,0.005679274,0.001391551,0.00275212,0.003064407,0.003048455,0.005206446],"category_scores_gemma":[0.08576729,0.0005807055,0.00171764,0.003230534,0.001087079,0.003447985,0.004176554,0.002599123,0.004009596],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002512689,"about_ca_system_score_gemma":0.002913707,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006742632,"about_ca_topic_score_gemma":0.00932659,"domain_scores_codex":[0.9733409,0.01367105,0.002680443,0.003573062,0.005885104,0.0008494642],"domain_scores_gemma":[0.9480521,0.03298598,0.001766004,0.006550076,0.009298692,0.001347155],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003400418,0.001960206,0.01480593,0.003646118,0.0009163238,0.0003372414,0.001039976,0.1244789,0.01553895,0.006246357,0.05451262,0.773117],"study_design_scores_gemma":[0.0007109735,0.00122742,0.007523361,0.0002592435,0.0001841194,0.0002909149,0.0004583939,0.9383442,0.02361646,0.0108182,0.01639188,0.0001747362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3489604,0.01707459,0.520606,0.001826466,0.001977628,0.003987523,0.00946657,0.06485207,0.03124868],"genre_scores_gemma":[0.5685363,0.001077531,0.4046025,0.0006342683,0.0003598067,0.001719653,0.01574033,0.001580686,0.005748854],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9776471,"threshold_uncertainty_score":0.1182151,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02706635386643755,"score_gpt":0.3433562945497067,"score_spread":0.3162899406832691,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}