{"id":"W4402502868","doi":"10.48550/arxiv.2408.08896","title":"LLMJudge: LLMs for Relevance Judgments","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; University of Waterloo; Università degli Studi di Padova; Universiteit van Amsterdam; University College London","keywords":"Relevance (law); Psychology; Political science; Cognitive psychology; Social psychology; Law","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02726006,0.002084066,0.001723917,0.004279318,0.002198576,0.007597441,0.004936648,0.005079692,0.04014352],"category_scores_gemma":[0.199176,0.001399801,0.002437368,0.003582794,0.00255123,0.01228804,0.01233879,0.006930022,0.03875458],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003053063,"about_ca_system_score_gemma":0.004372131,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003060634,"about_ca_topic_score_gemma":0.004998043,"domain_scores_codex":[0.9573224,0.02456403,0.003128821,0.004397242,0.009615577,0.0009719652],"domain_scores_gemma":[0.8904763,0.06445628,0.002696256,0.02767572,0.01236022,0.00233506],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00132536,0.0003417731,0.00137505,0.001709579,0.0001398203,0.0001954216,0.001113063,0.01061673,0.007876952,0.06742908,0.3691879,0.5386893],"study_design_scores_gemma":[0.0005555493,0.0002819778,0.001438415,0.0005423168,0.00005502341,0.0002982303,0.0004439591,0.2540005,0.01367151,0.3770271,0.3514697,0.0002157679],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00566837,0.001753763,0.8693114,0.004908406,0.001346352,0.0008520677,0.008135528,0.09641902,0.01160513],"genre_scores_gemma":[0.09489875,0.0007481835,0.8416609,0.002484132,0.0009897274,0.001858578,0.0251178,0.0198465,0.01239551],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04014352,"threshold_uncertainty_score":0.1441667,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05917345749622854,"score_gpt":0.2200694768184831,"score_spread":0.1608960193222546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}