{"id":"W4412377345","doi":"10.1145/3726302.3730159","title":"A Human-AI Comparative Analysis of Prompt Sensitivity in LLM-Based Relevance Judgment","year":2025,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Relevance (law); Sensitivity (control systems); Computer science; Artificial intelligence; Political science; Engineering; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000373095,0.0000847235,0.0003217451,0.0004059541,0.00003429833,0.00002636667,0.0002301231,0.00002966438,0.000009064196],"category_scores_gemma":[0.00001584155,0.00007815023,0.00007734758,0.001494662,0.00002762772,0.0001207532,0.0001158003,0.00009299178,0.00000185659],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008603781,"about_ca_system_score_gemma":0.00007710586,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005891728,"about_ca_topic_score_gemma":0.002936919,"domain_scores_codex":[0.9989775,0.00009525959,0.0002786472,0.0003338822,0.0001728221,0.0001418567],"domain_scores_gemma":[0.9992194,0.0001129644,0.00006640529,0.0004958219,0.00008327504,0.00002209044],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001347879,0.0007175767,0.05145178,0.00006623313,0.0004116817,0.00002258524,0.001935586,0.5940352,0.01170101,0.3351139,0.0002982816,0.004232624],"study_design_scores_gemma":[0.000217353,0.00001899508,0.04608241,0.00002961257,0.00003109606,6.095086e-8,0.00001789227,0.9401975,0.01247496,0.0008150388,0.00004418669,0.00007089741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3307816,0.00001322115,0.66528,0.0006879696,0.00002813782,0.0001135962,7.383909e-7,0.0000347053,0.003059995],"genre_scores_gemma":[0.9680625,2.791323e-7,0.03115222,0.0005070876,0.000002756223,0.000009450589,0.000001176816,8.989193e-7,0.0002635994],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6372809,"threshold_uncertainty_score":0.3186874,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0379838156217058,"score_gpt":0.3335992028802042,"score_spread":0.2956153872584985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}