{"id":"W7106813711","doi":"10.48448/svd4-cx32","title":"Batched Self-Consistency Improves LLM Relevance Assessment and Ranking","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Ranking (information retrieval); Relevance (law); Context (archaeology); Consistency (knowledge bases); Task (project management); Pointwise; Learning to rank","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006207884,0.002615801,0.002560426,0.002927503,0.001263262,0.002480597,0.003143008,0.002411306,0.003380739],"category_scores_gemma":[0.02447352,0.0006279142,0.001666082,0.00187946,0.0009580891,0.004709277,0.002302621,0.002803229,0.004136142],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001158652,"about_ca_system_score_gemma":0.002427444,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007956121,"about_ca_topic_score_gemma":0.01573716,"domain_scores_codex":[0.9933182,0.002155535,0.0006185232,0.001965842,0.001535419,0.0004064951],"domain_scores_gemma":[0.9873882,0.005976216,0.0007484048,0.003363207,0.002015979,0.0005080709],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002036368,0.001619225,0.01321138,0.0009103453,0.0006065121,0.000410051,0.0005584948,0.06847657,0.04488424,0.002163149,0.04512863,0.819995],"study_design_scores_gemma":[0.0005358119,0.001465421,0.007516292,0.00006397847,0.0003114128,0.0005234652,0.0003498567,0.9413353,0.03025677,0.008141327,0.009306247,0.0001941158],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3101435,0.01247235,0.5931378,0.001872497,0.001296496,0.001366981,0.002980938,0.06787775,0.008851635],"genre_scores_gemma":[0.6957114,0.0006319851,0.2847156,0.001063938,0.0007764163,0.0004314872,0.005596121,0.001559726,0.009513283],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007956121,"threshold_uncertainty_score":0.03283083,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02389677634670062,"score_gpt":0.3375571654782648,"score_spread":0.3136603891315641,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}