{"id":"W4416016892","doi":"10.1145/3746252.3761491","title":"JustEva: A Toolkit to Evaluate LLM Fairness in Legal Knowledge Inference","year":2025,"lang":"","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Knowledge-based systems; Knowledge representation and reasoning; Key (lock); Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01132867,0.001147929,0.001231544,0.003597833,0.001516053,0.004046615,0.002478445,0.002280403,0.0295356],"category_scores_gemma":[0.07470104,0.0008440064,0.001327042,0.001562451,0.001391494,0.006122624,0.004828345,0.002542073,0.004929611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001703346,"about_ca_system_score_gemma":0.004813306,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0127163,"about_ca_topic_score_gemma":0.0257271,"domain_scores_codex":[0.9939578,0.0030335,0.0005025959,0.0006304021,0.001535663,0.000340135],"domain_scores_gemma":[0.9767892,0.01657386,0.0007075553,0.003333397,0.001941476,0.0006545926],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003388189,0.001187863,0.0261713,0.0026166,0.0009760823,0.000542864,0.001557454,0.130544,0.00319734,0.2032288,0.2413103,0.3852791],"study_design_scores_gemma":[0.0004270065,0.0001512166,0.001835579,0.0001846916,0.0001121272,0.0001657171,0.0002264322,0.8415486,0.004130409,0.1106579,0.04047638,0.00008392923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1000957,0.001844424,0.6010239,0.002879094,0.00114153,0.001167401,0.01600765,0.2337601,0.04208011],"genre_scores_gemma":[0.4554182,0.0004116869,0.5124851,0.0008733571,0.0001436813,0.0007346111,0.01137498,0.00843849,0.01011989],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0295356,"threshold_uncertainty_score":0.09880638,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08378049121032433,"score_gpt":0.4606445458965511,"score_spread":0.3768640546862267,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}