{"id":"W4416016892","doi":"10.1145/3746252.3761491","title":"JustEva: A Toolkit to Evaluate LLM Fairness in Legal Knowledge Inference","year":2025,"lang":"","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Inference; Knowledge-based systems; Knowledge representation and reasoning; Key (lock); Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.003827336,0.0005178658,0.0007484648,0.0006650394,0.000858258,0.0007702102,0.001852578,0.0005142185,0.006774551],"category_scores_gemma":[0.00405685,0.0005349796,0.000222127,0.005835825,0.00107798,0.001085572,0.0008868154,0.0006746251,0.006677003],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001094993,"about_ca_system_score_gemma":0.003563275,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.0256825,"about_ca_topic_score_gemma":0.2082782,"domain_scores_codex":[0.9938208,0.001077935,0.001450059,0.001253078,0.0008630099,0.001535148],"domain_scores_gemma":[0.9963177,0.001463096,0.0001508776,0.0007590275,0.0008734156,0.0004358454],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00009919446,0.0006954904,0.004092888,0.00007082438,0.00003528159,0.00001636926,0.02950034,0.000884433,0.0001595078,0.836842,0.00447671,0.1231269],"study_design_scores_gemma":[0.0006885779,0.0007017204,0.007238178,0.001895377,0.0002256603,0.000001260189,0.08788516,0.02436652,0.009818418,0.1007194,0.7640823,0.002377394],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.1740456,0.0004748126,0.01136272,0.01129375,0.005421094,0.001702979,0.000007713777,0.0001478212,0.7955436],"genre_scores_gemma":[0.8918653,0.0001339337,0.0003096594,0.00110483,0.0003253427,0.0001519646,0.000001199992,0.0000205256,0.1060872],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7596056,"threshold_uncertainty_score":0.9997102,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08378049121032433,"score_gpt":0.4606445458965511,"score_spread":0.3768640546862267,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}