{"id":"W4389403907","doi":"10.2139/ssrn.4583531","title":"Legalbench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models","year":2023,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":135,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University; University of Toronto","funders":"","keywords":"Benchmark (surveying); Vocabulary; Process (computing); Computer science; Case-based reasoning; Knowledge management; Engineering ethics; Psychology; Data science; Artificial intelligence; Linguistics; Engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007955564,0.002651593,0.001356038,0.006633376,0.001683814,0.003962123,0.007362242,0.004898288,0.01085237],"category_scores_gemma":[0.06077308,0.001293523,0.002080319,0.00371635,0.001850994,0.01030716,0.005966471,0.003863925,0.004956863],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002873119,"about_ca_system_score_gemma":0.00541785,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03046308,"about_ca_topic_score_gemma":0.03642559,"domain_scores_codex":[0.9876864,0.004705986,0.001305015,0.002421824,0.003261408,0.0006193404],"domain_scores_gemma":[0.9490575,0.0337535,0.001676645,0.009029863,0.004419996,0.002062483],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002865974,0.01048382,0.04395874,0.00431159,0.002444491,0.001672514,0.003225089,0.263879,0.01709184,0.03806805,0.2909672,0.3210317],"study_design_scores_gemma":[0.0005963384,0.0006446578,0.004452156,0.0001149147,0.0002166698,0.0003610307,0.0008542198,0.9358537,0.01467113,0.01505218,0.0270327,0.0001502813],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.4455405,0.002404437,0.2582532,0.002417882,0.0009704358,0.002833924,0.06540216,0.1877026,0.03447489],"genre_scores_gemma":[0.5156087,0.0005448844,0.3375306,0.0006892821,0.0001263035,0.001391536,0.1324853,0.00566275,0.00596054],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.03046308,"threshold_uncertainty_score":0.06057149,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0462981910877303,"score_gpt":0.3537987956978807,"score_spread":0.3075006046101504,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}