{"id":"W4385571145","doi":"10.18653/v1/2023.acl-long.865","title":"LeXFiles and LegalLAMA: Facilitating English Multinational Legal Language Model Development","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Innovationsfonden","keywords":"Multinational corporation; Upstream (networking); Benchmark (surveying); Computer science; Downstream (manufacturing); Best practice; Natural language processing; Work (physics); Knowledge management; Artificial intelligence; Engineering; Political science; Operations management; Law","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003336857,0.0001109193,0.00009106878,0.0001510564,0.0001500292,0.0002425132,0.0003809419,0.0000459195,0.000006025089],"category_scores_gemma":[0.0002661621,0.00009509658,0.00001828372,0.0003117843,0.00002783048,0.0007280289,0.0004184899,0.0001170322,0.00001488843],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003190184,"about_ca_system_score_gemma":0.00008831119,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002384923,"about_ca_topic_score_gemma":0.00002017521,"domain_scores_codex":[0.999023,0.00001915075,0.0001623283,0.0003066936,0.0002717645,0.000217044],"domain_scores_gemma":[0.9995192,0.0001117079,0.00003742552,0.0001581208,0.0001213662,0.00005214319],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000005145145,0.00004785643,0.0002894102,0.0001335554,0.000025878,0.00008244158,0.05937082,0.0006954981,0.01623131,0.4862663,0.005945471,0.4309064],"study_design_scores_gemma":[0.0002444078,0.00001496196,0.0003296644,0.00003783913,0.000001638239,0.00000909129,0.001017832,0.9553288,0.03505324,0.006015155,0.001602307,0.000345022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1327857,0.0004636274,0.8594843,0.0004648644,0.00008800572,0.0001461518,0.000004497844,0.003430952,0.003131918],"genre_scores_gemma":[0.4047849,0.00000147181,0.5928504,0.0001033196,0.00001759141,0.00001545795,0.000007675478,0.00000468101,0.00221455],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9546334,"threshold_uncertainty_score":0.3877926,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01405885961727889,"score_gpt":0.2717166111702214,"score_spread":0.2576577515529425,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}