{"id":"W3093493237","doi":"10.1145/3340531.3412746","title":"The Utility of Context When Extracting Entities From Legal Documents","year":2020,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Cisco Systems (Canada)","funders":"","keywords":"Computer science; Sentence; Context (archaeology); Natural language processing; Process (computing); Named-entity recognition; Artificial intelligence; Sequence (biology); Information retrieval; Layer (electronics); Legal document; Information extraction; Task (project management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001404899,0.00005186807,0.0000805467,0.000005729608,0.00008078693,0.0001523237,0.000616742,0.00001732576,0.000214043],"category_scores_gemma":[0.00007506403,0.00003577699,0.00003683353,0.00003480658,0.00002632521,0.0003330002,0.000254882,0.00007670501,0.00001318266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000005338134,"about_ca_system_score_gemma":0.00002901108,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001779616,"about_ca_topic_score_gemma":0.00009379313,"domain_scores_codex":[0.9992932,0.00003996043,0.0001916817,0.0001788074,0.0001860411,0.0001103141],"domain_scores_gemma":[0.9993967,0.0001767979,0.00006428608,0.0002869325,0.00003450896,0.00004074146],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002762278,0.00004961809,0.01989669,0.00002552625,0.0001001612,0.0000119916,0.01723094,0.00004987256,0.002483757,0.2308886,0.009354937,0.7198802],"study_design_scores_gemma":[0.0003227829,0.00002745016,0.002599757,0.00001490661,0.00000607627,9.339282e-7,0.001713592,0.9188936,0.008929311,0.01367189,0.05368458,0.000135061],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1427852,0.0002290957,0.8332794,0.005942592,0.0003181236,0.00008263439,0.000002804986,0.0000637441,0.01729643],"genre_scores_gemma":[0.9801366,0.000003853341,0.01886224,0.0005040967,0.00004826502,0.000001527125,3.415717e-7,0.00000185317,0.0004412533],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9188438,"threshold_uncertainty_score":0.2690259,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04434289484593809,"score_gpt":0.2590443196216146,"score_spread":0.2147014247756765,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}