{"id":"W4399036020","doi":"10.1007/978-981-97-3076-6_8","title":"Overview of Benchmark Datasets and Methods for the Legal Information Extraction/Entailment Competition (COLIEE) 2024","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Benchmark (surveying); Information extraction; Extraction (chemistry); Artificial intelligence; Competition (biology); Information retrieval; Logical consequence; Data mining; Cartography; Chromatography; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0104934,0.0048182,0.002873975,0.01895258,0.004081279,0.005720751,0.007305077,0.004176748,0.03295147],"category_scores_gemma":[0.02822262,0.001560922,0.004169474,0.02002779,0.001385759,0.006316797,0.005996497,0.004074009,0.02912993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003522141,"about_ca_system_score_gemma":0.007344457,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02991418,"about_ca_topic_score_gemma":0.04848246,"domain_scores_codex":[0.9898278,0.002368029,0.001595767,0.00174931,0.003729641,0.0007294833],"domain_scores_gemma":[0.9827071,0.005650888,0.0006611801,0.005566657,0.004432837,0.0009812998],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005783308,0.0007611797,0.001460859,0.002975129,0.0002845242,0.000134223,0.00008940639,0.00365656,0.002288892,0.004475718,0.8437571,0.139538],"study_design_scores_gemma":[0.00148307,0.0006248996,0.008917121,0.001229414,0.0004691608,0.0008958717,0.0005093125,0.04589673,0.01702001,0.02149023,0.9012044,0.0002596559],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.02288263,0.01717671,0.04573726,0.003706537,0.001529055,0.003900104,0.8164091,0.04256788,0.04609068],"genre_scores_gemma":[0.005504964,0.001432727,0.05074035,0.000460608,0.0001037227,0.001045566,0.9349533,0.001259937,0.004498679],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9895066,"threshold_uncertainty_score":0.1102337,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06111280792826501,"score_gpt":0.4196401562609915,"score_spread":0.3585273483327265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}