{"id":"W4376166979","doi":"10.1145/3539618.3591852","title":"Extracting Complex Named Entities in Legal Documents via Weakly Supervised Object Detection","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Dependency (UML); Named-entity recognition; Artificial intelligence; Object (grammar); Information retrieval; Baseline (sea); Information extraction; Object detection; Natural language processing; Data mining; Machine learning; Pattern recognition (psychology); Task (project management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001478459,0.0009260424,0.0009875958,0.003928761,0.0008094674,0.002033999,0.001670572,0.001288027,0.00227263],"category_scores_gemma":[0.006162826,0.0003994407,0.0009309119,0.002378369,0.0008097277,0.003718415,0.00159365,0.00136096,0.004721077],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005200962,"about_ca_system_score_gemma":0.001058797,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002094553,"about_ca_topic_score_gemma":0.005492852,"domain_scores_codex":[0.9984413,0.0003111919,0.0001453912,0.0006752676,0.0003247734,0.0001020582],"domain_scores_gemma":[0.9943947,0.002255393,0.0008332564,0.001208922,0.001120385,0.0001874435],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002425228,0.000337345,0.01035375,0.0004994983,0.0001263762,0.0007808422,0.0006582345,0.01165585,0.1120414,0.005794727,0.01532739,0.842182],"study_design_scores_gemma":[0.00004018866,0.000218895,0.01258098,0.0001161248,0.0001499332,0.001371687,0.0005826861,0.7461513,0.1741637,0.02149306,0.04300195,0.0001293802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06321909,0.000810692,0.9148676,0.000388274,0.0001464782,0.000203099,0.001402913,0.01484802,0.004113845],"genre_scores_gemma":[0.2784051,0.0005201571,0.7058658,0.0002874685,0.0001679038,0.0002156335,0.006776892,0.0006600278,0.007100991],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003928761,"threshold_uncertainty_score":0.007818937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03620104001552275,"score_gpt":0.2766151920168823,"score_spread":0.2404141520013596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}