{"id":"W3108142956","doi":"10.3233/faia200860","title":"Sentence Embeddings and High-Speed Similarity Search for Fast Computer Assisted Annotation of Legal Documents","year":2020,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Topic Modeling","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Research Unit on Children's Psychosocial Maladjustment","funders":"","keywords":"Annotation; Computer science; Sentence; Natural language processing; Artificial intelligence; Similarity (geometry); Meaning (existential); Process (computing); Information retrieval; Interface (matter); Programming language; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001478056,0.0008132491,0.0006834982,0.002216819,0.0006437697,0.001172319,0.001020281,0.001059285,0.006520092],"category_scores_gemma":[0.00701798,0.0004532598,0.0006182598,0.002391097,0.0004348707,0.003943989,0.0018614,0.001220734,0.005075087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005655678,"about_ca_system_score_gemma":0.0008255607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002144639,"about_ca_topic_score_gemma":0.00367584,"domain_scores_codex":[0.9985006,0.0005825837,0.0001337056,0.0003645095,0.0003522775,0.0000662815],"domain_scores_gemma":[0.9973821,0.00113445,0.0002317788,0.0005331841,0.0006221112,0.00009641729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004449071,0.0002796344,0.001619126,0.0005819115,0.0001112334,0.0003629262,0.001024762,0.01069338,0.04611864,0.02998608,0.04980565,0.8589718],"study_design_scores_gemma":[0.00008747172,0.0002156792,0.002598392,0.00007222762,0.00007150981,0.0006722923,0.0005413854,0.8558443,0.03493674,0.05510345,0.04976516,0.00009145058],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02232435,0.000870114,0.9597813,0.0003161651,0.0002508434,0.0001954301,0.001029809,0.01206201,0.003169873],"genre_scores_gemma":[0.1685778,0.000386707,0.8196623,0.0001442574,0.0001725449,0.000223347,0.005344509,0.0008003323,0.004688113],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006520092,"threshold_uncertainty_score":0.0218119,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05354764576486624,"score_gpt":0.2979148153250844,"score_spread":0.2443671695602181,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}