{"id":"W2798978891","doi":"10.1145/3209978.3210015","title":"A Dataset and an Examination of Identifying Passages for Due Diligence","year":2018,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"Cisco Systems (Canada)","funders":"","keywords":"Computer science; Sequence labeling; Conditional random field; Due diligence; Task (project management); Sequence (biology); Artificial intelligence; Annotation; Sentence; Natural language processing; Diligence; Hidden Markov model; Information retrieval; Information extraction; Machine learning; Data science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001356654,0.0006903348,0.0003881865,0.004391117,0.001465858,0.001171424,0.001219966,0.001975792,0.005898084],"category_scores_gemma":[0.007288622,0.0002099323,0.0005939127,0.005269299,0.0005269992,0.001582405,0.001089988,0.001313684,0.005329594],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001551915,"about_ca_system_score_gemma":0.00173178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009575563,"about_ca_topic_score_gemma":0.03378055,"domain_scores_codex":[0.9984903,0.0003095558,0.0002741799,0.0003120789,0.0004838457,0.0001300907],"domain_scores_gemma":[0.9938263,0.002563688,0.0007848946,0.000920033,0.001479652,0.000425425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005706859,0.0007834687,0.02370211,0.002601608,0.00008580766,0.0011705,0.001619161,0.002326785,0.008892336,0.006046993,0.8665265,0.08567394],"study_design_scores_gemma":[0.0002365846,0.0002711144,0.07953374,0.0003810443,0.00006785765,0.001831775,0.002068034,0.008990068,0.008636673,0.00472993,0.893128,0.0001251769],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.09307116,0.002033154,0.005309568,0.002567849,0.0004982617,0.0007647757,0.8739376,0.002694682,0.01912299],"genre_scores_gemma":[0.04139895,0.0005213931,0.0148669,0.0004386642,0.0001530345,0.0006883255,0.9375662,0.0001493387,0.004217271],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.009575563,"threshold_uncertainty_score":0.01973104,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1640491185197676,"score_gpt":0.4630750503782688,"score_spread":0.2990259318585013,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}