{"id":"W2781014390","doi":"10.1145/3086512.3086550","title":"Two-step cascaded textual entailment for legal bar exam question answering","year":2017,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"National Institute of Informatics; Alberta Machine Intelligence Institute","keywords":"Textual entailment; Computer science; Logical consequence; Natural language processing; Artificial intelligence; Question answering; Negation; Information retrieval; Representation (politics); Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001785607,0.001394333,0.0009306663,0.00393179,0.000928756,0.001309243,0.002172071,0.00145842,0.009419363],"category_scores_gemma":[0.006913967,0.0005124154,0.001975168,0.001643706,0.0004967516,0.004347106,0.002141774,0.001654253,0.003851071],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001467531,"about_ca_system_score_gemma":0.00183287,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01081038,"about_ca_topic_score_gemma":0.016781,"domain_scores_codex":[0.9975801,0.0005692553,0.0002556326,0.0007060825,0.0007129086,0.0001759646],"domain_scores_gemma":[0.9975826,0.00126729,0.0001568668,0.0002803831,0.0005885941,0.0001243822],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009548871,0.001087504,0.004474956,0.001170172,0.0002546512,0.0008589515,0.001154224,0.01624732,0.07777888,0.01490988,0.02975319,0.8513553],"study_design_scores_gemma":[0.0001637247,0.0003461567,0.004445823,0.00006915901,0.000276597,0.0008340984,0.0003738475,0.8762411,0.0694732,0.02250142,0.0251729,0.0001019515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08829134,0.00126165,0.8666646,0.0009872236,0.000148636,0.001699799,0.00480516,0.02908941,0.007052113],"genre_scores_gemma":[0.301654,0.0003680539,0.6731078,0.0004406443,0.0001793445,0.000591613,0.01734925,0.000498847,0.005810486],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01081038,"threshold_uncertainty_score":0.03151089,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03522215394083796,"score_gpt":0.3092648581678765,"score_spread":0.2740427042270385,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}