{"id":"W4385570658","doi":"10.18653/v1/2023.findings-acl.858","title":"Exploring the Effectiveness of Prompt Engineering for Legal Reasoning Tasks","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Task (project management); Computer science; Textual entailment; Natural language processing; Artificial intelligence; Cluster analysis; Logical consequence; Shot (pellet); Zero (linguistics); Best practice; Natural language; Linguistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01084844,0.001892169,0.001027766,0.002169505,0.0007425599,0.001710079,0.001939032,0.001889152,0.002684212],"category_scores_gemma":[0.06019794,0.0006702933,0.0007575174,0.000970465,0.0006698348,0.008042772,0.002211946,0.003185619,0.001305215],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001644568,"about_ca_system_score_gemma":0.002294806,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007514655,"about_ca_topic_score_gemma":0.009684598,"domain_scores_codex":[0.9937264,0.003382778,0.0004541809,0.001451813,0.000754113,0.0002307351],"domain_scores_gemma":[0.9572973,0.03582625,0.001329777,0.002985833,0.001731188,0.0008295465],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003738729,0.002448092,0.02311364,0.001443058,0.0003671976,0.0003446601,0.001569115,0.1160422,0.02867679,0.003925805,0.01435714,0.8039735],"study_design_scores_gemma":[0.0003495881,0.001514392,0.005715054,0.00007956313,0.0001579025,0.0002314944,0.0004990008,0.9522281,0.02572614,0.008165303,0.005241294,0.00009218068],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6664236,0.00415181,0.258067,0.002044913,0.0003846612,0.0007185253,0.00154998,0.05970043,0.006959077],"genre_scores_gemma":[0.835659,0.0004414833,0.1589161,0.0003674059,0.00007977517,0.0001641623,0.00272394,0.0003655958,0.001282578],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01084844,"threshold_uncertainty_score":0.05737275,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06120192517994883,"score_gpt":0.2559378797659381,"score_spread":0.1947359545859893,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}