{"id":"W6929396995","doi":"10.48448/pggq-9g48","title":"Exploring the Effectiveness of Prompt Engineering for Legal Reasoning Tasks","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Task (project management); Textual entailment; Logical consequence; Natural language; Cluster analysis; Natural language understanding; Case-based reasoning; Automated reasoning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006909925,0.001789151,0.0008386065,0.002215533,0.0007467855,0.001613836,0.002008605,0.001931127,0.004727366],"category_scores_gemma":[0.03976563,0.0006020181,0.0007488446,0.0009122321,0.0006492395,0.007597682,0.002497569,0.002786979,0.002042797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001378428,"about_ca_system_score_gemma":0.002069189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007656992,"about_ca_topic_score_gemma":0.01133585,"domain_scores_codex":[0.995364,0.002221893,0.0003015917,0.001275391,0.0006522805,0.0001848188],"domain_scores_gemma":[0.9796252,0.01567234,0.0007300685,0.00212973,0.00129841,0.0005442449],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002388602,0.00184423,0.01248604,0.001086251,0.0002833953,0.0003135392,0.0008881895,0.07929812,0.02738671,0.003515549,0.01638935,0.8541201],"study_design_scores_gemma":[0.0002507654,0.001222471,0.00485415,0.00007859284,0.0001406909,0.00026497,0.0004249054,0.9425542,0.03396859,0.008986921,0.00716271,0.00009104358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.547976,0.004864702,0.3545766,0.002237591,0.0005087299,0.0007043608,0.001959679,0.07580733,0.01136498],"genre_scores_gemma":[0.7849285,0.0005445758,0.2059752,0.0004915696,0.00009265481,0.0001703429,0.004151814,0.0005444259,0.003100912],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007656992,"threshold_uncertainty_score":0.03654361,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05129019214394856,"score_gpt":0.3005461493868248,"score_spread":0.2492559572428763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}