{"id":"W4389979427","doi":"10.4337/cilj.2023.02.08","title":"Large language models and the treaty interpretation game","year":2023,"lang":"en","type":"article","venue":"Cambridge International Law Journal","topic":"Law, AI, and Intellectual Property","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Treaty; Interpretation (philosophy); Analogy; International law; Political science; Argument (complex analysis); Narrative; Law; Law and economics; Sociology; Epistemology; Computer science; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01716007,0.000723935,0.0005707608,0.001055149,0.004128356,0.009107533,0.0021309,0.003826328,0.01200776],"category_scores_gemma":[0.04398054,0.0006100752,0.0009207232,0.0009786497,0.01468177,0.01645458,0.008364301,0.006696663,0.001473708],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004184449,"about_ca_system_score_gemma":0.00348778,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00549941,"about_ca_topic_score_gemma":0.006542522,"domain_scores_codex":[0.9741938,0.02180306,0.0004325018,0.0009554777,0.001898871,0.0007163541],"domain_scores_gemma":[0.955222,0.03925995,0.001104856,0.002730533,0.0009091936,0.0007733899],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005192578,0.0000376626,0.0004702326,0.00005138655,0.000009452338,0.0001865721,0.008084022,0.003840118,0.0003967153,0.9680849,0.003618911,0.01516809],"study_design_scores_gemma":[0.00004484896,0.00005180707,0.0002959563,0.0001222674,0.0000143446,0.0001972694,0.004864589,0.04188786,0.0007737358,0.8845198,0.06718319,0.00004434337],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1009667,0.0008874728,0.4300169,0.04378992,0.0003534512,0.000237798,0.0002593548,0.001301832,0.4221865],"genre_scores_gemma":[0.9196303,0.000243003,0.05736126,0.001958728,0.00009735988,0.0002407999,0.0001625447,0.0002792619,0.02002677],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01716007,"threshold_uncertainty_score":0.09075224,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0172333229506089,"score_gpt":0.2573812544151795,"score_spread":0.2401479314645706,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}