{"id":"W4411688397","doi":"10.1109/raise66696.2025.00006","title":"Towards the LLM-Based Generation of Formal Specifications from Natural-Language Contracts: Early Experiments with Symboleo","year":2025,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Programming language; Natural language generation; Natural language; Formal methods; Natural (archaeology); Software engineering; Natural language processing; History","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009775162,0.0007288415,0.0005148712,0.000675017,0.0004723332,0.001345513,0.001528753,0.001363129,0.00557411],"category_scores_gemma":[0.06165989,0.0005930476,0.0006813506,0.000573296,0.00124868,0.002418412,0.002282237,0.001662261,0.001595415],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009011123,"about_ca_system_score_gemma":0.001715647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002034303,"about_ca_topic_score_gemma":0.00169232,"domain_scores_codex":[0.9900856,0.006585866,0.0006387411,0.0007639078,0.001651325,0.0002745916],"domain_scores_gemma":[0.9252666,0.06270477,0.001402983,0.006289354,0.003686607,0.0006497632],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003743362,0.003477689,0.02004012,0.005267635,0.0002438682,0.002779404,0.03007727,0.1139707,0.1252712,0.04298279,0.02031697,0.6318289],"study_design_scores_gemma":[0.001219392,0.003035604,0.009473083,0.000592678,0.0001804903,0.001838739,0.00685257,0.6563444,0.180026,0.02488855,0.1151913,0.0003572975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5890986,0.0003296556,0.3716306,0.001271572,0.0001676042,0.001593458,0.002447921,0.02003586,0.0134247],"genre_scores_gemma":[0.5310455,0.0001997466,0.458428,0.0003708988,0.00001696085,0.0005662972,0.003288141,0.002757264,0.003327239],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009775162,"threshold_uncertainty_score":0.0516966,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09567948888282997,"score_gpt":0.3632091950845656,"score_spread":0.2675297062017357,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}