{"id":"W4414113464","doi":"10.2139/ssrn.5472728","title":"From Law to Gherkin: A Human-Centred Quasi-Experiment on the Quality of LLM-Generated Behavioural Specifications from Food-Safety Regulations","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Food Supply Chain Traceability","field":"Agricultural and Biological Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"CLARITY; Quality (philosophy); Agile software development; Relevance (law); Completeness (order theory); Software; Domain (mathematical analysis); Software requirements specification","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02389095,0.001067943,0.001416776,0.0004260007,0.002167638,0.004400339,0.002425988,0.003935205,0.01822098],"category_scores_gemma":[0.07807053,0.00138852,0.000763062,0.0004480768,0.006496916,0.003600298,0.002763724,0.004779809,0.002406532],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001944119,"about_ca_system_score_gemma":0.002724638,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002242445,"about_ca_topic_score_gemma":0.001476765,"domain_scores_codex":[0.9795188,0.01332615,0.0008221652,0.00380841,0.001532743,0.0009917623],"domain_scores_gemma":[0.8703874,0.1021354,0.009691379,0.01232712,0.002811868,0.002646829],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"nonrandomized_trial","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.2682962,0.3029793,0.03219068,0.003769627,0.001006844,0.0007493675,0.1419254,0.01127671,0.08984758,0.04148366,0.008331898,0.09814278],"study_design_scores_gemma":[0.1775411,0.4390427,0.1614664,0.0008719692,0.001187823,0.0003092731,0.02066957,0.04495902,0.03717448,0.07017819,0.04525277,0.001346759],"study_design_candidate":"nonrandomized_trial","study_design_consensus":"nonrandomized_trial","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9792736,0.00007089818,0.008476429,0.0006119834,0.000322527,0.004691533,0.0003232655,0.0001506849,0.006079],"genre_scores_gemma":[0.9666658,0.00005276833,0.01275936,0.0009931306,0.000102743,0.01322606,0.0002461518,0.0001306148,0.005823277],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02389095,"threshold_uncertainty_score":0.1263489,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1130159363532756,"score_gpt":0.3045517953429158,"score_spread":0.1915358589896402,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}