{"id":"W4384009618","doi":"10.1109/msr59073.2023.00033","title":"On Codex Prompt Engineering for OCL Generation: An Empirical Study","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"Science and Engineering Research Council; Canadian Institute for Advanced Research","keywords":"Computer science; Programming language; Object Constraint Language; Task (project management); Unified Modeling Language; Natural language processing; Syntax; Artificial intelligence; Object (grammar); Software engineering; UML tool; Software","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004451714,0.00009369847,0.00009079408,0.0001847001,0.00006936973,0.0001632815,0.0005096447,0.00003162255,0.0000102778],"category_scores_gemma":[0.000408181,0.00008319679,0.00002624108,0.0006229765,0.000004238265,0.0002002187,0.0001253431,0.00009063973,0.0001035777],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003925693,"about_ca_system_score_gemma":0.00003707829,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000003193843,"about_ca_topic_score_gemma":0.000003088903,"domain_scores_codex":[0.9989101,0.00001820345,0.0001160142,0.0003546951,0.0003217522,0.000279242],"domain_scores_gemma":[0.9988959,0.0004577617,0.000007894353,0.0004736265,0.00006488856,0.00009992626],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003271846,0.00174308,0.04864866,0.0001277802,0.0001732727,0.0002183434,0.007605587,0.7134187,0.007483658,0.03921609,0.1454692,0.03586292],"study_design_scores_gemma":[0.0002610231,0.0005478534,0.0225301,0.000002438775,9.405711e-7,0.00000171849,0.00001302398,0.9748054,0.0008727803,0.00006547572,0.000782029,0.0001171741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4513201,0.000002030203,0.5466024,0.000294861,0.0002590193,0.0003932969,7.900488e-7,0.001110095,0.00001733733],"genre_scores_gemma":[0.9714481,2.869682e-7,0.02742282,0.00006676538,0.0002005932,0.0002394169,0.000006422119,0.00001904846,0.0005965965],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.520128,"threshold_uncertainty_score":0.3392667,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09103536348884478,"score_gpt":0.3692618199990915,"score_spread":0.2782264565102467,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}