{"id":"W4398243240","doi":"10.1016/j.datak.2024.102324","title":"Large language models: Expectations for semantics-driven systems engineering","year":2024,"lang":"en","type":"article","venue":"Data & Knowledge Engineering","topic":"Systems Engineering Methodologies and Applications","field":"Engineering","cited_by":49,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"Foundation for Research and Technology-Hellas","keywords":"Computer science; Semantics (computer science); Programming language; Modeling language; Operational semantics; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0244144,0.001301969,0.002715581,0.001996589,0.001956082,0.009456662,0.004732929,0.005453031,0.006802343],"category_scores_gemma":[0.1558973,0.001811753,0.002550144,0.001357311,0.008140259,0.02642926,0.004648992,0.00978821,0.002339423],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003920663,"about_ca_system_score_gemma":0.004205683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004967204,"about_ca_topic_score_gemma":0.003355515,"domain_scores_codex":[0.984323,0.007637258,0.0007624787,0.001633851,0.005026264,0.0006171371],"domain_scores_gemma":[0.8151162,0.1473886,0.003599877,0.01867912,0.012117,0.003099177],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001746812,0.0001374414,0.0009831304,0.0002416883,0.00007099189,0.0000779393,0.0004978883,0.02663882,0.000860552,0.9520926,0.003748097,0.01447607],"study_design_scores_gemma":[0.00003399764,0.00002493722,0.0001003003,0.00004243642,0.00001920198,0.00003431596,0.00008745705,0.07095242,0.0005204681,0.9251095,0.003055513,0.00001948689],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01461187,0.0009548896,0.9553972,0.01592303,0.0003209646,0.0002000412,0.0005269573,0.001043607,0.01102136],"genre_scores_gemma":[0.4954734,0.001790847,0.4873873,0.004724533,0.00148847,0.001367502,0.00203047,0.001225438,0.00451211],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0244144,"threshold_uncertainty_score":0.1291173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07354628162855453,"score_gpt":0.3157645917063224,"score_spread":0.2422183100777679,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}