{"id":"W4383052081","doi":"10.5281/zenodo.8111952","title":"Automated Domain Modeling with Large Language Models: A Comparative Study","year":2023,"lang":"en","type":"paratext","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Domain (mathematical analysis); Language model; Data modeling; Modeling language; Natural language processing; Programming language; Software engineering; Software; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008567585,0.0007333223,0.0008242375,0.002656506,0.0008391712,0.002888964,0.001934299,0.001056712,0.003797541],"category_scores_gemma":[0.02755051,0.0005777373,0.001371695,0.002164206,0.0007005559,0.005116811,0.0021967,0.001524122,0.001374683],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00173942,"about_ca_system_score_gemma":0.00168706,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007625041,"about_ca_topic_score_gemma":0.007706094,"domain_scores_codex":[0.9934198,0.004235207,0.0003863362,0.000517761,0.001311118,0.0001298095],"domain_scores_gemma":[0.9481485,0.04352617,0.0009390626,0.005330787,0.001768777,0.0002866185],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00196665,0.001316044,0.01442709,0.002323475,0.0005742488,0.0005704509,0.003398179,0.2106011,0.01157183,0.03995523,0.01031778,0.7029778],"study_design_scores_gemma":[0.000207328,0.000381104,0.004815202,0.0002215419,0.0003201147,0.0004455733,0.00117009,0.9190154,0.01560695,0.02029075,0.03744765,0.00007832541],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2485586,0.003786357,0.700635,0.001361038,0.0001057464,0.0008414802,0.003102086,0.02193088,0.01967874],"genre_scores_gemma":[0.6551037,0.001744689,0.3310472,0.0001645481,0.00004039534,0.0003197487,0.00738399,0.002094512,0.002101238],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008567585,"threshold_uncertainty_score":0.04531026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07540611999253258,"score_gpt":0.3019400788695598,"score_spread":0.2265339588770272,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}