{"id":"W4321793856","doi":"10.1016/j.jss.2023.111651","title":"Finding associations between natural and computer languages: A case-study of bilingual LDA applied to the bleeping computer forum posts","year":2023,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"Blackberry (Canada); Lakehead University; Queen's University","funders":"","keywords":"Perplexity; Computer science; Topic model; Natural language processing; Context (archaeology); Latent Dirichlet allocation; Artificial intelligence; Coherence (philosophical gambling strategy); Natural language; Software; Language model; Statistics; Mathematics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00109509,0.0001154571,0.0003434823,0.0003232411,0.0002080497,0.0002356625,0.0003478747,0.00005084048,1.78707e-7],"category_scores_gemma":[0.0001976948,0.00008040464,0.00004516938,0.0005540958,0.00001550874,0.000153291,0.0003749413,0.0002773504,0.000001542142],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004765238,"about_ca_system_score_gemma":0.00005085029,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001975504,"about_ca_topic_score_gemma":0.00003359597,"domain_scores_codex":[0.9986113,0.00007733839,0.0004431049,0.0001786825,0.0004399793,0.0002496265],"domain_scores_gemma":[0.9975026,0.001765423,0.0002186085,0.0002047055,0.0001993367,0.000109318],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0000130224,0.0001131549,0.8016459,0.0003113261,0.0007311606,0.002290366,0.07099503,0.02315484,0.00008441005,0.0003412454,0.003536682,0.09678292],"study_design_scores_gemma":[0.002843999,0.001794028,0.891982,0.0009165639,0.00009714188,0.004311695,0.02088873,0.07574502,0.000043303,0.00004564122,0.000672758,0.0006591629],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8919269,0.0002473391,0.1068576,0.0001592829,0.0004521013,0.0002831646,0.000008275804,0.0000647585,6.111461e-7],"genre_scores_gemma":[0.9934105,0.000003142825,0.00617002,0.00003140467,0.0003578375,0.000003582955,0.000001126409,0.00001144179,0.00001099353],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1014836,"threshold_uncertainty_score":0.3278806,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02679858553327135,"score_gpt":0.2989785756976393,"score_spread":0.2721799901643679,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}