{"id":"W4404002909","doi":"10.1145/3702980","title":"Trained without My Consent: Detecting Code Inclusion in Language Models Trained on Code","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Code (set theory); Inclusion (mineral); Programming language; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001227856,0.0002226504,0.0003054134,0.0004741094,0.0001511909,0.0000683422,0.0004059064,0.0001712808,0.000005821735],"category_scores_gemma":[0.0006502768,0.0002181754,0.00007540735,0.0003554612,0.00002749918,0.000223712,0.00007451566,0.0005583577,0.000002862496],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007700581,"about_ca_system_score_gemma":0.00006136093,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006364122,"about_ca_topic_score_gemma":0.00007467045,"domain_scores_codex":[0.998391,0.0002375049,0.0002917692,0.000572686,0.0001678634,0.0003391638],"domain_scores_gemma":[0.9962127,0.003155187,0.00002603022,0.0004905055,0.00001862989,0.00009694895],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002896898,0.00003684718,0.000007679268,0.0001518934,0.00004295256,0.00008812597,0.01619252,0.5917183,0.00665621,0.003558991,0.000002299471,0.3815153],"study_design_scores_gemma":[0.0005616405,0.0001194278,0.00004481192,0.0002656864,0.00001975932,0.0001219485,0.0001998391,0.9873437,0.005153067,0.005703669,0.0001839777,0.0002825211],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0821235,0.0004079715,0.9152251,0.0006614222,0.0005876715,0.0001527705,0.00001123271,0.000809609,0.0000206939],"genre_scores_gemma":[0.5374007,0.00003202147,0.4623131,0.0001232706,0.00002798087,0.00002632609,0.000001069789,0.00001982709,0.00005579373],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.4552771,"threshold_uncertainty_score":0.8896934,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1048751159032044,"score_gpt":0.3370346541083497,"score_spread":0.2321595382051452,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}