{"id":"W4404002909","doi":"10.1145/3702980","title":"Trained without My Consent: Detecting Code Inclusion in Language Models Trained on Code","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Code (set theory); Inclusion (mineral); Programming language; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01078884,0.0009933998,0.0007709222,0.001225433,0.0007571544,0.001926035,0.001865375,0.00167,0.004654708],"category_scores_gemma":[0.06894767,0.0005505059,0.0009628503,0.0008334244,0.001115152,0.00312864,0.003070513,0.002805435,0.004878815],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009410765,"about_ca_system_score_gemma":0.003624379,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006691647,"about_ca_topic_score_gemma":0.0131444,"domain_scores_codex":[0.9912145,0.003354246,0.000625081,0.002378978,0.001952435,0.0004748251],"domain_scores_gemma":[0.9630858,0.01817766,0.002245161,0.01164335,0.004089282,0.0007587372],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001762024,0.0006822518,0.09749233,0.000804162,0.0003620828,0.001839222,0.001993092,0.04194678,0.01933668,0.01028884,0.131549,0.6919435],"study_design_scores_gemma":[0.000240056,0.0004125081,0.01852102,0.000422033,0.0001264057,0.0009485592,0.0006448225,0.8564458,0.02645869,0.03739242,0.05823169,0.00015593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3016911,0.00190154,0.5823652,0.007719772,0.001071762,0.00144209,0.03316627,0.05346269,0.0171797],"genre_scores_gemma":[0.7038024,0.0003229602,0.2286772,0.002722151,0.0002080762,0.00124061,0.05291411,0.001663829,0.008448721],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01078884,"threshold_uncertainty_score":0.05705756,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1048751159032044,"score_gpt":0.3370346541083497,"score_spread":0.2321595382051452,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}