{"id":"W4393650160","doi":"10.5281/zenodo.10446176","title":"On Inter-dataset Code Duplication and Data Leakage in Large Language Models","year":2023,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University; McGill University","funders":"","keywords":"Computer science; Code (set theory); Leakage (economics); Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002344922,0.001990124,0.0009698951,0.00390733,0.001423974,0.001961066,0.002444668,0.001637304,0.009046121],"category_scores_gemma":[0.01447672,0.0004560944,0.001735667,0.006097935,0.0007356569,0.002345409,0.002370509,0.00204448,0.01358992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001785774,"about_ca_system_score_gemma":0.002643751,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01748415,"about_ca_topic_score_gemma":0.04430723,"domain_scores_codex":[0.9969172,0.0007790294,0.0002882986,0.0008760104,0.0009019757,0.0002374399],"domain_scores_gemma":[0.9923623,0.003277241,0.0004245881,0.002531594,0.001118671,0.0002857225],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0001424141,0.00007927966,0.003033172,0.0009089563,0.00009867211,0.0001337039,0.0001045365,0.003696861,0.0006924811,0.002928859,0.9709216,0.01725952],"study_design_scores_gemma":[0.0003971545,0.0001256567,0.01366035,0.0004869673,0.0001379138,0.0007461793,0.0003341642,0.0335048,0.005457853,0.01813847,0.9268971,0.0001134688],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.008795487,0.0007233294,0.005848454,0.0008426859,0.000224908,0.00009665389,0.9681166,0.01117177,0.004180044],"genre_scores_gemma":[0.006304898,0.0001648397,0.003569157,0.0001325645,0.00002230809,0.0001131441,0.988143,0.000482653,0.001067277],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9976551,"threshold_uncertainty_score":0.03476477,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07910873542123568,"score_gpt":0.3116455489015773,"score_spread":0.2325368134803416,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}