{"id":"W4384159091","doi":"10.1109/icpc58990.2023.00031","title":"Pathways to Leverage Transcompiler based Data Augmentation for Cross-Language Clone Detection","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Computer science; clone (Java method); Leverage (statistics); Source code; Exploit; Programming language; Software; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006246645,0.0000929041,0.00008267799,0.0002204746,0.0001028421,0.0002238953,0.0009891941,0.00003850951,0.00002590255],"category_scores_gemma":[0.0003149307,0.00008880335,0.00003258021,0.000804011,0.000009234299,0.0004726545,0.0001836796,0.00006986003,0.0001734794],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005228673,"about_ca_system_score_gemma":0.00004549727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004119055,"about_ca_topic_score_gemma":0.00002447753,"domain_scores_codex":[0.9987838,0.00002155407,0.000140407,0.0004487668,0.0003061332,0.0002993337],"domain_scores_gemma":[0.9983845,0.0005655949,0.00001361284,0.000884085,0.00005485957,0.00009730692],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001173095,0.000137889,0.001972812,0.0003422303,0.00005223579,0.00005740346,0.0038693,0.06118144,0.5477023,0.001001159,0.01127426,0.3722917],"study_design_scores_gemma":[0.00074601,0.0001284745,0.03538262,0.00001220614,0.000002207941,0.000001584662,0.00003246098,0.8411964,0.121243,0.00007406894,0.001004373,0.0001765415],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1863227,0.000006973048,0.8117904,0.0002313927,0.0003074119,0.0003719444,0.00004502789,0.0009052865,0.00001885945],"genre_scores_gemma":[0.9403679,6.613585e-7,0.05888342,0.0001557918,0.00006931408,0.0001097403,0.00009930623,0.00001666234,0.0002971886],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.780015,"threshold_uncertainty_score":0.3621295,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09413039926532048,"score_gpt":0.3641596172484101,"score_spread":0.2700292179830897,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}