{"id":"W4400065154","doi":"10.1145/3643787.3648028","title":"Aligning Programming Language and Natural Language: Exploring Design Choices in Multi-Modal Transformer-Based Embedding for Bug Localization","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Embedding; Computer science; Source code; Artificial intelligence; Natural language; Software; Machine learning; Natural language processing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001829582,0.0007236976,0.0004714859,0.000742585,0.0002604525,0.001021296,0.001068732,0.000704601,0.00184951],"category_scores_gemma":[0.009827462,0.0004047123,0.0006104247,0.0005641693,0.001302925,0.004584414,0.001761124,0.001398534,0.0004717881],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007960215,"about_ca_system_score_gemma":0.0006210062,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001870181,"about_ca_topic_score_gemma":0.002695899,"domain_scores_codex":[0.9989061,0.0005735647,0.0000508018,0.00024947,0.0001294681,0.00009051031],"domain_scores_gemma":[0.9960443,0.002804602,0.0003159766,0.000381033,0.0003383694,0.0001157868],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001163217,0.0006161261,0.01407255,0.0007437791,0.0001753143,0.0004785694,0.0021176,0.3398573,0.03782449,0.06620079,0.004224285,0.532526],"study_design_scores_gemma":[0.00001662911,0.00009748189,0.0005058941,0.00002608354,0.00002387992,0.00007315742,0.0001721757,0.9551525,0.003260968,0.03996529,0.0006924632,0.00001348911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1584368,0.0006005728,0.837885,0.0005349956,0.00002814046,0.00005606719,0.0001280308,0.001037106,0.001293321],"genre_scores_gemma":[0.8477858,0.0002267507,0.150013,0.0001438968,0.0000143106,0.00008303009,0.0003314985,0.0002439357,0.001157682],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001870181,"threshold_uncertainty_score":0.00967586,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06191044517153228,"score_gpt":0.3467190770583289,"score_spread":0.2848086318867967,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}