{"id":"W4410552808","doi":"10.1109/saner64311.2025.00062","title":"On the Performance of Large Language Models for Code Change Intent Classification","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Concordia University","funders":"","keywords":"Computer science; Programming language; Code (set theory); Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01012429,0.002396447,0.001412002,0.004514462,0.0009615696,0.002122412,0.001637408,0.001746767,0.001340798],"category_scores_gemma":[0.03763637,0.0005303816,0.001643008,0.002129982,0.0006950238,0.003393582,0.001522228,0.002792496,0.001483667],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001999503,"about_ca_system_score_gemma":0.001857275,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02124077,"about_ca_topic_score_gemma":0.01936934,"domain_scores_codex":[0.9934247,0.003369082,0.0004407094,0.001403785,0.001027185,0.0003345864],"domain_scores_gemma":[0.9538935,0.03850257,0.001807252,0.00204913,0.003103041,0.000644591],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001534953,0.001014286,0.04862115,0.0004693574,0.0004996267,0.0002800259,0.0004823016,0.3887499,0.004290618,0.003054671,0.01238616,0.538617],"study_design_scores_gemma":[0.00001493536,0.0001041457,0.001398864,0.00001645468,0.00002584106,0.0000308743,0.00004738208,0.9959642,0.000859939,0.001181774,0.0003404391,0.00001516414],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6775732,0.009071436,0.2887548,0.003351424,0.0006575971,0.0003898087,0.002519354,0.01148343,0.006198947],"genre_scores_gemma":[0.9261355,0.0006682954,0.0670052,0.0004776927,0.0002011611,0.0001611195,0.00337455,0.0002523869,0.001724193],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02124077,"threshold_uncertainty_score":0.05354297,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06768386307145074,"score_gpt":0.3200188008704129,"score_spread":0.2523349377989622,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}