{"id":"W7125588785","doi":"10.1109/cascon66301.2025.00025","title":"Can We Trust the AI Pair Programmer? Copilot for API Misuse Detection and Correction","year":2025,"lang":"","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Software; Precision and recall; Construct (python library); Coding (social sciences); Misuse detection; Secure coding; Code (set theory); Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006276026,0.001166244,0.0007110206,0.001973474,0.001085155,0.003181431,0.002449529,0.001044213,0.005795131],"category_scores_gemma":[0.05506434,0.001315404,0.0007501781,0.001156067,0.002178927,0.006602026,0.003750233,0.003242007,0.00549795],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001443262,"about_ca_system_score_gemma":0.003018401,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003212953,"about_ca_topic_score_gemma":0.004011324,"domain_scores_codex":[0.989148,0.003209416,0.0005309749,0.001641543,0.004753425,0.0007166299],"domain_scores_gemma":[0.9573258,0.0147831,0.004212412,0.01433293,0.007976017,0.001369865],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001127333,0.0004167565,0.04076128,0.0007996044,0.0001346589,0.00175917,0.00892724,0.007325319,0.04154293,0.02106074,0.09438059,0.7817643],"study_design_scores_gemma":[0.0002250773,0.0008146154,0.0249339,0.001062184,0.0002609292,0.006960373,0.00334135,0.3917195,0.1376822,0.0356745,0.3969216,0.0004037934],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1212196,0.0008384601,0.6930993,0.007120342,0.0004648384,0.0004335643,0.000755014,0.1533888,0.02268013],"genre_scores_gemma":[0.4579093,0.0004105362,0.5062608,0.00209053,0.0001127883,0.0002646288,0.001413245,0.01960414,0.01193397],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006276026,"threshold_uncertainty_score":0.0331912,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01999295301785185,"score_gpt":0.2875308711137979,"score_spread":0.267537918095946,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}