{"id":"W4410552841","doi":"10.1109/saner64311.2025.00053","title":"Evaluating the Effectiveness and Efficiency of Demonstration Retrievers in RAG for Coding Tasks","year":2025,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University; University of Manitoba","funders":"","keywords":"Computer science; Coding (social sciences); Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006836522,0.001374408,0.001052445,0.002043288,0.0004381302,0.001537763,0.002475638,0.001690613,0.002712943],"category_scores_gemma":[0.05055952,0.0006230147,0.0008276118,0.001723685,0.0009513326,0.004923032,0.001819171,0.00152854,0.001347489],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008142254,"about_ca_system_score_gemma":0.001801703,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006526607,"about_ca_topic_score_gemma":0.007371898,"domain_scores_codex":[0.9950093,0.002063108,0.0006787314,0.0007249886,0.001122465,0.0004014231],"domain_scores_gemma":[0.9393322,0.05143765,0.001833992,0.004815806,0.001797737,0.0007825923],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004946076,0.002113605,0.01288418,0.00275923,0.0005529862,0.0003346317,0.00171437,0.153102,0.03281093,0.002917452,0.007912464,0.777952],"study_design_scores_gemma":[0.001173948,0.005369435,0.01352732,0.0001667329,0.0006704027,0.0004313223,0.001569501,0.871964,0.09356041,0.004108951,0.007264134,0.0001938814],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9394748,0.001814891,0.04128676,0.0002872186,0.00005947073,0.0003997833,0.0008403934,0.01085548,0.004981204],"genre_scores_gemma":[0.8602225,0.0006487255,0.1315901,0.000170488,0.00003629089,0.0003818659,0.003534366,0.0008209759,0.002594697],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006836522,"threshold_uncertainty_score":0.0361554,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03623371557916335,"score_gpt":0.3680599931510121,"score_spread":0.3318262775718487,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}