{"id":"W4412438306","doi":"10.1145/3730436.3730529","title":"Comparative Analysis of Large Language Models for Context-Aware Code Completion using SAFIM Framework","year":2025,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Context (archaeology); Code (set theory); Programming language; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004699735,0.001548215,0.0009184373,0.002673483,0.0006654102,0.001597778,0.002094964,0.001103683,0.00199056],"category_scores_gemma":[0.02066975,0.0004121609,0.002156926,0.001562282,0.0005145472,0.002467893,0.001569245,0.001505223,0.001555145],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001622828,"about_ca_system_score_gemma":0.002340188,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01978764,"about_ca_topic_score_gemma":0.02724553,"domain_scores_codex":[0.9969258,0.001483927,0.0002022784,0.0006437335,0.0005488129,0.0001954281],"domain_scores_gemma":[0.9885279,0.007990384,0.00040002,0.001510749,0.001233514,0.0003374057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002578988,0.001337232,0.03823053,0.00268945,0.001140211,0.0006583681,0.001797954,0.5279652,0.007403161,0.01256732,0.05495936,0.3486722],"study_design_scores_gemma":[0.00006803028,0.0002254914,0.003519141,0.000083677,0.0001036728,0.0001128098,0.0003425929,0.9794709,0.003206494,0.004327078,0.008499643,0.00004057171],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6425819,0.005582437,0.2585764,0.001718527,0.0004587926,0.0007222912,0.02408163,0.05794869,0.008329381],"genre_scores_gemma":[0.7291416,0.001255057,0.1831756,0.00030916,0.0001158259,0.0007549335,0.07941332,0.002766269,0.003068084],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01978764,"threshold_uncertainty_score":0.03934491,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07661870692166582,"score_gpt":0.3764386418616509,"score_spread":0.299819934939985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}