{"id":"W4404570241","doi":"10.48550/arxiv.2411.10873","title":"Developer Challenges on Large Language Models: A Study of Stack Overflow and OpenAI Developer Forum Posts","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Stack (abstract data type); Computer science; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01993834,0.0006952473,0.0006995011,0.004920974,0.003538811,0.004077363,0.001381013,0.001587846,0.002273251],"category_scores_gemma":[0.1103437,0.0005657535,0.0004794261,0.003016403,0.002147094,0.0094971,0.005487877,0.001919485,0.0008368161],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002749292,"about_ca_system_score_gemma":0.002525886,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005798448,"about_ca_topic_score_gemma":0.008971903,"domain_scores_codex":[0.984608,0.008090131,0.0008827731,0.001603765,0.003825216,0.0009901101],"domain_scores_gemma":[0.833511,0.1246855,0.01962689,0.005592619,0.01106871,0.005515232],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0008119477,0.000801102,0.3848148,0.0007822407,0.00009087553,0.003150723,0.4484642,0.001458174,0.008949681,0.006478708,0.0147382,0.1294596],"study_design_scores_gemma":[0.0001374749,0.001104985,0.4451689,0.0007753039,0.0001586865,0.002914025,0.3731158,0.0539835,0.006350365,0.008212908,0.1075352,0.0005429078],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9887788,0.0002318743,0.00472852,0.001023255,0.00004565941,0.000125474,0.0003167452,0.0003800424,0.004369739],"genre_scores_gemma":[0.9899873,0.0001662936,0.00442204,0.0003576935,0.00007415884,0.0002521284,0.0008578036,0.0003951215,0.003487562],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01993834,"threshold_uncertainty_score":0.1054453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2948218354156552,"score_gpt":0.297872786374531,"score_spread":0.003050950958875753,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}