{"id":"W4400722606","doi":"10.1016/j.jss.2024.112158","title":"Reproducibility of issues reported in stack overflow questions: Challenges, impact &amp; estimation","year":2024,"lang":"en","type":"article","venue":"Journal of Systems and Software","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Reproducibility; Stack (abstract data type); Estimation; Computer science; Reliability engineering; Statistics; Mathematics; Engineering; Systems engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"reproducibility","study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003955449,0.0001008245,0.0003348207,0.0002094827,0.00003633458,0.000176429,0.0001304179,0.0000674818,0.000001609755],"category_scores_gemma":[0.00144798,0.00007497532,0.00008646763,0.000235796,0.00002782818,0.0005006798,0.0000399894,0.0001869055,0.000001033732],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008344391,"about_ca_system_score_gemma":0.0001435032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003825503,"about_ca_topic_score_gemma":0.00002260048,"domain_scores_codex":[0.9983121,0.0001389023,0.0007787484,0.0003741948,0.0002764766,0.0001195536],"domain_scores_gemma":[0.998332,0.0001572802,0.0003518686,0.0008491561,0.0002428087,0.000066943],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001487826,0.0006808675,0.1045215,0.009854851,0.0007620416,0.001569952,0.04538468,0.1108769,0.004097649,0.009516216,0.007242385,0.7053442],"study_design_scores_gemma":[0.002132155,0.002177985,0.4991866,0.04287547,0.0002358298,0.01671209,0.002280472,0.3865588,0.001164247,0.02447451,0.02068928,0.001512583],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7952474,0.08899307,0.114312,0.0004231834,0.0007945796,0.0001354747,0.000003348485,0.00006636398,0.0000245612],"genre_scores_gemma":[0.9889553,0.0008377089,0.01000747,0.000002809918,0.0001132115,0.000001272232,7.082086e-7,0.000006890542,0.00007465467],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7038316,"threshold_uncertainty_score":0.3057404,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04380599402015875,"score_gpt":0.3244064420340398,"score_spread":0.2806004480138811,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}