{"id":"W814172419","doi":"10.1109/saner.2016.105","title":"Analyzing the State of Static Analysis: A Large-Scale Evaluation in Open Source Software","year":2016,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":206,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Java; Python (programming language); Computer science; Open source software; Software; Software engineering; JavaScript; Open source; Static analysis; Population; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0285211,0.0007782911,0.00061886,0.005996292,0.00109946,0.001759128,0.001686032,0.0009144511,0.0007381137],"category_scores_gemma":[0.1397941,0.0004679528,0.0008979283,0.00476601,0.002615355,0.004397647,0.003559695,0.001745625,0.0003780293],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001975227,"about_ca_system_score_gemma":0.001593709,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005545626,"about_ca_topic_score_gemma":0.006617051,"domain_scores_codex":[0.9716178,0.01376209,0.001736431,0.003476279,0.008642665,0.0007646695],"domain_scores_gemma":[0.756635,0.1770995,0.01749156,0.02026133,0.02531491,0.003197693],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00190815,0.00337044,0.6419811,0.001965194,0.0008463771,0.0004271337,0.02130156,0.01090674,0.00885995,0.002763755,0.006896193,0.2987735],"study_design_scores_gemma":[0.0003177937,0.003105235,0.8750153,0.000671033,0.0005109677,0.0005368605,0.009835467,0.08266614,0.01160333,0.004469115,0.01106721,0.0002016291],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9919837,0.0004504752,0.005101203,0.0001642902,0.00001796568,0.0001616003,0.0004531488,0.000531563,0.001136042],"genre_scores_gemma":[0.9885465,0.0002105673,0.008680785,0.00007668323,0.00001784792,0.0001797569,0.001691569,0.0002274406,0.0003687942],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9714789,"threshold_uncertainty_score":0.1508358,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02635283887304471,"score_gpt":0.3156834690939992,"score_spread":0.2893306302209546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}