{"id":"W6911355662","doi":"10.5281/zenodo.10828315","title":"Replication Package for \"Effectiveness of ChatGPT for Static Analysis: How Far Are We?\"","year":2024,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Classical Antiquity Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Code (set theory); Replication (statistics); Software; Static analysis; Work (physics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01124932,0.001972415,0.001914436,0.003635201,0.001540234,0.002455477,0.003638288,0.00197049,0.2085008],"category_scores_gemma":[0.0943835,0.001431587,0.002555237,0.003334642,0.001244273,0.003652835,0.004808306,0.004094114,0.06799487],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00117058,"about_ca_system_score_gemma":0.003028568,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005261364,"about_ca_topic_score_gemma":0.004265343,"domain_scores_codex":[0.9902229,0.004515465,0.0007868212,0.001327721,0.002610865,0.0005361467],"domain_scores_gemma":[0.8736369,0.08485626,0.003389385,0.0241859,0.0122002,0.001731345],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001884096,0.0003373429,0.003831801,0.00211243,0.0004194055,0.000188681,0.001078932,0.001753571,0.003328942,0.01319054,0.8740799,0.09779427],"study_design_scores_gemma":[0.002541931,0.001396627,0.02451891,0.001547355,0.0006216817,0.0005761055,0.0005588158,0.02749391,0.01717531,0.04256559,0.8805199,0.0004837734],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"dataset","genre_scores_codex":[0.01518824,0.001065259,0.2451811,0.005977933,0.005512043,0.001821352,0.2629734,0.4075745,0.05470613],"genre_scores_gemma":[0.1476359,0.001087191,0.3455729,0.003755758,0.001939255,0.01385388,0.1782464,0.2361907,0.07171807],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.9887507,"threshold_uncertainty_score":0.6975048,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06964222350849456,"score_gpt":0.3350828381266961,"score_spread":0.2654406146182016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}