{"id":"W4408688463","doi":"10.1098/rsos.241038","title":"The reliability of replications: a study in computational reproductions","year":2025,"lang":"en","type":"article","venue":"Royal Society Open Science","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; Simon Fraser University","funders":"Agencia Nacional de Investigación y Desarrollo","keywords":"Transparency (behavior); Replication (statistics); Computer science; Decimal; Reliability (semiconductor); Opacity; Workflow; Code (set theory); Reproduction; Group (periodic table); Statistics; Psychology; Mathematics; Arithmetic; Programming language; Computer security; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5224916,0.0009884202,0.001533575,0.004246764,0.004899969,0.00673674,0.005089819,0.003933568,0.002154004],"category_scores_gemma":[0.9157522,0.001636959,0.002061968,0.003766078,0.01552137,0.008416415,0.005879583,0.004982929,0.0007732405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006205534,"about_ca_system_score_gemma":0.007517142,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003816778,"about_ca_topic_score_gemma":0.002503282,"domain_scores_codex":[0.2035883,0.6801349,0.03259148,0.02215898,0.05825001,0.003276457],"domain_scores_gemma":[0.01562139,0.8296679,0.04922832,0.0772096,0.02751579,0.0007569751],"domain_codex":"methods","domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002781038,0.001013898,0.4119231,0.00243667,0.001610121,0.00122759,0.3973638,0.00481467,0.003107135,0.03139179,0.00447294,0.1378572],"study_design_scores_gemma":[0.00142525,0.008901807,0.5266778,0.009345274,0.001703827,0.005022384,0.1628806,0.07395251,0.01785576,0.09406788,0.09693909,0.001227731],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7653807,0.00248764,0.1978542,0.006407094,0.00077021,0.004085578,0.0003901764,0.0006365884,0.02198776],"genre_scores_gemma":[0.9735045,0.0001382148,0.02333126,0.0005725855,0.0001398543,0.001432851,0.0001038015,0.0001713664,0.0006054956],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4775084,"threshold_uncertainty_score":0.5888529,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03172278573118792,"score_gpt":0.3656830597082144,"score_spread":0.3339602739770264,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}