{"id":"W2080505681","doi":"10.1016/j.infsof.2004.08.005","title":"Replicating software engineering experiments: a poisoned chalice or the Holy Grail","year":2004,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Engineering Research","field":"Computer Science","cited_by":84,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Holy Grail; Software; Engineering; Computer science; Software engineering; World Wide Web; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06323672,0.001442554,0.002498271,0.001459654,0.001512408,0.005189366,0.005056984,0.005020223,0.006548301],"category_scores_gemma":[0.2509927,0.0009783179,0.00107566,0.001090145,0.01105365,0.01065702,0.005380884,0.007050334,0.002960169],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001848042,"about_ca_system_score_gemma":0.002501081,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001703026,"about_ca_topic_score_gemma":0.001173542,"domain_scores_codex":[0.9632047,0.02727713,0.0008987893,0.003089546,0.00498331,0.0005463979],"domain_scores_gemma":[0.7263274,0.1126637,0.008899547,0.1414029,0.007861459,0.00284497],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.009841391,0.003174037,0.02708678,0.002080146,0.002782403,0.001188288,0.01381217,0.02785595,0.02514668,0.3657913,0.0900256,0.4312153],"study_design_scores_gemma":[0.0024513,0.004757293,0.008333061,0.0007904514,0.0005571673,0.0005020724,0.002216642,0.0360754,0.01486768,0.8349151,0.09408878,0.0004451532],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2994825,0.01331159,0.5141412,0.07973523,0.01472568,0.002998211,0.001943111,0.008039062,0.06562336],"genre_scores_gemma":[0.8377234,0.002544607,0.1193222,0.01864957,0.00222378,0.002252849,0.0006175344,0.001707586,0.01495835],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9367633,"threshold_uncertainty_score":0.3344318,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0136977408834442,"score_gpt":0.2550356574540044,"score_spread":0.2413379165705602,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}