{"id":"W2125759561","doi":"10.1145/1368088.1368115","title":"On the difficulty of replicating human subjects studies in software engineering","year":2008,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":67,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Replication (statistics); Comparability; Computer science; Context (archaeology); Literal (mathematical logic); Empirical research; Software; Test (biology); Artificial intelligence; Software engineering; Cognitive psychology; Programming language; Psychology; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6270509,0.003050905,0.003164387,0.003975099,0.005538221,0.007378999,0.008972805,0.00813703,0.003246535],"category_scores_gemma":[0.8422155,0.002213233,0.003584767,0.004296158,0.01660033,0.01455902,0.006195382,0.008059822,0.001673139],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003283343,"about_ca_system_score_gemma":0.005293241,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00262286,"about_ca_topic_score_gemma":0.002201772,"domain_scores_codex":[0.2396013,0.6667724,0.03068012,0.02134493,0.04012842,0.001472848],"domain_scores_gemma":[0.07145346,0.6813721,0.03503197,0.1765843,0.0345786,0.000979533],"domain_codex":"methods","domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.01220926,0.00997311,0.1158165,0.0180796,0.01510938,0.007730332,0.09153187,0.0214554,0.02435245,0.2728906,0.02704422,0.3838073],"study_design_scores_gemma":[0.01383297,0.04054641,0.06386396,0.0128036,0.005374427,0.0051576,0.02634814,0.04971648,0.03804159,0.5677426,0.1751449,0.001427415],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.101482,0.005847195,0.8341286,0.01117046,0.007403795,0.02278712,0.0007211345,0.000970443,0.01548944],"genre_scores_gemma":[0.3814714,0.001119511,0.5764634,0.008117389,0.001531282,0.02871246,0.0004515894,0.0004531544,0.001679842],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3729491,"threshold_uncertainty_score":0.4599127,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07825651788581889,"score_gpt":0.3125074568354242,"score_spread":0.2342509389496053,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}