{"id":"W4233143865","doi":"10.31234/osf.io/hwubj","title":"Promises and perils of experimentation: The mutual internal validity problem","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Experimental Behavioral Economics Studies","field":"Social Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"External validity; Internal validity; Triangulation; Epistemology; Test (biology); Psychology; Management science; Computer science; Cognitive psychology; Social psychology; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6858295,0.003469136,0.007472705,0.009339548,0.008434159,0.01628955,0.01042669,0.01713871,0.008577218],"category_scores_gemma":[0.8301145,0.003690082,0.005103519,0.006104158,0.08545662,0.04433795,0.0226797,0.02632189,0.002348756],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01371802,"about_ca_system_score_gemma":0.0174135,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002729256,"about_ca_topic_score_gemma":0.001664413,"domain_scores_codex":[0.1744178,0.6787802,0.0207339,0.02820461,0.09520854,0.002654997],"domain_scores_gemma":[0.05096407,0.8067997,0.02128147,0.1022008,0.01734278,0.001411317],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000946814,0.000312849,0.007313943,0.003259365,0.001669182,0.0003452337,0.01025995,0.002679746,0.0005217416,0.8951412,0.00887476,0.06867523],"study_design_scores_gemma":[0.0005138286,0.000274003,0.001374312,0.001893137,0.0002195895,0.0002256128,0.0009965071,0.003629438,0.0006274761,0.9676725,0.0224291,0.0001444704],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02191985,0.02752974,0.6139197,0.2010322,0.00569529,0.005056825,0.0009003828,0.001068204,0.1228778],"genre_scores_gemma":[0.5437679,0.008479786,0.3359775,0.06902,0.007630793,0.02676333,0.0005403062,0.0009623057,0.006858142],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3141705,"threshold_uncertainty_score":0.3874283,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.159344359733532,"score_gpt":0.4009219403315423,"score_spread":0.2415775805980103,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}