{"id":"W2991191531","doi":"10.1037/bul0000220","title":"Crowdsourcing hypothesis tests: Making transparent how design choices shape research results.","year":2020,"lang":"en","type":"article","venue":"Psychological Bulletin","topic":"Behavioral Health and Interventions","field":"Psychology","cited_by":180,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Memorial University of Newfoundland; University of British Columbia; Booth University College","funders":"Marsden Fund; Knut och Alice Wallenbergs Stiftelse; Institut Européen d'Administration des Affaires; Austrian Science Fund; Jan Wallanders och Tom Hedelius Stiftelse samt Tore Browaldhs Stiftelse","keywords":"PsycINFO; Psychology; Statistical hypothesis testing; Consistency (knowledge bases); Empirical research; Test (biology); Crowdsourcing; Statistical power; Research design; Social psychology; Null hypothesis; Cognitive psychology; Applied psychology; Econometrics; Statistics; Computer science; MEDLINE; Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6735176,0.002554327,0.003270292,0.006923426,0.003232324,0.01019935,0.008600037,0.006151709,0.00951313],"category_scores_gemma":[0.8584554,0.00239695,0.005967376,0.006773532,0.01514811,0.01044351,0.01489372,0.007806657,0.002153364],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00541618,"about_ca_system_score_gemma":0.01328668,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001486001,"about_ca_topic_score_gemma":0.002053993,"domain_scores_codex":[0.3105581,0.6284496,0.02303982,0.01591147,0.02065124,0.001389692],"domain_scores_gemma":[0.04901931,0.8584153,0.01884651,0.06066356,0.01146658,0.001588686],"domain_codex":"methods","domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0077782,0.0005173211,0.02874576,0.02479039,0.01307543,0.0006918666,0.03867307,0.009296437,0.007303467,0.09147607,0.04788028,0.7297717],"study_design_scores_gemma":[0.008233374,0.002494379,0.03358577,0.01144369,0.006222269,0.0003892237,0.006080197,0.04294953,0.01393825,0.7334287,0.1404342,0.0008003358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04243391,0.007852937,0.8519858,0.02908181,0.006924475,0.03201026,0.00392087,0.003480901,0.02230895],"genre_scores_gemma":[0.2156582,0.001584667,0.7053568,0.007403518,0.001025337,0.06615657,0.0007176119,0.0006687943,0.001428542],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3264824,"threshold_uncertainty_score":0.402611,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6072536598472963,"score_gpt":0.5019315143195294,"score_spread":0.1053221455277669,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}