{"id":"W4411549981","doi":"10.31222/osf.io/a9vhr_v1","title":"We Need to Talk about Mechanical Turk: What 22,989 Hypothesis Tests Tell Us about Publication Bias and p-Hacking in Online Experiments","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Misinformation and Its Impacts","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Wilfrid Laurier University","funders":"Canada Research Chairs","keywords":"Hacker; Computer science; Internet privacy; Psychology; Data science; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.1622967,0.0009193658,0.002554862,0.008049364,0.002694861,0.01165714,0.002225889,0.005591187,0.01122491],"category_scores_gemma":[0.6658506,0.0009637645,0.002639099,0.01122219,0.01213617,0.01911638,0.003031669,0.004280177,0.003805881],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002088161,"about_ca_system_score_gemma":0.003527156,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001411655,"about_ca_topic_score_gemma":0.001090186,"domain_scores_codex":[0.8488018,0.095016,0.01551677,0.01475962,0.02300733,0.002898546],"domain_scores_gemma":[0.1003468,0.7781289,0.05428428,0.05067239,0.01462381,0.001943909],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003338292,0.0003850514,0.479719,0.00844416,0.006789539,0.001403472,0.01289492,0.002981794,0.002826218,0.04698141,0.05989193,0.3743443],"study_design_scores_gemma":[0.0007192802,0.001237915,0.4098396,0.007856645,0.003324419,0.002713726,0.009569412,0.008909068,0.007953766,0.3872668,0.1598061,0.0008033055],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4960146,0.08496527,0.09648699,0.2220378,0.008402917,0.001031014,0.02380349,0.002223674,0.06503429],"genre_scores_gemma":[0.937786,0.009542187,0.01863077,0.0209838,0.005422299,0.000739805,0.003375544,0.0008046529,0.002714972],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9944088,"threshold_uncertainty_score":0.858317,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1213034299346663,"score_gpt":0.3724477654311586,"score_spread":0.2511443354964923,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}