{"id":"W4291009573","doi":"10.31222/osf.io/a9vhr","title":"We Need to Talk about Mechanical Turk: What 22,989 Hypothesis Tests Tell Us about Publication Bias and p-Hacking in Online Experiments","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Wilfrid Laurier University; University of Ottawa","funders":"Canada Research Chairs","keywords":"Credibility; Hacker; Set (abstract data type); Trustworthiness; Deception; Psychology; Marketing; Data science; Internet privacy; Computer science; Social psychology; Political science; Business; Computer security","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001095459,0.0005608512,0.0006744584,0.0007464219,0.0002440463,0.002839061,0.001569066,0.0003398735,0.0001930772],"category_scores_gemma":[0.0004931387,0.0005656431,0.0001589397,0.0008486999,0.00004799048,0.0007708112,0.004681147,0.0009093056,0.00002476606],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003626948,"about_ca_system_score_gemma":0.0002132916,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007568951,"about_ca_topic_score_gemma":0.0001844102,"domain_scores_codex":[0.995505,0.0003339502,0.000922038,0.001778056,0.0007616664,0.000699305],"domain_scores_gemma":[0.9968269,0.0004507718,0.0003750393,0.001834639,0.000166581,0.00034605],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009873251,0.00192623,0.005701085,0.000508732,0.0002305792,0.0002816986,0.01618305,0.01105036,0.03460981,0.008105778,0.004497307,0.9168066],"study_design_scores_gemma":[0.004959321,0.0009128232,0.09679256,0.00788643,0.0001832025,0.0005247191,0.01335551,0.6560333,0.09899911,0.01001758,0.1013333,0.009002086],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9625429,0.002096634,0.02622525,0.004834925,0.001463839,0.001142094,0.00001524167,0.0005759418,0.001103165],"genre_scores_gemma":[0.9230394,0.0007050307,0.07177526,0.002501521,0.0002017111,0.0002392071,0.00004989569,0.00007507626,0.001412858],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9078045,"threshold_uncertainty_score":0.9996795,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0625741574800946,"score_gpt":0.2973621482351371,"score_spread":0.2347879907550425,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}