{"id":"W4310641393","doi":"10.1177/17456916221134575","title":"Improving the Generalizability of Behavioral Science by Using Reality Checks: A Tool for Assessing Heterogeneity in Participants’ Consumership of Study Stimuli","year":2022,"lang":"en","type":"article","venue":"Perspectives on Psychological Science","topic":"Decision-Making and Behavioral Economics","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Scarborough Hospital; University of Toronto","funders":"","keywords":"Generalizability theory; Realism; Psychology; Relevance (law); External validity; Writ; Cognitive psychology; Ecological validity; Behavioural sciences; Social psychology; Nomothetic; Experimental psychology; Epistemology; Cognition; Developmental psychology; Psychotherapist","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":[],"category_scores_codex":[0.03311624,0.0001890201,0.0004915916,0.0003477348,0.001135418,0.0003214672,0.002845166,0.00004298082,0.00008619465],"category_scores_gemma":[0.005914487,0.000120768,0.0001594296,0.003373705,0.004488085,0.0006083165,0.0007276664,0.0003123064,9.474791e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005794726,"about_ca_system_score_gemma":0.0003062423,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004764442,"about_ca_topic_score_gemma":0.00003203445,"domain_scores_codex":[0.9936182,0.0007234253,0.001126306,0.001757723,0.002127929,0.0006464001],"domain_scores_gemma":[0.9959325,0.001327969,0.000756519,0.00129061,0.000555446,0.0001369633],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005402004,0.009901172,0.6740578,0.000003680915,0.000005489469,0.000005204975,0.00973036,0.005267802,0.2019118,0.0004476101,0.00003186748,0.09809697],"study_design_scores_gemma":[0.001767534,0.002907485,0.8088639,0.00001595824,0.00004460128,0.00001159488,0.1451602,0.02604054,0.005813813,0.008825495,0.00002265796,0.0005262304],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.997386,0.00004115384,0.001063995,0.000136463,0.0003661973,0.0008607756,0.0000825977,0.00001479958,0.00004803444],"genre_scores_gemma":[0.9985084,9.006198e-7,0.001351364,0.00005417678,0.000009988552,0.00006545353,3.361648e-7,0.000005758024,0.000003606734],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.196098,"threshold_uncertainty_score":0.9982211,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4827743902377341,"score_gpt":0.5657951970758375,"score_spread":0.08302080683810342,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}