{"id":"W4394687365","doi":"10.2196/50315","title":"Measuring the Reliability of a Gamified Stroop Task: Quantitative Experiment","year":2024,"lang":"en","type":"article","venue":"JMIR Serious Games","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria; University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Stroop effect; Intraclass correlation; Task (project management); Reliability (semiconductor); Cognition; Consistency (knowledge bases); Psychology; Cognitive psychology; Computer science; Psychometrics; Developmental psychology; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01054563,0.0007823932,0.0005551835,0.0006434259,0.0003841732,0.0007971336,0.0005974743,0.0007544506,0.001578808],"category_scores_gemma":[0.04223997,0.0004628775,0.0005020519,0.0004829367,0.001094879,0.000458291,0.0008120203,0.0007282281,0.0005890651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002324575,"about_ca_system_score_gemma":0.0003593368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003334014,"about_ca_topic_score_gemma":0.0004196473,"domain_scores_codex":[0.9889182,0.004651733,0.001207911,0.001979783,0.002839939,0.0004024431],"domain_scores_gemma":[0.9546802,0.02717545,0.005646244,0.004704842,0.007010388,0.0007827774],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01158179,0.01952579,0.2543528,0.001948999,0.001128284,0.0004257273,0.01662778,0.008833053,0.5095348,0.001543389,0.002030468,0.1724671],"study_design_scores_gemma":[0.0009563701,0.0689493,0.6837328,0.0002288917,0.0007275361,0.001142824,0.001826424,0.03247163,0.1998771,0.002681657,0.007003156,0.0004023344],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9676178,0.0001038744,0.02962722,0.00002825892,0.00003616219,0.0009689122,0.0001769158,0.0001021345,0.001338724],"genre_scores_gemma":[0.9640014,0.00006997505,0.032376,0.00005968098,0.00004434234,0.002372048,0.0002330303,0.00007482388,0.00076863],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01054563,"threshold_uncertainty_score":0.05577129,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05922230846763064,"score_gpt":0.3578098590581152,"score_spread":0.2985875505904845,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}