{"id":"W4362735211","doi":"10.1080/1091367x.2023.2199126","title":"Trust the “Process”? When Fundamental Motor Skill Scores are Reliably Unreliable","year":2023,"lang":"en","type":"article","venue":"Measurement in Physical Education and Exercise Science","topic":"Children's Physical and Motor Development","field":"Psychology","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Psychology; Gross motor skill; Motor skill; Reliability (semiconductor); Inter-rater reliability; Internal consistency; Physical medicine and rehabilitation; Applied psychology; Physical therapy; Clinical psychology; Psychometrics; Rating scale; Developmental psychology; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2236547,0.0008785538,0.002152272,0.004494981,0.0023009,0.006564439,0.002585037,0.003525132,0.001986862],"category_scores_gemma":[0.6182199,0.001341227,0.001167316,0.004074429,0.009393682,0.01045302,0.00515025,0.003872149,0.001237737],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002950058,"about_ca_system_score_gemma":0.004777323,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008917096,"about_ca_topic_score_gemma":0.008597557,"domain_scores_codex":[0.7708263,0.1406909,0.03351539,0.01296573,0.03755205,0.004449562],"domain_scores_gemma":[0.3609234,0.3859729,0.1027417,0.05928059,0.08766814,0.003413311],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001137816,0.000188776,0.6486231,0.002565847,0.001270344,0.002161536,0.06969017,0.00167752,0.003012015,0.0191964,0.01915814,0.2313184],"study_design_scores_gemma":[0.0003400568,0.00172499,0.5666955,0.01027012,0.001250821,0.007748382,0.08746942,0.03806806,0.01491352,0.1713368,0.09929573,0.0008867295],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6040819,0.009955931,0.2885718,0.04677433,0.004673341,0.001882289,0.00113397,0.001137373,0.04178894],"genre_scores_gemma":[0.9454008,0.001160578,0.04547174,0.004062889,0.0007581211,0.001267045,0.0003093966,0.0002964663,0.001273028],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7763453,"threshold_uncertainty_score":0.9573719,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02853696743147281,"score_gpt":0.310054374542902,"score_spread":0.2815174071114291,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}