{"id":"W6887749040","doi":"10.17605/osf.io/qrf2w","title":"Assessing the validity and reliability of a baseball pitch discrimination online task","year":2023,"lang":"en","type":"article","venue":"OSF Preprints (OSF Preprints)","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reliability (semiconductor); Task (project management); Action (physics); Validity; Task analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0308211,0.0005764804,0.0006769639,0.001864635,0.0008203498,0.001662223,0.0009715349,0.001721422,0.001505492],"category_scores_gemma":[0.1278567,0.0006743672,0.001312837,0.0009067165,0.001549864,0.001786945,0.001996757,0.001244429,0.001272308],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005526353,"about_ca_system_score_gemma":0.00112234,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003329142,"about_ca_topic_score_gemma":0.00339855,"domain_scores_codex":[0.9759851,0.009503819,0.002818343,0.003685194,0.007087141,0.0009204927],"domain_scores_gemma":[0.7154153,0.1904926,0.01624827,0.02756287,0.04774814,0.002532814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003664091,0.001133216,0.8916433,0.0001897003,0.001016797,0.0001316932,0.006204398,0.002962484,0.01448476,0.001851885,0.001482281,0.07523549],"study_design_scores_gemma":[0.0003059832,0.001983423,0.9541172,0.0001056042,0.000342987,0.0004537835,0.001385677,0.02492767,0.01039661,0.002385077,0.003460071,0.0001358243],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.975267,0.0002038721,0.01877755,0.0001079373,0.0001173089,0.0002753708,0.0003126964,0.0001055642,0.00483261],"genre_scores_gemma":[0.9934046,0.00004474145,0.005165332,0.00006475929,0.00004045779,0.0002182633,0.000330686,0.00006064017,0.0006705576],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0308211,"threshold_uncertainty_score":0.1629995,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1459806906473429,"score_gpt":0.3918010649520061,"score_spread":0.2458203743046632,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}