{"id":"W6887749040","doi":"10.17605/osf.io/qrf2w","title":"Assessing the validity and reliability of a baseball pitch discrimination online task","year":2023,"lang":"en","type":"article","venue":"OSF Preprints (OSF Preprints)","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reliability (semiconductor); Task (project management); Action (physics); Validity; Task analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["metaresearch","insufficient_payload"],"category_scores_codex":[0.05305212,0.0002212985,0.0004097574,0.0001767796,0.0003504895,0.0003908933,0.001540443,0.0001306466,0.05939811],"category_scores_gemma":[0.04450454,0.0001510354,0.0002025588,0.0008819092,0.0004409956,0.0006665411,0.001795066,0.0003701524,0.06939658],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001259249,"about_ca_system_score_gemma":0.0001348721,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001950151,"about_ca_topic_score_gemma":0.00007571572,"domain_scores_codex":[0.9919041,0.002438295,0.001237918,0.001912253,0.002154964,0.0003524905],"domain_scores_gemma":[0.9894068,0.005172845,0.0005295012,0.004096683,0.0006511583,0.0001429971],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00006730224,0.001098457,0.8549702,0.0001911434,0.00004628622,0.000007220523,0.004505678,0.006010968,0.01825735,0.0009034901,0.02942686,0.08451505],"study_design_scores_gemma":[0.0003084957,0.000002594782,0.9129551,0.00006890381,0.00004740844,0.000002797379,0.002108095,0.009761328,0.003621785,0.05477741,0.0161547,0.0001913496],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9424802,8.01323e-7,0.003831204,0.003459466,0.000314819,0.0007284104,0.00002865798,0.00007268623,0.04908377],"genre_scores_gemma":[0.9778039,0.00005451649,0.000981056,0.0001021502,0.00004230044,0.00005743197,0.00001559495,0.00001191645,0.02093117],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0843237,"threshold_uncertainty_score":0.9750821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1459806906473429,"score_gpt":0.3918010649520061,"score_spread":0.2458203743046632,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}