{"id":"W4220747158","doi":"10.1136/bmjsem-2021-001259","title":"Why machine learning (ML) has failed physical activity research and how we can improve","year":2022,"lang":"en","type":"article","venue":"BMJ Open Sport & Exercise Medicine","topic":"Physical Activity and Health","field":"Medicine","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan; University of Calgary; Memorial University of Newfoundland","funders":"","keywords":"Benchmark (surveying); Computer science; Field (mathematics); Software; Physical activity; Point (geometry); Measure (data warehouse); Machine learning; Artificial intelligence; Data science; Human–computer interaction; Data mining; Physical medicine and rehabilitation; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.002754113,0.0003447193,0.002200601,0.0002779695,0.001314044,0.00009043716,0.0004621816,0.00008409443,0.0005752852],"category_scores_gemma":[0.0002082425,0.0002715621,0.00009099725,0.0008875254,0.000674434,0.0003135257,0.001368882,0.0027724,0.00001249277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003230927,"about_ca_system_score_gemma":0.0006170984,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009431699,"about_ca_topic_score_gemma":0.0004406372,"domain_scores_codex":[0.9960612,0.0002280813,0.0002628215,0.0009187732,0.001707846,0.0008212955],"domain_scores_gemma":[0.9978024,0.000274997,0.0002399152,0.0007310489,0.0002024726,0.0007491978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.04551057,0.02105746,0.07594123,0.00426747,0.0006172541,0.00499932,0.0251927,0.00003434086,0.09607944,0.002962704,0.2400888,0.4832487],"study_design_scores_gemma":[0.02677833,0.02799073,0.1116807,0.002658477,0.0009217695,0.0004697721,0.01163006,0.01252572,0.008170752,0.003535475,0.7920226,0.001615685],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7559555,0.0002425349,0.00001734493,0.2364678,0.0001732159,0.004665386,0.00003400187,0.00008020883,0.002364004],"genre_scores_gemma":[0.9866436,0.0002658142,0.00007100381,0.0008934531,0.0008613538,0.001111256,0.0001324334,0.00006831156,0.009952774],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5519338,"threshold_uncertainty_score":0.9999861,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1753162058630499,"score_gpt":0.4336553102622821,"score_spread":0.2583391043992322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}