{"id":"W4327916713","doi":"10.1101/2023.03.20.23287486","title":"Characteristics of dynamic assessments of word reading skills and their implications for validity: A systematic review and meta-analysis","year":2023,"lang":"en","type":"review","venue":"medRxiv","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Rehabilitation Institute; University of Toronto","funders":"","keywords":"Reading (process); Psychology; Word (group theory); Test (biology); Meta-analysis; Natural language processing; Cognitive psychology; Computer science; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00154739,0.0003681814,0.006378239,0.0003056697,0.00005221155,0.00001766881,0.0003367982,0.0001840026,0.0001476045],"category_scores_gemma":[0.0002649439,0.0002188022,0.001260839,0.0009118429,0.00008073451,0.00003058283,0.00007040549,0.0001601017,0.000009145879],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002396535,"about_ca_system_score_gemma":0.00004234283,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004743996,"about_ca_topic_score_gemma":0.000001878416,"domain_scores_codex":[0.99683,0.0006597326,0.001660429,0.0005458333,0.0001051317,0.000198903],"domain_scores_gemma":[0.9944457,0.002629665,0.001901124,0.0007611784,0.0001757296,0.00008657992],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.000001114793,0.000335877,0.0004279793,0.867174,0.1225794,3.944274e-7,0.00008557725,4.508539e-9,3.786859e-7,0.001498937,0.00009350598,0.007802881],"study_design_scores_gemma":[0.00008065261,0.00008545518,0.01229016,0.02077767,0.9581883,0.00001690637,0.00003468659,8.701689e-7,3.834511e-8,0.001336587,0.006862213,0.0003264586],"study_design_candidate":"meta_analysis","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0001667945,0.9936519,0.0004863137,0.000196735,0.0001052647,0.002844893,0.002429482,0.00001935111,0.00009930506],"genre_scores_gemma":[0.0007793065,0.993301,0.0003260324,0.0001035997,0.00001368825,0.002891466,0.0004946428,0.00003779541,0.002052467],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8463963,"threshold_uncertainty_score":0.8922493,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2910477597929519,"score_gpt":0.5132938112503193,"score_spread":0.2222460514573673,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}