{"id":"W2967888896","doi":"10.1186/s40468-019-0089-4","title":"Critical review of validation models and practices in language testing: their limitations and future directions for validation research","year":2019,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Argument (complex analysis); Empirical research; Construct (python library); Computer science; Test (biology); Construct validity; Language assessment; Management science; Psychology; Data science; Psychometrics; Statistics; Mathematics education; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.447146,0.001935126,0.004411831,0.03158281,0.005530765,0.01526358,0.006665227,0.00573032,0.003200144],"category_scores_gemma":[0.7135997,0.001975636,0.003854088,0.02380938,0.01630698,0.0237427,0.007797544,0.01035305,0.001321395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0221,"about_ca_system_score_gemma":0.06731454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008780839,"about_ca_topic_score_gemma":0.00958327,"domain_scores_codex":[0.5562592,0.3161324,0.06128142,0.008355281,0.05580319,0.00216854],"domain_scores_gemma":[0.117742,0.705574,0.02763125,0.02380673,0.1235236,0.001722452],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002655216,0.00009912292,0.003267141,0.1231972,0.0009448603,0.0004537968,0.03095689,0.0008516443,0.0006031974,0.09072774,0.0700017,0.6786312],"study_design_scores_gemma":[0.0001205167,0.0002178603,0.004028926,0.4734623,0.00130573,0.0006501459,0.01991242,0.001425595,0.001688399,0.054178,0.4427722,0.0002378803],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.004839123,0.8135438,0.04342366,0.1149967,0.01007569,0.002256238,0.0004166136,0.0002141802,0.01023392],"genre_scores_gemma":[0.1050491,0.7312768,0.1056349,0.04156501,0.005088027,0.008264305,0.0006607135,0.0004891545,0.001972127],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.552854,"threshold_uncertainty_score":0.6817675,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2871225134574545,"score_gpt":0.4973541369068704,"score_spread":0.2102316234494159,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}