{"id":"W3087153552","doi":"10.1177/0265532220957298","title":"Hanyu Shuiping Kaoshi (HSK): A multi-level, multi-purpose proficiency test","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":50,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Language proficiency; Test (biology); Psychology; Argument (complex analysis); Language assessment; Scale (ratio); Mathematics education; Linguistics; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007883831,0.0003635658,0.000728913,0.004107976,0.0002901108,0.0008506479,0.0007496743,0.0005461299,0.002374853],"category_scores_gemma":[0.01831223,0.0001553423,0.000503607,0.002219226,0.0005447093,0.001055391,0.0008729197,0.0006051452,0.0005317453],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009432597,"about_ca_system_score_gemma":0.008148988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004707601,"about_ca_topic_score_gemma":0.00616891,"domain_scores_codex":[0.9964742,0.001218317,0.0006682827,0.0002312399,0.001332265,0.00007565485],"domain_scores_gemma":[0.9877635,0.005560908,0.001361621,0.0004230697,0.004363662,0.0005271061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001663854,0.0001927761,0.01618507,0.006537811,0.0002786633,0.0001877622,0.0001759719,0.0002011855,0.001458754,0.001056403,0.0088497,0.9647096],"study_design_scores_gemma":[0.0008867993,0.005226997,0.4557226,0.01818327,0.004295383,0.005888433,0.001075979,0.004776957,0.02178539,0.005644336,0.4761869,0.0003269927],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"methods","genre_scores_codex":[0.2014338,0.6605176,0.04249473,0.0206019,0.003905125,0.005587209,0.006327467,0.001364362,0.05776779],"genre_scores_gemma":[0.6571354,0.2474284,0.067031,0.003393615,0.00122001,0.003545743,0.006821753,0.0001395268,0.01328453],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.007883831,"threshold_uncertainty_score":0.04169416,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8227491867455936,"score_gpt":0.5099737195301289,"score_spread":0.3127754672154647,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}