{"id":"W2001358222","doi":"10.12735/ier.v1i1p01","title":"A New Method to Setting Standard for the Wide Range of Language Proficiency Levels","year":2013,"lang":"en","type":"article","venue":"International Education Research","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Range (aeronautics); Computer science; Natural language processing; Linguistics; Engineering; Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001740647,0.00007769366,0.00009301232,0.0002161479,0.0001179734,0.00007499403,0.0006429105,0.000050771,0.0131936],"category_scores_gemma":[0.001020064,0.00005314254,0.00005914173,0.0003592439,0.00004391387,0.00009262188,0.00006161951,0.0001944248,0.0004592197],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009666917,"about_ca_system_score_gemma":0.0005044662,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001829386,"about_ca_topic_score_gemma":0.00001349735,"domain_scores_codex":[0.9983675,0.0002191529,0.0002779328,0.0002746309,0.0006067631,0.0002540453],"domain_scores_gemma":[0.9957952,0.002364127,0.00008063548,0.0002502688,0.001389741,0.0001200352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0001086963,0.0003776513,0.01821528,0.00001373339,0.00008754843,1.957015e-7,0.01072819,0.000009070731,0.001685477,0.05416863,0.5127965,0.401809],"study_design_scores_gemma":[0.0005868301,0.0002768032,0.7568237,0.00005607079,0.000009219307,0.000006504362,0.02492142,0.00006249556,0.0009231919,0.03215021,0.184032,0.0001515318],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3879514,0.0006900005,0.2455339,0.2446572,0.006547213,0.004628099,0.0001971611,0.00005002365,0.109745],"genre_scores_gemma":[0.7943373,0.000003234324,0.08476368,0.002286844,0.0009597516,0.002318075,0.00003177731,0.00001921473,0.1152802],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7386084,"threshold_uncertainty_score":0.9877084,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2088285237702343,"score_gpt":0.6242796495812258,"score_spread":0.4154511258109915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}