{"id":"W2992174637","doi":"10.5539/elt.v13n1p31","title":"A Comparability Study of Text Difficulty and Task Characteristics of Parallel Academic IELTS Reading Tests","year":2019,"lang":"en","type":"article","venue":"English Language Teaching","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Comparability; Scope (computer science); Task (project management); Psychology; Test (biology); Reading (process); Construct (python library); Reading comprehension; Natural language processing; Cognitive psychology; Computer science; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01141077,0.0004223826,0.0004885531,0.004364149,0.0003024623,0.00133281,0.0008036317,0.0005753241,0.001546904],"category_scores_gemma":[0.1180808,0.0002120648,0.0006880672,0.002003096,0.000757464,0.001876649,0.002068242,0.0005179812,0.0004273091],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005346906,"about_ca_system_score_gemma":0.0004554105,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001003864,"about_ca_topic_score_gemma":0.0009413129,"domain_scores_codex":[0.9896289,0.003618446,0.002482706,0.001232899,0.002707056,0.0003299731],"domain_scores_gemma":[0.8799595,0.07540856,0.01713824,0.008061776,0.01729333,0.002138631],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003679931,0.000891412,0.8669068,0.0003684158,0.0005741853,0.0003627966,0.008640031,0.0009998591,0.01966706,0.0005413996,0.0004305464,0.09693763],"study_design_scores_gemma":[0.00006645354,0.002005002,0.9894253,0.00002986735,0.00007942555,0.000316262,0.001140346,0.001321358,0.004260957,0.0004529847,0.0008788153,0.00002328806],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9963139,0.0001233652,0.002048752,0.00001997003,0.0000136612,0.00006899933,0.0001745093,0.00002919221,0.001207708],"genre_scores_gemma":[0.9976104,0.00003369753,0.00136596,0.00001684633,0.00001472903,0.00008684631,0.0004472542,0.00002866841,0.0003955824],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01141077,"threshold_uncertainty_score":0.06034666,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01083677456860607,"score_gpt":0.2972518000888276,"score_spread":0.2864150255202215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}