{"id":"W1964735878","doi":"10.1002/j.2333-8504.2005.tb01990.x","title":"ANALYSIS OF DISCOURSE FEATURES AND VERIFICATION OF SCORING LEVELS FOR INDEPENDENT AND INTEGRATED PROTOTYPE WRITTEN TASKS FOR THE NEW TOEFL®","year":2005,"lang":"en","type":"article","venue":"ETS Research Report Series","topic":"Software Engineering Research","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Christian Studies; University of Toronto","funders":"","keywords":"Test of English as a Foreign Language; Computer science; Natural language processing; Psychology; Artificial intelligence; Mathematics education; Language assessment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006617932,0.000700636,0.00049584,0.003654033,0.000576063,0.001602847,0.0005935775,0.0006446332,0.002892792],"category_scores_gemma":[0.061495,0.000299939,0.0003242578,0.001147506,0.0009802959,0.001016068,0.001890846,0.0006020396,0.0005511401],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006505002,"about_ca_system_score_gemma":0.0003782703,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001086941,"about_ca_topic_score_gemma":0.001705191,"domain_scores_codex":[0.9955914,0.001354548,0.0007100308,0.000877798,0.001195198,0.000271036],"domain_scores_gemma":[0.8994058,0.06557185,0.01032521,0.006136184,0.01640728,0.002153632],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00393809,0.00116979,0.5940852,0.0005366832,0.0002360457,0.001297719,0.04221267,0.0009733316,0.1607037,0.000690133,0.0007542687,0.1934025],"study_design_scores_gemma":[0.00004753845,0.001240185,0.9771783,0.00002508488,0.00002865174,0.0003675392,0.0035165,0.001064474,0.01539782,0.0002040644,0.000891594,0.00003827122],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9983405,0.00003276816,0.0005350212,0.00001076389,0.000004505675,0.00004390237,0.0001070489,0.00002153974,0.0009038528],"genre_scores_gemma":[0.9959709,0.00002334089,0.002109016,0.00001115467,0.000008849248,0.000121383,0.0003013595,0.00001788547,0.001435976],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006617932,"threshold_uncertainty_score":0.03499937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.075963810214805,"score_gpt":0.3964993958148846,"score_spread":0.3205355856000797,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}