{"id":"W4389923528","doi":"10.21449/ijate.1394194","title":"Language models in automated essay scoring: Insights for the Turkish language","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Turkish; Transformative learning; Computer science; Language model; Artificial intelligence; Natural language processing; Intersection (aeronautics); Transformer; Linguistics; Sociology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007599398,0.00007845592,0.0001106344,0.000465956,0.00002811712,0.0002425978,0.001018761,0.00003735898,0.000004212897],"category_scores_gemma":[0.0001372648,0.00006084357,0.00005530272,0.000332686,0.000009985926,0.001091676,0.00009946679,0.0001830121,0.000002599705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004066423,"about_ca_system_score_gemma":0.0006739506,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007598846,"about_ca_topic_score_gemma":0.00004941729,"domain_scores_codex":[0.9987821,0.00006065214,0.0004391975,0.0001442078,0.0004429922,0.0001308722],"domain_scores_gemma":[0.9989404,0.000376723,0.0002372042,0.0001884939,0.0002278999,0.00002926889],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003674653,0.0006429817,0.006981761,0.00003502152,0.0001230209,0.0001032747,0.04177945,0.3662012,0.003641759,0.1514968,0.002376982,0.4265811],"study_design_scores_gemma":[0.0005118141,0.00002442918,0.0341858,0.0001559632,0.000003899799,0.00002046377,0.002923865,0.9516084,0.0002944002,0.009800757,0.0003855107,0.00008474191],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7406383,0.000315397,0.2491135,0.00450794,0.00427133,0.0002527768,0.000002439434,0.00007713098,0.0008212284],"genre_scores_gemma":[0.9795254,0.00006187551,0.01969401,0.000214112,0.000327195,0.00003938811,0.000009102637,0.00000694359,0.0001219313],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5854072,"threshold_uncertainty_score":0.2481129,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0407797704534206,"score_gpt":0.3934933832112348,"score_spread":0.3527136127578142,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}