{"id":"W4412886529","doi":"10.21203/rs.3.rs-7226688/v1","title":"Large Language Models as Mediators: Addressing Rater Disagreement in Turkish Essay Scoring","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Rubric; Turkish; Writing assessment; Task (project management); Test (biology); Active listening; Psychology; Consistency (knowledge bases); Benchmark (surveying); Applied psychology; Computer science; Cognitive psychology; Mathematics education; Linguistics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1075587,0.001859512,0.002024961,0.0026362,0.003091766,0.007873283,0.003719653,0.003453841,0.003633068],"category_scores_gemma":[0.3484689,0.00108703,0.001438503,0.003082151,0.002087604,0.008003675,0.007051114,0.004736295,0.001812372],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002134817,"about_ca_system_score_gemma":0.002962075,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002622127,"about_ca_topic_score_gemma":0.003219574,"domain_scores_codex":[0.8361216,0.1469199,0.003304284,0.007274671,0.004659792,0.001719775],"domain_scores_gemma":[0.588212,0.3664362,0.01128105,0.01756182,0.01425891,0.002250006],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01076247,0.001626271,0.2724015,0.001793432,0.004308912,0.001389714,0.03698622,0.06064055,0.0157314,0.06103264,0.01754913,0.5157778],"study_design_scores_gemma":[0.0006013841,0.0008702424,0.07047709,0.0004347627,0.002761425,0.0007227334,0.006907807,0.8139919,0.01909745,0.0734902,0.01031352,0.0003315243],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6047645,0.002974364,0.3749422,0.003915283,0.000454973,0.0005588299,0.0007285121,0.001345337,0.01031602],"genre_scores_gemma":[0.9750676,0.0001158205,0.02258132,0.0001307849,0.0001422325,0.00029666,0.0004597208,0.0002484139,0.0009574211],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1075587,"threshold_uncertainty_score":0.5688316,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09587885843140762,"score_gpt":0.4205200915898551,"score_spread":0.3246412331584475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}