{"id":"W4407944201","doi":"10.3390/app15052476","title":"A Small-Scale Evaluation of Large Language Models Used for Grammatical Error Correction in a German Children’s Literature Corpus: A Comparative Study","year":2025,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"German; Computer science; Linguistics; Natural language processing; Artificial intelligence; Philosophy","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004951828,0.00195008,0.0009059155,0.002887388,0.0005928888,0.001556586,0.00209464,0.00131125,0.002249447],"category_scores_gemma":[0.01779799,0.0007430557,0.001093981,0.001389159,0.0008223539,0.002587843,0.001696433,0.001566961,0.001889967],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001925591,"about_ca_system_score_gemma":0.001934343,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0289025,"about_ca_topic_score_gemma":0.03940212,"domain_scores_codex":[0.996627,0.001619554,0.0003371034,0.0007371248,0.000560744,0.0001185643],"domain_scores_gemma":[0.9865181,0.009260375,0.0004168204,0.001591453,0.001881126,0.0003320824],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002048781,0.001809637,0.03525067,0.004195768,0.001726355,0.001787136,0.004672378,0.1307817,0.04627968,0.002425672,0.03074406,0.7382782],"study_design_scores_gemma":[0.000658012,0.002669479,0.05551844,0.000771763,0.001433769,0.002056372,0.004335125,0.8082122,0.07870109,0.001632752,0.04364295,0.0003681345],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.927676,0.006126051,0.03920004,0.0005138405,0.0004220146,0.0007696268,0.006314825,0.01285475,0.006122862],"genre_scores_gemma":[0.8728984,0.002167757,0.09058771,0.0003156407,0.00009039292,0.0007537436,0.0274666,0.001205159,0.004514527],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0289025,"threshold_uncertainty_score":0.05746853,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04146972705040574,"score_gpt":0.3714450331222936,"score_spread":0.3299753060718879,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}