{"id":"W7104378963","doi":"10.18653/v1/2025.hcinlp-1.15","title":"Time Is Effort: Estimating Human Post-Editing Time for Grammar Error Correction Tool Evaluation","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Waterloo","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Government of Canada; Canadian Institute for Advanced Research","keywords":"Error detection and correction; Grammar; Error analysis; Human error","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0075583,0.001878831,0.0008315413,0.005813066,0.0007434762,0.002367586,0.001483632,0.001360825,0.002328483],"category_scores_gemma":[0.05992957,0.000365486,0.0007742133,0.003209154,0.0007094557,0.002549491,0.001644913,0.001065805,0.002430647],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001036459,"about_ca_system_score_gemma":0.00119858,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004486355,"about_ca_topic_score_gemma":0.01059103,"domain_scores_codex":[0.9854462,0.005113005,0.001672048,0.002755905,0.004458794,0.0005541768],"domain_scores_gemma":[0.9140602,0.05179284,0.008500405,0.007554193,0.01600034,0.002091976],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00301268,0.001089188,0.1345848,0.003815734,0.0007175409,0.0006987074,0.005609307,0.01785922,0.04741663,0.001978321,0.06606019,0.7171577],"study_design_scores_gemma":[0.0004882908,0.004331551,0.5008398,0.0006819749,0.0005075736,0.002018025,0.005427529,0.2510137,0.109814,0.007698093,0.1164294,0.0007500808],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7706803,0.00409239,0.1591397,0.0005526594,0.0006778038,0.001341344,0.02659888,0.0222914,0.01462538],"genre_scores_gemma":[0.7958253,0.0005805955,0.1533044,0.0002766604,0.0001723546,0.001437281,0.04030069,0.002046938,0.006055831],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0075583,"threshold_uncertainty_score":0.0399726,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01819553215468054,"score_gpt":0.3381604627708629,"score_spread":0.3199649306161824,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}