{"id":"W3208625967","doi":"10.1162/tacl_a_00427","title":"Lexically Aware Semi-Supervised Learning for OCR Post-Correction","year":2021,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Government of Canada; National Endowment for the Humanities; National Science Foundation","keywords":"Computer science; Decoding methods; Optical character recognition; Consistency (knowledge bases); Artificial intelligence; Natural language processing; Language model; Raw data; Vocabulary; Error detection and correction; Machine learning; Speech recognition; Image (mathematics); Algorithm","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001381945,0.0009173825,0.000897059,0.0007010094,0.0006017464,0.0008160943,0.002388001,0.001092281,0.002492286],"category_scores_gemma":[0.008213647,0.0004547873,0.0006499813,0.0004708325,0.0009261576,0.001442339,0.001165547,0.001737345,0.001604583],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000827758,"about_ca_system_score_gemma":0.001632644,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004704421,"about_ca_topic_score_gemma":0.01034698,"domain_scores_codex":[0.9984806,0.0004780066,0.0001302597,0.0003770059,0.0004385783,0.00009555593],"domain_scores_gemma":[0.9925141,0.003310873,0.0006402131,0.001239196,0.002166445,0.0001291434],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000274528,0.000377386,0.001638641,0.000234335,0.00009145754,0.000132698,0.0002239425,0.2345565,0.05812885,0.002787087,0.002455893,0.6990987],"study_design_scores_gemma":[0.000005090315,0.00003846314,0.000203661,0.000006975134,0.000005896439,0.00002896942,0.00001529363,0.9803652,0.01789188,0.001065004,0.0003643981,0.000009274471],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03763921,0.0001525507,0.9543724,0.000113303,0.00005872252,0.00008650655,0.00008709478,0.006595835,0.0008944007],"genre_scores_gemma":[0.5950214,0.00008055998,0.3996906,0.0001624607,0.00005552913,0.0002435841,0.000510118,0.0005287086,0.00370698],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004704421,"threshold_uncertainty_score":0.009354055,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01389359865853384,"score_gpt":0.2640636733909112,"score_spread":0.2501700747323774,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}