{"id":"W4408062992","doi":"10.5220/0013340100003905","title":"TokenOCR: An Attention Based Foundational Model for Intelligent Optical Character Recognition","year":2025,"lang":"en","type":"article","venue":"","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Government of Canada; Department of National Defence","funders":"","keywords":"Computer science; Character (mathematics); Character recognition; Optical character recognition; Artificial intelligence; Cognitive science; Natural language processing; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003921402,0.0001246121,0.0001207214,0.0002498892,0.0001300034,0.0002481822,0.0003531141,0.00009272039,0.000107364],"category_scores_gemma":[0.0000618639,0.0001216368,0.0001011019,0.0002063166,0.00003087214,0.0009053741,0.00006627096,0.00008627176,0.00007257473],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000820009,"about_ca_system_score_gemma":0.0001380036,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000003254437,"about_ca_topic_score_gemma":0.000009586176,"domain_scores_codex":[0.9988639,0.00003824993,0.0003024614,0.0004104216,0.000179725,0.0002052395],"domain_scores_gemma":[0.9990501,0.0001102832,0.00005494253,0.0002930956,0.0004176754,0.00007394786],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000383076,0.0003767595,0.00006683126,0.00004284313,0.00002017206,7.469283e-7,0.00004403167,0.00005685956,0.004896847,0.1400713,0.001905282,0.8524801],"study_design_scores_gemma":[0.0002741641,0.00008249592,0.0002707506,0.00003582784,0.00001079198,0.00000150472,0.00000610175,0.8747591,0.03740641,0.08620318,0.0007968749,0.000152848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00451496,0.000002913816,0.9887928,0.002230167,0.0001794662,0.0005282403,0.00001401991,0.0005101658,0.003227211],"genre_scores_gemma":[0.2430878,0.000004721287,0.7513253,0.003384077,0.00005593716,0.0004113311,0.0003590643,0.000009179425,0.001362638],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8747022,"threshold_uncertainty_score":0.4960205,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05316065354006574,"score_gpt":0.3177107171123,"score_spread":0.2645500635722343,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}