{"id":"W7160034663","doi":"10.1109/iccv51701.2025.02155","title":"A Token-Level Text Image Foundation Model for Document Understanding","year":2025,"lang":"","type":"article","venue":"","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Natural Science Foundation of China","keywords":"Foundation (evidence); Image (mathematics); Context (archaeology); Image processing; Feature (linguistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003661744,0.0006083773,0.0007224896,0.000804253,0.0003828829,0.001046787,0.001710646,0.0008289313,0.004666323],"category_scores_gemma":[0.001181792,0.000392221,0.0009931703,0.0008972898,0.0003325072,0.002358868,0.0007414133,0.001302847,0.002714708],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007280448,"about_ca_system_score_gemma":0.001186033,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01060455,"about_ca_topic_score_gemma":0.01255021,"domain_scores_codex":[0.9997655,0.00002232689,0.00001750958,0.0000914454,0.0000663406,0.00003682304],"domain_scores_gemma":[0.9995703,0.0001082967,0.00004203628,0.00009858466,0.0001507374,0.00002997935],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007678532,0.0002483152,0.001535117,0.0002860586,0.0001457023,0.0003532433,0.000158073,0.2643059,0.07487697,0.03213071,0.01090794,0.6142842],"study_design_scores_gemma":[0.000007097516,0.00003828445,0.0002716943,0.00000899254,0.0000252158,0.00004713929,0.00000869916,0.9856564,0.007272315,0.005063949,0.001589291,0.00001090585],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01329728,0.00023885,0.9813161,0.0001010389,0.0000539864,0.00005005956,0.000537734,0.003302908,0.001101934],"genre_scores_gemma":[0.4808992,0.0008344852,0.4967226,0.0001885062,0.0001162919,0.0002455081,0.003536327,0.0008917841,0.0165653],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01060455,"threshold_uncertainty_score":0.02108568,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09104862361110418,"score_gpt":0.3344530283190299,"score_spread":0.2434044047079257,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}