{"id":"W4405272702","doi":"10.1109/ubmk63289.2024.10773601","title":"Word Image Representation at Local and Global Levels Based on Vision Transformers","year":2024,"lang":"en","type":"article","venue":"","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure","funders":"","keywords":"Computer science; Transformer; Computer vision; Artificial intelligence; Word (group theory); Image (mathematics); Representation (politics); Natural language processing; Linguistics; Engineering; Electrical engineering; Political science; Voltage","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001398315,0.0006910998,0.0004503383,0.0005131201,0.000130954,0.0007417748,0.0008209027,0.0004531509,0.002345026],"category_scores_gemma":[0.0005479301,0.0002160131,0.000456355,0.0004692685,0.000343703,0.001324118,0.0005219382,0.0006730738,0.001435644],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003931241,"about_ca_system_score_gemma":0.0005096843,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002993589,"about_ca_topic_score_gemma":0.004659954,"domain_scores_codex":[0.9998748,0.00001037806,0.00000761769,0.000051697,0.00003514325,0.00002024916],"domain_scores_gemma":[0.9998624,0.0000260173,0.00001777459,0.00002967002,0.00005234062,0.00001183296],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002779756,0.0001347935,0.001139823,0.0001621917,0.00006753093,0.0001641821,0.00007737942,0.06950504,0.1914571,0.006434704,0.003011978,0.7275673],"study_design_scores_gemma":[0.00001300861,0.0001790898,0.001349246,0.00001669244,0.00005422606,0.0001942866,0.00003450801,0.9010642,0.09006114,0.003988019,0.00302624,0.000019269],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0664437,0.0004853148,0.9248512,0.0001168744,0.0001099023,0.0000754745,0.0002351518,0.0034255,0.004256792],"genre_scores_gemma":[0.7760928,0.0007529727,0.2082691,0.0001770424,0.00006506866,0.00009579098,0.0008203432,0.0001944477,0.01353253],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002993589,"threshold_uncertainty_score":0.007844865,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01798399319047412,"score_gpt":0.3111256265403669,"score_spread":0.2931416333498928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}