{"id":"W1501272469","doi":"","title":"Sparse descriptor for lexicon reduction in handwritten Arabic documents","year":2012,"lang":"en","type":"article","venue":"Espace ÉTS (ETS)","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; École de Technologie Supérieure","funders":"","keywords":"Lexicon; Artificial intelligence; Arabic; Computer science; Histogram; Pattern recognition (psychology); Word (group theory); Natural language processing; Skeleton (computer programming); Pixel; Reduction (mathematics); Image (mathematics); Mathematics; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007853696,0.000223394,0.0002488557,0.0003496305,0.0001177615,0.0001903263,0.0005233483,0.0001488606,0.00006031929],"category_scores_gemma":[0.0001125729,0.000224395,0.0001005053,0.000452408,0.00005178382,0.002042591,0.0001450328,0.0001885815,0.0002224573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001945093,"about_ca_system_score_gemma":0.00006351556,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006341274,"about_ca_topic_score_gemma":0.00003371593,"domain_scores_codex":[0.9982038,0.000113828,0.0003133843,0.0004442554,0.0002750791,0.0006497251],"domain_scores_gemma":[0.9989846,0.00006226749,0.0001398632,0.0005157157,0.0001049352,0.0001925954],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003484679,0.002296972,0.02188492,0.0003112945,0.0001514586,0.00002951229,0.01472757,0.00002792475,0.139724,0.09664143,0.1387147,0.5851417],"study_design_scores_gemma":[0.004746118,0.0007773525,0.02270249,0.0005645822,0.00006215175,0.000244509,0.0007550558,0.004582164,0.7270972,0.03725775,0.1990957,0.002114892],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3364649,0.0008127139,0.6467808,0.005364773,0.002919361,0.00220034,0.00001413288,0.001195312,0.004247677],"genre_scores_gemma":[0.9067735,0.00005094423,0.08955888,0.0002858464,0.0003291676,0.0003749506,0.00001183464,0.00002882971,0.002586028],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5873732,"threshold_uncertainty_score":0.9150565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02709086643507978,"score_gpt":0.2802605596387316,"score_spread":0.2531696932036518,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}