{"id":"W4388195444","doi":"10.1016/j.knosys.2023.111080","title":"Table detection for visually rich document images","year":2023,"lang":"en","type":"article","venue":"Knowledge-Based Systems","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"Mitacs","keywords":"Computer science; Information loss; Intersection (aeronautics); Artificial intelligence; Task (project management); Ground truth; Image (mathematics); Table (database); Data mining; Term (time); Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001503185,0.0006986374,0.0007500327,0.002015969,0.0003351836,0.001292225,0.0008093347,0.0007150992,0.009361632],"category_scores_gemma":[0.0008120865,0.0002747717,0.0004957257,0.001197525,0.0001569288,0.0009142421,0.0004751753,0.0004117158,0.004298835],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005123221,"about_ca_system_score_gemma":0.0004930611,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005240534,"about_ca_topic_score_gemma":0.008780565,"domain_scores_codex":[0.999813,0.00001078285,0.000011987,0.00005361001,0.00006958809,0.00004099848],"domain_scores_gemma":[0.9996213,0.000117539,0.00004585485,0.00006227291,0.0001224425,0.00003053423],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005569304,0.00008997713,0.001505891,0.0003135093,0.00004585194,0.0004291855,0.00005309953,0.004848601,0.2765442,0.0008863242,0.01065169,0.7040747],"study_design_scores_gemma":[0.00006367786,0.0003520363,0.01039386,0.0001055736,0.0001456464,0.001705821,0.0002521195,0.4599018,0.4979132,0.003078558,0.0260181,0.00006966678],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1613789,0.002792609,0.7945351,0.0002972812,0.0003956809,0.0002584278,0.003949818,0.02651282,0.00987932],"genre_scores_gemma":[0.5093541,0.001502644,0.4659847,0.0002180483,0.0001338314,0.0000919236,0.004542947,0.0005724645,0.01759933],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009361632,"threshold_uncertainty_score":0.03131777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02134654737885951,"score_gpt":0.2956310126386067,"score_spread":0.2742844652597472,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}