{"id":"W3146762582","doi":"10.18653/v1/2022.nlpcss-1.19","title":"OLALA: Object-Level Active Learning for Efficient Document Layout Annotation","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Harvard Data Science Initiative, Harvard University; Harvard Catalyst; National Science Foundation","keywords":"Annotation; Computer science; Automatic image annotation; Artificial intelligence; Object (grammar); Information retrieval; Process (computing); Active learning (machine learning); Image retrieval; Machine learning; Image (mathematics); Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0008281915,0.0003412892,0.0003633822,0.0003885537,0.0003756181,0.0004169062,0.001167856,0.0001962425,0.000350522],"category_scores_gemma":[0.0001707737,0.000349606,0.0002673744,0.0002797917,0.00003088428,0.0002010814,0.00245357,0.0008897526,0.00004118641],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005099454,"about_ca_system_score_gemma":0.0003039617,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001102354,"about_ca_topic_score_gemma":0.000009736071,"domain_scores_codex":[0.9972406,0.0002347946,0.0004552569,0.00106998,0.0005976647,0.0004017243],"domain_scores_gemma":[0.9982009,0.0003025827,0.0003961125,0.0006472024,0.0003542856,0.00009894966],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000969328,0.0005182066,0.0001041193,0.0004692452,0.0003121848,0.00002504812,0.01120255,0.05268479,0.0009761363,0.05107788,0.007609104,0.8749238],"study_design_scores_gemma":[0.002914045,0.001642194,0.002904014,0.0004873264,0.0001694671,0.00004172869,0.002364909,0.6538634,0.1523476,0.1113236,0.06816313,0.003778717],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006646287,0.0000458294,0.9800655,0.0005780223,0.0007460134,0.001764808,0.00004578927,0.001327849,0.008779899],"genre_scores_gemma":[0.5636327,0.00004110451,0.4243805,0.0005252004,0.0001850668,0.004126269,0.0004447947,0.00006218256,0.006602136],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8711451,"threshold_uncertainty_score":0.9998956,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0334919703407659,"score_gpt":0.3054129292682622,"score_spread":0.2719209589274963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}