{"id":"W3061846580","doi":"10.48550/arxiv.2008.07680","title":"An Annotated Corpus of Webtables for Information Extraction Tasks","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Information retrieval; Annotation; Relationship extraction; Paragraph; Information extraction; Natural language processing; Task (project management); Benchmark (surveying); Table (database); Sentence; Artificial intelligence; Context (archaeology); Data mining; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001275477,0.001042713,0.0007142268,0.007687509,0.001547062,0.001656754,0.001623387,0.001478933,0.01185267],"category_scores_gemma":[0.01012268,0.0004827264,0.000815624,0.0113582,0.0006403762,0.002640476,0.001820845,0.001195776,0.01093466],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001102519,"about_ca_system_score_gemma":0.002525213,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01145517,"about_ca_topic_score_gemma":0.02470538,"domain_scores_codex":[0.9976625,0.0004994734,0.0003121386,0.0007125176,0.0006851763,0.0001281762],"domain_scores_gemma":[0.9924566,0.003069074,0.0005148007,0.001590081,0.001997074,0.0003725008],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006971241,0.0005048395,0.01084625,0.007202821,0.0002146482,0.002237097,0.002215736,0.004885963,0.02067156,0.01100057,0.7351073,0.204416],"study_design_scores_gemma":[0.000107238,0.0001155426,0.03076,0.0007124179,0.0001190851,0.001550549,0.001338167,0.009085068,0.0164633,0.006578849,0.9330496,0.0001201016],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.05293377,0.004267576,0.03895419,0.0008855636,0.0006357269,0.000699511,0.8662207,0.01106884,0.02433412],"genre_scores_gemma":[0.03205942,0.0007985174,0.03805425,0.0001756521,0.00009139688,0.000721174,0.9236726,0.0006519242,0.003775116],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01185267,"threshold_uncertainty_score":0.03965116,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09069754288968365,"score_gpt":0.2105369483483559,"score_spread":0.1198394054586722,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}