{"id":"W4387846857","doi":"10.1145/3583780.3615046","title":"SAND: Semantic Annotation of Numeric Data in Web Tables","year":2023,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Information retrieval; Semantic Web; Annotation; Semantic Web Stack; Semantic annotation; Graph; Margin (machine learning); Knowledge graph; Social Semantic Web; Schema (genetic algorithms); Natural language processing; Artificial intelligence; Machine learning; Theoretical computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001667247,0.0008993847,0.0006836584,0.007594217,0.0009883944,0.002842667,0.001399412,0.001150107,0.005126316],"category_scores_gemma":[0.009161503,0.000485559,0.001028524,0.006293756,0.0009189948,0.005075597,0.002805642,0.0011531,0.002531028],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001244439,"about_ca_system_score_gemma":0.002125194,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01062925,"about_ca_topic_score_gemma":0.01888213,"domain_scores_codex":[0.9974981,0.0004889703,0.000292536,0.0005762072,0.001020259,0.0001239699],"domain_scores_gemma":[0.9945083,0.002009208,0.0007010821,0.001483698,0.001121204,0.000176509],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009177626,0.000325705,0.03409998,0.002891867,0.0002726867,0.001664529,0.004266237,0.04582574,0.03373033,0.1320528,0.1393416,0.6046107],"study_design_scores_gemma":[0.00006404888,0.0001061068,0.0146334,0.0006824444,0.0001633683,0.001133845,0.002639145,0.2742575,0.05787477,0.1318529,0.5163843,0.0002081904],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02772805,0.0007624301,0.8749163,0.0008568551,0.0002886484,0.0004414385,0.0448832,0.03775886,0.0123642],"genre_scores_gemma":[0.2009404,0.0008115098,0.7384462,0.0004629672,0.00009557977,0.0004293654,0.05107197,0.001766284,0.005975802],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01062925,"threshold_uncertainty_score":0.02113473,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3297667459228645,"score_gpt":0.4672587250817268,"score_spread":0.1374919791588622,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}