{"id":"W2795302121","doi":"10.1109/icde.2018.00093","title":"Seeping Semantics: Linking Datasets Using Word Embeddings for Data Discovery","year":2018,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":89,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Schema (genetic algorithms); Knowledge graph; Semantics (computer science); Linked data; Information retrieval; Semantic Web; Word (group theory); RDF; Distributional semantics; Natural language processing; Semantic similarity; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004435755,0.001795644,0.001191804,0.01192542,0.001277448,0.003274195,0.001859354,0.001649603,0.002998935],"category_scores_gemma":[0.02616591,0.0008620199,0.002161444,0.01185927,0.001024685,0.01253505,0.006335652,0.001698093,0.001803649],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007318583,"about_ca_system_score_gemma":0.001849453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003607118,"about_ca_topic_score_gemma":0.007783853,"domain_scores_codex":[0.9957366,0.001571976,0.0006330385,0.0009750287,0.0009039929,0.0001793107],"domain_scores_gemma":[0.9913467,0.004072045,0.00086973,0.00249244,0.0009294658,0.000289587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005347961,0.0005686339,0.01548976,0.001469069,0.0006515134,0.0004100766,0.002524017,0.01720436,0.007378708,0.04172926,0.01615389,0.8958858],"study_design_scores_gemma":[0.0002011823,0.0005898735,0.006553609,0.0004940627,0.0005088464,0.001072826,0.003896886,0.4631914,0.01733811,0.4306353,0.0752804,0.0002374408],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02644702,0.0009132492,0.955642,0.0006655185,0.0001960143,0.000512068,0.00446223,0.009083256,0.002078697],"genre_scores_gemma":[0.07890739,0.0003946026,0.9115608,0.0002003725,0.00005021035,0.0004034221,0.007163353,0.0004346729,0.0008851273],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01192542,"threshold_uncertainty_score":0.02345878,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4932316803399396,"score_gpt":0.519707133187662,"score_spread":0.0264754528477224,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}