{"id":"W4393183687","doi":"10.1145/3639279","title":"DTT: An Example-Driven Tabular Transformer for Joinability by Leveraging Large Language Models","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on Management of Data","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Transformer; Computer science; Engineering; Electrical engineering; Voltage","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001350108,0.00101697,0.0007322612,0.001430152,0.0005445267,0.001773986,0.002506195,0.0009190601,0.01073176],"category_scores_gemma":[0.006887335,0.0005499555,0.001753532,0.001671582,0.000907535,0.006842368,0.002464308,0.00242467,0.005227981],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001117516,"about_ca_system_score_gemma":0.002265301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006429025,"about_ca_topic_score_gemma":0.01094178,"domain_scores_codex":[0.9990103,0.000188743,0.00008800172,0.0002983102,0.0003332073,0.00008137789],"domain_scores_gemma":[0.9977513,0.0008313867,0.0001592505,0.0008287902,0.0003334803,0.00009578584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005623537,0.0004250442,0.004315146,0.0004880039,0.0001202508,0.0003965413,0.0005003671,0.1052744,0.01323588,0.05652421,0.06684957,0.7513082],"study_design_scores_gemma":[0.00005408286,0.00008074307,0.0002798055,0.00003499546,0.0000325621,0.0001611911,0.00009108228,0.9183519,0.01312005,0.04855638,0.01920556,0.00003163437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01483866,0.0002956834,0.9390623,0.0003892163,0.0001514832,0.0001771698,0.003214503,0.0384702,0.003400696],"genre_scores_gemma":[0.2596592,0.0004729463,0.7078319,0.0006330098,0.0001268195,0.0003470774,0.01802761,0.003564385,0.009337069],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01073176,"threshold_uncertainty_score":0.03590131,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1034483500304466,"score_gpt":0.3135535429041299,"score_spread":0.2101051928736833,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}