{"id":"W4393327679","doi":"10.1016/j.knosys.2024.111678","title":"Better entity matching with transformers through ensembles","year":2024,"lang":"en","type":"article","venue":"Knowledge-Based Systems","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"National Research Council Canada; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Transformer; Computer science; Matching (statistics); Artificial intelligence; Data mining; Engineering; Mathematics; Electrical engineering; Statistics; Voltage","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004668405,0.0008203154,0.001608372,0.003002418,0.0009831013,0.003387756,0.001617782,0.001183905,0.004134153],"category_scores_gemma":[0.02268433,0.0007105339,0.001671008,0.004244291,0.0009197753,0.01127105,0.003856178,0.001914878,0.001341263],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000948277,"about_ca_system_score_gemma":0.002050785,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005668989,"about_ca_topic_score_gemma":0.008642,"domain_scores_codex":[0.9956363,0.00144875,0.0004101607,0.00100573,0.001154289,0.0003448183],"domain_scores_gemma":[0.9891873,0.004662211,0.0004486483,0.004111233,0.001331986,0.0002585085],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007704294,0.0004227483,0.009874784,0.0002947068,0.0005143831,0.0003156723,0.0004594744,0.1744857,0.01198602,0.1089182,0.009344106,0.6826138],"study_design_scores_gemma":[0.0000340373,0.00007485404,0.000894268,0.00004529842,0.0001660743,0.0001474424,0.0001088961,0.8247283,0.01200303,0.1571428,0.004624852,0.00003010291],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03019156,0.0003478447,0.9649071,0.0003242312,0.00008206352,0.00007849749,0.0003604502,0.002091489,0.001616741],"genre_scores_gemma":[0.6343021,0.0005912122,0.360061,0.0002349615,0.0001215408,0.00007209874,0.001672411,0.0003958539,0.002548841],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005668989,"threshold_uncertainty_score":0.02468914,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.122142341677316,"score_gpt":0.3903886435702951,"score_spread":0.2682463018929792,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}