{"id":"W4387846873","doi":"10.1145/3583780.3615172","title":"Product Entity Matching via Tabular Data","year":2023,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Serialization; ENCODE; Transformer; Data mining; Matching (statistics); Benchmark (surveying); Entity linking; Task (project management); Product (mathematics); Artificial intelligence; Information retrieval; Natural language processing; Machine learning; Knowledge base; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002849416,0.00149443,0.001586989,0.005550422,0.001026671,0.003012413,0.00410716,0.002422271,0.007774111],"category_scores_gemma":[0.01459211,0.0007735272,0.002086285,0.009952033,0.0008076017,0.0117569,0.003394957,0.002339577,0.00660895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001954485,"about_ca_system_score_gemma":0.002432981,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01129754,"about_ca_topic_score_gemma":0.01841898,"domain_scores_codex":[0.997107,0.0006162824,0.0002525582,0.001256469,0.0006064667,0.0001612683],"domain_scores_gemma":[0.9943658,0.001905063,0.0004883831,0.002438616,0.000662289,0.0001399125],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009884211,0.000994813,0.01229932,0.001614418,0.0003865395,0.0005565758,0.000421163,0.1063364,0.007741526,0.02577467,0.1257705,0.7171157],"study_design_scores_gemma":[0.0001202652,0.0001954705,0.002001052,0.0001129319,0.0001008915,0.0004989314,0.0003232616,0.8632475,0.01348909,0.06385812,0.05597343,0.00007906794],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07297166,0.005481836,0.763299,0.002162581,0.0004722139,0.0008358103,0.07870391,0.06721416,0.008858738],"genre_scores_gemma":[0.2189506,0.001127125,0.5921454,0.001132133,0.0001405915,0.0004794726,0.1766217,0.0008794554,0.008523518],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01129754,"threshold_uncertainty_score":0.026007,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4807826226288973,"score_gpt":0.492133428387495,"score_spread":0.01135080575859776,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}