{"id":"W7126398324","doi":"10.21428/594757db.564cdb72","title":"Comparing Traditional and Deep Learning Approaches for Product Matching: Performance on Unseen Entities","year":2025,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Deep learning; Feature learning; Autoencoder; Matching (statistics); Transformer; Representation (politics); Feature (linguistics); Training set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001967291,0.00007619485,0.0001433133,0.0001281933,0.0003128644,0.0003271636,0.0002588725,0.0000148971,0.00005581655],"category_scores_gemma":[0.0002651717,0.00005688895,0.00003220201,0.0001545602,0.00006769116,0.0003437066,0.00008278264,0.0000789222,0.0000227606],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001351819,"about_ca_system_score_gemma":0.00001252087,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001174698,"about_ca_topic_score_gemma":0.00002913619,"domain_scores_codex":[0.9988751,0.00006333978,0.0002454614,0.0003289299,0.0003577845,0.0001293589],"domain_scores_gemma":[0.9991367,0.0005644966,0.00005638179,0.0001845018,0.00003317085,0.00002478283],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000062714,0.00008775246,0.009160319,0.00009826811,0.0000335882,1.944603e-7,0.000871093,0.005503851,0.000008549478,0.8427467,0.007370448,0.1340565],"study_design_scores_gemma":[0.001051052,0.0002440089,0.2380364,0.00009899207,0.00003888267,0.000002257409,0.01506792,0.162349,0.0007479158,0.2622742,0.3197055,0.0003838025],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6825736,0.0001207999,0.134731,0.003091281,0.0003340344,0.0004861569,0.000009943363,0.00008488228,0.1785683],"genre_scores_gemma":[0.9732118,0.00001434911,0.002184322,0.0002769667,0.00004024998,0.00003214327,0.00002796823,0.000002716862,0.02420953],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5804725,"threshold_uncertainty_score":0.3154846,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4048250354532657,"score_gpt":0.3676819528872744,"score_spread":0.0371430825659913,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}