{"id":"W4306317280","doi":"10.1145/3511808.3557673","title":"Probing the Robustness of Pre-trained Language Models for Entity Matching","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 31st ACM International Conference on Information &amp; Knowledge Management","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Robustness (evolution); Computer science; Spurious relationship; Machine learning; Software deployment; Artificial intelligence; Training set; Data modeling; Data mining; Database; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01451214,0.001844272,0.001342107,0.00141216,0.0007871662,0.002051615,0.003386759,0.002768425,0.001631414],"category_scores_gemma":[0.05660156,0.0009723156,0.001380274,0.001205,0.001427272,0.005574816,0.002873233,0.004939722,0.001456217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001886142,"about_ca_system_score_gemma":0.001594037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01177434,"about_ca_topic_score_gemma":0.01012354,"domain_scores_codex":[0.9945226,0.003037516,0.0003481266,0.001288086,0.0004356004,0.0003681639],"domain_scores_gemma":[0.9729321,0.02054076,0.0008278341,0.00400708,0.001359307,0.000332793],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005812267,0.0002535181,0.008761658,0.0002139393,0.0003999042,0.0001435439,0.0002010664,0.8857884,0.00315513,0.003477528,0.00369897,0.09332519],"study_design_scores_gemma":[0.00001451461,0.00005488487,0.0005541435,0.00001743771,0.00002330017,0.00003685615,0.00003816586,0.9941232,0.002288756,0.002405498,0.0004303328,0.00001303004],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4196748,0.003356611,0.561591,0.002892599,0.0004462067,0.0002984833,0.00159216,0.005446408,0.004701816],"genre_scores_gemma":[0.9158394,0.0005412792,0.07734504,0.0007388259,0.0001499876,0.0001688178,0.00310438,0.0003184832,0.001793719],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01451214,"threshold_uncertainty_score":0.07674843,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1525473395701562,"score_gpt":0.3846648573911663,"score_spread":0.2321175178210101,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}