{"id":"W3176788992","doi":"10.1109/icde51399.2021.00116","title":"Automating Entity Matching Model Development","year":2021,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Pipeline (software); Bottleneck; Benchmark (surveying); Computer science; Process (computing); Artificial intelligence; Machine learning; Matching (statistics); Task (project management); Engineering; Systems engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009628436,0.001714465,0.001619564,0.003534954,0.001481521,0.003198789,0.004584682,0.002106563,0.007065902],"category_scores_gemma":[0.02682208,0.00126883,0.002995771,0.003765033,0.001181574,0.008584389,0.005552212,0.003449859,0.005044009],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001877125,"about_ca_system_score_gemma":0.003855443,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005882613,"about_ca_topic_score_gemma":0.009301748,"domain_scores_codex":[0.9939976,0.002586809,0.0003350348,0.001693318,0.001089169,0.0002981109],"domain_scores_gemma":[0.9883195,0.005228576,0.0005264407,0.004307766,0.001420996,0.0001967073],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000270244,0.0003712824,0.007345866,0.0006573062,0.00026644,0.0004183559,0.0005760167,0.1152743,0.008717493,0.04264997,0.04201042,0.7814423],"study_design_scores_gemma":[0.00004972992,0.0000693948,0.0009434025,0.00006430119,0.00007664898,0.0003488619,0.0002531575,0.8833266,0.01716389,0.06137499,0.03627868,0.00005039459],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009527304,0.0003384913,0.9709768,0.0006275515,0.00005761081,0.0002106745,0.001057851,0.01554398,0.001659777],"genre_scores_gemma":[0.1156253,0.0003404389,0.8712591,0.0006026545,0.00004950752,0.0002386939,0.007411773,0.001421303,0.003051174],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009628436,"threshold_uncertainty_score":0.05092061,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3130788376251375,"score_gpt":0.4498523623238104,"score_spread":0.1367735246986729,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}