{"id":"W4399174291","doi":"10.1145/3654963","title":"OTClean: Data Cleaning for Conditional Independence Violations using Optimal Transport","year":2024,"lang":"en","type":"article","venue":"Proceedings of the ACM on Management of Data","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Universitas Brawijaya","keywords":"Computer science; Scalability; Mathematical optimization; Independence (probability theory); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009386571,0.001605415,0.002095251,0.001930419,0.001767949,0.002945765,0.003806895,0.002607432,0.002827798],"category_scores_gemma":[0.0320976,0.001065605,0.002668152,0.002563363,0.003632285,0.005997322,0.007385392,0.005768319,0.0011272],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001953521,"about_ca_system_score_gemma":0.00621663,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005121819,"about_ca_topic_score_gemma":0.004157647,"domain_scores_codex":[0.9936463,0.002311984,0.0004228577,0.001250626,0.001959227,0.0004090894],"domain_scores_gemma":[0.9832876,0.008246134,0.001619218,0.004411487,0.002053997,0.0003816006],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000361282,0.0001622498,0.002603657,0.0004787413,0.0002278738,0.0003282394,0.0005967455,0.6314083,0.01255909,0.09293812,0.009314306,0.2490214],"study_design_scores_gemma":[0.00002374032,0.00008232993,0.0002940556,0.00003725482,0.00002075147,0.0001242762,0.0001145595,0.9252272,0.00764299,0.06267842,0.003722261,0.00003199781],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00174141,0.0000663337,0.997221,0.0001358009,0.00002032716,0.00003859871,0.00005106481,0.000540071,0.0001854129],"genre_scores_gemma":[0.1593981,0.0002945233,0.8351722,0.0004357428,0.0001231696,0.0004202596,0.0009657904,0.000968841,0.002221215],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009386571,"threshold_uncertainty_score":0.04964149,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1243438065986618,"score_gpt":0.3641923593716973,"score_spread":0.2398485527730355,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}