{"id":"W4309617049","doi":"10.1145/3571281","title":"Efficiently Cleaning Structured Event Logs: A Graph Repair Approach","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Database Systems","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Beijing National Research Center For Information Science And Technology; National Natural Science Foundation of China","keywords":"Computer science; Workflow; Event (particle physics); Pruning; Data mining; Graph; Profiling (computer programming); Information retrieval; Theoretical computer science; Database; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002421997,0.001803952,0.001477332,0.003763601,0.001701707,0.001418174,0.003188455,0.001676271,0.001925949],"category_scores_gemma":[0.01445816,0.0008608645,0.002058743,0.003386246,0.001402858,0.003526282,0.002380341,0.002210292,0.00107113],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00105662,"about_ca_system_score_gemma":0.002783895,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008777151,"about_ca_topic_score_gemma":0.01302549,"domain_scores_codex":[0.9968985,0.0008022708,0.0002484814,0.0007986059,0.00104486,0.0002073245],"domain_scores_gemma":[0.9857898,0.006698771,0.001362722,0.003988374,0.001871797,0.0002885988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004402264,0.0005072728,0.007456467,0.001112413,0.0002321999,0.001051079,0.001064945,0.2851672,0.02856133,0.01839515,0.02607188,0.6299399],"study_design_scores_gemma":[0.00006552428,0.0001579421,0.001329349,0.0000515255,0.000122218,0.0005202333,0.0004784882,0.9183147,0.0157549,0.0504079,0.01274408,0.0000532913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01294394,0.0003162917,0.978907,0.0004995849,0.00006408377,0.0002850334,0.0008466721,0.005584656,0.0005527845],"genre_scores_gemma":[0.1093605,0.0002941921,0.8830175,0.000243858,0.0000641066,0.0002193214,0.004326333,0.0006688405,0.001805258],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008777151,"threshold_uncertainty_score":0.01745212,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1167959570941646,"score_gpt":0.3589782289255208,"score_spread":0.2421822718313563,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}