{"id":"W3081234131","doi":"10.48550/arxiv.2008.10549","title":"On sampling from data with duplicate records","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Data deduplication; Sampling (signal processing); Data mining; Set (abstract data type); Task (project management); Complement (music); Mathematical proof; Data set; Process (computing); Sample (material); Algorithm; Database; Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science","insufficient_payload"],"consensus_categories":["open_science"],"category_scores_codex":[0.001338673,0.0003304889,0.0005115792,0.000245782,0.0001743886,0.0004457552,0.006796081,0.0001776117,0.0006677337],"category_scores_gemma":[0.0008432708,0.0002875064,0.0001104066,0.00078538,0.000149235,0.0004928602,0.009293856,0.0006649972,0.001989133],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009077667,"about_ca_system_score_gemma":0.0001406107,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001280723,"about_ca_topic_score_gemma":0.0007045329,"domain_scores_codex":[0.9957086,0.0002864811,0.0003894616,0.002858339,0.0004673794,0.0002897839],"domain_scores_gemma":[0.9921615,0.001237496,0.0005057939,0.005713188,0.0001350082,0.0002469989],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002321169,0.0006121873,0.006199637,0.0001151558,0.001161197,0.001540402,0.0007300357,0.2354335,0.00002863823,0.5212761,0.2049723,0.0256096],"study_design_scores_gemma":[0.000761324,0.0001345933,0.003672158,0.0001605626,0.0002512955,5.881921e-7,0.0008598709,0.1028568,0.00001678534,0.7549387,0.1355874,0.0007599114],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1977017,0.0000272448,0.773966,0.002325878,0.0008703587,0.0005884669,0.0039599,0.0002567621,0.0203037],"genre_scores_gemma":[0.99189,0.0001098029,0.002014634,0.0009804431,0.0001504311,7.465139e-7,0.001363718,0.00002488965,0.0034654],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7941883,"threshold_uncertainty_score":0.9999577,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6244679694520039,"score_gpt":0.3281598972520859,"score_spread":0.2963080721999179,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}