{"id":"W3081234131","doi":"10.48550/arxiv.2008.10549","title":"On sampling from data with duplicate records","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Data deduplication; Sampling (signal processing); Data mining; Set (abstract data type); Task (project management); Complement (music); Mathematical proof; Data set; Process (computing); Sample (material); Algorithm; Database; Mathematics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04383586,0.001294517,0.002585879,0.002976133,0.001920656,0.003303645,0.004150317,0.002552241,0.0008500707],"category_scores_gemma":[0.1720881,0.001072141,0.001318015,0.003983992,0.003304449,0.004681507,0.003944127,0.002477954,0.0004628568],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001767099,"about_ca_system_score_gemma":0.002470954,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003446328,"about_ca_topic_score_gemma":0.003055244,"domain_scores_codex":[0.9718595,0.0179075,0.001357004,0.002913807,0.005271686,0.0006904701],"domain_scores_gemma":[0.7612144,0.1948305,0.008099731,0.0267623,0.007731534,0.00136152],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002743299,0.0007589539,0.05251346,0.001253802,0.0009130762,0.001337628,0.002533772,0.5167314,0.008417041,0.1217584,0.009197002,0.2818422],"study_design_scores_gemma":[0.0002197123,0.000246121,0.002694437,0.00009627349,0.00008487878,0.0006097568,0.0004155066,0.8998994,0.00594078,0.08712835,0.002609247,0.000055498],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05565191,0.0007735906,0.9406738,0.000870612,0.00007880998,0.000427894,0.0002667313,0.0005339644,0.000722605],"genre_scores_gemma":[0.4634541,0.0008510718,0.5302734,0.0008994615,0.0004050716,0.001013502,0.001744676,0.0001690759,0.001189651],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9561641,"threshold_uncertainty_score":0.231829,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6244679694520039,"score_gpt":0.3281598972520859,"score_spread":0.2963080721999179,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}