{"id":"W3016677020","doi":"10.1109/access.2020.2988120","title":"Sampling for Big Data Profiling: A Survey","year":2020,"lang":"en","type":"article","venue":"IEEE Access","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":32,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"National Natural Science Foundation of China","keywords":"Big data; Profiling (computer programming); Computer science; Data mining; Metadata; Data science; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"not_applicable","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"review","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009235054,0.002320996,0.003272566,0.006415248,0.001461857,0.005181494,0.003898319,0.002340372,0.002397871],"category_scores_gemma":[0.02893629,0.001326239,0.00219915,0.01348204,0.001909497,0.009626735,0.002763436,0.002442328,0.00179735],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001425382,"about_ca_system_score_gemma":0.002566389,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003118262,"about_ca_topic_score_gemma":0.001324687,"domain_scores_codex":[0.9881016,0.003786268,0.001172672,0.00182629,0.004706936,0.0004061808],"domain_scores_gemma":[0.9721709,0.01894911,0.001303959,0.002916946,0.003985537,0.0006735311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002499871,0.0001893047,0.01084749,0.006084358,0.000208054,0.0002008488,0.0005145262,0.009525438,0.001670759,0.04009861,0.02344772,0.9069629],"study_design_scores_gemma":[0.0001181491,0.000847649,0.0166855,0.007554927,0.0005013512,0.003457722,0.002967736,0.1571342,0.0112756,0.1731241,0.6258777,0.0004553836],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.008745376,0.6270015,0.3370054,0.00833048,0.002023998,0.0005949237,0.001279431,0.001844104,0.01317484],"genre_scores_gemma":[0.1316944,0.6642568,0.1865781,0.003711644,0.005896421,0.0008457823,0.003492766,0.0005711463,0.002952925],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.009235054,"threshold_uncertainty_score":0.04884022,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9190001080756124,"score_gpt":0.5835216452442223,"score_spread":0.33547846283139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}