{"id":"W2091214390","doi":"10.1109/tkde.2012.220","title":"Bias Correction in a Small Sample from Big Data","year":2012,"lang":"en","type":"article","venue":"IEEE Transactions on Knowledge and Data Engineering","topic":"Complex Network Analysis Techniques","field":"Physics and Astronomy","cited_by":71,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Windsor","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Estimator; Sample size determination; Computer science; Sampling bias; Big data; Sample (material); Simple random sample; Population size; Sampling (signal processing); Statistics; Reciprocal; Population; Point estimation; Data mining; Mathematics; Telecommunications","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05336294,0.0009854879,0.00174783,0.002073823,0.001121138,0.002306399,0.002341002,0.001940022,0.001703827],"category_scores_gemma":[0.2569131,0.0008615612,0.001024927,0.002924754,0.002859082,0.004554228,0.002894252,0.002922851,0.0006555266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001225923,"about_ca_system_score_gemma":0.001474776,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002482749,"about_ca_topic_score_gemma":0.001973321,"domain_scores_codex":[0.9725424,0.01825464,0.001158369,0.003457364,0.004024589,0.0005627336],"domain_scores_gemma":[0.8127549,0.1548711,0.007563912,0.01717614,0.006913932,0.0007199163],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001083659,0.0002604633,0.06111751,0.001768613,0.00161141,0.001844329,0.002093448,0.1871778,0.005235617,0.3958166,0.01871389,0.3232767],"study_design_scores_gemma":[0.0001958502,0.0002881341,0.01055522,0.0003641165,0.0002723013,0.000863247,0.0003766689,0.5855778,0.006137665,0.3788186,0.01642889,0.0001215587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01303547,0.001106972,0.9824145,0.001067668,0.000458809,0.0001576751,0.0001773003,0.0003687177,0.001212827],"genre_scores_gemma":[0.5584185,0.001590315,0.4293938,0.003085807,0.001200574,0.001028674,0.0008122456,0.0003820747,0.004088202],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05336294,"threshold_uncertainty_score":0.2822136,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1054743786026927,"score_gpt":0.291973674795889,"score_spread":0.1864992961931962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}