{"id":"W7143899963","doi":"10.71465/ajdsa906","title":"Understanding Data Bias: Challenges and Solutions in Data Science","year":2020,"lang":"","type":"article","venue":"American Journal of Data Science and Analysis","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Data collection; Noisy data; Big data; Data analysis; Data modeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4607471,0.001777129,0.006003413,0.008839672,0.00942101,0.02445631,0.007647936,0.01968492,0.004068536],"category_scores_gemma":[0.7098413,0.00249273,0.002598613,0.01234284,0.06610866,0.04335421,0.02036359,0.02963466,0.001178555],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01348739,"about_ca_system_score_gemma":0.0318178,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006196372,"about_ca_topic_score_gemma":0.004427556,"domain_scores_codex":[0.4769716,0.3948701,0.03134593,0.02712338,0.06596321,0.003725864],"domain_scores_gemma":[0.1005423,0.8161301,0.02461507,0.03046182,0.02571876,0.002532007],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001226856,0.0000664578,0.005985573,0.002483791,0.0003892399,0.0003530692,0.006330281,0.002283697,0.0002129519,0.8797716,0.01957023,0.08243028],"study_design_scores_gemma":[0.00004751855,0.00002198046,0.0004029751,0.002007282,0.0000459663,0.0001741716,0.001052998,0.001972973,0.0001621313,0.9739758,0.02007555,0.00006074562],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.004840122,0.04238585,0.300213,0.6361772,0.005216477,0.000591935,0.0005624418,0.000233919,0.009778994],"genre_scores_gemma":[0.3626566,0.0527437,0.3430435,0.2092663,0.02450847,0.003754275,0.0008152603,0.0006058192,0.002606159],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5392529,"threshold_uncertainty_score":0.6649948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8680911416230382,"score_gpt":0.5203191606530488,"score_spread":0.3477719809699894,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}