{"id":"W3125002781","doi":"10.5539/gjhs.v13n3p23","title":"Data Cleaning Needs and Issues: A Case Study of the National Reproductive Health Assessment (RHA) Data from Solomon Islands","year":2021,"lang":"en","type":"article","venue":"Global Journal of Health Science","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Data collection; Reproductive health; Standardization; Data entry; Reliability (semiconductor); Process (computing); Research data; Medicine; Computer science; Environmental health; Database; Data science; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04136124,0.0001097472,0.0004439347,0.0001562799,0.0007762265,0.0004133706,0.004222566,0.00001817011,0.00002322566],"category_scores_gemma":[0.004006505,0.00006940771,0.00002588677,0.002399192,0.0003628612,0.002018164,0.005234852,0.0002302502,0.000001942746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002453907,"about_ca_system_score_gemma":0.004058692,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003098854,"about_ca_topic_score_gemma":0.002485126,"domain_scores_codex":[0.9919685,0.0009033443,0.001582953,0.0008398364,0.004388697,0.0003166351],"domain_scores_gemma":[0.9941612,0.0003166852,0.001950143,0.002310756,0.001045287,0.0002158841],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001776742,0.004257038,0.2886874,0.0001140109,0.0002776987,0.0008854782,0.02775365,0.0008345645,0.00008407448,0.005276958,0.3620891,0.3095623],"study_design_scores_gemma":[0.00229247,0.001434104,0.3550991,0.0002996509,0.00005654976,0.005535294,0.5798387,0.009016417,0.00001454791,0.01651241,0.0296115,0.0002892101],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9359757,0.002704979,0.005931756,0.04905825,0.002014976,0.0005636304,0.003012459,0.000008967091,0.0007292969],"genre_scores_gemma":[0.991123,0.0001607892,0.006628746,0.001909917,0.0001331878,4.630876e-7,0.00001893168,0.000002229136,0.00002276734],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.552085,"threshold_uncertainty_score":0.9871203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4862324880655968,"score_gpt":0.5819926248280866,"score_spread":0.09576013676248973,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}