{"id":"W3217141020","doi":"10.23889/ijpds.v6i1.1680","title":"Data Harmonization and Data Pooling from Cohort Studies: A Practical Approach for Data Management","year":2021,"lang":"en","type":"article","venue":"International Journal for Population Data Science","topic":"Gestational Diabetes Research and Management","field":"Medicine","cited_by":58,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; Alberta Health Services; University of Calgary","funders":"","keywords":"Pooling; Harmonization; Matching (statistics); Computer science; Variable (mathematics); Construct (python library); Data mining; Sample (material); Econometrics; Data science; Statistics; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.6267604,0.003607483,0.006567854,0.02001115,0.005663777,0.01561963,0.01146567,0.005296494,0.009628922],"category_scores_gemma":[0.7226273,0.005222689,0.009083062,0.03611077,0.007521647,0.01418403,0.02174757,0.0111064,0.00417192],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006762436,"about_ca_system_score_gemma":0.0298472,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004940528,"about_ca_topic_score_gemma":0.004186812,"domain_scores_codex":[0.2984462,0.5712999,0.07746933,0.02146066,0.02908644,0.002237434],"domain_scores_gemma":[0.2513324,0.3649854,0.04765401,0.2640867,0.06852608,0.003415294],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002186935,0.0004177278,0.01370543,0.01258626,0.008146875,0.0005646981,0.01936169,0.008919519,0.003076594,0.1683048,0.1147648,0.6479646],"study_design_scores_gemma":[0.002518883,0.000999396,0.01514942,0.01177507,0.003482406,0.0006909788,0.006762168,0.02886116,0.009201242,0.3983869,0.5210009,0.001171517],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002349287,0.0009805474,0.9500679,0.006066301,0.001235405,0.02886035,0.004972332,0.001850962,0.00361685],"genre_scores_gemma":[0.01496223,0.0004615817,0.9266344,0.00170523,0.0005849844,0.05137939,0.002814615,0.0005848078,0.0008727812],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3732396,"threshold_uncertainty_score":0.4602709,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3638508337347413,"score_gpt":0.5330212202993609,"score_spread":0.1691703865646197,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}